[llvm] AMDGPU: VALU data fast-forwarding needs no s_delay (PR #205481)
Reem Elkhouly via llvm-commits
llvm-commits at lists.llvm.org
Sun Sep 13 19:29:55 PDT 2026
https://github.com/amd-relkhoul updated https://github.com/llvm/llvm-project/pull/205481
>From 896bf43f9dfa9cd99f8de58dd34facf5b37bb148 Mon Sep 17 00:00:00 2001
From: Reem Elkhouly <reem.elkhouly at amd.com>
Date: Thu, 3 Sep 2026 12:55:18 +0900
Subject: [PATCH 1/3] [AMDGPU] VALU data fast-forwarding needs no s_delay
---
.../Target/AMDGPU/AMDGPUInsertDelayAlu.cpp | 215 +++++++++++++++++-
...delay-alu-skipped-with-fast-forwarding.mir | 176 ++++++++++++++
2 files changed, 383 insertions(+), 8 deletions(-)
create mode 100644 llvm/test/CodeGen/AMDGPU/insert-delay-alu-skipped-with-fast-forwarding.mir
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUInsertDelayAlu.cpp b/llvm/lib/Target/AMDGPU/AMDGPUInsertDelayAlu.cpp
index 5b70cde09d9ca8..76de039e07e950 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUInsertDelayAlu.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUInsertDelayAlu.cpp
@@ -12,6 +12,7 @@
//===----------------------------------------------------------------------===//
#include "AMDGPU.h"
+#include "AMDGPULaneMaskUtils.h"
#include "GCNSubtarget.h"
#include "SIInstrInfo.h"
#include "SIMachineFunctionInfo.h"
@@ -117,9 +118,13 @@ class AMDGPUInsertDelayAlu {
// until it completes.
uint8_t SALUCycles = 0;
+ // Set when this entry was produced by a data fast-forward producer
+ bool IsFFProducer = false;
+
DelayInfo() = default;
- DelayInfo(DelayType Type, unsigned Cycles) {
+ DelayInfo(DelayType Type, unsigned Cycles, bool IsFFProducer = false)
+ : IsFFProducer(IsFFProducer) {
switch (Type) {
default:
llvm_unreachable("unexpected type");
@@ -143,7 +148,8 @@ class AMDGPUInsertDelayAlu {
bool operator==(const DelayInfo &RHS) const {
return VALUCycles == RHS.VALUCycles && VALUNum == RHS.VALUNum &&
TRANSCycles == RHS.TRANSCycles && TRANSNum == RHS.TRANSNum &&
- TRANSNumVALU == RHS.TRANSNumVALU && SALUCycles == RHS.SALUCycles;
+ TRANSNumVALU == RHS.TRANSNumVALU && SALUCycles == RHS.SALUCycles &&
+ IsFFProducer == RHS.IsFFProducer;
}
bool operator!=(const DelayInfo &RHS) const { return !(*this == RHS); }
@@ -157,6 +163,7 @@ class AMDGPUInsertDelayAlu {
TRANSNum = std::min(TRANSNum, RHS.TRANSNum);
TRANSNumVALU = std::min(TRANSNumVALU, RHS.TRANSNumVALU);
SALUCycles = std::max(SALUCycles, RHS.SALUCycles);
+ IsFFProducer = IsFFProducer && RHS.IsFFProducer;
}
// Update this DelayInfo after issuing an instruction of the specified type.
@@ -344,6 +351,169 @@ class AMDGPUInsertDelayAlu {
return (Imm & 0x780) ? nullptr : DelayAlu;
}
+ static bool isFastForwardProducer(const MachineInstr &MI,
+ const MachineOperand &MO) {
+ assert((MO.isReg() && MO.isDef()) && "Expected a register definition");
+
+ if (!SIInstrInfo::isVALU(MI, /*AllowLDSDMA=*/false))
+ return false;
+
+ const TargetRegisterInfo *TRI =
+ MI.getMF()->getSubtarget().getRegisterInfo();
+ Register Reg = MO.getReg();
+ if (AMDGPU::isSGPR(Reg, TRI)) {
+ switch (MI.getOpcode()) {
+ // VOP2 carry-out producers (implicit VCC)
+ case AMDGPU::V_ADD_CO_U32_e32:
+ case AMDGPU::V_SUB_CO_U32_e32:
+ case AMDGPU::V_SUBREV_CO_U32_e32:
+ case AMDGPU::V_ADDC_U32_e32:
+ case AMDGPU::V_SUBB_U32_e32:
+ case AMDGPU::V_SUBBREV_U32_e32:
+ // VOP3 carry-out producers (explicit VCC/SGPR)
+ case AMDGPU::V_ADD_CO_U32_e64:
+ case AMDGPU::V_SUB_CO_U32_e64:
+ case AMDGPU::V_SUBREV_CO_U32_e64:
+ case AMDGPU::V_ADDC_U32_e64:
+ case AMDGPU::V_SUBB_U32_e64:
+ case AMDGPU::V_SUBBREV_U32_e64:
+ case AMDGPU::V_DIV_SCALE_F32_e64:
+ case AMDGPU::V_DIV_SCALE_F64_e64:
+ return true;
+ default:
+ if (MI.isCompare())
+ return true;
+ }
+ }
+ return false;
+ }
+
+ static unsigned int getVOPDComponentOpCode(const MachineInstr &MI,
+ unsigned OpNo,
+ const MachineOperand &MO) {
+ Register Reg = MO.getReg();
+ auto MIOpCode = MI.getOpcode();
+ // Get the component instruction descriptors for VOPD.
+ auto [OpX, OpY] = AMDGPU::getVOPDComponents(MIOpCode);
+ const MCInstrInfo *MCII = MI.getMF()->getTarget().getMCInstrInfo();
+
+ if (MO.isImplicit()) {
+ const TargetRegisterInfo *TRI =
+ MI.getMF()->getSubtarget().getRegisterInfo();
+ for (unsigned CompOp : {OpX, OpY}) {
+ const MCInstrDesc &CompDesc = MCII->get(CompOp);
+ bool UsesReg = any_of(CompDesc.implicit_uses(), [&](MCPhysReg R) {
+ return TRI->regsOverlap(R, Reg);
+ });
+ if (UsesReg) {
+ MIOpCode = CompOp;
+ break;
+ }
+ }
+ } else {
+ // Explicit source: identify which component/source THIS operand is by
+ // matching its operand index (OpNo). The same register can appear in
+ // multiple slots, so matching by register value is not reliable.
+ const auto &InstInfo = AMDGPU::getVOPDInstInfo(MIOpCode, MCII);
+
+ // Map a parsed source index (0/1/2) to the matching named operand for a
+ // standalone component opcode.
+ static constexpr AMDGPU::OpName SrcNames[] = {
+ AMDGPU::OpName::src0, AMDGPU::OpName::src1, AMDGPU::OpName::src2};
+
+ bool Found = false;
+ for (auto CompIdx : AMDGPU::VOPD::COMPONENTS) {
+ const auto &CInfo = InstInfo[CompIdx];
+ unsigned CompSrcOperandsNum = CInfo.getCompParsedSrcOperandsNum();
+ for (unsigned CompSrcIdx = 0; CompSrcIdx < CompSrcOperandsNum;
+ ++CompSrcIdx) {
+ if (CInfo.getIndexOfSrcInParsedOperands(CompSrcIdx) != OpNo)
+ continue;
+
+ // Re-target analysis to this component's standalone opcode.
+ MIOpCode = (CompIdx == AMDGPU::VOPD::X) ? OpX : OpY;
+
+ // Translate the parsed src index into MIOpCode's operand-index
+ assert(CompSrcIdx < std::size(SrcNames) &&
+ "unexpected VOPD src index");
+ int NamedIdx =
+ AMDGPU::getNamedOperandIdx(MIOpCode, SrcNames[CompSrcIdx]);
+ if (NamedIdx >= 0)
+ OpNo = static_cast<unsigned>(NamedIdx);
+ Found = true;
+ break;
+ }
+ if (Found)
+ break;
+ }
+ }
+ return MIOpCode;
+ }
+
+ static bool isFastForwardConsumer(const MachineInstr &MI,
+ const MachineOperand &MO, Register VccReg,
+ Register ExecReg, unsigned OpNo) {
+ assert((MO.isReg() && MO.isUse()) && "Expected a register use");
+
+ if (!SIInstrInfo::isVALU(MI, /*AllowLDSDMA=*/false))
+ return false;
+
+ const TargetRegisterInfo *TRI =
+ MI.getMF()->getSubtarget().getRegisterInfo();
+ Register Reg = MO.getReg();
+ auto MIOpCode = MI.getOpcode();
+
+ if (AMDGPU::isVOPD(MIOpCode)) {
+ MIOpCode = getVOPDComponentOpCode(MI, OpNo, MO);
+ }
+
+ if (TRI->isSubRegisterEq(Reg, VccReg)) {
+ switch (MIOpCode) {
+ case AMDGPU::V_DIV_FMAS_F32_e64:
+ case AMDGPU::V_DIV_FMAS_F64_e64:
+ return false;
+
+ // VOP2 implicit carry-in consumers
+ case AMDGPU::V_ADDC_U32_e32:
+ case AMDGPU::V_SUBB_U32_e32:
+ case AMDGPU::V_SUBBREV_U32_e32:
+ // VOP2 implicit condition mask consumers
+ case AMDGPU::V_CNDMASK_B32_e32:
+ case AMDGPU::V_CNDMASK_B16_fake16_e32:
+ case AMDGPU::V_CNDMASK_B16_t16_e32:
+ return MO.isImplicit();
+ }
+ } else if (TRI->isSubRegisterEq(Reg, ExecReg)) {
+ switch (MIOpCode) {
+ case AMDGPU::V_READLANE_B32:
+ case AMDGPU::V_READFIRSTLANE_B32:
+ case AMDGPU::V_WRITELANE_B32:
+ return false;
+ default:
+ // Any other VALU
+ return MO.isImplicit();
+ }
+ }
+
+ if (AMDGPU::isSGPR(Reg, TRI)) {
+ switch (MIOpCode) {
+ // VOP3 explicit carry-in consumers
+ case AMDGPU::V_ADDC_U32_e64:
+ case AMDGPU::V_SUBB_U32_e64:
+ case AMDGPU::V_SUBBREV_U32_e64:
+ // VOP3 condition mask consumers
+ case AMDGPU::V_CNDMASK_B32_e64:
+ case AMDGPU::V_CNDMASK_B16_fake16_e64:
+ case AMDGPU::V_CNDMASK_B16_t16_e64:
+ int Src2Idx =
+ AMDGPU::getNamedOperandIdx(MIOpCode, AMDGPU::OpName::src2);
+ assert(Src2Idx >= 0 && "Unexpected source index");
+ return OpNo == static_cast<unsigned>(Src2Idx);
+ }
+ }
+ return false;
+ }
+
bool runOnMachineBasicBlock(MachineBasicBlock &MBB, bool Emit) {
DelayState State;
for (auto *Pred : MBB.predecessors())
@@ -392,19 +562,42 @@ class AMDGPUInsertDelayAlu {
State = DelayState();
} else if (Type != OTHER) {
DelayInfo Delay;
+ Register VccReg = AMDGPU::LaneMaskConstants::get(*ST).VccReg;
+ Register ExecReg = AMDGPU::LaneMaskConstants::get(*ST).ExecReg;
+
// C-reuse: back-to-back WMMAs into the same C register forward the
// accumulator in place, so the tied srcC read has no dependency. WMMA
// implies GFX11+, so no explicit subtarget check is needed.
bool IsWMMACReuse =
PrevWMMAVDst.isValid() && (SII->isWMMA(MI) || SII->isSWMMAC(MI));
- // TODO: Scan implicit uses too?
- for (const auto &Op : MI.explicit_uses()) {
+
+ for (const auto &Op : MI.all_uses()) {
if (Op.isReg()) {
// One of the operands of the writelane is also the output operand.
// This creates the insertion of redundant delays. Hence, we have to
// ignore this operand.
if (MI.getOpcode() == AMDGPU::V_WRITELANE_B32 && Op.isTied())
continue;
+
+ Register Reg = Op.getReg();
+ unsigned OperandNo = MI.getOperandNo(&Op);
+ // Skip operands that are not part of the instruction definition
+ if (OperandNo >= MI.getDesc().getNumOperands() &&
+ !MI.getDesc().hasImplicitUseOfPhysReg(
+ Reg == VccReg ? AMDGPU::VCC : Reg))
+ continue;
+
+ // Suppress the delay for VCC/EXEC/SGPR operands when both the
+ // producer and consumer are in the hardware fast-forward set.
+ if (isFastForwardConsumer(MI, Op, VccReg, ExecReg, OperandNo) &&
+ llvm::all_of(TRI->regunits(Reg), [&](MCRegUnit Unit) {
+ auto It = State.find(Unit);
+ if (It != State.end())
+ return It->second.IsFFProducer;
+ return true; // no wait for this regunit if it has no delay
+ })) {
+ continue;
+ }
// Skip the tied srcC of a C-reuse edge.
if (IsWMMACReuse && Op.isTied() && Op.getReg() == PrevWMMAVDst)
continue;
@@ -436,12 +629,18 @@ class AMDGPUInsertDelayAlu {
}
if (Type != OTHER) {
- // TODO: Scan implicit defs too?
- for (const auto &Op : MI.defs()) {
+ for (const auto &Op : MI.all_defs()) {
+ unsigned OperandNo = MI.getOperandNo(&Op);
+ Register Reg = Op.getReg();
+ if (OperandNo >= MI.getDesc().getNumOperands() &&
+ !MI.getDesc().hasImplicitDefOfPhysReg(Reg))
+ continue;
+
unsigned Latency = SchedModel->computeOperandLatency(
&MI, Op.getOperandNo(), nullptr, 0);
- for (MCRegUnit Unit : TRI->regunits(Op.getReg()))
- State[Unit] = DelayInfo(Type, Latency);
+ DelayInfo Info(Type, Latency, isFastForwardProducer(MI, Op));
+ for (MCRegUnit Unit : TRI->regunits(Reg))
+ State[Unit] = Info;
}
}
diff --git a/llvm/test/CodeGen/AMDGPU/insert-delay-alu-skipped-with-fast-forwarding.mir b/llvm/test/CodeGen/AMDGPU/insert-delay-alu-skipped-with-fast-forwarding.mir
new file mode 100644
index 00000000000000..c7c78a2c12f99f
--- /dev/null
+++ b/llvm/test/CodeGen/AMDGPU/insert-delay-alu-skipped-with-fast-forwarding.mir
@@ -0,0 +1,176 @@
+# NOTE: Assertions have been autogenerated by utils/update_mir_test_checks.py
+# RUN: llc -mtriple=amdgcn -mcpu=gfx1310 -mattr=+wavefrontsize32 -verify-machineinstrs -run-pass=amdgpu-insert-delay-alu %s -o - | FileCheck %s
+
+# VALU hardware fast-forwarding for certain VALU -> VALU dependency
+# patterns that involve carry/mask registers. No s_delay_alu is needed for:
+# { V_CMP, add/sub with carry-out } -> { v_cndmask*, add/sub with carry-in }
+#
+# Each producer and each consumer is exercised at least once below.
+# Producers: V_ADD_CO_U32_e64, V_SUB_CO_U32_e64, V_SUBREV_CO_U32_e64,
+# V_ADD_CO_CI_U32_e64, V_SUB_CO_CI_U32_e64, V_SUBREV_CO_CI_U32_e64,
+# V_CMP* (all VOPC, exercised via V_CMP_EQ_I32_e64)
+# Consumers (explicit e64): V_ADDC_U32_e64, V_SUBB_U32_e64, V_SUBREV_CO_CI_U32_e64,
+# V_CNDMASK_B32_e64
+# Consumers (implicit e32): V_ADDC_U32_e32, V_SUBB_U32_e32, V_SUBBREV_U32_e32,
+# V_CNDMASK_B32_e32
+
+# Case 1a: V_ADD_CO_U32_e64 (producer) -> V_ADDC_U32_e64 (consumer, explicit).
+# No s_delay_alu expected after fix.
+---
+name: add_co_to_addc_e64
+body: |
+ bb.0:
+ ; CHECK-LABEL: name: add_co_to_addc_e64
+ ; CHECK: $vgpr2, $vcc_lo = V_ADD_CO_U32_e64 $vgpr2, $vgpr4, 0, implicit $exec
+ ; CHECK-NEXT: $vgpr3, $vcc_lo = V_ADDC_U32_e64 $vgpr3, $vgpr5, $vcc_lo, 0, implicit $exec
+ $vgpr2, $vcc_lo = V_ADD_CO_U32_e64 $vgpr2, $vgpr4, 0, implicit $exec
+ $vgpr3, $vcc_lo = V_ADDC_U32_e64 $vgpr3, $vgpr5, $vcc_lo, 0, implicit $exec
+...
+
+# Case 1b: V_SUB_CO_U32_e64 (producer) -> V_SUBB_U32_e64 (consumer, explicit).
+# No s_delay_alu expected after fix.
+---
+name: sub_co_to_subb_e64
+body: |
+ bb.0:
+ ; CHECK-LABEL: name: sub_co_to_subb_e64
+ ; CHECK: $vgpr2, $vcc_lo = V_SUB_CO_U32_e64 $vgpr2, $vgpr4, 0, implicit $exec
+ ; CHECK-NEXT: $vgpr3, $vcc_lo = V_SUBB_U32_e64 $vgpr3, $vgpr5, $vcc_lo, 0, implicit $exec
+ $vgpr2, $vcc_lo = V_SUB_CO_U32_e64 $vgpr2, $vgpr4, 0, implicit $exec
+ $vgpr3, $vcc_lo = V_SUBB_U32_e64 $vgpr3, $vgpr5, $vcc_lo, 0, implicit $exec
+...
+
+# Case 1c: V_SUBREV_CO_U32_e64 (producer) -> V_ADDC_U32_e64 (consumer, explicit).
+# No s_delay_alu expected after fix.
+---
+name: subrev_co_to_addc_e64
+body: |
+ bb.0:
+ ; CHECK-LABEL: name: subrev_co_to_addc_e64
+ ; CHECK: $vgpr2, $vcc_lo = V_SUBREV_CO_U32_e64 $vgpr2, $vgpr4, 0, implicit $exec
+ ; CHECK-NEXT: $vgpr3, $vcc_lo = V_ADDC_U32_e64 $vgpr3, $vgpr5, $vcc_lo, 0, implicit $exec
+ $vgpr2, $vcc_lo = V_SUBREV_CO_U32_e64 $vgpr2, $vgpr4, 0, implicit $exec
+ $vgpr3, $vcc_lo = V_ADDC_U32_e64 $vgpr3, $vgpr5, $vcc_lo, 0, implicit $exec
+...
+
+# Case 1d: V_ADDC_U32_e64 (dual producer+consumer) chained.
+# No s_delay_alu expected after fix (neither carry-in step nor carry-out step).
+---
+name: add_co_ci_chain_e64
+body: |
+ bb.0:
+ ; CHECK-LABEL: name: add_co_ci_chain_e64
+ ; CHECK: $vgpr0, $vcc_lo = V_ADD_CO_U32_e64 $vgpr0, $vgpr1, 0, implicit $exec
+ ; CHECK-NEXT: $vgpr2, $vcc_lo = V_ADDC_U32_e64 $vgpr2, $vgpr3, $vcc_lo, 0, implicit $exec
+ ; CHECK-NEXT: $vgpr4, $vcc_lo = V_ADDC_U32_e64 $vgpr4, $vgpr5, $vcc_lo, 0, implicit $exec
+ $vgpr0, $vcc_lo = V_ADD_CO_U32_e64 $vgpr0, $vgpr1, 0, implicit $exec
+ $vgpr2, $vcc_lo = V_ADDC_U32_e64 $vgpr2, $vgpr3, $vcc_lo, 0, implicit $exec
+ $vgpr4, $vcc_lo = V_ADDC_U32_e64 $vgpr4, $vgpr5, $vcc_lo, 0, implicit $exec
+...
+
+# Case 1e: V_SUBB_U32_e64 (dual producer+consumer) chained.
+# No s_delay_alu expected after fix.
+---
+name: sub_co_ci_chain_e64
+body: |
+ bb.0:
+ ; CHECK-LABEL: name: sub_co_ci_chain_e64
+ ; CHECK: $vgpr0, $vcc_lo = V_SUB_CO_U32_e64 $vgpr0, $vgpr1, 0, implicit $exec
+ ; CHECK-NEXT: $vgpr2, $vcc_lo = V_SUBB_U32_e64 $vgpr2, $vgpr3, $vcc_lo, 0, implicit $exec
+ ; CHECK-NEXT: $vgpr4, $vcc_lo = V_SUBB_U32_e64 $vgpr4, $vgpr5, $vcc_lo, 0, implicit $exec
+ $vgpr0, $vcc_lo = V_SUB_CO_U32_e64 $vgpr0, $vgpr1, 0, implicit $exec
+ $vgpr2, $vcc_lo = V_SUBB_U32_e64 $vgpr2, $vgpr3, $vcc_lo, 0, implicit $exec
+ $vgpr4, $vcc_lo = V_SUBB_U32_e64 $vgpr4, $vgpr5, $vcc_lo, 0, implicit $exec
+...
+
+# Case 1f: V_SUBBREV_U32_e64 (dual producer+consumer) chained.
+# No s_delay_alu expected after fix.
+---
+name: subrev_co_ci_chain_e64
+body: |
+ bb.0:
+ ; CHECK-LABEL: name: subrev_co_ci_chain_e64
+ ; CHECK: $vgpr0, $vcc_lo = V_SUB_CO_U32_e64 $vgpr0, $vgpr1, 0, implicit $exec
+ ; CHECK-NEXT: $vgpr2, $vcc_lo = V_SUBBREV_U32_e64 $vgpr2, $vgpr3, $vcc_lo, 0, implicit $exec
+ ; CHECK-NEXT: $vgpr4, $vcc_lo = V_SUBBREV_U32_e64 $vgpr4, $vgpr5, $vcc_lo, 0, implicit $exec
+ $vgpr0, $vcc_lo = V_SUB_CO_U32_e64 $vgpr0, $vgpr1, 0, implicit $exec
+ $vgpr2, $vcc_lo = V_SUBBREV_U32_e64 $vgpr2, $vgpr3, $vcc_lo, 0, implicit $exec
+ $vgpr4, $vcc_lo = V_SUBBREV_U32_e64 $vgpr4, $vgpr5, $vcc_lo, 0, implicit $exec
+...
+
+# Case 2a: V_CMP_EQ_I32_e64 (producer) -> V_CNDMASK_B32_e64 (consumer, explicit).
+# No s_delay_alu expected after fix.
+---
+name: cmp_to_cndmask_b32_e64
+body: |
+ bb.0:
+ ; CHECK-LABEL: name: cmp_to_cndmask_b32_e64
+ ; CHECK: $vcc_lo = V_CMP_EQ_I32_e64 $vgpr0, $vgpr1, implicit $exec
+ ; CHECK-NEXT: $vgpr2 = V_CNDMASK_B32_e64 0, $vgpr3, 0, $vgpr4, $vcc_lo, implicit $exec
+ $vcc_lo = V_CMP_EQ_I32_e64 $vgpr0, $vgpr1, implicit $exec
+ $vgpr2 = V_CNDMASK_B32_e64 0, $vgpr3, 0, $vgpr4, $vcc_lo, implicit $exec
+...
+
+# Case 3a: V_ADD_CO_U32_e64 (producer) -> V_ADDC_U32_e32 (consumer, implicit VCC).
+# VCC is implicit in e32 - pass does not track it yet (TODO). No delay emitted
+# today and none expected after fix either; included to guard against regression
+# once implicit tracking is added.
+---
+name: add_co_to_addc_e32
+body: |
+ bb.0:
+ ; CHECK-LABEL: name: add_co_to_addc_e32
+ ; CHECK: $vgpr2, $vcc_lo = V_ADD_CO_U32_e64 $vgpr2, $vgpr4, 0, implicit $exec
+ ; CHECK-NEXT: $vgpr3 = V_ADDC_U32_e32 $vgpr3, $vgpr5, implicit-def $vcc_lo, implicit $vcc_lo, implicit $exec
+ $vgpr2, $vcc_lo = V_ADD_CO_U32_e64 $vgpr2, $vgpr4, 0, implicit $exec
+ $vgpr3 = V_ADDC_U32_e32 $vgpr3, $vgpr5, implicit-def $vcc, implicit $vcc, implicit $exec
+...
+
+# Case 3b: V_SUB_CO_U32_e64 (producer) -> V_SUBB_U32_e32 (consumer, implicit VCC).
+---
+name: sub_co_to_subb_e32
+body: |
+ bb.0:
+ ; CHECK-LABEL: name: sub_co_to_subb_e32
+ ; CHECK: $vgpr2, $vcc_lo = V_SUB_CO_U32_e64 $vgpr2, $vgpr4, 0, implicit $exec
+ ; CHECK-NEXT: $vgpr3 = V_SUBB_U32_e32 $vgpr3, $vgpr5, implicit-def $vcc_lo, implicit $vcc_lo, implicit $exec
+ $vgpr2, $vcc_lo = V_SUB_CO_U32_e64 $vgpr2, $vgpr4, 0, implicit $exec
+ $vgpr3 = V_SUBB_U32_e32 $vgpr3, $vgpr5, implicit-def $vcc, implicit $vcc, implicit $exec
+...
+
+# Case 3c: V_SUB_CO_U32_e64 (producer) -> V_SUBBREV_U32_e32 (consumer, implicit VCC).
+---
+name: sub_co_to_subbrev_e32
+body: |
+ bb.0:
+ ; CHECK-LABEL: name: sub_co_to_subbrev_e32
+ ; CHECK: $vgpr2, $vcc_lo = V_SUB_CO_U32_e64 $vgpr2, $vgpr4, 0, implicit $exec
+ ; CHECK-NEXT: $vgpr3 = V_SUBBREV_U32_e32 $vgpr3, $vgpr5, implicit-def $vcc_lo, implicit $vcc_lo, implicit $exec
+ $vgpr2, $vcc_lo = V_SUB_CO_U32_e64 $vgpr2, $vgpr4, 0, implicit $exec
+ $vgpr3 = V_SUBBREV_U32_e32 $vgpr3, $vgpr5, implicit-def $vcc, implicit $vcc, implicit $exec
+...
+
+# Case 3d: V_CMP_EQ_I32_e64 (producer) -> V_CNDMASK_B32_e32 (consumer, implicit VCC).
+---
+name: cmp_to_cndmask_b32_e32
+body: |
+ bb.0:
+ ; CHECK-LABEL: name: cmp_to_cndmask_b32_e32
+ ; CHECK: $vcc_lo = V_CMP_EQ_I32_e64 $vgpr0, $vgpr1, implicit $exec
+ ; CHECK-NEXT: $vgpr2 = V_CNDMASK_B32_e32 $vgpr3, $vgpr4, implicit $vcc_lo, implicit $exec
+ $vcc_lo = V_CMP_EQ_I32_e64 $vgpr0, $vgpr1, implicit $exec
+ $vgpr2 = V_CNDMASK_B32_e32 $vgpr3, $vgpr4, implicit $vcc, implicit $exec
+...
+
+# Case 4a: carry-out (vcc_lo) -> V_DIV_FMAS_F32 negative case (delay expected)
+---
+name: carry_to_div_fmas_no_fastforward
+body: |
+ bb.0:
+ ; CHECK-LABEL: name: carry_to_div_fmas_no_fastforward
+ ; CHECK: $vgpr0, $vcc_lo = V_ADD_CO_U32_e64 $vgpr0, $vgpr1, 0, implicit $exec
+ ; CHECK-NEXT: S_DELAY_ALU .id0_VALU_DEP_1
+ ; CHECK-NEXT: $vgpr2 = V_DIV_FMAS_F32_e64 0, $vgpr2, 0, $vgpr3, 0, $vgpr4, 0, 0, implicit $mode, implicit $vcc_lo, implicit $exec
+ $vgpr0, $vcc_lo = V_ADD_CO_U32_e64 $vgpr0, $vgpr1, 0, implicit $exec
+ $vgpr2 = V_DIV_FMAS_F32_e64 0, $vgpr2, 0, $vgpr3, 0, $vgpr4, 0, 0, implicit $mode, implicit $vcc, implicit $exec
+...
>From bfcab7571227059fa08ded9f207a8ca0c12a25ff Mon Sep 17 00:00:00 2001
From: Reem Elkhouly <reem.elkhouly at amd.com>
Date: Thu, 3 Sep 2026 16:48:56 +0900
Subject: [PATCH 2/3] Update tests where s_delay_alu is added or removed
---
.../CodeGen/AMDGPU/GlobalISel/addsubu64.ll | 2 -
.../AMDGPU/GlobalISel/atomicrmw_fmax.ll | 22 +-
.../AMDGPU/GlobalISel/atomicrmw_fmin.ll | 22 +-
.../AMDGPU/GlobalISel/atomicrmw_udec_wrap.ll | 12 +-
.../AMDGPU/GlobalISel/atomicrmw_uinc_wrap.ll | 20 +-
.../GlobalISel/buffer-load-byte-short.ll | 7 +-
.../AMDGPU/GlobalISel/extractelement.i128.ll | 3 +-
.../AMDGPU/GlobalISel/extractelement.i16.ll | 6 +-
.../AMDGPU/GlobalISel/extractelement.i8.ll | 18 +-
llvm/test/CodeGen/AMDGPU/GlobalISel/fabs.ll | 6 +-
llvm/test/CodeGen/AMDGPU/GlobalISel/fcmp.ll | 42 +
.../CodeGen/AMDGPU/GlobalISel/fdiv.f32.ll | 110 ++-
.../CodeGen/AMDGPU/GlobalISel/fdiv.f64.ll | 18 +-
llvm/test/CodeGen/AMDGPU/GlobalISel/fneg.ll | 6 +-
llvm/test/CodeGen/AMDGPU/GlobalISel/fpow.ll | 39 +-
.../CodeGen/AMDGPU/GlobalISel/fptrunc.bf16.ll | 95 ++-
.../test/CodeGen/AMDGPU/GlobalISel/fptrunc.ll | 25 +-
llvm/test/CodeGen/AMDGPU/GlobalISel/fshl.ll | 44 +-
llvm/test/CodeGen/AMDGPU/GlobalISel/fshr.ll | 49 +-
llvm/test/CodeGen/AMDGPU/GlobalISel/icmp.ll | 42 +
.../GlobalISel/inst-select-copy-scc-vcc.ll | 2 +-
.../llvm.amdgcn.image.gather4.a16.dim.ll | 20 +
.../GlobalISel/llvm.amdgcn.interp.inreg.ll | 60 +-
.../GlobalISel/llvm.amdgcn.intersect_ray.ll | 24 +-
.../CodeGen/AMDGPU/GlobalISel/mubuf-global.ll | 33 +-
.../AMDGPU/GlobalISel/mul-known-bits.i64.ll | 2 +
llvm/test/CodeGen/AMDGPU/GlobalISel/mul.ll | 28 +-
.../regbanklegalize-amdgcn.s.buffer.load.ll | 27 +-
...klegalize-amdgcn.s.buffer.load.subdword.ll | 22 +-
llvm/test/CodeGen/AMDGPU/add.ll | 4 +
llvm/test/CodeGen/AMDGPU/add_i1.ll | 3 +-
llvm/test/CodeGen/AMDGPU/add_u64.ll | 6 -
llvm/test/CodeGen/AMDGPU/addrspacecast-gas.ll | 8 +-
llvm/test/CodeGen/AMDGPU/amd.endpgm.ll | 2 +
.../CodeGen/AMDGPU/amdgcn-call-whole-wave.ll | 2 +
.../CodeGen/AMDGPU/amdgcn.bitcast.1024bit.ll | 306 ++++---
.../CodeGen/AMDGPU/amdgcn.bitcast.128bit.ll | 257 ++++--
.../CodeGen/AMDGPU/amdgcn.bitcast.160bit.ll | 48 +-
.../CodeGen/AMDGPU/amdgcn.bitcast.16bit.ll | 36 +-
.../CodeGen/AMDGPU/amdgcn.bitcast.192bit.ll | 130 ++-
.../CodeGen/AMDGPU/amdgcn.bitcast.224bit.ll | 48 +-
.../CodeGen/AMDGPU/amdgcn.bitcast.256bit.ll | 248 ++++--
.../CodeGen/AMDGPU/amdgcn.bitcast.288bit.ll | 48 +-
.../CodeGen/AMDGPU/amdgcn.bitcast.320bit.ll | 175 ++--
.../CodeGen/AMDGPU/amdgcn.bitcast.32bit.ll | 165 +++-
.../CodeGen/AMDGPU/amdgcn.bitcast.352bit.ll | 48 +-
.../CodeGen/AMDGPU/amdgcn.bitcast.384bit.ll | 135 ++-
.../CodeGen/AMDGPU/amdgcn.bitcast.448bit.ll | 140 ++--
.../CodeGen/AMDGPU/amdgcn.bitcast.48bit.ll | 48 +-
.../CodeGen/AMDGPU/amdgcn.bitcast.512bit.ll | 294 ++++---
.../CodeGen/AMDGPU/amdgcn.bitcast.576bit.ll | 215 +++--
.../CodeGen/AMDGPU/amdgcn.bitcast.640bit.ll | 191 +++--
.../CodeGen/AMDGPU/amdgcn.bitcast.64bit.ll | 248 ++++--
.../CodeGen/AMDGPU/amdgcn.bitcast.704bit.ll | 186 +++--
.../CodeGen/AMDGPU/amdgcn.bitcast.768bit.ll | 186 +++--
.../CodeGen/AMDGPU/amdgcn.bitcast.832bit.ll | 193 +++--
.../CodeGen/AMDGPU/amdgcn.bitcast.896bit.ll | 193 +++--
.../CodeGen/AMDGPU/amdgcn.bitcast.960bit.ll | 200 +++--
.../CodeGen/AMDGPU/amdgcn.bitcast.96bit.ll | 127 ++-
.../test/CodeGen/AMDGPU/amdgpu-cs-chain-cc.ll | 5 +-
.../CodeGen/AMDGPU/amdgpu-nsa-threshold.ll | 12 +
.../AMDGPU/arbitrary-fp-to-float-fp8-hw.ll | 5 +-
.../CodeGen/AMDGPU/asyncmark-gfx12plus.ll | 2 +-
.../AMDGPU/atomic_optimizations_buffer.ll | 54 +-
...mizations_dpp_lds_expected_active_lanes.ll | 28 +-
.../atomic_optimizations_global_pointer.ll | 631 +++++++++-----
.../atomic_optimizations_local_pointer.ll | 291 +++++--
.../atomic_optimizations_pixelshader.ll | 36 +-
.../AMDGPU/atomic_optimizations_raw_buffer.ll | 48 +-
.../atomic_optimizations_struct_buffer.ll | 48 +-
llvm/test/CodeGen/AMDGPU/atomicrmw-expand.ll | 3 +-
.../test/CodeGen/AMDGPU/atomicrmw_usub_sat.ll | 92 +-
.../CodeGen/AMDGPU/atomics-system-scope.ll | 40 +-
.../AMDGPU/barrier-signal-wait-latency.ll | 8 +-
llvm/test/CodeGen/AMDGPU/bf16-conversions.ll | 16 +-
llvm/test/CodeGen/AMDGPU/bf16.ll | 98 +--
.../AMDGPU/branch-relaxation-gfx1250.ll | 46 +-
.../branch-relaxation-inst-size-gfx10.ll | 10 +-
.../branch-relaxation-inst-size-gfx11.ll | 1 +
llvm/test/CodeGen/AMDGPU/branch-relaxation.ll | 47 +-
.../buffer-fat-pointer-atomicrmw-fadd.ll | 134 +--
.../buffer-fat-pointer-atomicrmw-fmax.ll | 114 ++-
.../buffer-fat-pointer-atomicrmw-fmin.ll | 114 ++-
.../CodeGen/AMDGPU/calling-conventions.ll | 18 +-
.../test/CodeGen/AMDGPU/carryout-selection.ll | 77 +-
llvm/test/CodeGen/AMDGPU/clmul.ll | 58 ++
.../test/CodeGen/AMDGPU/code-size-estimate.ll | 10 +-
.../CodeGen/AMDGPU/combine-add-zext-xor.ll | 24 +-
.../AMDGPU/commute-compares-scalar-float.ll | 64 +-
llvm/test/CodeGen/AMDGPU/ctpop64.ll | 2 +-
llvm/test/CodeGen/AMDGPU/d16-write-vgpr32.ll | 4 +-
.../CodeGen/AMDGPU/dagcombine-fmul-sel.ll | 20 +-
.../AMDGPU/ds-atomic-barrier-convergent.ll | 6 +-
llvm/test/CodeGen/AMDGPU/ds_read2-gfx1250.ll | 16 +-
llvm/test/CodeGen/AMDGPU/ds_write2.ll | 11 +-
.../dynamic-vgpr-reserve-stack-for-cwsr.ll | 12 +-
.../test/CodeGen/AMDGPU/dynamic_stackalloc.ll | 111 +--
.../AMDGPU/expand-mov-b64-globaladdr.ll | 6 +-
.../expand-scalar-carry-out-select-user.ll | 4 +-
.../CodeGen/AMDGPU/extract-subvector-16bit.ll | 22 +-
.../CodeGen/AMDGPU/extract_vector_elt-f16.ll | 24 +-
llvm/test/CodeGen/AMDGPU/fcopysign.bf16.ll | 68 +-
llvm/test/CodeGen/AMDGPU/fcopysign.f16.ll | 37 +-
llvm/test/CodeGen/AMDGPU/fdiv.bf16.ll | 8 +-
.../CodeGen/AMDGPU/flat-atomicrmw-fadd.ll | 331 +++++---
.../CodeGen/AMDGPU/flat-atomicrmw-fmax.ll | 368 +++++---
.../CodeGen/AMDGPU/flat-atomicrmw-fmin.ll | 368 +++++---
.../CodeGen/AMDGPU/flat-atomicrmw-fsub.ll | 393 ++++++---
.../AMDGPU/flat-load-saddr-to-vaddr.ll | 1 +
.../test/CodeGen/AMDGPU/flat-saddr-atomics.ll | 449 +++++-----
llvm/test/CodeGen/AMDGPU/flat-saddr-load.ll | 238 +++---
llvm/test/CodeGen/AMDGPU/flat-saddr-store.ll | 120 ++-
llvm/test/CodeGen/AMDGPU/flat_atomics.ll | 3 -
llvm/test/CodeGen/AMDGPU/flat_atomics_i64.ll | 592 ++++++++-----
llvm/test/CodeGen/AMDGPU/float-sopc-vopc.ll | 265 +++---
.../float-to-arbitrary-fp-fp8-f16-hw.ll | 31 +-
.../AMDGPU/float-to-arbitrary-fp-fp8-hw.ll | 139 ++--
.../AMDGPU/float-to-arbitrary-fp-widen.ll | 5 +-
llvm/test/CodeGen/AMDGPU/fmax3-maximumnum.ll | 64 +-
llvm/test/CodeGen/AMDGPU/fmax_legacy.f16.ll | 8 +-
llvm/test/CodeGen/AMDGPU/fmed3.bf16.ll | 11 +-
llvm/test/CodeGen/AMDGPU/fmin3-minimumnum.ll | 64 +-
llvm/test/CodeGen/AMDGPU/fmin_legacy.f16.ll | 8 +-
.../AMDGPU/fmul-2-combine-multi-use.ll | 11 +-
llvm/test/CodeGen/AMDGPU/fneg-combines.f16.ll | 12 +-
.../CodeGen/AMDGPU/fneg-modifier-casting.ll | 27 +-
llvm/test/CodeGen/AMDGPU/fold-gep-offset.ll | 19 +-
.../AMDGPU/fold-int-pow2-with-fmul-or-fdiv.ll | 4 +-
llvm/test/CodeGen/AMDGPU/fp_to_sint.ll | 8 +-
llvm/test/CodeGen/AMDGPU/fptrunc.f16.ll | 246 ++++--
llvm/test/CodeGen/AMDGPU/fptrunc.ll | 30 +-
llvm/test/CodeGen/AMDGPU/fract-match.ll | 260 +++++-
llvm/test/CodeGen/AMDGPU/frem.ll | 787 ++++++++++++------
.../CodeGen/AMDGPU/frexp-inf-nan-combine.ll | 8 +-
.../AMDGPU/function-esm2-prologue-epilogue.ll | 4 +-
.../AMDGPU/gfx12_scalar_subword_loads.ll | 8 +-
.../CodeGen/AMDGPU/global-atomicrmw-fadd.ll | 420 ++++++----
.../CodeGen/AMDGPU/global-atomicrmw-fmax.ll | 332 +++++---
.../CodeGen/AMDGPU/global-atomicrmw-fmin.ll | 332 +++++---
.../CodeGen/AMDGPU/global-atomicrmw-fsub.ll | 365 +++++---
.../global-saddr-atomics-min-max-system.ll | 208 ++---
llvm/test/CodeGen/AMDGPU/global-saddr-load.ll | 36 +-
.../AMDGPU/global_atomics_scan_fadd.ll | 249 ++++--
.../AMDGPU/global_atomics_scan_fmax.ll | 150 ++--
.../AMDGPU/global_atomics_scan_fmin.ll | 150 ++--
.../AMDGPU/global_atomics_scan_fsub.ll | 301 ++++---
.../AMDGPU/group-image-instructions.ll | 1 +
llvm/test/CodeGen/AMDGPU/i1-to-bf16.ll | 68 +-
llvm/test/CodeGen/AMDGPU/idiv-licm.ll | 15 +-
.../AMDGPU/indirect-call-known-callees.ll | 2 +-
.../CodeGen/AMDGPU/insert-delay-alu-bug.ll | 18 +-
llvm/test/CodeGen/AMDGPU/insert-delay-alu.mir | 1 -
.../AMDGPU/insert_vector_elt.v2bf16.ll | 12 +
.../CodeGen/AMDGPU/insert_vector_elt.v2i16.ll | 17 +-
.../insert_waitcnt_for_precise_memory.ll | 20 +-
.../CodeGen/AMDGPU/integer-mad-patterns.ll | 93 +--
.../AMDGPU/integer-select-src-modifiers.ll | 20 +-
.../AMDGPU/intrinsic-amdgcn-s-alloc-vgpr.ll | 37 +-
.../issue130120-eliminate-frame-index.ll | 1 +
...e92561-restore-undef-scc-verifier-error.ll | 3 +
.../test/CodeGen/AMDGPU/lds-misaligned-bug.ll | 12 -
llvm/test/CodeGen/AMDGPU/literal64.ll | 12 +-
.../AMDGPU/llvm.amdgcn.av.load.b128.ll | 438 +++-------
.../AMDGPU/llvm.amdgcn.av.store.b128.ll | 41 +-
.../AMDGPU/llvm.amdgcn.bvh8_intersect_ray.ll | 3 +-
.../llvm.amdgcn.cluster.load.async.to.lds.ll | 16 -
llvm/test/CodeGen/AMDGPU/llvm.amdgcn.dead.ll | 19 +-
.../AMDGPU/llvm.amdgcn.dual_intersect_ray.ll | 3 +-
.../llvm.amdgcn.global.load.async.to.lds.ll | 12 +-
...llvm.amdgcn.global.store.async.from.lds.ll | 4 -
.../AMDGPU/llvm.amdgcn.init.whole.wave-w32.ll | 31 +-
.../AMDGPU/llvm.amdgcn.init.whole.wave-w64.ll | 4 +-
.../AMDGPU/llvm.amdgcn.interp.inreg.ll | 60 +-
.../AMDGPU/llvm.amdgcn.intersect_ray.ll | 15 +-
.../CodeGen/AMDGPU/llvm.amdgcn.is.private.ll | 6 +-
.../CodeGen/AMDGPU/llvm.amdgcn.is.shared.ll | 5 +-
.../CodeGen/AMDGPU/llvm.amdgcn.permlane.ll | 4 +-
.../llvm.amdgcn.raw.atomic.buffer.load.ll | 49 +-
.../llvm.amdgcn.raw.ptr.atomic.buffer.load.ll | 49 +-
.../AMDGPU/llvm.amdgcn.readfirstlane.m0.ll | 1 +
.../CodeGen/AMDGPU/llvm.amdgcn.reduce.add.ll | 168 ++--
.../CodeGen/AMDGPU/llvm.amdgcn.reduce.and.ll | 176 ++--
.../CodeGen/AMDGPU/llvm.amdgcn.reduce.fadd.ll | 201 +++--
.../CodeGen/AMDGPU/llvm.amdgcn.reduce.fmax.ll | 164 ++--
.../CodeGen/AMDGPU/llvm.amdgcn.reduce.fmin.ll | 164 ++--
.../CodeGen/AMDGPU/llvm.amdgcn.reduce.fsub.ll | 203 +++--
.../CodeGen/AMDGPU/llvm.amdgcn.reduce.max.ll | 158 ++--
.../CodeGen/AMDGPU/llvm.amdgcn.reduce.min.ll | 158 ++--
.../CodeGen/AMDGPU/llvm.amdgcn.reduce.or.ll | 176 ++--
.../CodeGen/AMDGPU/llvm.amdgcn.reduce.sub.ll | 148 ++--
.../CodeGen/AMDGPU/llvm.amdgcn.reduce.umax.ll | 160 ++--
.../CodeGen/AMDGPU/llvm.amdgcn.reduce.umin.ll | 160 ++--
.../CodeGen/AMDGPU/llvm.amdgcn.reduce.xor.ll | 174 ++--
.../CodeGen/AMDGPU/llvm.amdgcn.s.barrier.ll | 6 +-
.../llvm.amdgcn.s.barrier.signal.isfirst.ll | 6 +
.../AMDGPU/llvm.amdgcn.s.prefetch.data.ll | 9 +-
.../AMDGPU/llvm.amdgcn.s.prefetch.inst.ll | 6 +-
.../AMDGPU/llvm.amdgcn.s.ttracedata.ll | 3 +
.../llvm.amdgcn.set.inactive.chain.arg.ll | 69 +-
.../llvm.amdgcn.struct.atomic.buffer.load.ll | 55 +-
.../AMDGPU/llvm.amdgcn.struct.buffer.store.ll | 2 +
...vm.amdgcn.struct.ptr.atomic.buffer.load.ll | 55 +-
.../AMDGPU/llvm.amdgcn.tensor.load.store.ll | 13 +
.../AMDGPU/llvm.amdgcn.wave.shuffle.ll | 2 +
llvm/test/CodeGen/AMDGPU/llvm.exp2.bf16.ll | 288 +++----
.../test/CodeGen/AMDGPU/llvm.fptrunc.round.ll | 271 ++++--
llvm/test/CodeGen/AMDGPU/llvm.is.fpclass.ll | 4 +-
llvm/test/CodeGen/AMDGPU/llvm.log.ll | 146 ++--
llvm/test/CodeGen/AMDGPU/llvm.log10.ll | 146 ++--
llvm/test/CodeGen/AMDGPU/llvm.log2.ll | 40 +-
llvm/test/CodeGen/AMDGPU/llvm.maximum.f16.ll | 8 +-
llvm/test/CodeGen/AMDGPU/llvm.maximum.f64.ll | 10 +-
llvm/test/CodeGen/AMDGPU/llvm.minimum.f16.ll | 8 +-
llvm/test/CodeGen/AMDGPU/llvm.minimum.f64.ll | 10 +-
llvm/test/CodeGen/AMDGPU/llvm.mulo.ll | 34 +-
llvm/test/CodeGen/AMDGPU/llvm.round.ll | 44 +-
llvm/test/CodeGen/AMDGPU/llvm.sponentry.ll | 34 +-
llvm/test/CodeGen/AMDGPU/load-atomic-flat.ll | 4 +-
.../AMDGPU/load-constant-always-uniform.ll | 4 +-
.../CodeGen/AMDGPU/load-saddr-offset-imm.ll | 4 +-
.../CodeGen/AMDGPU/local-atomicrmw-fadd.ll | 144 +++-
.../CodeGen/AMDGPU/local-atomicrmw-fmax.ll | 135 ++-
.../CodeGen/AMDGPU/local-atomicrmw-fmin.ll | 135 ++-
.../CodeGen/AMDGPU/local-atomicrmw-fsub.ll | 171 ++--
.../test/CodeGen/AMDGPU/loop-prefetch-data.ll | 36 +-
.../lower-work-group-id-intrinsics-opt.ll | 14 +-
.../AMDGPU/lower-work-group-id-intrinsics.ll | 10 +-
llvm/test/CodeGen/AMDGPU/lrint.ll | 15 +-
llvm/test/CodeGen/AMDGPU/lround.ll | 153 ++--
llvm/test/CodeGen/AMDGPU/mad_64_32.ll | 27 +-
llvm/test/CodeGen/AMDGPU/madak.ll | 2 +
.../match-perm-extract-vector-elt-bug.ll | 2 +-
llvm/test/CodeGen/AMDGPU/maximumnum.bf16.ll | 340 ++++----
llvm/test/CodeGen/AMDGPU/min.ll | 2 +
llvm/test/CodeGen/AMDGPU/minimumnum.bf16.ll | 340 ++++----
llvm/test/CodeGen/AMDGPU/mixed-vmem-types.ll | 15 +-
...uf-legalize-operands-non-ptr-intrinsics.ll | 6 +-
.../CodeGen/AMDGPU/mubuf-legalize-operands.ll | 6 +-
llvm/test/CodeGen/AMDGPU/mul.ll | 41 +-
.../CodeGen/AMDGPU/narrow_math_for_and.ll | 7 +-
.../CodeGen/AMDGPU/no-dup-inst-prefetch.ll | 3 +-
llvm/test/CodeGen/AMDGPU/offset-split-flat.ll | 153 +---
.../CodeGen/AMDGPU/offset-split-global.ll | 84 --
.../AMDGPU/optimize-ds-bvh-stack-pre-ra.ll | 3 +
...eemit-peephole-scc-liveness-issue215745.ll | 5 +-
...-alloca-vector-dynamic-idx-bitcasts-llc.ll | 6 +
.../AMDGPU/promote-constOffset-to-imm.ll | 115 ++-
.../AMDGPU/pseudo-scalar-transcendental.ll | 49 +-
llvm/test/CodeGen/AMDGPU/ptradd-sdag.ll | 8 +-
.../AMDGPU/reassoc-mul-add-1-to-mad.ll | 25 +-
.../CodeGen/AMDGPU/release-vgprs-spill.ll | 1 -
...rename-independent-subregs-unused-lanes.ll | 6 +-
llvm/test/CodeGen/AMDGPU/repeated-divisor.ll | 8 +-
.../AMDGPU/required-export-priority.ll | 8 +
.../AMDGPU/s-barrier-signal-var-gep.ll | 17 +-
llvm/test/CodeGen/AMDGPU/s-barrier.ll | 14 +-
llvm/test/CodeGen/AMDGPU/s-cluster-barrier.ll | 3 +-
llvm/test/CodeGen/AMDGPU/s-wakeup-barrier.ll | 3 +-
llvm/test/CodeGen/AMDGPU/sad.ll | 34 +-
llvm/test/CodeGen/AMDGPU/saddo.ll | 4 +-
.../CodeGen/AMDGPU/scale-offset-global.ll | 4 +-
llvm/test/CodeGen/AMDGPU/scmp.ll | 32 +-
.../CodeGen/AMDGPU/scratch-pointer-sink.ll | 2 +
.../AMDGPU/select-fabs-fneg-extract.v2f16.ll | 121 ++-
.../AMDGPU/select-flags-to-fmin-fmax.ll | 42 +-
llvm/test/CodeGen/AMDGPU/select.f16.ll | 8 +-
...pre-emit-peephole-redundant-mode-writes.ll | 18 +-
.../AMDGPU/simulated-trap-pseudo-expand.ll | 4 +-
llvm/test/CodeGen/AMDGPU/skip-if-dead.ll | 44 +-
llvm/test/CodeGen/AMDGPU/ssubo.ll | 15 +-
llvm/test/CodeGen/AMDGPU/store-atomic-flat.ll | 1 -
.../CodeGen/AMDGPU/strict_fptrunc_bf16.ll | 8 +-
llvm/test/CodeGen/AMDGPU/sub.ll | 3 -
llvm/test/CodeGen/AMDGPU/sub_i1.ll | 3 +-
llvm/test/CodeGen/AMDGPU/sub_u64.ll | 7 -
llvm/test/CodeGen/AMDGPU/trap-abis.ll | 3 +
llvm/test/CodeGen/AMDGPU/uaddo.ll | 1 +
llvm/test/CodeGen/AMDGPU/ucmp.ll | 35 +-
.../umin-sub-to-usubo-select-combine.ll | 6 +-
.../AMDGPU/uniform-inside-divergent-cfg.ll | 7 +
llvm/test/CodeGen/AMDGPU/uniform-select.ll | 13 +-
.../AMDGPU/uniform-vgpr-to-sgpr-return.ll | 4 +-
llvm/test/CodeGen/AMDGPU/usubo.ll | 1 +
llvm/test/CodeGen/AMDGPU/v_cndmask.ll | 42 +-
llvm/test/CodeGen/AMDGPU/v_swap_b16.ll | 18 +-
llvm/test/CodeGen/AMDGPU/vector-reduce-add.ll | 53 +-
.../CodeGen/AMDGPU/vector-reduce-fmaximum.ll | 6 +-
.../CodeGen/AMDGPU/vector-reduce-fminimum.ll | 6 +-
.../test/CodeGen/AMDGPU/vector-reduce-smax.ll | 47 +-
.../test/CodeGen/AMDGPU/vector-reduce-smin.ll | 47 +-
.../test/CodeGen/AMDGPU/vector-reduce-umax.ll | 47 +-
.../test/CodeGen/AMDGPU/vector-reduce-umin.ll | 47 +-
.../AMDGPU/vgpr-lowering-gfx1250-t16.mir | 5 +
.../CodeGen/AMDGPU/vgpr-lowering-gfx1250.mir | 40 +-
.../AMDGPU/vgpr-mark-last-scratch-load.ll | 5 +-
.../CodeGen/AMDGPU/whole-wave-functions.ll | 35 +-
.../AMDGPU/workgroup-id-in-arch-sgprs.ll | 15 +-
.../CodeGen/AMDGPU/workitem-intrinsic-opts.ll | 1 +
298 files changed, 13584 insertions(+), 8551 deletions(-)
diff --git a/llvm/test/CodeGen/AMDGPU/GlobalISel/addsubu64.ll b/llvm/test/CodeGen/AMDGPU/GlobalISel/addsubu64.ll
index d71605f59d45ff..aa7cd00cf5852f 100644
--- a/llvm/test/CodeGen/AMDGPU/GlobalISel/addsubu64.ll
+++ b/llvm/test/CodeGen/AMDGPU/GlobalISel/addsubu64.ll
@@ -39,7 +39,6 @@ define amdgpu_ps void @v_add_u64(ptr addrspace(1) %out, i64 %a, i64 %b) {
; GCN-LABEL: v_add_u64:
; GCN: ; %bb.0: ; %entry
; GCN-NEXT: v_add_co_u32 v2, vcc_lo, v2, v4
-; GCN-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GCN-NEXT: v_add_co_ci_u32_e64 v3, null, v3, v5, vcc_lo
; GCN-NEXT: global_store_b64 v[0:1], v[2:3], off
; GCN-NEXT: s_endpgm
@@ -86,7 +85,6 @@ define amdgpu_ps void @v_sub_u64(ptr addrspace(1) %out, i64 %a, i64 %b) {
; GCN-LABEL: v_sub_u64:
; GCN: ; %bb.0: ; %entry
; GCN-NEXT: v_sub_co_u32 v2, vcc_lo, v2, v4
-; GCN-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GCN-NEXT: v_sub_co_ci_u32_e64 v3, null, v3, v5, vcc_lo
; GCN-NEXT: global_store_b64 v[0:1], v[2:3], off
; GCN-NEXT: s_endpgm
diff --git a/llvm/test/CodeGen/AMDGPU/GlobalISel/atomicrmw_fmax.ll b/llvm/test/CodeGen/AMDGPU/GlobalISel/atomicrmw_fmax.ll
index 5444a36a542cb9..fc60b23e4bf805 100644
--- a/llvm/test/CodeGen/AMDGPU/GlobalISel/atomicrmw_fmax.ll
+++ b/llvm/test/CodeGen/AMDGPU/GlobalISel/atomicrmw_fmax.ll
@@ -622,9 +622,11 @@ define double @global_agent_atomic_fmax_ret_f64__amdgpu_no_fine_grained_memory(p
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB6_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: v_dual_mov_b32 v0, v4 :: v_dual_mov_b32 v1, v5
; GFX12-NEXT: s_setpc_b64 s[30:31]
;
@@ -658,11 +660,12 @@ define double @global_agent_atomic_fmax_ret_f64__amdgpu_no_fine_grained_memory(p
; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: v_cmp_eq_u64_e32 vcc_lo, v[4:5], v[6:7]
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB6_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: v_dual_mov_b32 v0, v4 :: v_dual_mov_b32 v1, v5
; GFX11-NEXT: s_setpc_b64 s[30:31]
;
@@ -779,6 +782,7 @@ define void @global_agent_atomic_fmax_noret_f64__amdgpu_no_fine_grained_memory(p
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB7_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -814,7 +818,7 @@ define void @global_agent_atomic_fmax_noret_f64__amdgpu_no_fine_grained_memory(p
; GFX11-NEXT: v_cmp_eq_u64_e32 vcc_lo, v[2:3], v[4:5]
; GFX11-NEXT: v_dual_mov_b32 v5, v3 :: v_dual_mov_b32 v4, v2
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB7_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -1212,9 +1216,11 @@ define double @flat_agent_atomic_fmax_ret_f64__amdgpu_no_fine_grained_memory(ptr
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB10_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: v_dual_mov_b32 v0, v4 :: v_dual_mov_b32 v1, v5
; GFX12-NEXT: s_setpc_b64 s[30:31]
;
@@ -1248,11 +1254,12 @@ define double @flat_agent_atomic_fmax_ret_f64__amdgpu_no_fine_grained_memory(ptr
; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: v_cmp_eq_u64_e32 vcc_lo, v[4:5], v[6:7]
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB10_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: v_dual_mov_b32 v0, v4 :: v_dual_mov_b32 v1, v5
; GFX11-NEXT: s_setpc_b64 s[30:31]
;
@@ -1367,6 +1374,7 @@ define void @flat_agent_atomic_fmax_noret_f64__amdgpu_no_fine_grained_memory(ptr
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB11_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -1402,7 +1410,7 @@ define void @flat_agent_atomic_fmax_noret_f64__amdgpu_no_fine_grained_memory(ptr
; GFX11-NEXT: v_cmp_eq_u64_e32 vcc_lo, v[2:3], v[4:5]
; GFX11-NEXT: v_dual_mov_b32 v5, v3 :: v_dual_mov_b32 v4, v2
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB11_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -1851,6 +1859,7 @@ define double @buffer_fat_ptr_agent_atomic_fmax_ret_f64__amdgpu_no_fine_grained_
; GFX12-NEXT: s_or_b32 s4, vcc_lo, s4
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB14_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s4
@@ -1896,7 +1905,7 @@ define double @buffer_fat_ptr_agent_atomic_fmax_ret_f64__amdgpu_no_fine_grained_
; GFX11-NEXT: v_cmp_eq_u64_e32 vcc_lo, v[0:1], v[4:5]
; GFX11-NEXT: v_dual_mov_b32 v5, v1 :: v_dual_mov_b32 v4, v0
; GFX11-NEXT: s_or_b32 s4, vcc_lo, s4
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
; GFX11-NEXT: s_cbranch_execnz .LBB14_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -2037,6 +2046,7 @@ define void @buffer_fat_ptr_agent_atomic_fmax_noret_f64__amdgpu_no_fine_grained_
; GFX12-NEXT: s_or_b32 s4, vcc_lo, s4
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB15_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s4
@@ -2081,7 +2091,7 @@ define void @buffer_fat_ptr_agent_atomic_fmax_noret_f64__amdgpu_no_fine_grained_
; GFX11-NEXT: v_cmp_eq_u64_e32 vcc_lo, v[7:8], v[2:3]
; GFX11-NEXT: v_dual_mov_b32 v2, v7 :: v_dual_mov_b32 v3, v8
; GFX11-NEXT: s_or_b32 s4, vcc_lo, s4
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
; GFX11-NEXT: s_cbranch_execnz .LBB15_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
diff --git a/llvm/test/CodeGen/AMDGPU/GlobalISel/atomicrmw_fmin.ll b/llvm/test/CodeGen/AMDGPU/GlobalISel/atomicrmw_fmin.ll
index 508e8e1da7b5e5..91d5804253e666 100644
--- a/llvm/test/CodeGen/AMDGPU/GlobalISel/atomicrmw_fmin.ll
+++ b/llvm/test/CodeGen/AMDGPU/GlobalISel/atomicrmw_fmin.ll
@@ -622,9 +622,11 @@ define double @global_agent_atomic_fmin_ret_f64__amdgpu_no_fine_grained_memory(p
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB6_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: v_dual_mov_b32 v0, v4 :: v_dual_mov_b32 v1, v5
; GFX12-NEXT: s_setpc_b64 s[30:31]
;
@@ -658,11 +660,12 @@ define double @global_agent_atomic_fmin_ret_f64__amdgpu_no_fine_grained_memory(p
; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: v_cmp_eq_u64_e32 vcc_lo, v[4:5], v[6:7]
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB6_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: v_dual_mov_b32 v0, v4 :: v_dual_mov_b32 v1, v5
; GFX11-NEXT: s_setpc_b64 s[30:31]
;
@@ -779,6 +782,7 @@ define void @global_agent_atomic_fmin_noret_f64__amdgpu_no_fine_grained_memory(p
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB7_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -814,7 +818,7 @@ define void @global_agent_atomic_fmin_noret_f64__amdgpu_no_fine_grained_memory(p
; GFX11-NEXT: v_cmp_eq_u64_e32 vcc_lo, v[2:3], v[4:5]
; GFX11-NEXT: v_dual_mov_b32 v5, v3 :: v_dual_mov_b32 v4, v2
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB7_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -1212,9 +1216,11 @@ define double @flat_agent_atomic_fmin_ret_f64__amdgpu_no_fine_grained_memory(ptr
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB10_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: v_dual_mov_b32 v0, v4 :: v_dual_mov_b32 v1, v5
; GFX12-NEXT: s_setpc_b64 s[30:31]
;
@@ -1248,11 +1254,12 @@ define double @flat_agent_atomic_fmin_ret_f64__amdgpu_no_fine_grained_memory(ptr
; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: v_cmp_eq_u64_e32 vcc_lo, v[4:5], v[6:7]
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB10_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: v_dual_mov_b32 v0, v4 :: v_dual_mov_b32 v1, v5
; GFX11-NEXT: s_setpc_b64 s[30:31]
;
@@ -1367,6 +1374,7 @@ define void @flat_agent_atomic_fmin_noret_f64__amdgpu_no_fine_grained_memory(ptr
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB11_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -1402,7 +1410,7 @@ define void @flat_agent_atomic_fmin_noret_f64__amdgpu_no_fine_grained_memory(ptr
; GFX11-NEXT: v_cmp_eq_u64_e32 vcc_lo, v[2:3], v[4:5]
; GFX11-NEXT: v_dual_mov_b32 v5, v3 :: v_dual_mov_b32 v4, v2
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB11_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -1851,6 +1859,7 @@ define double @buffer_fat_ptr_agent_atomic_fmin_ret_f64__amdgpu_no_fine_grained_
; GFX12-NEXT: s_or_b32 s4, vcc_lo, s4
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB14_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s4
@@ -1896,7 +1905,7 @@ define double @buffer_fat_ptr_agent_atomic_fmin_ret_f64__amdgpu_no_fine_grained_
; GFX11-NEXT: v_cmp_eq_u64_e32 vcc_lo, v[0:1], v[4:5]
; GFX11-NEXT: v_dual_mov_b32 v5, v1 :: v_dual_mov_b32 v4, v0
; GFX11-NEXT: s_or_b32 s4, vcc_lo, s4
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
; GFX11-NEXT: s_cbranch_execnz .LBB14_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -2037,6 +2046,7 @@ define void @buffer_fat_ptr_agent_atomic_fmin_noret_f64__amdgpu_no_fine_grained_
; GFX12-NEXT: s_or_b32 s4, vcc_lo, s4
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB15_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s4
@@ -2081,7 +2091,7 @@ define void @buffer_fat_ptr_agent_atomic_fmin_noret_f64__amdgpu_no_fine_grained_
; GFX11-NEXT: v_cmp_eq_u64_e32 vcc_lo, v[7:8], v[2:3]
; GFX11-NEXT: v_dual_mov_b32 v2, v7 :: v_dual_mov_b32 v3, v8
; GFX11-NEXT: s_or_b32 s4, vcc_lo, s4
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
; GFX11-NEXT: s_cbranch_execnz .LBB15_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
diff --git a/llvm/test/CodeGen/AMDGPU/GlobalISel/atomicrmw_udec_wrap.ll b/llvm/test/CodeGen/AMDGPU/GlobalISel/atomicrmw_udec_wrap.ll
index 1a8ae1180d0aa3..e67025285ceee5 100644
--- a/llvm/test/CodeGen/AMDGPU/GlobalISel/atomicrmw_udec_wrap.ll
+++ b/llvm/test/CodeGen/AMDGPU/GlobalISel/atomicrmw_udec_wrap.ll
@@ -1584,14 +1584,14 @@ define amdgpu_kernel void @flat_atomic_dec_ret_i32_offset_addr64(ptr %out, ptr %
; GFX11-NEXT: v_dual_mov_b32 v1, s3 :: v_dual_lshlrev_b32 v2, 2, v0
; GFX11-NEXT: v_mov_b32_e32 v0, s2
; GFX11-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-NEXT: flat_atomic_dec_u32 v3, v[0:1], v3 offset:20 glc
; GFX11-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
; GFX11-NEXT: buffer_gl1_inv
; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: v_dual_mov_b32 v0, s0 :: v_dual_mov_b32 v1, s1
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-NEXT: flat_store_b32 v[0:1], v3
@@ -1695,7 +1695,7 @@ define amdgpu_kernel void @flat_atomic_dec_noret_i32_offset_addr64(ptr %ptr) #1
; GFX11-NEXT: v_lshlrev_b32_e32 v2, 2, v0
; GFX11-NEXT: v_mov_b32_e32 v0, s0
; GFX11-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_4)
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-NEXT: v_mov_b32_e32 v2, 42
; GFX11-NEXT: flat_atomic_dec_u32 v[0:1], v2 offset:20
@@ -2313,14 +2313,14 @@ define amdgpu_kernel void @flat_atomic_dec_ret_i64_offset_addr64(ptr %out, ptr %
; GFX11-NEXT: v_dual_mov_b32 v1, s3 :: v_dual_lshlrev_b32 v4, 3, v0
; GFX11-NEXT: v_mov_b32_e32 v0, s2
; GFX11-NEXT: v_add_co_u32 v0, vcc_lo, v0, v4
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-NEXT: flat_atomic_dec_u64 v[0:1], v[0:1], v[2:3] offset:40 glc
; GFX11-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
; GFX11-NEXT: buffer_gl1_inv
; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: v_dual_mov_b32 v2, s0 :: v_dual_mov_b32 v3, s1
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-NEXT: v_add_co_u32 v2, vcc_lo, v2, v4
; GFX11-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX11-NEXT: flat_store_b64 v[2:3], v[0:1]
@@ -2429,7 +2429,7 @@ define amdgpu_kernel void @flat_atomic_dec_noret_i64_offset_addr64(ptr %ptr) #1
; GFX11-NEXT: v_dual_mov_b32 v1, s1 :: v_dual_lshlrev_b32 v4, 3, v0
; GFX11-NEXT: v_mov_b32_e32 v0, s0
; GFX11-NEXT: v_add_co_u32 v0, vcc_lo, v0, v4
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-NEXT: flat_atomic_dec_u64 v[0:1], v[2:3] offset:40
; GFX11-NEXT: s_waitcnt lgkmcnt(0)
diff --git a/llvm/test/CodeGen/AMDGPU/GlobalISel/atomicrmw_uinc_wrap.ll b/llvm/test/CodeGen/AMDGPU/GlobalISel/atomicrmw_uinc_wrap.ll
index bb26e2131473fe..ec351e88608ba0 100644
--- a/llvm/test/CodeGen/AMDGPU/GlobalISel/atomicrmw_uinc_wrap.ll
+++ b/llvm/test/CodeGen/AMDGPU/GlobalISel/atomicrmw_uinc_wrap.ll
@@ -3040,14 +3040,14 @@ define amdgpu_kernel void @flat_atomic_inc_ret_i32_offset_addr64(ptr %out, ptr %
; GFX11-NEXT: v_dual_mov_b32 v1, s3 :: v_dual_lshlrev_b32 v2, 2, v0
; GFX11-NEXT: v_mov_b32_e32 v0, s2
; GFX11-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-NEXT: flat_atomic_inc_u32 v3, v[0:1], v3 offset:20 glc
; GFX11-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
; GFX11-NEXT: buffer_gl1_inv
; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: v_dual_mov_b32 v0, s0 :: v_dual_mov_b32 v1, s1
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-NEXT: flat_store_b32 v[0:1], v3
@@ -3062,7 +3062,7 @@ define amdgpu_kernel void @flat_atomic_inc_ret_i32_offset_addr64(ptr %out, ptr %
; GFX12-NEXT: v_dual_mov_b32 v1, s3 :: v_dual_lshlrev_b32 v2, 2, v0
; GFX12-NEXT: v_mov_b32_e32 v0, s2
; GFX12-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
-; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_4) | instid1(VALU_DEP_1)
+; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_4) | instid1(VALU_DEP_1)
; GFX12-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX12-NEXT: flat_atomic_inc_u32 v3, v[0:1], v3 offset:20 th:TH_ATOMIC_RETURN scope:SCOPE_DEV
; GFX12-NEXT: s_wait_loadcnt_dscnt 0x0
@@ -3173,7 +3173,7 @@ define amdgpu_kernel void @flat_atomic_inc_noret_i32_offset_addr64(ptr %ptr) #1
; GFX11-NEXT: v_lshlrev_b32_e32 v2, 2, v0
; GFX11-NEXT: v_mov_b32_e32 v0, s0
; GFX11-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_4)
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-NEXT: v_mov_b32_e32 v2, 42
; GFX11-NEXT: flat_atomic_inc_u32 v[0:1], v2 offset:20
@@ -3192,7 +3192,7 @@ define amdgpu_kernel void @flat_atomic_inc_noret_i32_offset_addr64(ptr %ptr) #1
; GFX12-NEXT: v_lshlrev_b32_e32 v2, 2, v0
; GFX12-NEXT: v_mov_b32_e32 v0, s0
; GFX12-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
-; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_4)
; GFX12-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX12-NEXT: v_mov_b32_e32 v2, 42
; GFX12-NEXT: flat_atomic_inc_u32 v[0:1], v2 offset:20 scope:SCOPE_DEV
@@ -4117,14 +4117,14 @@ define amdgpu_kernel void @flat_atomic_inc_ret_i64_offset_addr64(ptr %out, ptr %
; GFX11-NEXT: v_dual_mov_b32 v1, s3 :: v_dual_lshlrev_b32 v4, 3, v0
; GFX11-NEXT: v_mov_b32_e32 v0, s2
; GFX11-NEXT: v_add_co_u32 v0, vcc_lo, v0, v4
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-NEXT: flat_atomic_inc_u64 v[0:1], v[0:1], v[2:3] offset:40 glc
; GFX11-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
; GFX11-NEXT: buffer_gl1_inv
; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: v_dual_mov_b32 v2, s0 :: v_dual_mov_b32 v3, s1
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-NEXT: v_add_co_u32 v2, vcc_lo, v2, v4
; GFX11-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX11-NEXT: flat_store_b64 v[2:3], v[0:1]
@@ -4140,7 +4140,7 @@ define amdgpu_kernel void @flat_atomic_inc_ret_i64_offset_addr64(ptr %out, ptr %
; GFX12-NEXT: v_dual_mov_b32 v1, s3 :: v_dual_lshlrev_b32 v4, 3, v0
; GFX12-NEXT: v_mov_b32_e32 v0, s2
; GFX12-NEXT: v_add_co_u32 v0, vcc_lo, v0, v4
-; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_4) | instid1(VALU_DEP_1)
+; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_4) | instid1(VALU_DEP_1)
; GFX12-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX12-NEXT: flat_atomic_inc_u64 v[0:1], v[0:1], v[2:3] offset:40 th:TH_ATOMIC_RETURN scope:SCOPE_DEV
; GFX12-NEXT: s_wait_loadcnt_dscnt 0x0
@@ -4256,7 +4256,7 @@ define amdgpu_kernel void @flat_atomic_inc_noret_i64_offset_addr64(ptr %ptr) #1
; GFX11-NEXT: v_dual_mov_b32 v1, s1 :: v_dual_lshlrev_b32 v4, 3, v0
; GFX11-NEXT: v_mov_b32_e32 v0, s0
; GFX11-NEXT: v_add_co_u32 v0, vcc_lo, v0, v4
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-NEXT: flat_atomic_inc_u64 v[0:1], v[2:3] offset:40
; GFX11-NEXT: s_waitcnt lgkmcnt(0)
@@ -4275,7 +4275,7 @@ define amdgpu_kernel void @flat_atomic_inc_noret_i64_offset_addr64(ptr %ptr) #1
; GFX12-NEXT: v_dual_mov_b32 v1, s1 :: v_dual_lshlrev_b32 v4, 3, v0
; GFX12-NEXT: v_mov_b32_e32 v0, s0
; GFX12-NEXT: v_add_co_u32 v0, vcc_lo, v0, v4
-; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX12-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX12-NEXT: flat_atomic_inc_u64 v[0:1], v[2:3] offset:40 scope:SCOPE_DEV
; GFX12-NEXT: s_wait_storecnt_dscnt 0x0
diff --git a/llvm/test/CodeGen/AMDGPU/GlobalISel/buffer-load-byte-short.ll b/llvm/test/CodeGen/AMDGPU/GlobalISel/buffer-load-byte-short.ll
index b062f7ccea0686..ce2fbadff2af26 100644
--- a/llvm/test/CodeGen/AMDGPU/GlobalISel/buffer-load-byte-short.ll
+++ b/llvm/test/CodeGen/AMDGPU/GlobalISel/buffer-load-byte-short.ll
@@ -162,7 +162,7 @@ define amdgpu_ps void @test_buffer_load_u8_waterfall_rsrc(<4 x i32> %rsrc, i32 %
; GFX12-NEXT: v_cmp_eq_u64_e32 vcc_lo, s[4:5], v[0:1]
; GFX12-NEXT: v_cmp_eq_u64_e64 s1, s[6:7], v[2:3]
; GFX12-NEXT: s_and_b32 s1, vcc_lo, s1
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_and_saveexec_b32 s1, s1
; GFX12-NEXT: buffer_load_u8 v1, v4, s[4:7], s0 offen
; GFX12-NEXT: ; implicit-def: $vgpr0
@@ -172,6 +172,7 @@ define amdgpu_ps void @test_buffer_load_u8_waterfall_rsrc(<4 x i32> %rsrc, i32 %
; GFX12-NEXT: s_cbranch_execnz .LBB8_1
; GFX12-NEXT: ; %bb.2:
; GFX12-NEXT: s_mov_b32 exec_lo, s2
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: v_mov_b32_e32 v0, 0
; GFX12-NEXT: s_wait_loadcnt 0x0
; GFX12-NEXT: global_store_b32 v0, v1, s[8:9]
@@ -196,9 +197,11 @@ define amdgpu_ps void @test_buffer_load_i8_waterfall_soffset(<4 x i32> inreg %rs
; GFX12-NEXT: ; implicit-def: $vgpr1
; GFX12-NEXT: ; implicit-def: $vgpr0
; GFX12-NEXT: s_xor_b32 exec_lo, exec_lo, s7
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB9_1
; GFX12-NEXT: ; %bb.2:
; GFX12-NEXT: s_mov_b32 exec_lo, s6
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: v_mov_b32_e32 v0, 0
; GFX12-NEXT: s_wait_loadcnt 0x0
; GFX12-NEXT: global_store_b32 v0, v2, s[4:5]
@@ -235,9 +238,11 @@ define amdgpu_ps void @test_buffer_load_u16_waterfall_both(<4 x i32> %rsrc, i32
; GFX12-NEXT: ; implicit-def: $vgpr4
; GFX12-NEXT: ; implicit-def: $vgpr2_vgpr3
; GFX12-NEXT: s_xor_b32 exec_lo, exec_lo, s2
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB10_1
; GFX12-NEXT: ; %bb.2:
; GFX12-NEXT: s_mov_b32 exec_lo, s8
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: v_mov_b32_e32 v0, 0
; GFX12-NEXT: s_wait_loadcnt 0x0
; GFX12-NEXT: global_store_b32 v0, v1, s[0:1]
diff --git a/llvm/test/CodeGen/AMDGPU/GlobalISel/extractelement.i128.ll b/llvm/test/CodeGen/AMDGPU/GlobalISel/extractelement.i128.ll
index 56fb1c61a3ea7e..245da3ee0e035b 100644
--- a/llvm/test/CodeGen/AMDGPU/GlobalISel/extractelement.i128.ll
+++ b/llvm/test/CodeGen/AMDGPU/GlobalISel/extractelement.i128.ll
@@ -160,7 +160,7 @@ define amdgpu_ps i128 @extractelement_vgpr_v4i128_sgpr_idx(ptr addrspace(1) %ptr
; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_dual_mov_b32 v3, s1 :: v_dual_mov_b32 v2, s0
; GFX11-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, v1, v3, vcc_lo
; GFX11-NEXT: global_load_b128 v[0:3], v[0:1], off
; GFX11-NEXT: s_waitcnt vmcnt(0)
@@ -229,7 +229,6 @@ define i128 @extractelement_vgpr_v4i128_vgpr_idx(ptr addrspace(1) %ptr, i32 %idx
; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_lshlrev_b32_e32 v2, 4, v2
; GFX11-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-NEXT: global_load_b128 v[0:3], v[0:1], off
; GFX11-NEXT: s_waitcnt vmcnt(0)
diff --git a/llvm/test/CodeGen/AMDGPU/GlobalISel/extractelement.i16.ll b/llvm/test/CodeGen/AMDGPU/GlobalISel/extractelement.i16.ll
index 33823dcbe927eb..72c161e937d235 100644
--- a/llvm/test/CodeGen/AMDGPU/GlobalISel/extractelement.i16.ll
+++ b/llvm/test/CodeGen/AMDGPU/GlobalISel/extractelement.i16.ll
@@ -133,7 +133,7 @@ define amdgpu_ps i16 @extractelement_vgpr_v4i16_sgpr_idx(ptr addrspace(1) %ptr,
; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_dual_mov_b32 v3, s1 :: v_dual_mov_b32 v2, s0
; GFX11-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, v1, v3, vcc_lo
; GFX11-NEXT: global_load_u16 v0, v[0:1], off
; GFX11-NEXT: s_waitcnt vmcnt(0)
@@ -199,7 +199,6 @@ define i16 @extractelement_vgpr_v4i16_vgpr_idx(ptr addrspace(1) %ptr, i32 %idx)
; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_lshlrev_b32_e32 v2, 1, v2
; GFX11-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-NEXT: global_load_u16 v0, v[0:1], off
; GFX11-NEXT: s_waitcnt vmcnt(0)
@@ -777,7 +776,7 @@ define amdgpu_ps i16 @extractelement_vgpr_v8i16_sgpr_idx(ptr addrspace(1) %ptr,
; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_dual_mov_b32 v3, s1 :: v_dual_mov_b32 v2, s0
; GFX11-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, v1, v3, vcc_lo
; GFX11-NEXT: global_load_u16 v0, v[0:1], off
; GFX11-NEXT: s_waitcnt vmcnt(0)
@@ -843,7 +842,6 @@ define i16 @extractelement_vgpr_v8i16_vgpr_idx(ptr addrspace(1) %ptr, i32 %idx)
; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_lshlrev_b32_e32 v2, 1, v2
; GFX11-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-NEXT: global_load_u16 v0, v[0:1], off
; GFX11-NEXT: s_waitcnt vmcnt(0)
diff --git a/llvm/test/CodeGen/AMDGPU/GlobalISel/extractelement.i8.ll b/llvm/test/CodeGen/AMDGPU/GlobalISel/extractelement.i8.ll
index faf2898562d573..838ac18234d499 100644
--- a/llvm/test/CodeGen/AMDGPU/GlobalISel/extractelement.i8.ll
+++ b/llvm/test/CodeGen/AMDGPU/GlobalISel/extractelement.i8.ll
@@ -131,7 +131,7 @@ define amdgpu_ps i8 @extractelement_vgpr_v4i8_sgpr_idx(ptr addrspace(1) %ptr, i3
; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_ashr_i32 s1, s0, 31
; GFX11-NEXT: v_dual_mov_b32 v2, s0 :: v_dual_mov_b32 v3, s1
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, v1, v3, vcc_lo
; GFX11-NEXT: global_load_u8 v0, v[0:1], off
@@ -195,7 +195,7 @@ define i8 @extractelement_vgpr_v4i8_vgpr_idx(ptr addrspace(1) %ptr, i32 %idx) {
; GFX11: ; %bb.0:
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: v_and_b32_e32 v2, 3, v2
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX11-NEXT: v_ashrrev_i32_e32 v3, 31, v2
; GFX11-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, v1, v3, vcc_lo
@@ -267,7 +267,7 @@ define amdgpu_ps i8 @extractelement_sgpr_v4i8_vgpr_idx(ptr addrspace(4) inreg %p
; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-NEXT: v_ashrrev_i32_e32 v3, 31, v2
; GFX11-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, v1, v3, vcc_lo
; GFX11-NEXT: global_load_u8 v0, v[0:1], off
; GFX11-NEXT: s_waitcnt vmcnt(0)
@@ -784,7 +784,7 @@ define amdgpu_ps i8 @extractelement_vgpr_v8i8_sgpr_idx(ptr addrspace(1) %ptr, i3
; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_ashr_i32 s1, s0, 31
; GFX11-NEXT: v_dual_mov_b32 v2, s0 :: v_dual_mov_b32 v3, s1
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, v1, v3, vcc_lo
; GFX11-NEXT: global_load_u8 v0, v[0:1], off
@@ -848,7 +848,7 @@ define i8 @extractelement_vgpr_v8i8_vgpr_idx(ptr addrspace(1) %ptr, i32 %idx) {
; GFX11: ; %bb.0:
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: v_and_b32_e32 v2, 7, v2
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX11-NEXT: v_ashrrev_i32_e32 v3, 31, v2
; GFX11-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, v1, v3, vcc_lo
@@ -920,7 +920,7 @@ define amdgpu_ps i8 @extractelement_sgpr_v8i8_vgpr_idx(ptr addrspace(4) inreg %p
; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-NEXT: v_ashrrev_i32_e32 v3, 31, v2
; GFX11-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, v1, v3, vcc_lo
; GFX11-NEXT: global_load_u8 v0, v[0:1], off
; GFX11-NEXT: s_waitcnt vmcnt(0)
@@ -1821,7 +1821,7 @@ define amdgpu_ps i8 @extractelement_vgpr_v16i8_sgpr_idx(ptr addrspace(1) %ptr, i
; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_ashr_i32 s1, s0, 31
; GFX11-NEXT: v_dual_mov_b32 v2, s0 :: v_dual_mov_b32 v3, s1
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, v1, v3, vcc_lo
; GFX11-NEXT: global_load_u8 v0, v[0:1], off
@@ -1885,7 +1885,7 @@ define i8 @extractelement_vgpr_v16i8_vgpr_idx(ptr addrspace(1) %ptr, i32 %idx) {
; GFX11: ; %bb.0:
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: v_and_b32_e32 v2, 15, v2
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX11-NEXT: v_ashrrev_i32_e32 v3, 31, v2
; GFX11-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, v1, v3, vcc_lo
@@ -1957,7 +1957,7 @@ define amdgpu_ps i8 @extractelement_sgpr_v16i8_vgpr_idx(ptr addrspace(4) inreg %
; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-NEXT: v_ashrrev_i32_e32 v3, 31, v2
; GFX11-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, v1, v3, vcc_lo
; GFX11-NEXT: global_load_u8 v0, v[0:1], off
; GFX11-NEXT: s_waitcnt vmcnt(0)
diff --git a/llvm/test/CodeGen/AMDGPU/GlobalISel/fabs.ll b/llvm/test/CodeGen/AMDGPU/GlobalISel/fabs.ll
index fb81eba869caa5..61c4c3b60b2b4b 100644
--- a/llvm/test/CodeGen/AMDGPU/GlobalISel/fabs.ll
+++ b/llvm/test/CodeGen/AMDGPU/GlobalISel/fabs.ll
@@ -46,8 +46,8 @@ define amdgpu_ps void @s_fabs_f16_salu_use(half inreg %in, i32 inreg %val, ptr a
; GFX12: ; %bb.0:
; GFX12-NEXT: s_and_b32 s0, s0, 0x7fff
; GFX12-NEXT: s_cmp_eq_u32 s1, 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s0, s0, 0
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: v_mov_b32_e32 v2, s0
; GFX12-NEXT: global_store_b16 v[0:1], v2, off
; GFX12-NEXT: s_endpgm
@@ -102,8 +102,8 @@ define amdgpu_ps void @s_fabs_f32_salu_use(float inreg %in, i32 inreg %val, ptr
; GFX12: ; %bb.0:
; GFX12-NEXT: s_bitset0_b32 s0, 31
; GFX12-NEXT: s_cmp_eq_u32 s1, 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s0, s0, 0
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: v_mov_b32_e32 v2, s0
; GFX12-NEXT: global_store_b32 v[0:1], v2, off
; GFX12-NEXT: s_endpgm
@@ -285,8 +285,8 @@ define amdgpu_ps void @s_fabs_v2f32_salu_use(<2 x float> inreg %in, i32 inreg %v
; GFX12-NEXT: s_bitset0_b32 s0, 31
; GFX12-NEXT: s_bitset0_b32 s1, 31
; GFX12-NEXT: s_cmp_eq_u32 s2, 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b64 s[0:1], s[0:1], 0
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: v_dual_mov_b32 v3, s1 :: v_dual_mov_b32 v2, s0
; GFX12-NEXT: global_store_b64 v[0:1], v[2:3], off
; GFX12-NEXT: s_endpgm
diff --git a/llvm/test/CodeGen/AMDGPU/GlobalISel/fcmp.ll b/llvm/test/CodeGen/AMDGPU/GlobalISel/fcmp.ll
index 6011841643f155..8ac2b9e45392c3 100644
--- a/llvm/test/CodeGen/AMDGPU/GlobalISel/fcmp.ll
+++ b/llvm/test/CodeGen/AMDGPU/GlobalISel/fcmp.ll
@@ -102,59 +102,73 @@ define void @fcmp_f16_uniform(half inreg %a, half inreg %b, ptr %p) {
; GFX12-NEXT: s_wait_bvhcnt 0x0
; GFX12-NEXT: s_wait_kmcnt 0x0
; GFX12-NEXT: s_cmp_eq_f16 s0, s1
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_3) | instskip(SKIP_1) | instid1(SALU_CYCLE_3)
; GFX12-NEXT: s_cselect_b32 s2, 1, 0
; GFX12-NEXT: s_cmp_gt_f16 s0, s1
; GFX12-NEXT: s_cselect_b32 s3, 1, 0
; GFX12-NEXT: s_cmp_ge_f16 s0, s1
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_3) | instskip(SKIP_1) | instid1(SALU_CYCLE_3)
; GFX12-NEXT: s_cselect_b32 s4, 1, 0
; GFX12-NEXT: s_cmp_lt_f16 s0, s1
; GFX12-NEXT: s_cselect_b32 s5, 1, 0
; GFX12-NEXT: s_cmp_le_f16 s0, s1
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_3) | instskip(SKIP_1) | instid1(SALU_CYCLE_3)
; GFX12-NEXT: s_cselect_b32 s6, 1, 0
; GFX12-NEXT: s_cmp_lg_f16 s0, s1
; GFX12-NEXT: s_cselect_b32 s7, 1, 0
; GFX12-NEXT: s_cmp_o_f16 s0, s1
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_3) | instskip(SKIP_1) | instid1(SALU_CYCLE_3)
; GFX12-NEXT: s_cselect_b32 s8, 1, 0
; GFX12-NEXT: s_cmp_nlg_f16 s0, s1
; GFX12-NEXT: s_cselect_b32 s9, 1, 0
; GFX12-NEXT: s_cmp_nle_f16 s0, s1
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_3) | instskip(SKIP_1) | instid1(SALU_CYCLE_3)
; GFX12-NEXT: s_cselect_b32 s10, 1, 0
; GFX12-NEXT: s_cmp_nlt_f16 s0, s1
; GFX12-NEXT: s_cselect_b32 s11, 1, 0
; GFX12-NEXT: s_cmp_nge_f16 s0, s1
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_3) | instskip(SKIP_1) | instid1(SALU_CYCLE_3)
; GFX12-NEXT: s_cselect_b32 s12, 1, 0
; GFX12-NEXT: s_cmp_ngt_f16 s0, s1
; GFX12-NEXT: s_cselect_b32 s13, 1, 0
; GFX12-NEXT: s_cmp_neq_f16 s0, s1
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_3) | instskip(SKIP_1) | instid1(SALU_CYCLE_3)
; GFX12-NEXT: s_cselect_b32 s14, 1, 0
; GFX12-NEXT: s_cmp_u_f16 s0, s1
; GFX12-NEXT: s_cselect_b32 s0, 1, 0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_cmp_lg_u32 s2, 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s1, 1, 0
; GFX12-NEXT: s_cmp_lg_u32 s3, 0
; GFX12-NEXT: s_cselect_b32 s2, 1, 0
; GFX12-NEXT: s_cmp_lg_u32 s4, 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s3, 1, 0
; GFX12-NEXT: s_cmp_lg_u32 s5, 0
; GFX12-NEXT: s_cselect_b32 s4, 1, 0
; GFX12-NEXT: s_cmp_lg_u32 s6, 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s5, 1, 0
; GFX12-NEXT: s_cmp_lg_u32 s7, 0
; GFX12-NEXT: s_cselect_b32 s6, 1, 0
; GFX12-NEXT: s_cmp_lg_u32 s8, 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s7, 1, 0
; GFX12-NEXT: s_cmp_lg_u32 s9, 0
; GFX12-NEXT: s_cselect_b32 s8, 1, 0
; GFX12-NEXT: s_cmp_lg_u32 s10, 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s9, 1, 0
; GFX12-NEXT: s_cmp_lg_u32 s11, 0
; GFX12-NEXT: s_cselect_b32 s10, 1, 0
; GFX12-NEXT: s_cmp_lg_u32 s12, 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s11, 1, 0
; GFX12-NEXT: s_cmp_lg_u32 s13, 0
; GFX12-NEXT: s_cselect_b32 s12, 1, 0
; GFX12-NEXT: s_cmp_lg_u32 s14, 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s13, 1, 0
; GFX12-NEXT: s_cmp_lg_u32 s0, 0
; GFX12-NEXT: s_cselect_b32 s0, 1, 0
@@ -487,59 +501,73 @@ define void @fcmp_f32_uniform(float inreg %a, float inreg %b, ptr %p) {
; GFX12-NEXT: s_wait_bvhcnt 0x0
; GFX12-NEXT: s_wait_kmcnt 0x0
; GFX12-NEXT: s_cmp_eq_f32 s0, s1
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_3) | instskip(SKIP_1) | instid1(SALU_CYCLE_3)
; GFX12-NEXT: s_cselect_b32 s2, 1, 0
; GFX12-NEXT: s_cmp_gt_f32 s0, s1
; GFX12-NEXT: s_cselect_b32 s3, 1, 0
; GFX12-NEXT: s_cmp_ge_f32 s0, s1
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_3) | instskip(SKIP_1) | instid1(SALU_CYCLE_3)
; GFX12-NEXT: s_cselect_b32 s4, 1, 0
; GFX12-NEXT: s_cmp_lt_f32 s0, s1
; GFX12-NEXT: s_cselect_b32 s5, 1, 0
; GFX12-NEXT: s_cmp_le_f32 s0, s1
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_3) | instskip(SKIP_1) | instid1(SALU_CYCLE_3)
; GFX12-NEXT: s_cselect_b32 s6, 1, 0
; GFX12-NEXT: s_cmp_lg_f32 s0, s1
; GFX12-NEXT: s_cselect_b32 s7, 1, 0
; GFX12-NEXT: s_cmp_o_f32 s0, s1
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_3) | instskip(SKIP_1) | instid1(SALU_CYCLE_3)
; GFX12-NEXT: s_cselect_b32 s8, 1, 0
; GFX12-NEXT: s_cmp_nlg_f32 s0, s1
; GFX12-NEXT: s_cselect_b32 s9, 1, 0
; GFX12-NEXT: s_cmp_nle_f32 s0, s1
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_3) | instskip(SKIP_1) | instid1(SALU_CYCLE_3)
; GFX12-NEXT: s_cselect_b32 s10, 1, 0
; GFX12-NEXT: s_cmp_nlt_f32 s0, s1
; GFX12-NEXT: s_cselect_b32 s11, 1, 0
; GFX12-NEXT: s_cmp_nge_f32 s0, s1
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_3) | instskip(SKIP_1) | instid1(SALU_CYCLE_3)
; GFX12-NEXT: s_cselect_b32 s12, 1, 0
; GFX12-NEXT: s_cmp_ngt_f32 s0, s1
; GFX12-NEXT: s_cselect_b32 s13, 1, 0
; GFX12-NEXT: s_cmp_neq_f32 s0, s1
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_3) | instskip(SKIP_1) | instid1(SALU_CYCLE_3)
; GFX12-NEXT: s_cselect_b32 s14, 1, 0
; GFX12-NEXT: s_cmp_u_f32 s0, s1
; GFX12-NEXT: s_cselect_b32 s0, 1, 0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_cmp_lg_u32 s2, 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s1, 1, 0
; GFX12-NEXT: s_cmp_lg_u32 s3, 0
; GFX12-NEXT: s_cselect_b32 s2, 1, 0
; GFX12-NEXT: s_cmp_lg_u32 s4, 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s3, 1, 0
; GFX12-NEXT: s_cmp_lg_u32 s5, 0
; GFX12-NEXT: s_cselect_b32 s4, 1, 0
; GFX12-NEXT: s_cmp_lg_u32 s6, 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s5, 1, 0
; GFX12-NEXT: s_cmp_lg_u32 s7, 0
; GFX12-NEXT: s_cselect_b32 s6, 1, 0
; GFX12-NEXT: s_cmp_lg_u32 s8, 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s7, 1, 0
; GFX12-NEXT: s_cmp_lg_u32 s9, 0
; GFX12-NEXT: s_cselect_b32 s8, 1, 0
; GFX12-NEXT: s_cmp_lg_u32 s10, 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s9, 1, 0
; GFX12-NEXT: s_cmp_lg_u32 s11, 0
; GFX12-NEXT: s_cselect_b32 s10, 1, 0
; GFX12-NEXT: s_cmp_lg_u32 s12, 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s11, 1, 0
; GFX12-NEXT: s_cmp_lg_u32 s13, 0
; GFX12-NEXT: s_cselect_b32 s12, 1, 0
; GFX12-NEXT: s_cmp_lg_u32 s14, 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s13, 1, 0
; GFX12-NEXT: s_cmp_lg_u32 s0, 0
; GFX12-NEXT: s_cselect_b32 s0, 1, 0
@@ -886,59 +914,73 @@ define void @fcmp_f64_uniform(double inreg %a, double inreg %b, ptr %p) {
; GFX12-NEXT: v_cmp_neq_f64_e64 s16, s[0:1], s[2:3]
; GFX12-NEXT: v_cmp_u_f64_e64 s0, s[0:1], s[2:3]
; GFX12-NEXT: s_cmp_lg_u32 s4, 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s4, 1, 0
; GFX12-NEXT: s_cmp_lg_u32 s5, 0
; GFX12-NEXT: s_cselect_b32 s1, 1, 0
; GFX12-NEXT: s_cmp_lg_u32 s6, 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s2, 1, 0
; GFX12-NEXT: s_cmp_lg_u32 s7, 0
; GFX12-NEXT: s_cselect_b32 s3, 1, 0
; GFX12-NEXT: s_cmp_lg_u32 s8, 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s5, 1, 0
; GFX12-NEXT: s_cmp_lg_u32 s9, 0
; GFX12-NEXT: s_cselect_b32 s6, 1, 0
; GFX12-NEXT: s_cmp_lg_u32 s10, 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s7, 1, 0
; GFX12-NEXT: s_cmp_lg_u32 s11, 0
; GFX12-NEXT: s_cselect_b32 s8, 1, 0
; GFX12-NEXT: s_cmp_lg_u32 s12, 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s9, 1, 0
; GFX12-NEXT: s_cmp_lg_u32 s13, 0
; GFX12-NEXT: s_cselect_b32 s10, 1, 0
; GFX12-NEXT: s_cmp_lg_u32 s14, 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s11, 1, 0
; GFX12-NEXT: s_cmp_lg_u32 s15, 0
; GFX12-NEXT: s_cselect_b32 s12, 1, 0
; GFX12-NEXT: s_cmp_lg_u32 s16, 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s13, 1, 0
; GFX12-NEXT: s_cmp_lg_u32 s0, 0
; GFX12-NEXT: s_cselect_b32 s0, 1, 0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_cmp_lg_u32 s4, 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s4, 1, 0
; GFX12-NEXT: s_cmp_lg_u32 s1, 0
; GFX12-NEXT: s_cselect_b32 s1, 1, 0
; GFX12-NEXT: s_cmp_lg_u32 s2, 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s2, 1, 0
; GFX12-NEXT: s_cmp_lg_u32 s3, 0
; GFX12-NEXT: s_cselect_b32 s3, 1, 0
; GFX12-NEXT: s_cmp_lg_u32 s5, 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s5, 1, 0
; GFX12-NEXT: s_cmp_lg_u32 s6, 0
; GFX12-NEXT: s_cselect_b32 s6, 1, 0
; GFX12-NEXT: s_cmp_lg_u32 s7, 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s7, 1, 0
; GFX12-NEXT: s_cmp_lg_u32 s8, 0
; GFX12-NEXT: s_cselect_b32 s8, 1, 0
; GFX12-NEXT: s_cmp_lg_u32 s9, 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s9, 1, 0
; GFX12-NEXT: s_cmp_lg_u32 s10, 0
; GFX12-NEXT: s_cselect_b32 s10, 1, 0
; GFX12-NEXT: s_cmp_lg_u32 s11, 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s11, 1, 0
; GFX12-NEXT: s_cmp_lg_u32 s12, 0
; GFX12-NEXT: s_cselect_b32 s12, 1, 0
; GFX12-NEXT: s_cmp_lg_u32 s13, 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s13, 1, 0
; GFX12-NEXT: s_cmp_lg_u32 s0, 0
; GFX12-NEXT: s_cselect_b32 s0, 1, 0
diff --git a/llvm/test/CodeGen/AMDGPU/GlobalISel/fdiv.f32.ll b/llvm/test/CodeGen/AMDGPU/GlobalISel/fdiv.f32.ll
index 276333bd234bff..ad886ebb37bbaa 100644
--- a/llvm/test/CodeGen/AMDGPU/GlobalISel/fdiv.f32.ll
+++ b/llvm/test/CodeGen/AMDGPU/GlobalISel/fdiv.f32.ll
@@ -196,8 +196,9 @@ define float @v_fdiv_f32(float %a, float %b) #1 {
; GFX11-FLUSH-NEXT: v_fmac_f32_e32 v5, v6, v3
; GFX11-FLUSH-NEXT: v_fma_f32 v2, -v2, v5, v4
; GFX11-FLUSH-NEXT: s_denorm_mode 0
-; GFX11-FLUSH-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX11-FLUSH-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-FLUSH-NEXT: v_div_fmas_f32 v2, v2, v3, v5
+; GFX11-FLUSH-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-FLUSH-NEXT: v_div_fixup_f32 v0, v2, v1, v0
; GFX11-FLUSH-NEXT: s_setpc_b64 s[30:31]
%fdiv = fdiv float %a, %b
@@ -374,9 +375,10 @@ define float @v_fdiv_f32_dynamic_denorm(float %a, float %b) #0 {
; GFX11-IEEE-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-IEEE-NEXT: v_fma_f32 v6, -v2, v5, v4
; GFX11-IEEE-NEXT: v_fmac_f32_e32 v5, v6, v3
-; GFX11-IEEE-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
+; GFX11-IEEE-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-IEEE-NEXT: v_fma_f32 v2, -v2, v5, v4
; GFX11-IEEE-NEXT: s_setreg_b32 hwreg(HW_REG_MODE, 4, 2), s0
+; GFX11-IEEE-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-IEEE-NEXT: v_div_fmas_f32 v2, v2, v3, v5
; GFX11-IEEE-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-IEEE-NEXT: v_div_fixup_f32 v0, v2, v1, v0
@@ -398,9 +400,10 @@ define float @v_fdiv_f32_dynamic_denorm(float %a, float %b) #0 {
; GFX11-FLUSH-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-FLUSH-NEXT: v_fma_f32 v6, -v2, v5, v4
; GFX11-FLUSH-NEXT: v_fmac_f32_e32 v5, v6, v3
-; GFX11-FLUSH-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
+; GFX11-FLUSH-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-FLUSH-NEXT: v_fma_f32 v2, -v2, v5, v4
; GFX11-FLUSH-NEXT: s_setreg_b32 hwreg(HW_REG_MODE, 4, 2), s0
+; GFX11-FLUSH-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-FLUSH-NEXT: v_div_fmas_f32 v2, v2, v3, v5
; GFX11-FLUSH-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-FLUSH-NEXT: v_div_fixup_f32 v0, v2, v1, v0
@@ -511,13 +514,13 @@ define float @v_fdiv_f32_ulp25(float %a, float %b) #1 {
; GFX11-FLUSH: ; %bb.0:
; GFX11-FLUSH-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-FLUSH-NEXT: v_cmp_lt_f32_e64 s0, 0x6f800000, |v1|
-; GFX11-FLUSH-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-FLUSH-NEXT: v_cndmask_b32_e64 v2, 1.0, 0x2f800000, s0
+; GFX11-FLUSH-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-FLUSH-NEXT: v_mul_f32_e32 v1, v1, v2
-; GFX11-FLUSH-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_2) | instid1(VALU_DEP_1)
; GFX11-FLUSH-NEXT: v_rcp_f32_e32 v1, v1
; GFX11-FLUSH-NEXT: s_waitcnt_depctr depctr_va_vdst(0)
; GFX11-FLUSH-NEXT: v_mul_f32_e32 v0, v0, v1
+; GFX11-FLUSH-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-FLUSH-NEXT: v_mul_f32_e32 v0, v2, v0
; GFX11-FLUSH-NEXT: s_setpc_b64 s[30:31]
%fdiv = fdiv float %a, %b, !fpmath !0
@@ -813,8 +816,9 @@ define float @v_rcp_f32(float %x) #1 {
; GFX11-FLUSH-NEXT: v_fmac_f32_e32 v4, v5, v2
; GFX11-FLUSH-NEXT: v_fma_f32 v1, -v1, v4, v3
; GFX11-FLUSH-NEXT: s_denorm_mode 0
-; GFX11-FLUSH-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX11-FLUSH-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-FLUSH-NEXT: v_div_fmas_f32 v1, v1, v2, v4
+; GFX11-FLUSH-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-FLUSH-NEXT: v_div_fixup_f32 v0, v1, v0, 1.0
; GFX11-FLUSH-NEXT: s_setpc_b64 s[30:31]
%fdiv = fdiv float 1.0, %x
@@ -997,8 +1001,9 @@ define float @v_rcp_f32_arcp(float %x) #1 {
; GFX11-FLUSH-NEXT: v_fmac_f32_e32 v4, v5, v2
; GFX11-FLUSH-NEXT: v_fma_f32 v1, -v1, v4, v3
; GFX11-FLUSH-NEXT: s_denorm_mode 0
-; GFX11-FLUSH-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX11-FLUSH-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-FLUSH-NEXT: v_div_fmas_f32 v1, v1, v2, v4
+; GFX11-FLUSH-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-FLUSH-NEXT: v_div_fixup_f32 v0, v1, v0, 1.0
; GFX11-FLUSH-NEXT: s_setpc_b64 s[30:31]
%fdiv = fdiv arcp float 1.0, %x
@@ -1440,10 +1445,10 @@ define <2 x float> @v_fdiv_v2f32(<2 x float> %a, <2 x float> %b) #1 {
; GFX11-IEEE-NEXT: v_fma_f32 v5, -v5, v11, v8
; GFX11-IEEE-NEXT: v_div_fmas_f32 v4, v4, v6, v9
; GFX11-IEEE-NEXT: s_mov_b32 vcc_lo, s0
-; GFX11-IEEE-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX11-IEEE-NEXT: s_delay_alu instid0(VALU_DEP_2) | instid1(SALU_CYCLE_1)
; GFX11-IEEE-NEXT: v_div_fmas_f32 v5, v5, v7, v11
+; GFX11-IEEE-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-IEEE-NEXT: v_div_fixup_f32 v0, v4, v2, v0
-; GFX11-IEEE-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX11-IEEE-NEXT: v_div_fixup_f32 v1, v5, v3, v1
; GFX11-IEEE-NEXT: s_setpc_b64 s[30:31]
;
@@ -1465,26 +1470,28 @@ define <2 x float> @v_fdiv_v2f32(<2 x float> %a, <2 x float> %b) #1 {
; GFX11-FLUSH-NEXT: v_fmac_f32_e32 v7, v8, v5
; GFX11-FLUSH-NEXT: v_fma_f32 v4, -v4, v7, v6
; GFX11-FLUSH-NEXT: s_denorm_mode 0
+; GFX11-FLUSH-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-FLUSH-NEXT: v_div_scale_f32 v6, null, v3, v3, v1
-; GFX11-FLUSH-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-FLUSH-NEXT: v_div_fmas_f32 v4, v4, v5, v7
+; GFX11-FLUSH-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-FLUSH-NEXT: v_rcp_f32_e32 v8, v6
-; GFX11-FLUSH-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_4) | instid1(VALU_DEP_1)
; GFX11-FLUSH-NEXT: v_div_fixup_f32 v0, v4, v2, v0
; GFX11-FLUSH-NEXT: v_div_scale_f32 v2, vcc_lo, v1, v3, v1
; GFX11-FLUSH-NEXT: s_denorm_mode 3
; GFX11-FLUSH-NEXT: s_waitcnt_depctr depctr_va_vdst(0)
; GFX11-FLUSH-NEXT: v_fma_f32 v4, -v6, v8, 1.0
-; GFX11-FLUSH-NEXT: v_fmac_f32_e32 v8, v4, v8
; GFX11-FLUSH-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX11-FLUSH-NEXT: v_fmac_f32_e32 v8, v4, v8
; GFX11-FLUSH-NEXT: v_mul_f32_e32 v4, v2, v8
-; GFX11-FLUSH-NEXT: v_fma_f32 v5, -v6, v4, v2
; GFX11-FLUSH-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX11-FLUSH-NEXT: v_fma_f32 v5, -v6, v4, v2
; GFX11-FLUSH-NEXT: v_fmac_f32_e32 v4, v5, v8
+; GFX11-FLUSH-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-FLUSH-NEXT: v_fma_f32 v2, -v6, v4, v2
; GFX11-FLUSH-NEXT: s_denorm_mode 0
-; GFX11-FLUSH-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX11-FLUSH-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-FLUSH-NEXT: v_div_fmas_f32 v2, v2, v8, v4
+; GFX11-FLUSH-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-FLUSH-NEXT: v_div_fixup_f32 v1, v2, v3, v1
; GFX11-FLUSH-NEXT: s_setpc_b64 s[30:31]
%fdiv = fdiv <2 x float> %a, %b
@@ -1637,7 +1644,6 @@ define <2 x float> @v_fdiv_v2f32_ulp25(<2 x float> %a, <2 x float> %b) #1 {
; GFX11-FLUSH: ; %bb.0:
; GFX11-FLUSH-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-FLUSH-NEXT: v_cmp_lt_f32_e64 s0, 0x6f800000, |v2|
-; GFX11-FLUSH-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-FLUSH-NEXT: v_cndmask_b32_e64 v4, 1.0, 0x2f800000, s0
; GFX11-FLUSH-NEXT: v_cmp_lt_f32_e64 s0, 0x6f800000, |v3|
; GFX11-FLUSH-NEXT: v_cndmask_b32_e64 v5, 1.0, 0x2f800000, s0
@@ -1919,10 +1925,10 @@ define <2 x float> @v_rcp_v2f32(<2 x float> %x) #1 {
; GFX11-IEEE-NEXT: v_fma_f32 v3, -v3, v9, v6
; GFX11-IEEE-NEXT: v_div_fmas_f32 v2, v2, v4, v7
; GFX11-IEEE-NEXT: s_mov_b32 vcc_lo, s0
-; GFX11-IEEE-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX11-IEEE-NEXT: s_delay_alu instid0(VALU_DEP_2) | instid1(SALU_CYCLE_1)
; GFX11-IEEE-NEXT: v_div_fmas_f32 v3, v3, v5, v9
+; GFX11-IEEE-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-IEEE-NEXT: v_div_fixup_f32 v0, v2, v0, 1.0
-; GFX11-IEEE-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX11-IEEE-NEXT: v_div_fixup_f32 v1, v3, v1, 1.0
; GFX11-IEEE-NEXT: s_setpc_b64 s[30:31]
;
@@ -1944,26 +1950,28 @@ define <2 x float> @v_rcp_v2f32(<2 x float> %x) #1 {
; GFX11-FLUSH-NEXT: v_fmac_f32_e32 v5, v6, v3
; GFX11-FLUSH-NEXT: v_fma_f32 v2, -v2, v5, v4
; GFX11-FLUSH-NEXT: s_denorm_mode 0
+; GFX11-FLUSH-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-FLUSH-NEXT: v_div_scale_f32 v4, null, v1, v1, 1.0
-; GFX11-FLUSH-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-FLUSH-NEXT: v_div_fmas_f32 v2, v2, v3, v5
+; GFX11-FLUSH-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-FLUSH-NEXT: v_rcp_f32_e32 v6, v4
-; GFX11-FLUSH-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_4) | instid1(VALU_DEP_1)
; GFX11-FLUSH-NEXT: v_div_fixup_f32 v0, v2, v0, 1.0
; GFX11-FLUSH-NEXT: v_div_scale_f32 v2, vcc_lo, 1.0, v1, 1.0
; GFX11-FLUSH-NEXT: s_denorm_mode 3
; GFX11-FLUSH-NEXT: s_waitcnt_depctr depctr_va_vdst(0)
; GFX11-FLUSH-NEXT: v_fma_f32 v3, -v4, v6, 1.0
-; GFX11-FLUSH-NEXT: v_fmac_f32_e32 v6, v3, v6
; GFX11-FLUSH-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX11-FLUSH-NEXT: v_fmac_f32_e32 v6, v3, v6
; GFX11-FLUSH-NEXT: v_mul_f32_e32 v3, v2, v6
-; GFX11-FLUSH-NEXT: v_fma_f32 v5, -v4, v3, v2
; GFX11-FLUSH-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX11-FLUSH-NEXT: v_fma_f32 v5, -v4, v3, v2
; GFX11-FLUSH-NEXT: v_fmac_f32_e32 v3, v5, v6
+; GFX11-FLUSH-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-FLUSH-NEXT: v_fma_f32 v2, -v4, v3, v2
; GFX11-FLUSH-NEXT: s_denorm_mode 0
-; GFX11-FLUSH-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX11-FLUSH-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-FLUSH-NEXT: v_div_fmas_f32 v2, v2, v6, v3
+; GFX11-FLUSH-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-FLUSH-NEXT: v_div_fixup_f32 v1, v2, v1, 1.0
; GFX11-FLUSH-NEXT: s_setpc_b64 s[30:31]
%fdiv = fdiv <2 x float> <float 1.0, float 1.0>, %x
@@ -2235,10 +2243,10 @@ define <2 x float> @v_rcp_v2f32_arcp(<2 x float> %x) #1 {
; GFX11-IEEE-NEXT: v_fma_f32 v3, -v3, v9, v6
; GFX11-IEEE-NEXT: v_div_fmas_f32 v2, v2, v4, v7
; GFX11-IEEE-NEXT: s_mov_b32 vcc_lo, s0
-; GFX11-IEEE-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX11-IEEE-NEXT: s_delay_alu instid0(VALU_DEP_2) | instid1(SALU_CYCLE_1)
; GFX11-IEEE-NEXT: v_div_fmas_f32 v3, v3, v5, v9
+; GFX11-IEEE-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-IEEE-NEXT: v_div_fixup_f32 v0, v2, v0, 1.0
-; GFX11-IEEE-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX11-IEEE-NEXT: v_div_fixup_f32 v1, v3, v1, 1.0
; GFX11-IEEE-NEXT: s_setpc_b64 s[30:31]
;
@@ -2260,26 +2268,28 @@ define <2 x float> @v_rcp_v2f32_arcp(<2 x float> %x) #1 {
; GFX11-FLUSH-NEXT: v_fmac_f32_e32 v5, v6, v3
; GFX11-FLUSH-NEXT: v_fma_f32 v2, -v2, v5, v4
; GFX11-FLUSH-NEXT: s_denorm_mode 0
+; GFX11-FLUSH-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-FLUSH-NEXT: v_div_scale_f32 v4, null, v1, v1, 1.0
-; GFX11-FLUSH-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-FLUSH-NEXT: v_div_fmas_f32 v2, v2, v3, v5
+; GFX11-FLUSH-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-FLUSH-NEXT: v_rcp_f32_e32 v6, v4
-; GFX11-FLUSH-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_4) | instid1(VALU_DEP_1)
; GFX11-FLUSH-NEXT: v_div_fixup_f32 v0, v2, v0, 1.0
; GFX11-FLUSH-NEXT: v_div_scale_f32 v2, vcc_lo, 1.0, v1, 1.0
; GFX11-FLUSH-NEXT: s_denorm_mode 3
; GFX11-FLUSH-NEXT: s_waitcnt_depctr depctr_va_vdst(0)
; GFX11-FLUSH-NEXT: v_fma_f32 v3, -v4, v6, 1.0
-; GFX11-FLUSH-NEXT: v_fmac_f32_e32 v6, v3, v6
; GFX11-FLUSH-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX11-FLUSH-NEXT: v_fmac_f32_e32 v6, v3, v6
; GFX11-FLUSH-NEXT: v_mul_f32_e32 v3, v2, v6
-; GFX11-FLUSH-NEXT: v_fma_f32 v5, -v4, v3, v2
; GFX11-FLUSH-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX11-FLUSH-NEXT: v_fma_f32 v5, -v4, v3, v2
; GFX11-FLUSH-NEXT: v_fmac_f32_e32 v3, v5, v6
+; GFX11-FLUSH-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-FLUSH-NEXT: v_fma_f32 v2, -v4, v3, v2
; GFX11-FLUSH-NEXT: s_denorm_mode 0
-; GFX11-FLUSH-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX11-FLUSH-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-FLUSH-NEXT: v_div_fmas_f32 v2, v2, v6, v3
+; GFX11-FLUSH-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-FLUSH-NEXT: v_div_fixup_f32 v1, v2, v1, 1.0
; GFX11-FLUSH-NEXT: s_setpc_b64 s[30:31]
%fdiv = fdiv arcp <2 x float> <float 1.0, float 1.0>, %x
@@ -2775,9 +2785,10 @@ define float @v_fdiv_f32_dynamic__nnan_ninf(float %x, float %y, float %z) #0 {
; GFX11-IEEE-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-IEEE-NEXT: v_fma_f32 v6, -v2, v5, v4
; GFX11-IEEE-NEXT: v_fmac_f32_e32 v5, v6, v3
-; GFX11-IEEE-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
+; GFX11-IEEE-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-IEEE-NEXT: v_fma_f32 v2, -v2, v5, v4
; GFX11-IEEE-NEXT: s_setreg_b32 hwreg(HW_REG_MODE, 4, 2), s0
+; GFX11-IEEE-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-IEEE-NEXT: v_div_fmas_f32 v2, v2, v3, v5
; GFX11-IEEE-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-IEEE-NEXT: v_div_fixup_f32 v0, v2, v1, v0
@@ -2799,9 +2810,10 @@ define float @v_fdiv_f32_dynamic__nnan_ninf(float %x, float %y, float %z) #0 {
; GFX11-FLUSH-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-FLUSH-NEXT: v_fma_f32 v6, -v2, v5, v4
; GFX11-FLUSH-NEXT: v_fmac_f32_e32 v5, v6, v3
-; GFX11-FLUSH-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
+; GFX11-FLUSH-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-FLUSH-NEXT: v_fma_f32 v2, -v2, v5, v4
; GFX11-FLUSH-NEXT: s_setreg_b32 hwreg(HW_REG_MODE, 4, 2), s0
+; GFX11-FLUSH-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-FLUSH-NEXT: v_div_fmas_f32 v2, v2, v3, v5
; GFX11-FLUSH-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-FLUSH-NEXT: v_div_fixup_f32 v0, v2, v1, v0
@@ -3288,9 +3300,10 @@ define float @v_fdiv_neglhs_f32_dynamic(float %x, float %y) #0 {
; GFX11-IEEE-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-IEEE-NEXT: v_fma_f32 v6, -v3, v5, v2
; GFX11-IEEE-NEXT: v_fmac_f32_e32 v5, v6, v4
-; GFX11-IEEE-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
+; GFX11-IEEE-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-IEEE-NEXT: v_fma_f32 v2, -v3, v5, v2
; GFX11-IEEE-NEXT: s_setreg_b32 hwreg(HW_REG_MODE, 4, 2), s0
+; GFX11-IEEE-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-IEEE-NEXT: v_div_fmas_f32 v2, v2, v4, v5
; GFX11-IEEE-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-IEEE-NEXT: v_div_fixup_f32 v0, v2, v1, -v0
@@ -3314,9 +3327,10 @@ define float @v_fdiv_neglhs_f32_dynamic(float %x, float %y) #0 {
; GFX11-FLUSH-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-FLUSH-NEXT: v_fma_f32 v6, -v3, v5, v2
; GFX11-FLUSH-NEXT: v_fmac_f32_e32 v5, v6, v4
-; GFX11-FLUSH-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
+; GFX11-FLUSH-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-FLUSH-NEXT: v_fma_f32 v2, -v3, v5, v2
; GFX11-FLUSH-NEXT: s_setreg_b32 hwreg(HW_REG_MODE, 4, 2), s0
+; GFX11-FLUSH-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-FLUSH-NEXT: v_div_fmas_f32 v2, v2, v4, v5
; GFX11-FLUSH-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-FLUSH-NEXT: v_div_fixup_f32 v0, v2, v1, -v0
@@ -3684,9 +3698,10 @@ define float @v_fdiv_negrhs_f32_dynamic(float %x, float %y) #0 {
; GFX11-IEEE-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-IEEE-NEXT: v_fma_f32 v6, -v3, v5, v2
; GFX11-IEEE-NEXT: v_fmac_f32_e32 v5, v6, v4
-; GFX11-IEEE-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
+; GFX11-IEEE-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-IEEE-NEXT: v_fma_f32 v2, -v3, v5, v2
; GFX11-IEEE-NEXT: s_setreg_b32 hwreg(HW_REG_MODE, 4, 2), s0
+; GFX11-IEEE-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-IEEE-NEXT: v_div_fmas_f32 v2, v2, v4, v5
; GFX11-IEEE-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-IEEE-NEXT: v_div_fixup_f32 v0, v2, -v1, v0
@@ -3710,9 +3725,10 @@ define float @v_fdiv_negrhs_f32_dynamic(float %x, float %y) #0 {
; GFX11-FLUSH-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-FLUSH-NEXT: v_fma_f32 v6, -v3, v5, v2
; GFX11-FLUSH-NEXT: v_fmac_f32_e32 v5, v6, v4
-; GFX11-FLUSH-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
+; GFX11-FLUSH-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-FLUSH-NEXT: v_fma_f32 v2, -v3, v5, v2
; GFX11-FLUSH-NEXT: s_setreg_b32 hwreg(HW_REG_MODE, 4, 2), s0
+; GFX11-FLUSH-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-FLUSH-NEXT: v_div_fmas_f32 v2, v2, v4, v5
; GFX11-FLUSH-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-FLUSH-NEXT: v_div_fixup_f32 v0, v2, -v1, v0
@@ -3966,9 +3982,10 @@ define float @v_fdiv_f32_constrhs0_dynamic(float %x) #0 {
; GFX11-IEEE-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-IEEE-NEXT: v_fma_f32 v5, -v1, v4, v3
; GFX11-IEEE-NEXT: v_fmac_f32_e32 v4, v5, v2
-; GFX11-IEEE-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
+; GFX11-IEEE-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-IEEE-NEXT: v_fma_f32 v1, -v1, v4, v3
; GFX11-IEEE-NEXT: s_setreg_b32 hwreg(HW_REG_MODE, 4, 2), s0
+; GFX11-IEEE-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-IEEE-NEXT: v_div_fmas_f32 v1, v1, v2, v4
; GFX11-IEEE-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-IEEE-NEXT: v_div_fixup_f32 v0, v1, 0x4640e400, v0
@@ -3990,9 +4007,10 @@ define float @v_fdiv_f32_constrhs0_dynamic(float %x) #0 {
; GFX11-FLUSH-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-FLUSH-NEXT: v_fma_f32 v5, -v1, v4, v3
; GFX11-FLUSH-NEXT: v_fmac_f32_e32 v4, v5, v2
-; GFX11-FLUSH-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
+; GFX11-FLUSH-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-FLUSH-NEXT: v_fma_f32 v1, -v1, v4, v3
; GFX11-FLUSH-NEXT: s_setreg_b32 hwreg(HW_REG_MODE, 4, 2), s0
+; GFX11-FLUSH-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-FLUSH-NEXT: v_div_fmas_f32 v1, v1, v2, v4
; GFX11-FLUSH-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-FLUSH-NEXT: v_div_fixup_f32 v0, v1, 0x4640e400, v0
@@ -4336,9 +4354,10 @@ define float @v_fdiv_f32_constlhs0_dynamic(float %x) #0 {
; GFX11-IEEE-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-IEEE-NEXT: v_fma_f32 v5, -v1, v4, v3
; GFX11-IEEE-NEXT: v_fmac_f32_e32 v4, v5, v2
-; GFX11-IEEE-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
+; GFX11-IEEE-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-IEEE-NEXT: v_fma_f32 v1, -v1, v4, v3
; GFX11-IEEE-NEXT: s_setreg_b32 hwreg(HW_REG_MODE, 4, 2), s0
+; GFX11-IEEE-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-IEEE-NEXT: v_div_fmas_f32 v1, v1, v2, v4
; GFX11-IEEE-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-IEEE-NEXT: v_div_fixup_f32 v0, v1, v0, 0x4640e400
@@ -4360,9 +4379,10 @@ define float @v_fdiv_f32_constlhs0_dynamic(float %x) #0 {
; GFX11-FLUSH-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-FLUSH-NEXT: v_fma_f32 v5, -v1, v4, v3
; GFX11-FLUSH-NEXT: v_fmac_f32_e32 v4, v5, v2
-; GFX11-FLUSH-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
+; GFX11-FLUSH-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-FLUSH-NEXT: v_fma_f32 v1, -v1, v4, v3
; GFX11-FLUSH-NEXT: s_setreg_b32 hwreg(HW_REG_MODE, 4, 2), s0
+; GFX11-FLUSH-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-FLUSH-NEXT: v_div_fmas_f32 v1, v1, v2, v4
; GFX11-FLUSH-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-FLUSH-NEXT: v_div_fixup_f32 v0, v1, v0, 0x4640e400
@@ -4701,9 +4721,10 @@ define float @v_fdiv_f32_dynamic_nodenorm_x(float nofpclass(sub) %x, float %y) #
; GFX11-IEEE-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-IEEE-NEXT: v_fma_f32 v6, -v2, v5, v4
; GFX11-IEEE-NEXT: v_fmac_f32_e32 v5, v6, v3
-; GFX11-IEEE-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
+; GFX11-IEEE-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-IEEE-NEXT: v_fma_f32 v2, -v2, v5, v4
; GFX11-IEEE-NEXT: s_setreg_b32 hwreg(HW_REG_MODE, 4, 2), s0
+; GFX11-IEEE-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-IEEE-NEXT: v_div_fmas_f32 v2, v2, v3, v5
; GFX11-IEEE-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-IEEE-NEXT: v_div_fixup_f32 v0, v2, v1, v0
@@ -4725,9 +4746,10 @@ define float @v_fdiv_f32_dynamic_nodenorm_x(float nofpclass(sub) %x, float %y) #
; GFX11-FLUSH-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-FLUSH-NEXT: v_fma_f32 v6, -v2, v5, v4
; GFX11-FLUSH-NEXT: v_fmac_f32_e32 v5, v6, v3
-; GFX11-FLUSH-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
+; GFX11-FLUSH-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-FLUSH-NEXT: v_fma_f32 v2, -v2, v5, v4
; GFX11-FLUSH-NEXT: s_setreg_b32 hwreg(HW_REG_MODE, 4, 2), s0
+; GFX11-FLUSH-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-FLUSH-NEXT: v_div_fmas_f32 v2, v2, v3, v5
; GFX11-FLUSH-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-FLUSH-NEXT: v_div_fixup_f32 v0, v2, v1, v0
@@ -5082,9 +5104,10 @@ define float @v_fdiv_f32_dynamic_nodenorm_y(float %x, float nofpclass(sub) %y) #
; GFX11-IEEE-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-IEEE-NEXT: v_fma_f32 v6, -v2, v5, v4
; GFX11-IEEE-NEXT: v_fmac_f32_e32 v5, v6, v3
-; GFX11-IEEE-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
+; GFX11-IEEE-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-IEEE-NEXT: v_fma_f32 v2, -v2, v5, v4
; GFX11-IEEE-NEXT: s_setreg_b32 hwreg(HW_REG_MODE, 4, 2), s0
+; GFX11-IEEE-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-IEEE-NEXT: v_div_fmas_f32 v2, v2, v3, v5
; GFX11-IEEE-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-IEEE-NEXT: v_div_fixup_f32 v0, v2, v1, v0
@@ -5106,9 +5129,10 @@ define float @v_fdiv_f32_dynamic_nodenorm_y(float %x, float nofpclass(sub) %y) #
; GFX11-FLUSH-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-FLUSH-NEXT: v_fma_f32 v6, -v2, v5, v4
; GFX11-FLUSH-NEXT: v_fmac_f32_e32 v5, v6, v3
-; GFX11-FLUSH-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
+; GFX11-FLUSH-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-FLUSH-NEXT: v_fma_f32 v2, -v2, v5, v4
; GFX11-FLUSH-NEXT: s_setreg_b32 hwreg(HW_REG_MODE, 4, 2), s0
+; GFX11-FLUSH-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-FLUSH-NEXT: v_div_fmas_f32 v2, v2, v3, v5
; GFX11-FLUSH-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-FLUSH-NEXT: v_div_fixup_f32 v0, v2, v1, v0
diff --git a/llvm/test/CodeGen/AMDGPU/GlobalISel/fdiv.f64.ll b/llvm/test/CodeGen/AMDGPU/GlobalISel/fdiv.f64.ll
index 6440de340ec86b..14f8d2b08e0fbe 100644
--- a/llvm/test/CodeGen/AMDGPU/GlobalISel/fdiv.f64.ll
+++ b/llvm/test/CodeGen/AMDGPU/GlobalISel/fdiv.f64.ll
@@ -884,9 +884,10 @@ define <2 x double> @v_fdiv_v2f64(<2 x double> %a, <2 x double> %b) {
; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-NEXT: v_fma_f64 v[8:9], -v[8:9], v[18:19], v[20:21]
; GFX11-NEXT: v_fma_f64 v[10:11], -v[10:11], v[22:23], v[16:17]
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_2)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX11-NEXT: v_div_fmas_f64 v[8:9], v[8:9], v[12:13], v[18:19]
; GFX11-NEXT: s_mov_b32 vcc_lo, s0
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: v_div_fmas_f64 v[10:11], v[10:11], v[14:15], v[22:23]
; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-NEXT: v_div_fixup_f64 v[0:1], v[8:9], v[4:5], v[0:1]
@@ -1117,9 +1118,10 @@ define <2 x double> @v_fdiv_v2f64_ulp25(<2 x double> %a, <2 x double> %b) {
; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-NEXT: v_fma_f64 v[8:9], -v[8:9], v[18:19], v[20:21]
; GFX11-NEXT: v_fma_f64 v[10:11], -v[10:11], v[22:23], v[16:17]
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_2)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX11-NEXT: v_div_fmas_f64 v[8:9], v[8:9], v[12:13], v[18:19]
; GFX11-NEXT: s_mov_b32 vcc_lo, s0
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: v_div_fmas_f64 v[10:11], v[10:11], v[14:15], v[22:23]
; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-NEXT: v_div_fixup_f64 v[0:1], v[8:9], v[4:5], v[0:1]
@@ -1277,9 +1279,10 @@ define <2 x double> @v_rcp_v2f64(<2 x double> %x) {
; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-NEXT: v_fma_f64 v[4:5], -v[4:5], v[14:15], v[16:17]
; GFX11-NEXT: v_fma_f64 v[6:7], -v[6:7], v[18:19], v[12:13]
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_2)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX11-NEXT: v_div_fmas_f64 v[4:5], v[4:5], v[8:9], v[14:15]
; GFX11-NEXT: s_mov_b32 vcc_lo, s0
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: v_div_fmas_f64 v[6:7], v[6:7], v[10:11], v[18:19]
; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-NEXT: v_div_fixup_f64 v[0:1], v[4:5], v[0:1], 1.0
@@ -1437,9 +1440,10 @@ define <2 x double> @v_rcp_v2f64_arcp(<2 x double> %x) {
; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-NEXT: v_fma_f64 v[4:5], -v[4:5], v[14:15], v[16:17]
; GFX11-NEXT: v_fma_f64 v[6:7], -v[6:7], v[18:19], v[12:13]
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_2)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX11-NEXT: v_div_fmas_f64 v[4:5], v[4:5], v[8:9], v[14:15]
; GFX11-NEXT: s_mov_b32 vcc_lo, s0
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: v_div_fmas_f64 v[6:7], v[6:7], v[10:11], v[18:19]
; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-NEXT: v_div_fixup_f64 v[0:1], v[4:5], v[0:1], 1.0
@@ -1664,9 +1668,10 @@ define <2 x double> @v_rcp_v2f64_ulp25(<2 x double> %x) {
; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-NEXT: v_fma_f64 v[4:5], -v[4:5], v[14:15], v[16:17]
; GFX11-NEXT: v_fma_f64 v[6:7], -v[6:7], v[18:19], v[12:13]
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_2)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX11-NEXT: v_div_fmas_f64 v[4:5], v[4:5], v[8:9], v[14:15]
; GFX11-NEXT: s_mov_b32 vcc_lo, s0
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: v_div_fmas_f64 v[6:7], v[6:7], v[10:11], v[18:19]
; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-NEXT: v_div_fixup_f64 v[0:1], v[4:5], v[0:1], 1.0
@@ -1897,9 +1902,10 @@ define <2 x double> @v_fdiv_v2f64_arcp_ulp25(<2 x double> %a, <2 x double> %b) {
; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-NEXT: v_fma_f64 v[8:9], -v[8:9], v[18:19], v[20:21]
; GFX11-NEXT: v_fma_f64 v[10:11], -v[10:11], v[22:23], v[16:17]
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_2)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX11-NEXT: v_div_fmas_f64 v[8:9], v[8:9], v[12:13], v[18:19]
; GFX11-NEXT: s_mov_b32 vcc_lo, s0
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: v_div_fmas_f64 v[10:11], v[10:11], v[14:15], v[22:23]
; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-NEXT: v_div_fixup_f64 v[0:1], v[8:9], v[4:5], v[0:1]
diff --git a/llvm/test/CodeGen/AMDGPU/GlobalISel/fneg.ll b/llvm/test/CodeGen/AMDGPU/GlobalISel/fneg.ll
index 0720794f941023..1b324f455f08b8 100644
--- a/llvm/test/CodeGen/AMDGPU/GlobalISel/fneg.ll
+++ b/llvm/test/CodeGen/AMDGPU/GlobalISel/fneg.ll
@@ -46,8 +46,8 @@ define amdgpu_ps void @s_fneg_f16_salu_use(half inreg %in, i32 inreg %val, ptr a
; GFX12: ; %bb.0:
; GFX12-NEXT: s_xor_b32 s0, s0, 0x8000
; GFX12-NEXT: s_cmp_eq_u32 s1, 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s0, s0, 0
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: v_mov_b32_e32 v2, s0
; GFX12-NEXT: global_store_b16 v[0:1], v2, off
; GFX12-NEXT: s_endpgm
@@ -102,8 +102,8 @@ define amdgpu_ps void @s_fneg_f32_salu_use(float inreg %in, i32 inreg %val, ptr
; GFX12: ; %bb.0:
; GFX12-NEXT: s_xor_b32 s0, s0, 0x80000000
; GFX12-NEXT: s_cmp_eq_u32 s1, 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s0, s0, 0
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: v_mov_b32_e32 v2, s0
; GFX12-NEXT: global_store_b32 v[0:1], v2, off
; GFX12-NEXT: s_endpgm
@@ -285,8 +285,8 @@ define amdgpu_ps void @s_fneg_v2f32_salu_use(<2 x float> inreg %in, i32 inreg %v
; GFX12-NEXT: s_xor_b32 s0, s0, 0x80000000
; GFX12-NEXT: s_xor_b32 s1, s1, 0x80000000
; GFX12-NEXT: s_cmp_eq_u32 s2, 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b64 s[0:1], s[0:1], 0
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: v_dual_mov_b32 v3, s1 :: v_dual_mov_b32 v2, s0
; GFX12-NEXT: global_store_b64 v[0:1], v[2:3], off
; GFX12-NEXT: s_endpgm
diff --git a/llvm/test/CodeGen/AMDGPU/GlobalISel/fpow.ll b/llvm/test/CodeGen/AMDGPU/GlobalISel/fpow.ll
index b84ec70d3e35de..2be5bd8e983a25 100644
--- a/llvm/test/CodeGen/AMDGPU/GlobalISel/fpow.ll
+++ b/llvm/test/CodeGen/AMDGPU/GlobalISel/fpow.ll
@@ -279,24 +279,23 @@ define <2 x float> @v_pow_v2f32(<2 x float> %x, <2 x float> %y) {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: v_cmp_gt_f32_e64 s0, 0x800000, v1
; GFX11-NEXT: v_cmp_gt_f32_e32 vcc_lo, 0x800000, v0
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX11-NEXT: v_cndmask_b32_e64 v5, 0, 1, s0
; GFX11-NEXT: v_cndmask_b32_e64 v4, 0, 1, vcc_lo
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_lshlrev_b32_e32 v5, 5, v5
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX11-NEXT: v_ldexp_f32 v1, v1, v5
; GFX11-NEXT: v_cndmask_b32_e64 v5, 0, 0x42000000, s0
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_2) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_log_f32_e32 v1, v1
; GFX11-NEXT: s_waitcnt_depctr depctr_va_vdst(0)
; GFX11-NEXT: v_dual_sub_f32 v1, v1, v5 :: v_dual_lshlrev_b32 v4, 5, v4
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_3)
; GFX11-NEXT: v_ldexp_f32 v0, v0, v4
; GFX11-NEXT: v_cndmask_b32_e64 v4, 0, 0x42000000, vcc_lo
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_3)
; GFX11-NEXT: v_mul_dx9_zero_f32_e32 v1, v1, v3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_log_f32_e32 v0, v0
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_3) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_cmp_gt_f32_e64 s0, 0xc2fc0000, v1
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_2) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_cndmask_b32_e64 v3, 0, 0x42800000, s0
; GFX11-NEXT: s_waitcnt_depctr depctr_va_vdst(0)
; GFX11-NEXT: v_dual_sub_f32 v0, v0, v4 :: v_dual_add_f32 v1, v1, v3
@@ -1107,22 +1106,22 @@ define float @v_pow_f32_fabs_lhs(float %x, float %y) {
; GFX11: ; %bb.0:
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: v_cmp_gt_f32_e64 s0, 0x800000, |v0|
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_cndmask_b32_e64 v2, 0, 1, s0
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_lshlrev_b32_e32 v2, 5, v2
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX11-NEXT: v_ldexp_f32 v0, |v0|, v2
; GFX11-NEXT: v_cndmask_b32_e64 v2, 0, 0x42000000, s0
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_2) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_log_f32_e32 v0, v0
; GFX11-NEXT: s_waitcnt_depctr depctr_va_vdst(0)
; GFX11-NEXT: v_sub_f32_e32 v0, v0, v2
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_mul_dx9_zero_f32_e32 v0, v0, v1
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_cmp_gt_f32_e32 vcc_lo, 0xc2fc0000, v0
; GFX11-NEXT: v_cndmask_b32_e64 v1, 0, 0x42800000, vcc_lo
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX11-NEXT: v_add_f32_e32 v0, v0, v1
; GFX11-NEXT: v_cndmask_b32_e64 v1, 0, 0xffffffc0, vcc_lo
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX11-NEXT: v_exp_f32_e32 v0, v0
; GFX11-NEXT: s_waitcnt_depctr depctr_va_vdst(0)
; GFX11-NEXT: v_ldexp_f32 v0, v0, v1
@@ -1349,22 +1348,22 @@ define float @v_pow_f32_fabs_lhs_rhs(float %x, float %y) {
; GFX11: ; %bb.0:
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: v_cmp_gt_f32_e64 s0, 0x800000, |v0|
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_cndmask_b32_e64 v2, 0, 1, s0
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_lshlrev_b32_e32 v2, 5, v2
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX11-NEXT: v_ldexp_f32 v0, |v0|, v2
; GFX11-NEXT: v_cndmask_b32_e64 v2, 0, 0x42000000, s0
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_2) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_log_f32_e32 v0, v0
; GFX11-NEXT: s_waitcnt_depctr depctr_va_vdst(0)
; GFX11-NEXT: v_sub_f32_e32 v0, v0, v2
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_mul_dx9_zero_f32_e64 v0, v0, |v1|
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_cmp_gt_f32_e32 vcc_lo, 0xc2fc0000, v0
; GFX11-NEXT: v_cndmask_b32_e64 v1, 0, 0x42800000, vcc_lo
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX11-NEXT: v_add_f32_e32 v0, v0, v1
; GFX11-NEXT: v_cndmask_b32_e64 v1, 0, 0xffffffc0, vcc_lo
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX11-NEXT: v_exp_f32_e32 v0, v0
; GFX11-NEXT: s_waitcnt_depctr depctr_va_vdst(0)
; GFX11-NEXT: v_ldexp_f32 v0, v0, v1
@@ -1479,6 +1478,7 @@ define amdgpu_ps float @v_pow_f32_sgpr_vgpr(float inreg %x, float %y) {
; GFX11: ; %bb.0:
; GFX11-NEXT: v_cmp_gt_f32_e64 s1, 0x800000, s0
; GFX11-NEXT: s_cmp_lg_u32 s1, 0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cselect_b32 s2, 1, 0
; GFX11-NEXT: s_cselect_b32 s1, 1, 0
; GFX11-NEXT: s_lshl_b32 s2, s2, 5
@@ -1721,6 +1721,7 @@ define amdgpu_ps float @v_pow_f32_sgpr_sgpr(float inreg %x, float inreg %y) {
; GFX11: ; %bb.0:
; GFX11-NEXT: v_cmp_gt_f32_e64 s2, 0x800000, s0
; GFX11-NEXT: s_cmp_lg_u32 s2, 0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cselect_b32 s3, 1, 0
; GFX11-NEXT: s_cselect_b32 s2, 1, 0
; GFX11-NEXT: s_lshl_b32 s3, s3, 5
@@ -1732,13 +1733,13 @@ define amdgpu_ps float @v_pow_f32_sgpr_sgpr(float inreg %x, float inreg %y) {
; GFX11-NEXT: s_waitcnt_depctr depctr_va_vdst(0)
; GFX11-NEXT: v_subrev_f32_e32 v0, s0, v0
; GFX11-NEXT: v_mul_dx9_zero_f32_e32 v0, s1, v0
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: v_cmp_gt_f32_e32 vcc_lo, 0xc2fc0000, v0
; GFX11-NEXT: s_cmp_lg_u32 vcc_lo, 0
; GFX11-NEXT: s_cselect_b32 s0, 0x42800000, 0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_f32_e32 v0, s0, v0
; GFX11-NEXT: s_cselect_b32 s0, 0xffffffc0, 0
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_exp_f32_e32 v0, v0
; GFX11-NEXT: s_waitcnt_depctr depctr_va_vdst(0)
; GFX11-NEXT: v_ldexp_f32 v0, v0, s0
@@ -1843,22 +1844,22 @@ define float @v_pow_f32_fneg_lhs(float %x, float %y) {
; GFX11: ; %bb.0:
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: v_cmp_gt_f32_e64 s0, 0x800000, -v0
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_cndmask_b32_e64 v2, 0, 1, s0
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_lshlrev_b32_e32 v2, 5, v2
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX11-NEXT: v_ldexp_f32 v0, -v0, v2
; GFX11-NEXT: v_cndmask_b32_e64 v2, 0, 0x42000000, s0
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_2) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_log_f32_e32 v0, v0
; GFX11-NEXT: s_waitcnt_depctr depctr_va_vdst(0)
; GFX11-NEXT: v_sub_f32_e32 v0, v0, v2
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_mul_dx9_zero_f32_e32 v0, v0, v1
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_cmp_gt_f32_e32 vcc_lo, 0xc2fc0000, v0
; GFX11-NEXT: v_cndmask_b32_e64 v1, 0, 0x42800000, vcc_lo
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX11-NEXT: v_add_f32_e32 v0, v0, v1
; GFX11-NEXT: v_cndmask_b32_e64 v1, 0, 0xffffffc0, vcc_lo
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX11-NEXT: v_exp_f32_e32 v0, v0
; GFX11-NEXT: s_waitcnt_depctr depctr_va_vdst(0)
; GFX11-NEXT: v_ldexp_f32 v0, v0, v1
diff --git a/llvm/test/CodeGen/AMDGPU/GlobalISel/fptrunc.bf16.ll b/llvm/test/CodeGen/AMDGPU/GlobalISel/fptrunc.bf16.ll
index f9e9b145aa955a..7670c16cb03d63 100644
--- a/llvm/test/CodeGen/AMDGPU/GlobalISel/fptrunc.bf16.ll
+++ b/llvm/test/CodeGen/AMDGPU/GlobalISel/fptrunc.bf16.ll
@@ -15,9 +15,10 @@ define amdgpu_ps bfloat @fptrunc_f32_to_bf16_s(float inreg %a) {
; GFX11-NEXT: s_bitset1_b32 s0, 22
; GFX11-NEXT: s_addk_i32 s1, 0x7fff
; GFX11-NEXT: s_cmp_lg_u32 s2, 0
-; GFX11-NEXT: s_cselect_b32 s0, s0, s1
; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
+; GFX11-NEXT: s_cselect_b32 s0, s0, s1
; GFX11-NEXT: s_lshr_b32 s0, s0, 16
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: v_mov_b32_e32 v0, s0
; GFX11-NEXT: ; return to shader part epilog
;
@@ -26,12 +27,12 @@ define amdgpu_ps bfloat @fptrunc_f32_to_bf16_s(float inreg %a) {
; GFX12-NEXT: s_bfe_u32 s1, s0, 0x10010
; GFX12-NEXT: s_or_b32 s2, s0, 0x400000
; GFX12-NEXT: s_add_co_i32 s1, s1, s0
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_3)
; GFX12-NEXT: s_addk_co_i32 s1, 0x7fff
; GFX12-NEXT: s_cmp_u_f32 s0, 0
; GFX12-NEXT: s_cselect_b32 s0, s2, s1
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_lshr_b32 s0, s0, 16
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: v_mov_b32_e32 v0, s0
; GFX12-NEXT: ; return to shader part epilog
;
@@ -44,12 +45,12 @@ define amdgpu_ps bfloat @fptrunc_f32_to_bf16_s(float inreg %a) {
; GFX1250-NEXT: s_bfe_u32 s1, s0, 0x10010
; GFX1250-NEXT: s_or_b32 s2, s0, 0x400000
; GFX1250-NEXT: s_add_co_i32 s1, s1, s0
-; GFX1250-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
+; GFX1250-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_3)
; GFX1250-NEXT: s_addk_co_i32 s1, 0x7fff
; GFX1250-NEXT: s_cmp_u_f32 s0, 0
; GFX1250-NEXT: s_cselect_b32 s0, s2, s1
+; GFX1250-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1250-NEXT: s_lshr_b32 s0, s0, 16
-; GFX1250-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1250-NEXT: v_mov_b32_e32 v0, s0
; GFX1250-NEXT: ; return to shader part epilog
%result = fptrunc float %a to bfloat
@@ -150,26 +151,28 @@ define amdgpu_ps bfloat @fptrunc_f64_to_bf16_s(double inreg %a) {
; GFX11-NEXT: v_cmp_ngt_f64_e64 s0, |s[0:1]|, |v[0:1]|
; GFX11-NEXT: v_readfirstlane_b32 s1, v2
; GFX11-NEXT: s_cmp_lg_u32 vcc_lo, 0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_cselect_b32 s2, 1, 0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_4) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_or_b32 s2, s2, s1
; GFX11-NEXT: s_cmp_lg_u32 s0, 0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_cselect_b32 s0, -1, 0
; GFX11-NEXT: s_and_b32 s2, s2, 1
; GFX11-NEXT: s_or_b32 s0, s0, 1
; GFX11-NEXT: s_add_i32 s0, s1, s0
; GFX11-NEXT: s_cmp_lg_u32 s2, 0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_cselect_b32 s0, s1, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_bfe_u32 s1, s0, 0x10010
; GFX11-NEXT: v_cmp_u_f32_e64 s2, s0, 0
; GFX11-NEXT: s_add_i32 s1, s1, s0
; GFX11-NEXT: s_bitset1_b32 s0, 22
; GFX11-NEXT: s_addk_i32 s1, 0x7fff
; GFX11-NEXT: s_cmp_lg_u32 s2, 0
-; GFX11-NEXT: s_cselect_b32 s0, s0, s1
; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
+; GFX11-NEXT: s_cselect_b32 s0, s0, s1
; GFX11-NEXT: s_lshr_b32 s0, s0, 16
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: v_mov_b32_e32 v0, s0
; GFX11-NEXT: ; return to shader part epilog
;
@@ -182,10 +185,11 @@ define amdgpu_ps bfloat @fptrunc_f64_to_bf16_s(double inreg %a) {
; GFX12-NEXT: v_cmp_ngt_f64_e64 s0, |s[0:1]|, |v[0:1]|
; GFX12-NEXT: v_readfirstlane_b32 s1, v2
; GFX12-NEXT: s_cmp_lg_u32 vcc_lo, 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s2, 1, 0
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_or_b32 s2, s2, s1
; GFX12-NEXT: s_cmp_lg_u32 s0, 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s0, -1, 0
; GFX12-NEXT: s_and_b32 s2, s2, 1
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
@@ -204,6 +208,7 @@ define amdgpu_ps bfloat @fptrunc_f64_to_bf16_s(double inreg %a) {
; GFX12-NEXT: s_addk_co_i32 s1, 0x7fff
; GFX12-NEXT: s_cmp_u_f32 s0, 0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_2)
; GFX12-NEXT: s_cselect_b32 s0, s2, s1
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_lshr_b32 s0, s0, 16
@@ -224,20 +229,22 @@ define amdgpu_ps bfloat @fptrunc_f64_to_bf16_s(double inreg %a) {
; GFX1250-NEXT: v_cmp_ngt_f64_e64 s0, |s[0:1]|, |v[0:1]|
; GFX1250-NEXT: v_readfirstlane_b32 s1, v2
; GFX1250-NEXT: s_cmp_lg_u32 vcc_lo, 0
+; GFX1250-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1250-NEXT: s_cselect_b32 s2, 1, 0
-; GFX1250-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_4) | instid1(SALU_CYCLE_1)
; GFX1250-NEXT: s_or_b32 s2, s2, s1
; GFX1250-NEXT: s_cmp_lg_u32 s0, 0
+; GFX1250-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
; GFX1250-NEXT: s_cselect_b32 s0, -1, 0
; GFX1250-NEXT: s_and_b32 s2, s2, 1
; GFX1250-NEXT: s_or_b32 s0, s0, 1
; GFX1250-NEXT: s_add_co_i32 s0, s1, s0
; GFX1250-NEXT: s_cmp_lg_u32 s2, 0
+; GFX1250-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1250-NEXT: s_cselect_b32 s0, s1, s0
-; GFX1250-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
; GFX1250-NEXT: s_bfe_u32 s1, s0, 0x10010
; GFX1250-NEXT: s_or_b32 s2, s0, 0x400000
; GFX1250-NEXT: s_add_co_i32 s1, s1, s0
+; GFX1250-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_3)
; GFX1250-NEXT: s_addk_co_i32 s1, 0x7fff
; GFX1250-NEXT: s_cmp_u_f32 s0, 0
; GFX1250-NEXT: s_cselect_b32 s0, s2, s1
@@ -258,22 +265,21 @@ define amdgpu_ps bfloat @fptrunc_f64_to_bf16_v(double %a) {
; GFX11-FAKE16-NEXT: v_cmp_ngt_f64_e64 s0, |v[0:1]|, |v[2:3]|
; GFX11-FAKE16-NEXT: v_cmp_nlg_f64_e32 vcc_lo, v[0:1], v[2:3]
; GFX11-FAKE16-NEXT: v_and_b32_e32 v1, 1, v4
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v0, 0, -1, s0
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-FAKE16-NEXT: v_cmp_ne_u32_e64 s0, 0, v1
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_or_b32_e32 v0, 1, v0
; GFX11-FAKE16-NEXT: s_or_b32 vcc_lo, vcc_lo, s0
-; GFX11-FAKE16-NEXT: v_add_nc_u32_e32 v0, v4, v0
; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX11-FAKE16-NEXT: v_add_nc_u32_e32 v0, v4, v0
; GFX11-FAKE16-NEXT: v_cndmask_b32_e32 v0, v0, v4, vcc_lo
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_2) | instid1(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_bfe_u32 v1, v0, 16, 1
; GFX11-FAKE16-NEXT: v_or_b32_e32 v2, 0x400000, v0
; GFX11-FAKE16-NEXT: v_cmp_u_f32_e32 vcc_lo, 0, v0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_add3_u32 v1, v1, v0, 0x7fff
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_cndmask_b32_e32 v0, v1, v2, vcc_lo
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v0, 16, v0
; GFX11-FAKE16-NEXT: ; return to shader part epilog
;
@@ -285,22 +291,21 @@ define amdgpu_ps bfloat @fptrunc_f64_to_bf16_v(double %a) {
; GFX11-TRUE16-NEXT: v_cmp_ngt_f64_e64 s0, |v[0:1]|, |v[2:3]|
; GFX11-TRUE16-NEXT: v_cmp_nlg_f64_e32 vcc_lo, v[0:1], v[2:3]
; GFX11-TRUE16-NEXT: v_and_b32_e32 v1, 1, v4
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-TRUE16-NEXT: v_cndmask_b32_e64 v0, 0, -1, s0
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-TRUE16-NEXT: v_cmp_ne_u32_e64 s0, 0, v1
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_or_b32_e32 v0, 1, v0
; GFX11-TRUE16-NEXT: s_or_b32 vcc_lo, vcc_lo, s0
-; GFX11-TRUE16-NEXT: v_add_nc_u32_e32 v0, v4, v0
; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX11-TRUE16-NEXT: v_add_nc_u32_e32 v0, v4, v0
; GFX11-TRUE16-NEXT: v_cndmask_b32_e32 v0, v0, v4, vcc_lo
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_2) | instid1(VALU_DEP_3)
; GFX11-TRUE16-NEXT: v_bfe_u32 v1, v0, 16, 1
; GFX11-TRUE16-NEXT: v_or_b32_e32 v2, 0x400000, v0
; GFX11-TRUE16-NEXT: v_cmp_u_f32_e32 vcc_lo, 0, v0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_add3_u32 v1, v1, v0, 0x7fff
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_cndmask_b32_e32 v0, v1, v2, vcc_lo
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v0.l, v0.h
; GFX11-TRUE16-NEXT: ; return to shader part epilog
;
@@ -311,12 +316,12 @@ define amdgpu_ps bfloat @fptrunc_f64_to_bf16_v(double %a) {
; GFX12-FAKE16-NEXT: v_cvt_f64_f32_e32 v[2:3], v4
; GFX12-FAKE16-NEXT: v_cmp_ngt_f64_e64 s0, |v[0:1]|, |v[2:3]|
; GFX12-FAKE16-NEXT: v_cmp_nlg_f64_e32 vcc_lo, v[0:1], v[2:3]
-; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX12-FAKE16-NEXT: v_cndmask_b32_e64 v0, 0, -1, s0
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX12-FAKE16-NEXT: v_or_b32_e32 v0, 1, v0
-; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX12-FAKE16-NEXT: v_add_nc_u32_e32 v0, v4, v0
; GFX12-FAKE16-NEXT: v_and_b32_e32 v1, 1, v4
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: v_cmp_ne_u32_e64 s0, 0, v1
; GFX12-FAKE16-NEXT: s_or_b32 vcc_lo, vcc_lo, s0
; GFX12-FAKE16-NEXT: v_cndmask_b32_e32 v0, v0, v4, vcc_lo
@@ -338,12 +343,12 @@ define amdgpu_ps bfloat @fptrunc_f64_to_bf16_v(double %a) {
; GFX12-TRUE16-NEXT: v_cvt_f64_f32_e32 v[2:3], v4
; GFX12-TRUE16-NEXT: v_cmp_ngt_f64_e64 s0, |v[0:1]|, |v[2:3]|
; GFX12-TRUE16-NEXT: v_cmp_nlg_f64_e32 vcc_lo, v[0:1], v[2:3]
-; GFX12-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX12-TRUE16-NEXT: v_cndmask_b32_e64 v0, 0, -1, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX12-TRUE16-NEXT: v_or_b32_e32 v0, 1, v0
-; GFX12-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX12-TRUE16-NEXT: v_add_nc_u32_e32 v0, v4, v0
; GFX12-TRUE16-NEXT: v_and_b32_e32 v1, 1, v4
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: v_cmp_ne_u32_e64 s0, 0, v1
; GFX12-TRUE16-NEXT: s_or_b32 vcc_lo, vcc_lo, s0
; GFX12-TRUE16-NEXT: v_cndmask_b32_e32 v0, v0, v4, vcc_lo
@@ -369,12 +374,12 @@ define amdgpu_ps bfloat @fptrunc_f64_to_bf16_v(double %a) {
; GFX1250-FAKE16-NEXT: v_cvt_f64_f32_e32 v[2:3], v4
; GFX1250-FAKE16-NEXT: v_cmp_ngt_f64_e64 s0, |v[0:1]|, |v[2:3]|
; GFX1250-FAKE16-NEXT: v_cmp_nlg_f64_e32 vcc_lo, v[0:1], v[2:3]
-; GFX1250-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1250-FAKE16-NEXT: v_cndmask_b32_e64 v0, 0, -1, s0
+; GFX1250-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1250-FAKE16-NEXT: v_or_b32_e32 v0, 1, v0
-; GFX1250-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX1250-FAKE16-NEXT: v_add_nc_u32_e32 v0, v4, v0
; GFX1250-FAKE16-NEXT: v_and_b32_e32 v1, 1, v4
+; GFX1250-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX1250-FAKE16-NEXT: v_cmp_ne_u32_e64 s0, 0, v1
; GFX1250-FAKE16-NEXT: s_or_b32 vcc_lo, vcc_lo, s0
; GFX1250-FAKE16-NEXT: v_cndmask_b32_e32 v0, v0, v4, vcc_lo
@@ -399,12 +404,12 @@ define amdgpu_ps bfloat @fptrunc_f64_to_bf16_v(double %a) {
; GFX1250-TRUE16-NEXT: v_cvt_f64_f32_e32 v[2:3], v4
; GFX1250-TRUE16-NEXT: v_cmp_ngt_f64_e64 s0, |v[0:1]|, |v[2:3]|
; GFX1250-TRUE16-NEXT: v_cmp_nlg_f64_e32 vcc_lo, v[0:1], v[2:3]
-; GFX1250-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1250-TRUE16-NEXT: v_cndmask_b32_e64 v0, 0, -1, s0
+; GFX1250-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1250-TRUE16-NEXT: v_or_b32_e32 v0, 1, v0
-; GFX1250-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX1250-TRUE16-NEXT: v_add_nc_u32_e32 v0, v4, v0
; GFX1250-TRUE16-NEXT: v_and_b32_e32 v1, 1, v4
+; GFX1250-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX1250-TRUE16-NEXT: v_cmp_ne_u32_e64 s0, 0, v1
; GFX1250-TRUE16-NEXT: s_or_b32 vcc_lo, vcc_lo, s0
; GFX1250-TRUE16-NEXT: v_cndmask_b32_e32 v0, v0, v4, vcc_lo
@@ -433,14 +438,14 @@ define amdgpu_ps <2 x bfloat> @fptrunc_v2f32_to_v2bf16_s(<2 x float> inreg %a) {
; GFX11-FAKE16-NEXT: v_cmp_u_f32_e64 s3, s1, 0
; GFX11-FAKE16-NEXT: s_cselect_b32 s0, s0, s2
; GFX11-FAKE16-NEXT: s_bfe_u32 s2, s1, 0x10010
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_4) | instid1(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_3) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_add_i32 s2, s2, s1
; GFX11-FAKE16-NEXT: s_bitset1_b32 s1, 22
; GFX11-FAKE16-NEXT: s_addk_i32 s2, 0x7fff
; GFX11-FAKE16-NEXT: s_cmp_lg_u32 s3, 0
; GFX11-FAKE16-NEXT: s_cselect_b32 s1, s1, s2
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_pack_hh_b32_b16 s0, s0, s1
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v0, s0
; GFX11-FAKE16-NEXT: ; return to shader part epilog
;
@@ -460,11 +465,11 @@ define amdgpu_ps <2 x bfloat> @fptrunc_v2f32_to_v2bf16_s(<2 x float> inreg %a) {
; GFX11-TRUE16-NEXT: s_bitset1_b32 s1, 22
; GFX11-TRUE16-NEXT: s_addk_i32 s2, 0x7fff
; GFX11-TRUE16-NEXT: s_cmp_lg_u32 s3, 0
-; GFX11-TRUE16-NEXT: s_cselect_b32 s1, s1, s2
; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_cselect_b32 s1, s1, s2
; GFX11-TRUE16-NEXT: s_lshr_b32 s1, s1, 16
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_pack_ll_b32_b16 s0, s0, s1
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_mov_b32_e32 v0, s0
; GFX11-TRUE16-NEXT: ; return to shader part epilog
;
@@ -473,19 +478,19 @@ define amdgpu_ps <2 x bfloat> @fptrunc_v2f32_to_v2bf16_s(<2 x float> inreg %a) {
; GFX12-FAKE16-NEXT: s_bfe_u32 s2, s0, 0x10010
; GFX12-FAKE16-NEXT: s_or_b32 s3, s0, 0x400000
; GFX12-FAKE16-NEXT: s_add_co_i32 s2, s2, s0
-; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_3)
; GFX12-FAKE16-NEXT: s_addk_co_i32 s2, 0x7fff
; GFX12-FAKE16-NEXT: s_cmp_u_f32 s0, 0
; GFX12-FAKE16-NEXT: s_cselect_b32 s0, s3, s2
; GFX12-FAKE16-NEXT: s_bfe_u32 s2, s1, 0x10010
; GFX12-FAKE16-NEXT: s_or_b32 s3, s1, 0x400000
; GFX12-FAKE16-NEXT: s_add_co_i32 s2, s2, s1
-; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_3)
; GFX12-FAKE16-NEXT: s_addk_co_i32 s2, 0x7fff
; GFX12-FAKE16-NEXT: s_cmp_u_f32 s1, 0
; GFX12-FAKE16-NEXT: s_cselect_b32 s1, s3, s2
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: s_pack_hh_b32_b16 s0, s0, s1
-; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: v_mov_b32_e32 v0, s0
; GFX12-FAKE16-NEXT: ; return to shader part epilog
;
@@ -494,7 +499,7 @@ define amdgpu_ps <2 x bfloat> @fptrunc_v2f32_to_v2bf16_s(<2 x float> inreg %a) {
; GFX12-TRUE16-NEXT: s_bfe_u32 s2, s0, 0x10010
; GFX12-TRUE16-NEXT: s_or_b32 s3, s0, 0x400000
; GFX12-TRUE16-NEXT: s_add_co_i32 s2, s2, s0
-; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_3)
; GFX12-TRUE16-NEXT: s_addk_co_i32 s2, 0x7fff
; GFX12-TRUE16-NEXT: s_cmp_u_f32 s0, 0
; GFX12-TRUE16-NEXT: s_cselect_b32 s0, s3, s2
@@ -504,11 +509,11 @@ define amdgpu_ps <2 x bfloat> @fptrunc_v2f32_to_v2bf16_s(<2 x float> inreg %a) {
; GFX12-TRUE16-NEXT: s_or_b32 s3, s1, 0x400000
; GFX12-TRUE16-NEXT: s_addk_co_i32 s2, 0x7fff
; GFX12-TRUE16-NEXT: s_cmp_u_f32 s1, 0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_3) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_cselect_b32 s1, s3, s2
-; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_lshr_b32 s1, s1, 16
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_pack_ll_b32_b16 s0, s0, s1
-; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: v_mov_b32_e32 v0, s0
; GFX12-TRUE16-NEXT: ; return to shader part epilog
;
@@ -521,19 +526,19 @@ define amdgpu_ps <2 x bfloat> @fptrunc_v2f32_to_v2bf16_s(<2 x float> inreg %a) {
; GFX1250-FAKE16-NEXT: s_bfe_u32 s2, s0, 0x10010
; GFX1250-FAKE16-NEXT: s_or_b32 s3, s0, 0x400000
; GFX1250-FAKE16-NEXT: s_add_co_i32 s2, s2, s0
-; GFX1250-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1250-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_3)
; GFX1250-FAKE16-NEXT: s_addk_co_i32 s2, 0x7fff
; GFX1250-FAKE16-NEXT: s_cmp_u_f32 s0, 0
; GFX1250-FAKE16-NEXT: s_cselect_b32 s0, s3, s2
; GFX1250-FAKE16-NEXT: s_bfe_u32 s2, s1, 0x10010
; GFX1250-FAKE16-NEXT: s_or_b32 s3, s1, 0x400000
; GFX1250-FAKE16-NEXT: s_add_co_i32 s2, s2, s1
-; GFX1250-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
+; GFX1250-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_3)
; GFX1250-FAKE16-NEXT: s_addk_co_i32 s2, 0x7fff
; GFX1250-FAKE16-NEXT: s_cmp_u_f32 s1, 0
; GFX1250-FAKE16-NEXT: s_cselect_b32 s1, s3, s2
+; GFX1250-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1250-FAKE16-NEXT: s_pack_hh_b32_b16 s0, s0, s1
-; GFX1250-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1250-FAKE16-NEXT: v_mov_b32_e32 v0, s0
; GFX1250-FAKE16-NEXT: ; return to shader part epilog
;
@@ -546,7 +551,7 @@ define amdgpu_ps <2 x bfloat> @fptrunc_v2f32_to_v2bf16_s(<2 x float> inreg %a) {
; GFX1250-TRUE16-NEXT: s_bfe_u32 s2, s0, 0x10010
; GFX1250-TRUE16-NEXT: s_or_b32 s3, s0, 0x400000
; GFX1250-TRUE16-NEXT: s_add_co_i32 s2, s2, s0
-; GFX1250-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1250-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_3)
; GFX1250-TRUE16-NEXT: s_addk_co_i32 s2, 0x7fff
; GFX1250-TRUE16-NEXT: s_cmp_u_f32 s0, 0
; GFX1250-TRUE16-NEXT: s_cselect_b32 s0, s3, s2
@@ -556,11 +561,11 @@ define amdgpu_ps <2 x bfloat> @fptrunc_v2f32_to_v2bf16_s(<2 x float> inreg %a) {
; GFX1250-TRUE16-NEXT: s_or_b32 s3, s1, 0x400000
; GFX1250-TRUE16-NEXT: s_addk_co_i32 s2, 0x7fff
; GFX1250-TRUE16-NEXT: s_cmp_u_f32 s1, 0
+; GFX1250-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_3) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1250-TRUE16-NEXT: s_cselect_b32 s1, s3, s2
-; GFX1250-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1250-TRUE16-NEXT: s_lshr_b32 s1, s1, 16
+; GFX1250-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1250-TRUE16-NEXT: s_pack_ll_b32_b16 s0, s0, s1
-; GFX1250-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1250-TRUE16-NEXT: v_mov_b32_e32 v0, s0
; GFX1250-TRUE16-NEXT: ; return to shader part epilog
%result = fptrunc <2 x float> %a to <2 x bfloat>
diff --git a/llvm/test/CodeGen/AMDGPU/GlobalISel/fptrunc.ll b/llvm/test/CodeGen/AMDGPU/GlobalISel/fptrunc.ll
index d34b222eb56ea9..c515d8516d1599 100644
--- a/llvm/test/CodeGen/AMDGPU/GlobalISel/fptrunc.ll
+++ b/llvm/test/CodeGen/AMDGPU/GlobalISel/fptrunc.ll
@@ -31,8 +31,8 @@ define amdgpu_ps half @fptrunc_f32_to_f16_uniform(float inreg %a) {
; GFX1250-NEXT: v_nop
; GFX1250-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 2, 2), 0 ; msbs: dst=0 src0=0 src1=0 src2=0
+; GFX1250-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_3)
; GFX1250-NEXT: s_cvt_f16_f32 s0, s0
-; GFX1250-NEXT: s_delay_alu instid0(SALU_CYCLE_3)
; GFX1250-NEXT: v_mov_b32_e32 v0, s0
; GFX1250-NEXT: ; return to shader part epilog
%result = fptrunc float %a to half
@@ -67,6 +67,7 @@ define amdgpu_ps half @fptrunc_f32_to_f16_div(float %a) {
; GFX1250-FAKE16-NEXT: v_nop
; GFX1250-FAKE16-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-FAKE16-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 2, 2), 0 ; msbs: dst=0 src0=0 src1=0 src2=0
+; GFX1250-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1250-FAKE16-NEXT: v_cvt_f16_f32_e32 v0, v0
; GFX1250-FAKE16-NEXT: ; return to shader part epilog
;
@@ -77,6 +78,7 @@ define amdgpu_ps half @fptrunc_f32_to_f16_div(float %a) {
; GFX1250-TRUE16-NEXT: v_nop
; GFX1250-TRUE16-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-TRUE16-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 2, 2), 0 ; msbs: dst=0 src0=0 src1=0 src2=0
+; GFX1250-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1250-TRUE16-NEXT: v_cvt_f16_f32_e32 v0.l, v0
; GFX1250-TRUE16-NEXT: ; return to shader part epilog
%result = fptrunc float %a to half
@@ -153,24 +155,27 @@ define amdgpu_ps half @fptrunc_f64_to_f16_uniform(double inreg %a) {
; GFX11-NEXT: s_lshl_b32 s4, s6, s4
; GFX11-NEXT: s_or_b32 s0, s0, s7
; GFX11-NEXT: s_cmp_lg_u32 s4, s5
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_cselect_b32 s4, 1, 0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_or_b32 s4, s6, s4
; GFX11-NEXT: s_cmp_lt_i32 s2, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_cselect_b32 s0, s4, s0
; GFX11-NEXT: s_and_b32 s4, s0, 7
; GFX11-NEXT: s_lshr_b32 s0, s0, 2
; GFX11-NEXT: s_cmp_eq_u32 s4, 3
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_cselect_b32 s5, 1, 0
; GFX11-NEXT: s_cmp_gt_i32 s4, 5
; GFX11-NEXT: s_cselect_b32 s4, 1, 0
; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_or_b32 s4, s5, s4
; GFX11-NEXT: s_cmp_lg_u32 s4, 0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_cselect_b32 s4, 1, 0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_add_i32 s0, s0, s4
; GFX11-NEXT: s_cmp_gt_i32 s2, 30
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_cselect_b32 s0, 0x7c00, s0
; GFX11-NEXT: s_cmpk_eq_i32 s2, 0x40f
; GFX11-NEXT: s_cselect_b32 s0, s3, s0
@@ -205,24 +210,27 @@ define amdgpu_ps half @fptrunc_f64_to_f16_uniform(double inreg %a) {
; GFX12-NEXT: s_lshl_b32 s4, s6, s4
; GFX12-NEXT: s_or_b32 s0, s0, s7
; GFX12-NEXT: s_cmp_lg_u32 s4, s5
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s4, 1, 0
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_or_b32 s4, s6, s4
; GFX12-NEXT: s_cmp_lt_i32 s2, 1
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s0, s4, s0
; GFX12-NEXT: s_and_b32 s4, s0, 7
; GFX12-NEXT: s_lshr_b32 s0, s0, 2
; GFX12-NEXT: s_cmp_eq_u32 s4, 3
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s5, 1, 0
; GFX12-NEXT: s_cmp_gt_i32 s4, 5
; GFX12-NEXT: s_cselect_b32 s4, 1, 0
; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_or_b32 s4, s5, s4
; GFX12-NEXT: s_cmp_lg_u32 s4, 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s4, 1, 0
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_add_co_i32 s0, s0, s4
; GFX12-NEXT: s_cmp_gt_i32 s2, 30
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s0, 0x7c00, s0
; GFX12-NEXT: s_cmp_eq_u32 s2, 0x40f
; GFX12-NEXT: s_cselect_b32 s0, s3, s0
@@ -261,24 +269,27 @@ define amdgpu_ps half @fptrunc_f64_to_f16_uniform(double inreg %a) {
; GFX1250-NEXT: s_lshl_b32 s4, s6, s4
; GFX1250-NEXT: s_or_b32 s0, s0, s7
; GFX1250-NEXT: s_cmp_lg_u32 s4, s5
+; GFX1250-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1250-NEXT: s_cselect_b32 s4, 1, 0
-; GFX1250-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
; GFX1250-NEXT: s_or_b32 s4, s6, s4
; GFX1250-NEXT: s_cmp_lt_i32 s2, 1
+; GFX1250-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1250-NEXT: s_cselect_b32 s0, s4, s0
; GFX1250-NEXT: s_and_b32 s4, s0, 7
; GFX1250-NEXT: s_lshr_b32 s0, s0, 2
; GFX1250-NEXT: s_cmp_eq_u32 s4, 3
+; GFX1250-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX1250-NEXT: s_cselect_b32 s5, 1, 0
; GFX1250-NEXT: s_cmp_gt_i32 s4, 5
; GFX1250-NEXT: s_cselect_b32 s4, 1, 0
; GFX1250-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1250-NEXT: s_or_b32 s4, s5, s4
; GFX1250-NEXT: s_cmp_lg_u32 s4, 0
+; GFX1250-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1250-NEXT: s_cselect_b32 s4, 1, 0
-; GFX1250-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1250-NEXT: s_add_co_i32 s0, s0, s4
; GFX1250-NEXT: s_cmp_gt_i32 s2, 30
+; GFX1250-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX1250-NEXT: s_cselect_b32 s0, 0x7c00, s0
; GFX1250-NEXT: s_cmp_eq_u32 s2, 0x40f
; GFX1250-NEXT: s_cselect_b32 s0, s3, s0
diff --git a/llvm/test/CodeGen/AMDGPU/GlobalISel/fshl.ll b/llvm/test/CodeGen/AMDGPU/GlobalISel/fshl.ll
index dd6e3bd7ebb70b..ccaf7f91fe886a 100644
--- a/llvm/test/CodeGen/AMDGPU/GlobalISel/fshl.ll
+++ b/llvm/test/CodeGen/AMDGPU/GlobalISel/fshl.ll
@@ -165,13 +165,15 @@ define amdgpu_ps i7 @s_fshl_i7(i7 inreg %lhs, i7 inreg %rhs, i7 inreg %amt) {
; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_mul_i32 s3, s3, -7
; GFX11-NEXT: s_add_i32 s2, s2, s3
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_4) | instid1(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_cmp_ge_u32 s2, 7
; GFX11-NEXT: s_cselect_b32 s3, 1, 0
; GFX11-NEXT: s_add_i32 s4, s2, -7
; GFX11-NEXT: s_cmp_lg_u32 s3, 0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_cselect_b32 s2, s4, s2
; GFX11-NEXT: s_cmp_ge_u32 s2, 7
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_cselect_b32 s3, 1, 0
; GFX11-NEXT: s_add_i32 s4, s2, -7
; GFX11-NEXT: s_cmp_lg_u32 s3, 0
@@ -1822,13 +1824,15 @@ define amdgpu_ps i24 @s_fshl_i24(i24 inreg %lhs, i24 inreg %rhs, i24 inreg %amt)
; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_mulk_i32 s3, 0xffe8
; GFX11-NEXT: s_add_i32 s2, s2, s3
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_4) | instid1(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_cmp_ge_u32 s2, 24
; GFX11-NEXT: s_cselect_b32 s3, 1, 0
; GFX11-NEXT: s_sub_i32 s4, s2, 24
; GFX11-NEXT: s_cmp_lg_u32 s3, 0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_cselect_b32 s2, s4, s2
; GFX11-NEXT: s_cmp_ge_u32 s2, 24
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_cselect_b32 s3, 1, 0
; GFX11-NEXT: s_sub_i32 s4, s2, 24
; GFX11-NEXT: s_cmp_lg_u32 s3, 0
@@ -2564,13 +2568,15 @@ define amdgpu_ps i48 @s_fshl_v2i24(i48 inreg %lhs.arg, i48 inreg %rhs.arg, i48 i
; GFX11-TRUE16-NEXT: s_mulk_i32 s4, 0xffe8
; GFX11-TRUE16-NEXT: s_or_b32 s0, s0, s2
; GFX11-TRUE16-NEXT: s_add_i32 s5, s5, s4
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_4) | instid1(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_cmp_ge_u32 s5, 24
; GFX11-TRUE16-NEXT: s_cselect_b32 s2, 1, 0
; GFX11-TRUE16-NEXT: s_sub_i32 s4, s5, 24
; GFX11-TRUE16-NEXT: s_cmp_lg_u32 s2, 0
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_cselect_b32 s2, s4, s5
; GFX11-TRUE16-NEXT: s_cmp_ge_u32 s2, 24
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_cselect_b32 s4, 1, 0
; GFX11-TRUE16-NEXT: s_sub_i32 s5, s2, 24
; GFX11-TRUE16-NEXT: s_cmp_lg_u32 s4, 0
@@ -2668,15 +2674,17 @@ define amdgpu_ps i48 @s_fshl_v2i24(i48 inreg %lhs.arg, i48 inreg %rhs.arg, i48 i
; GFX11-FAKE16-NEXT: s_add_i32 s6, s6, s7
; GFX11-FAKE16-NEXT: s_or_b32 s4, s4, s5
; GFX11-FAKE16-NEXT: s_cmp_ge_u32 s6, 24
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_cselect_b32 s5, 1, 0
; GFX11-FAKE16-NEXT: s_sub_i32 s7, s6, 24
; GFX11-FAKE16-NEXT: s_cmp_lg_u32 s5, 0
; GFX11-FAKE16-NEXT: s_cselect_b32 s5, s7, s6
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_cmp_ge_u32 s5, 24
; GFX11-FAKE16-NEXT: s_cselect_b32 s6, 1, 0
; GFX11-FAKE16-NEXT: s_sub_i32 s7, s5, 24
; GFX11-FAKE16-NEXT: s_cmp_lg_u32 s6, 0
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_cselect_b32 s5, s7, s5
; GFX11-FAKE16-NEXT: s_lshr_b32 s2, s2, 1
; GFX11-FAKE16-NEXT: s_sub_i32 s6, 23, s5
@@ -2686,13 +2694,15 @@ define amdgpu_ps i48 @s_fshl_v2i24(i48 inreg %lhs.arg, i48 inreg %rhs.arg, i48 i
; GFX11-FAKE16-NEXT: s_mulk_i32 s5, 0xffe8
; GFX11-FAKE16-NEXT: s_or_b32 s0, s0, s2
; GFX11-FAKE16-NEXT: s_add_i32 s4, s4, s5
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_4) | instid1(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_cmp_ge_u32 s4, 24
; GFX11-FAKE16-NEXT: s_cselect_b32 s2, 1, 0
; GFX11-FAKE16-NEXT: s_sub_i32 s5, s4, 24
; GFX11-FAKE16-NEXT: s_cmp_lg_u32 s2, 0
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_cselect_b32 s2, s5, s4
; GFX11-FAKE16-NEXT: s_cmp_ge_u32 s2, 24
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_cselect_b32 s4, 1, 0
; GFX11-FAKE16-NEXT: s_sub_i32 s5, s2, 24
; GFX11-FAKE16-NEXT: s_cmp_lg_u32 s4, 0
@@ -6436,6 +6446,7 @@ define amdgpu_ps i128 @s_fshl_i128(i128 inreg %lhs, i128 inreg %rhs, i128 inreg
; GFX11-NEXT: s_sub_i32 s11, s9, 64
; GFX11-NEXT: s_sub_i32 s12, 64, s9
; GFX11-NEXT: s_cmp_lt_u32 s9, 64
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_cselect_b32 s18, 1, 0
; GFX11-NEXT: s_cmp_eq_u32 s9, 0
; GFX11-NEXT: s_cselect_b32 s9, 1, 0
@@ -6445,6 +6456,7 @@ define amdgpu_ps i128 @s_fshl_i128(i128 inreg %lhs, i128 inreg %rhs, i128 inreg
; GFX11-NEXT: s_or_b64 s[12:13], s[12:13], s[14:15]
; GFX11-NEXT: s_lshl_b64 s[0:1], s[0:1], s11
; GFX11-NEXT: s_cmp_lg_u32 s18, 0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_cselect_b64 s[14:15], s[16:17], 0
; GFX11-NEXT: s_cselect_b64 s[0:1], s[12:13], s[0:1]
; GFX11-NEXT: s_cmp_lg_u32 s9, 0
@@ -6458,6 +6470,7 @@ define amdgpu_ps i128 @s_fshl_i128(i128 inreg %lhs, i128 inreg %rhs, i128 inreg
; GFX11-NEXT: s_sub_i32 s12, s6, 64
; GFX11-NEXT: s_sub_i32 s8, 64, s6
; GFX11-NEXT: s_cmp_lt_u32 s6, 64
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_cselect_b32 s13, 1, 0
; GFX11-NEXT: s_cmp_eq_u32 s6, 0
; GFX11-NEXT: s_cselect_b32 s16, 1, 0
@@ -6467,10 +6480,12 @@ define amdgpu_ps i128 @s_fshl_i128(i128 inreg %lhs, i128 inreg %rhs, i128 inreg
; GFX11-NEXT: s_or_b64 s[6:7], s[6:7], s[8:9]
; GFX11-NEXT: s_lshr_b64 s[4:5], s[4:5], s12
; GFX11-NEXT: s_cmp_lg_u32 s13, 0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_cselect_b64 s[4:5], s[6:7], s[4:5]
; GFX11-NEXT: s_cmp_lg_u32 s16, 0
; GFX11-NEXT: s_cselect_b64 s[0:1], s[0:1], s[4:5]
; GFX11-NEXT: s_cmp_lg_u32 s13, 0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cselect_b64 s[4:5], s[10:11], 0
; GFX11-NEXT: s_or_b64 s[0:1], s[14:15], s[0:1]
; GFX11-NEXT: s_or_b64 s[2:3], s[2:3], s[4:5]
@@ -7227,6 +7242,7 @@ define amdgpu_ps <4 x float> @v_fshl_i128_svs(i128 inreg %lhs, i128 %rhs, i128 i
; GFX11-NEXT: s_sub_i32 s12, s5, 64
; GFX11-NEXT: s_sub_i32 s6, 64, s5
; GFX11-NEXT: s_cmp_lt_u32 s5, 64
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cselect_b32 s13, 1, 0
; GFX11-NEXT: s_cmp_eq_u32 s5, 0
; GFX11-NEXT: v_lshl_or_b32 v1, v2, 31, v1
@@ -7241,9 +7257,9 @@ define amdgpu_ps <4 x float> @v_fshl_i128_svs(i128 inreg %lhs, i128 %rhs, i128 i
; GFX11-NEXT: s_cselect_b64 s[8:9], s[10:11], 0
; GFX11-NEXT: s_cselect_b64 s[0:1], s[6:7], s[0:1]
; GFX11-NEXT: s_cmp_lg_u32 s5, 0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_cselect_b64 s[0:1], s[2:3], s[0:1]
; GFX11-NEXT: s_and_not1_b32 s2, 0x7f, s4
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_sub_i32 s4, 64, s2
; GFX11-NEXT: v_lshrrev_b64 v[4:5], s2, v[0:1]
; GFX11-NEXT: v_lshlrev_b64 v[6:7], s4, v[2:3]
@@ -7257,22 +7273,22 @@ define amdgpu_ps <4 x float> @v_fshl_i128_svs(i128 inreg %lhs, i128 %rhs, i128 i
; GFX11-NEXT: v_or_b32_e32 v5, v5, v7
; GFX11-NEXT: s_cselect_b32 s5, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s4, 0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_4) | instid1(VALU_DEP_1)
; GFX11-NEXT: s_cselect_b32 vcc_lo, exec_lo, 0
; GFX11-NEXT: s_cmp_lg_u32 s5, 0
; GFX11-NEXT: v_dual_cndmask_b32 v2, v2, v4 :: v_dual_cndmask_b32 v3, v3, v5
; GFX11-NEXT: s_cselect_b32 vcc_lo, exec_lo, 0
; GFX11-NEXT: s_cmp_lg_u32 s4, 0
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_4) | instid1(VALU_DEP_3)
; GFX11-NEXT: v_dual_cndmask_b32 v4, v2, v0 :: v_dual_cndmask_b32 v5, v3, v1
; GFX11-NEXT: s_cselect_b32 vcc_lo, exec_lo, 0
; GFX11-NEXT: v_dual_mov_b32 v0, s8 :: v_dual_mov_b32 v1, s9
; GFX11-NEXT: v_dual_cndmask_b32 v6, 0, v8 :: v_dual_mov_b32 v3, s1
; GFX11-NEXT: v_dual_mov_b32 v2, s0 :: v_dual_cndmask_b32 v7, 0, v9
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_4)
; GFX11-NEXT: v_or_b32_e32 v0, v0, v4
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_3)
; GFX11-NEXT: v_or_b32_e32 v1, v1, v5
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_4)
; GFX11-NEXT: v_or_b32_e32 v2, v2, v6
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_4)
; GFX11-NEXT: v_or_b32_e32 v3, v3, v7
; GFX11-NEXT: ; return to shader part epilog
%result = call i128 @llvm.fshl.i128(i128 %lhs, i128 %rhs, i128 %amt)
@@ -8063,6 +8079,7 @@ define amdgpu_ps <2 x i128> @s_fshl_v2i128(<2 x i128> inreg %lhs, <2 x i128> inr
; GFX11-NEXT: s_sub_i32 s19, s17, 64
; GFX11-NEXT: s_sub_i32 s21, 64, s17
; GFX11-NEXT: s_cmp_lt_u32 s17, 64
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_cselect_b32 s28, 1, 0
; GFX11-NEXT: s_cmp_eq_u32 s17, 0
; GFX11-NEXT: s_cselect_b32 s17, 1, 0
@@ -8072,6 +8089,7 @@ define amdgpu_ps <2 x i128> @s_fshl_v2i128(<2 x i128> inreg %lhs, <2 x i128> inr
; GFX11-NEXT: s_or_b64 s[22:23], s[22:23], s[24:25]
; GFX11-NEXT: s_lshl_b64 s[0:1], s[0:1], s19
; GFX11-NEXT: s_cmp_lg_u32 s28, 0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_cselect_b64 s[24:25], s[26:27], 0
; GFX11-NEXT: s_cselect_b64 s[0:1], s[22:23], s[0:1]
; GFX11-NEXT: s_cmp_lg_u32 s17, 0
@@ -8085,6 +8103,7 @@ define amdgpu_ps <2 x i128> @s_fshl_v2i128(<2 x i128> inreg %lhs, <2 x i128> inr
; GFX11-NEXT: s_sub_i32 s21, s10, 64
; GFX11-NEXT: s_sub_i32 s16, 64, s10
; GFX11-NEXT: s_cmp_lt_u32 s10, 64
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_cselect_b32 s26, 1, 0
; GFX11-NEXT: s_cmp_eq_u32 s10, 0
; GFX11-NEXT: s_cselect_b32 s27, 1, 0
@@ -8094,10 +8113,12 @@ define amdgpu_ps <2 x i128> @s_fshl_v2i128(<2 x i128> inreg %lhs, <2 x i128> inr
; GFX11-NEXT: s_or_b64 s[10:11], s[10:11], s[16:17]
; GFX11-NEXT: s_lshr_b64 s[8:9], s[8:9], s21
; GFX11-NEXT: s_cmp_lg_u32 s26, 0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_cselect_b64 s[8:9], s[10:11], s[8:9]
; GFX11-NEXT: s_cmp_lg_u32 s27, 0
; GFX11-NEXT: s_cselect_b64 s[0:1], s[0:1], s[8:9]
; GFX11-NEXT: s_cmp_lg_u32 s26, 0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cselect_b64 s[8:9], s[22:23], 0
; GFX11-NEXT: s_and_b32 s10, s20, 0x7f
; GFX11-NEXT: s_or_b64 s[0:1], s[24:25], s[0:1]
@@ -8105,6 +8126,7 @@ define amdgpu_ps <2 x i128> @s_fshl_v2i128(<2 x i128> inreg %lhs, <2 x i128> inr
; GFX11-NEXT: s_sub_i32 s19, s10, 64
; GFX11-NEXT: s_sub_i32 s8, 64, s10
; GFX11-NEXT: s_cmp_lt_u32 s10, 64
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_cselect_b32 s21, 1, 0
; GFX11-NEXT: s_cmp_eq_u32 s10, 0
; GFX11-NEXT: s_cselect_b32 s22, 1, 0
@@ -8114,6 +8136,7 @@ define amdgpu_ps <2 x i128> @s_fshl_v2i128(<2 x i128> inreg %lhs, <2 x i128> inr
; GFX11-NEXT: s_or_b64 s[8:9], s[8:9], s[10:11]
; GFX11-NEXT: s_lshl_b64 s[4:5], s[4:5], s19
; GFX11-NEXT: s_cmp_lg_u32 s21, 0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_cselect_b64 s[10:11], s[16:17], 0
; GFX11-NEXT: s_cselect_b64 s[4:5], s[8:9], s[4:5]
; GFX11-NEXT: s_cmp_lg_u32 s22, 0
@@ -8127,6 +8150,7 @@ define amdgpu_ps <2 x i128> @s_fshl_v2i128(<2 x i128> inreg %lhs, <2 x i128> inr
; GFX11-NEXT: s_sub_i32 s18, s12, 64
; GFX11-NEXT: s_sub_i32 s14, 64, s12
; GFX11-NEXT: s_cmp_lt_u32 s12, 64
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_cselect_b32 s19, 1, 0
; GFX11-NEXT: s_cmp_eq_u32 s12, 0
; GFX11-NEXT: s_cselect_b32 s20, 1, 0
@@ -8136,10 +8160,12 @@ define amdgpu_ps <2 x i128> @s_fshl_v2i128(<2 x i128> inreg %lhs, <2 x i128> inr
; GFX11-NEXT: s_or_b64 s[12:13], s[12:13], s[14:15]
; GFX11-NEXT: s_lshr_b64 s[8:9], s[8:9], s18
; GFX11-NEXT: s_cmp_lg_u32 s19, 0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_cselect_b64 s[8:9], s[12:13], s[8:9]
; GFX11-NEXT: s_cmp_lg_u32 s20, 0
; GFX11-NEXT: s_cselect_b64 s[4:5], s[4:5], s[8:9]
; GFX11-NEXT: s_cmp_lg_u32 s19, 0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cselect_b64 s[8:9], s[16:17], 0
; GFX11-NEXT: s_or_b64 s[4:5], s[10:11], s[4:5]
; GFX11-NEXT: s_or_b64 s[6:7], s[6:7], s[8:9]
diff --git a/llvm/test/CodeGen/AMDGPU/GlobalISel/fshr.ll b/llvm/test/CodeGen/AMDGPU/GlobalISel/fshr.ll
index 6bd74d89deb57e..d21217608eaefa 100644
--- a/llvm/test/CodeGen/AMDGPU/GlobalISel/fshr.ll
+++ b/llvm/test/CodeGen/AMDGPU/GlobalISel/fshr.ll
@@ -166,13 +166,15 @@ define amdgpu_ps i7 @s_fshr_i7(i7 inreg %lhs, i7 inreg %rhs, i7 inreg %amt) {
; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_mul_i32 s3, s3, -7
; GFX11-NEXT: s_add_i32 s2, s2, s3
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_4) | instid1(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_cmp_ge_u32 s2, 7
; GFX11-NEXT: s_cselect_b32 s3, 1, 0
; GFX11-NEXT: s_add_i32 s4, s2, -7
; GFX11-NEXT: s_cmp_lg_u32 s3, 0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_cselect_b32 s2, s4, s2
; GFX11-NEXT: s_cmp_ge_u32 s2, 7
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_cselect_b32 s3, 1, 0
; GFX11-NEXT: s_add_i32 s4, s2, -7
; GFX11-NEXT: s_cmp_lg_u32 s3, 0
@@ -1837,13 +1839,15 @@ define amdgpu_ps i24 @s_fshr_i24(i24 inreg %lhs, i24 inreg %rhs, i24 inreg %amt)
; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_mulk_i32 s3, 0xffe8
; GFX11-NEXT: s_add_i32 s2, s2, s3
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_4) | instid1(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_cmp_ge_u32 s2, 24
; GFX11-NEXT: s_cselect_b32 s3, 1, 0
; GFX11-NEXT: s_sub_i32 s4, s2, 24
; GFX11-NEXT: s_cmp_lg_u32 s3, 0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_cselect_b32 s2, s4, s2
; GFX11-NEXT: s_cmp_ge_u32 s2, 24
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_cselect_b32 s3, 1, 0
; GFX11-NEXT: s_sub_i32 s4, s2, 24
; GFX11-NEXT: s_cmp_lg_u32 s3, 0
@@ -2586,15 +2590,17 @@ define amdgpu_ps i48 @s_fshr_v2i24(i48 inreg %lhs.arg, i48 inreg %rhs.arg, i48 i
; GFX11-TRUE16-NEXT: s_add_i32 s5, s5, s8
; GFX11-TRUE16-NEXT: s_or_b32 s0, s0, s2
; GFX11-TRUE16-NEXT: s_cmp_ge_u32 s5, 24
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_cselect_b32 s2, 1, 0
; GFX11-TRUE16-NEXT: s_sub_i32 s4, s5, 24
; GFX11-TRUE16-NEXT: s_cmp_lg_u32 s2, 0
; GFX11-TRUE16-NEXT: s_cselect_b32 s2, s4, s5
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_cmp_ge_u32 s2, 24
; GFX11-TRUE16-NEXT: s_cselect_b32 s4, 1, 0
; GFX11-TRUE16-NEXT: s_sub_i32 s5, s2, 24
; GFX11-TRUE16-NEXT: s_cmp_lg_u32 s4, 0
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_cselect_b32 s2, s5, s2
; GFX11-TRUE16-NEXT: s_lshl_b32 s1, s1, 1
; GFX11-TRUE16-NEXT: s_sub_i32 s4, 23, s2
@@ -2685,15 +2691,17 @@ define amdgpu_ps i48 @s_fshr_v2i24(i48 inreg %lhs.arg, i48 inreg %rhs.arg, i48 i
; GFX11-FAKE16-NEXT: s_and_b32 s1, 0xffff, s1
; GFX11-FAKE16-NEXT: s_or_b32 s4, s4, s5
; GFX11-FAKE16-NEXT: s_cmp_ge_u32 s8, 24
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_cselect_b32 s5, 1, 0
; GFX11-FAKE16-NEXT: s_sub_i32 s9, s8, 24
; GFX11-FAKE16-NEXT: s_cmp_lg_u32 s5, 0
; GFX11-FAKE16-NEXT: s_cselect_b32 s5, s9, s8
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_cmp_ge_u32 s5, 24
; GFX11-FAKE16-NEXT: s_cselect_b32 s8, 1, 0
; GFX11-FAKE16-NEXT: s_sub_i32 s9, s5, 24
; GFX11-FAKE16-NEXT: s_cmp_lg_u32 s8, 0
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_cselect_b32 s5, s9, s5
; GFX11-FAKE16-NEXT: s_lshl_b32 s6, s6, 17
; GFX11-FAKE16-NEXT: s_lshl_b32 s0, s0, 1
@@ -2706,15 +2714,17 @@ define amdgpu_ps i48 @s_fshr_v2i24(i48 inreg %lhs.arg, i48 inreg %rhs.arg, i48 i
; GFX11-FAKE16-NEXT: s_add_i32 s4, s4, s6
; GFX11-FAKE16-NEXT: s_or_b32 s0, s0, s2
; GFX11-FAKE16-NEXT: s_cmp_ge_u32 s4, 24
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_cselect_b32 s2, 1, 0
; GFX11-FAKE16-NEXT: s_sub_i32 s5, s4, 24
; GFX11-FAKE16-NEXT: s_cmp_lg_u32 s2, 0
; GFX11-FAKE16-NEXT: s_cselect_b32 s2, s5, s4
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_cmp_ge_u32 s2, 24
; GFX11-FAKE16-NEXT: s_cselect_b32 s4, 1, 0
; GFX11-FAKE16-NEXT: s_sub_i32 s5, s2, 24
; GFX11-FAKE16-NEXT: s_cmp_lg_u32 s4, 0
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_cselect_b32 s2, s5, s2
; GFX11-FAKE16-NEXT: s_lshl_b32 s4, s7, 17
; GFX11-FAKE16-NEXT: s_lshl_b32 s1, s1, 1
@@ -6146,6 +6156,7 @@ define amdgpu_ps i128 @s_fshr_i128(i128 inreg %lhs, i128 inreg %rhs, i128 inreg
; GFX11-NEXT: s_sub_i32 s16, s9, 64
; GFX11-NEXT: s_sub_i32 s10, 64, s9
; GFX11-NEXT: s_cmp_lt_u32 s9, 64
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_cselect_b32 s17, 1, 0
; GFX11-NEXT: s_cmp_eq_u32 s9, 0
; GFX11-NEXT: s_cselect_b32 s9, 1, 0
@@ -6155,17 +6166,19 @@ define amdgpu_ps i128 @s_fshr_i128(i128 inreg %lhs, i128 inreg %rhs, i128 inreg
; GFX11-NEXT: s_or_b64 s[10:11], s[10:11], s[12:13]
; GFX11-NEXT: s_lshl_b64 s[0:1], s[0:1], s16
; GFX11-NEXT: s_cmp_lg_u32 s17, 0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_cselect_b64 s[12:13], s[14:15], 0
; GFX11-NEXT: s_cselect_b64 s[0:1], s[10:11], s[0:1]
; GFX11-NEXT: s_cmp_lg_u32 s9, 0
; GFX11-NEXT: s_cselect_b64 s[2:3], s[2:3], s[0:1]
; GFX11-NEXT: s_and_b32 s0, s8, 0x7f
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_sub_i32 s14, s0, 64
; GFX11-NEXT: s_sub_i32 s9, 64, s0
; GFX11-NEXT: s_cmp_lt_u32 s0, 64
; GFX11-NEXT: s_cselect_b32 s15, 1, 0
; GFX11-NEXT: s_cmp_eq_u32 s0, 0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cselect_b32 s16, 1, 0
; GFX11-NEXT: s_lshr_b64 s[0:1], s[4:5], s8
; GFX11-NEXT: s_lshl_b64 s[10:11], s[6:7], s9
@@ -6173,10 +6186,12 @@ define amdgpu_ps i128 @s_fshr_i128(i128 inreg %lhs, i128 inreg %rhs, i128 inreg
; GFX11-NEXT: s_or_b64 s[0:1], s[0:1], s[10:11]
; GFX11-NEXT: s_lshr_b64 s[6:7], s[6:7], s14
; GFX11-NEXT: s_cmp_lg_u32 s15, 0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_cselect_b64 s[0:1], s[0:1], s[6:7]
; GFX11-NEXT: s_cmp_lg_u32 s16, 0
; GFX11-NEXT: s_cselect_b64 s[0:1], s[4:5], s[0:1]
; GFX11-NEXT: s_cmp_lg_u32 s15, 0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cselect_b64 s[4:5], s[8:9], 0
; GFX11-NEXT: s_or_b64 s[0:1], s[12:13], s[0:1]
; GFX11-NEXT: s_or_b64 s[2:3], s[2:3], s[4:5]
@@ -6941,6 +6956,7 @@ define amdgpu_ps <4 x float> @v_fshr_i128_svs(i128 inreg %lhs, i128 %rhs, i128 i
; GFX11-NEXT: s_sub_i32 s12, s5, 64
; GFX11-NEXT: s_sub_i32 s6, 64, s5
; GFX11-NEXT: s_cmp_lt_u32 s5, 64
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_cselect_b32 s13, 1, 0
; GFX11-NEXT: s_cmp_eq_u32 s5, 0
; GFX11-NEXT: s_cselect_b32 s5, 1, 0
@@ -6950,6 +6966,7 @@ define amdgpu_ps <4 x float> @v_fshr_i128_svs(i128 inreg %lhs, i128 %rhs, i128 i
; GFX11-NEXT: s_or_b64 s[6:7], s[6:7], s[8:9]
; GFX11-NEXT: s_lshl_b64 s[0:1], s[0:1], s12
; GFX11-NEXT: s_cmp_lg_u32 s13, 0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_cselect_b64 s[8:9], s[10:11], 0
; GFX11-NEXT: s_cselect_b64 s[0:1], s[6:7], s[0:1]
; GFX11-NEXT: s_cmp_lg_u32 s5, 0
@@ -6969,22 +6986,22 @@ define amdgpu_ps <4 x float> @v_fshr_i128_svs(i128 inreg %lhs, i128 %rhs, i128 i
; GFX11-NEXT: v_or_b32_e32 v5, v5, v7
; GFX11-NEXT: s_cselect_b32 s5, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s4, 0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_4) | instid1(VALU_DEP_1)
; GFX11-NEXT: s_cselect_b32 vcc_lo, exec_lo, 0
; GFX11-NEXT: s_cmp_lg_u32 s5, 0
; GFX11-NEXT: v_dual_cndmask_b32 v2, v2, v4 :: v_dual_cndmask_b32 v3, v3, v5
; GFX11-NEXT: s_cselect_b32 vcc_lo, exec_lo, 0
; GFX11-NEXT: s_cmp_lg_u32 s4, 0
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_4) | instid1(VALU_DEP_3)
; GFX11-NEXT: v_dual_cndmask_b32 v4, v2, v0 :: v_dual_cndmask_b32 v5, v3, v1
; GFX11-NEXT: s_cselect_b32 vcc_lo, exec_lo, 0
; GFX11-NEXT: v_dual_mov_b32 v0, s8 :: v_dual_mov_b32 v1, s9
; GFX11-NEXT: v_dual_cndmask_b32 v6, 0, v8 :: v_dual_mov_b32 v3, s1
; GFX11-NEXT: v_dual_mov_b32 v2, s0 :: v_dual_cndmask_b32 v7, 0, v9
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_4)
; GFX11-NEXT: v_or_b32_e32 v0, v0, v4
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_3)
; GFX11-NEXT: v_or_b32_e32 v1, v1, v5
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_4)
; GFX11-NEXT: v_or_b32_e32 v2, v2, v6
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_4)
; GFX11-NEXT: v_or_b32_e32 v3, v3, v7
; GFX11-NEXT: ; return to shader part epilog
%result = call i128 @llvm.fshr.i128(i128 %lhs, i128 %rhs, i128 %amt)
@@ -7777,6 +7794,7 @@ define amdgpu_ps <2 x i128> @s_fshr_v2i128(<2 x i128> inreg %lhs, <2 x i128> inr
; GFX11-NEXT: s_sub_i32 s26, s17, 64
; GFX11-NEXT: s_sub_i32 s18, 64, s17
; GFX11-NEXT: s_cmp_lt_u32 s17, 64
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_cselect_b32 s27, 1, 0
; GFX11-NEXT: s_cmp_eq_u32 s17, 0
; GFX11-NEXT: s_cselect_b32 s17, 1, 0
@@ -7786,17 +7804,19 @@ define amdgpu_ps <2 x i128> @s_fshr_v2i128(<2 x i128> inreg %lhs, <2 x i128> inr
; GFX11-NEXT: s_or_b64 s[18:19], s[18:19], s[22:23]
; GFX11-NEXT: s_lshl_b64 s[0:1], s[0:1], s26
; GFX11-NEXT: s_cmp_lg_u32 s27, 0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_cselect_b64 s[22:23], s[24:25], 0
; GFX11-NEXT: s_cselect_b64 s[0:1], s[18:19], s[0:1]
; GFX11-NEXT: s_cmp_lg_u32 s17, 0
; GFX11-NEXT: s_cselect_b64 s[2:3], s[2:3], s[0:1]
; GFX11-NEXT: s_and_b32 s0, s16, 0x7f
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_sub_i32 s21, s0, 64
; GFX11-NEXT: s_sub_i32 s17, 64, s0
; GFX11-NEXT: s_cmp_lt_u32 s0, 64
; GFX11-NEXT: s_cselect_b32 s24, 1, 0
; GFX11-NEXT: s_cmp_eq_u32 s0, 0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cselect_b32 s25, 1, 0
; GFX11-NEXT: s_lshr_b64 s[0:1], s[8:9], s16
; GFX11-NEXT: s_lshl_b64 s[18:19], s[10:11], s17
@@ -7804,10 +7824,12 @@ define amdgpu_ps <2 x i128> @s_fshr_v2i128(<2 x i128> inreg %lhs, <2 x i128> inr
; GFX11-NEXT: s_or_b64 s[0:1], s[0:1], s[18:19]
; GFX11-NEXT: s_lshr_b64 s[10:11], s[10:11], s21
; GFX11-NEXT: s_cmp_lg_u32 s24, 0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_cselect_b64 s[0:1], s[0:1], s[10:11]
; GFX11-NEXT: s_cmp_lg_u32 s25, 0
; GFX11-NEXT: s_cselect_b64 s[0:1], s[8:9], s[0:1]
; GFX11-NEXT: s_cmp_lg_u32 s24, 0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cselect_b64 s[8:9], s[16:17], 0
; GFX11-NEXT: s_lshl_b64 s[6:7], s[6:7], 1
; GFX11-NEXT: s_or_b64 s[2:3], s[2:3], s[8:9]
@@ -7820,6 +7842,7 @@ define amdgpu_ps <2 x i128> @s_fshr_v2i128(<2 x i128> inreg %lhs, <2 x i128> inr
; GFX11-NEXT: s_sub_i32 s18, s8, 64
; GFX11-NEXT: s_sub_i32 s9, 64, s8
; GFX11-NEXT: s_cmp_lt_u32 s8, 64
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_cselect_b32 s19, 1, 0
; GFX11-NEXT: s_cmp_eq_u32 s8, 0
; GFX11-NEXT: s_cselect_b32 s21, 1, 0
@@ -7829,17 +7852,19 @@ define amdgpu_ps <2 x i128> @s_fshr_v2i128(<2 x i128> inreg %lhs, <2 x i128> inr
; GFX11-NEXT: s_or_b64 s[8:9], s[8:9], s[10:11]
; GFX11-NEXT: s_lshl_b64 s[4:5], s[4:5], s18
; GFX11-NEXT: s_cmp_lg_u32 s19, 0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_cselect_b64 s[10:11], s[16:17], 0
; GFX11-NEXT: s_cselect_b64 s[4:5], s[8:9], s[4:5]
; GFX11-NEXT: s_cmp_lg_u32 s21, 0
; GFX11-NEXT: s_cselect_b64 s[6:7], s[6:7], s[4:5]
; GFX11-NEXT: s_and_b32 s4, s20, 0x7f
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_sub_i32 s18, s4, 64
; GFX11-NEXT: s_sub_i32 s8, 64, s4
; GFX11-NEXT: s_cmp_lt_u32 s4, 64
; GFX11-NEXT: s_cselect_b32 s19, 1, 0
; GFX11-NEXT: s_cmp_eq_u32 s4, 0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cselect_b32 s21, 1, 0
; GFX11-NEXT: s_lshr_b64 s[4:5], s[12:13], s20
; GFX11-NEXT: s_lshl_b64 s[8:9], s[14:15], s8
@@ -7847,10 +7872,12 @@ define amdgpu_ps <2 x i128> @s_fshr_v2i128(<2 x i128> inreg %lhs, <2 x i128> inr
; GFX11-NEXT: s_or_b64 s[4:5], s[4:5], s[8:9]
; GFX11-NEXT: s_lshr_b64 s[8:9], s[14:15], s18
; GFX11-NEXT: s_cmp_lg_u32 s19, 0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_cselect_b64 s[4:5], s[4:5], s[8:9]
; GFX11-NEXT: s_cmp_lg_u32 s21, 0
; GFX11-NEXT: s_cselect_b64 s[4:5], s[12:13], s[4:5]
; GFX11-NEXT: s_cmp_lg_u32 s19, 0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cselect_b64 s[8:9], s[16:17], 0
; GFX11-NEXT: s_or_b64 s[4:5], s[10:11], s[4:5]
; GFX11-NEXT: s_or_b64 s[6:7], s[6:7], s[8:9]
diff --git a/llvm/test/CodeGen/AMDGPU/GlobalISel/icmp.ll b/llvm/test/CodeGen/AMDGPU/GlobalISel/icmp.ll
index 516e2bd0673023..0645871d7bf173 100644
--- a/llvm/test/CodeGen/AMDGPU/GlobalISel/icmp.ll
+++ b/llvm/test/CodeGen/AMDGPU/GlobalISel/icmp.ll
@@ -82,44 +82,54 @@ define void @icmp_i16_uniform(i16 inreg %a, i16 inreg %b, ptr addrspace(1) %p) {
; GFX12-NEXT: s_sext_i32_i16 s1, s1
; GFX12-NEXT: s_cselect_b32 s4, 1, 0
; GFX12-NEXT: s_cmp_lg_u32 s2, s3
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s5, 1, 0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_cmp_lt_i32 s0, s1
; GFX12-NEXT: s_cselect_b32 s6, 1, 0
; GFX12-NEXT: s_cmp_gt_i32 s0, s1
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s7, 1, 0
; GFX12-NEXT: s_cmp_le_i32 s0, s1
; GFX12-NEXT: s_cselect_b32 s8, 1, 0
; GFX12-NEXT: s_cmp_ge_i32 s0, s1
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s0, 1, 0
; GFX12-NEXT: s_cmp_lt_u32 s2, s3
; GFX12-NEXT: s_cselect_b32 s1, 1, 0
; GFX12-NEXT: s_cmp_gt_u32 s2, s3
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s9, 1, 0
; GFX12-NEXT: s_cmp_le_u32 s2, s3
; GFX12-NEXT: s_cselect_b32 s10, 1, 0
; GFX12-NEXT: s_cmp_ge_u32 s2, s3
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s2, 1, 0
; GFX12-NEXT: s_cmp_lg_u32 s4, 0
; GFX12-NEXT: s_cselect_b32 s3, 1, 0
; GFX12-NEXT: s_cmp_lg_u32 s5, 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s4, 1, 0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_cmp_lg_u32 s6, 0
; GFX12-NEXT: s_cselect_b32 s5, 1, 0
; GFX12-NEXT: s_cmp_lg_u32 s7, 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s6, 1, 0
; GFX12-NEXT: s_cmp_lg_u32 s8, 0
; GFX12-NEXT: s_cselect_b32 s7, 1, 0
; GFX12-NEXT: s_cmp_lg_u32 s0, 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s0, 1, 0
; GFX12-NEXT: s_cmp_lg_u32 s1, 0
; GFX12-NEXT: s_cselect_b32 s1, 1, 0
; GFX12-NEXT: s_cmp_lg_u32 s9, 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s8, 1, 0
; GFX12-NEXT: s_cmp_lg_u32 s10, 0
; GFX12-NEXT: s_cselect_b32 s9, 1, 0
; GFX12-NEXT: s_cmp_lg_u32 s2, 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s2, 1, 0
; GFX12-NEXT: s_add_co_i32 s3, s3, s4
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
@@ -369,43 +379,53 @@ define void @icmp_i32_uniform(i32 inreg %a, i32 inreg %b, ptr addrspace(1) %p) {
; GFX12-NEXT: s_wait_bvhcnt 0x0
; GFX12-NEXT: s_wait_kmcnt 0x0
; GFX12-NEXT: s_cmp_eq_u32 s0, s1
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s2, 1, 0
; GFX12-NEXT: s_cmp_lg_u32 s0, s1
; GFX12-NEXT: s_cselect_b32 s3, 1, 0
; GFX12-NEXT: s_cmp_lt_i32 s0, s1
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s4, 1, 0
; GFX12-NEXT: s_cmp_gt_i32 s0, s1
; GFX12-NEXT: s_cselect_b32 s5, 1, 0
; GFX12-NEXT: s_cmp_le_i32 s0, s1
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s6, 1, 0
; GFX12-NEXT: s_cmp_ge_i32 s0, s1
; GFX12-NEXT: s_cselect_b32 s7, 1, 0
; GFX12-NEXT: s_cmp_lt_u32 s0, s1
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s8, 1, 0
; GFX12-NEXT: s_cmp_gt_u32 s0, s1
; GFX12-NEXT: s_cselect_b32 s9, 1, 0
; GFX12-NEXT: s_cmp_le_u32 s0, s1
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s10, 1, 0
; GFX12-NEXT: s_cmp_ge_u32 s0, s1
; GFX12-NEXT: s_cselect_b32 s0, 1, 0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_cmp_lg_u32 s2, 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s1, 1, 0
; GFX12-NEXT: s_cmp_lg_u32 s3, 0
; GFX12-NEXT: s_cselect_b32 s2, 1, 0
; GFX12-NEXT: s_cmp_lg_u32 s4, 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s3, 1, 0
; GFX12-NEXT: s_cmp_lg_u32 s5, 0
; GFX12-NEXT: s_cselect_b32 s4, 1, 0
; GFX12-NEXT: s_cmp_lg_u32 s6, 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s5, 1, 0
; GFX12-NEXT: s_cmp_lg_u32 s7, 0
; GFX12-NEXT: s_cselect_b32 s6, 1, 0
; GFX12-NEXT: s_cmp_lg_u32 s8, 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s7, 1, 0
; GFX12-NEXT: s_cmp_lg_u32 s9, 0
; GFX12-NEXT: s_cselect_b32 s8, 1, 0
; GFX12-NEXT: s_cmp_lg_u32 s10, 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s9, 1, 0
; GFX12-NEXT: s_cmp_lg_u32 s0, 0
; GFX12-NEXT: s_cselect_b32 s0, 1, 0
@@ -756,27 +776,33 @@ define void @icmp_p3_uniform(ptr addrspace(3) inreg %a, ptr addrspace(3) inreg %
; GFX12-NEXT: s_wait_bvhcnt 0x0
; GFX12-NEXT: s_wait_kmcnt 0x0
; GFX12-NEXT: s_cmp_eq_u32 s0, s1
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s2, 1, 0
; GFX12-NEXT: s_cmp_lg_u32 s0, s1
; GFX12-NEXT: s_cselect_b32 s3, 1, 0
; GFX12-NEXT: s_cmp_lt_u32 s0, s1
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s4, 1, 0
; GFX12-NEXT: s_cmp_gt_u32 s0, s1
; GFX12-NEXT: s_cselect_b32 s5, 1, 0
; GFX12-NEXT: s_cmp_le_u32 s0, s1
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s6, 1, 0
; GFX12-NEXT: s_cmp_ge_u32 s0, s1
; GFX12-NEXT: s_cselect_b32 s0, 1, 0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_cmp_lg_u32 s2, 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s1, 1, 0
; GFX12-NEXT: s_cmp_lg_u32 s3, 0
; GFX12-NEXT: s_cselect_b32 s2, 1, 0
; GFX12-NEXT: s_cmp_lg_u32 s4, 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s3, 1, 0
; GFX12-NEXT: s_cmp_lg_u32 s5, 0
; GFX12-NEXT: s_cselect_b32 s4, 1, 0
; GFX12-NEXT: s_cmp_lg_u32 s6, 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s5, 1, 0
; GFX12-NEXT: s_cmp_lg_u32 s0, 0
; GFX12-NEXT: s_cselect_b32 s0, 1, 0
@@ -945,27 +971,33 @@ define void @icmp_p5_uniform(ptr addrspace(5) inreg %a, ptr addrspace(5) inreg %
; GFX12-NEXT: s_wait_bvhcnt 0x0
; GFX12-NEXT: s_wait_kmcnt 0x0
; GFX12-NEXT: s_cmp_eq_u32 s0, s1
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s2, 1, 0
; GFX12-NEXT: s_cmp_lg_u32 s0, s1
; GFX12-NEXT: s_cselect_b32 s3, 1, 0
; GFX12-NEXT: s_cmp_lt_u32 s0, s1
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s4, 1, 0
; GFX12-NEXT: s_cmp_gt_u32 s0, s1
; GFX12-NEXT: s_cselect_b32 s5, 1, 0
; GFX12-NEXT: s_cmp_le_u32 s0, s1
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s6, 1, 0
; GFX12-NEXT: s_cmp_ge_u32 s0, s1
; GFX12-NEXT: s_cselect_b32 s0, 1, 0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_cmp_lg_u32 s2, 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s1, 1, 0
; GFX12-NEXT: s_cmp_lg_u32 s3, 0
; GFX12-NEXT: s_cselect_b32 s2, 1, 0
; GFX12-NEXT: s_cmp_lg_u32 s4, 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s3, 1, 0
; GFX12-NEXT: s_cmp_lg_u32 s5, 0
; GFX12-NEXT: s_cselect_b32 s4, 1, 0
; GFX12-NEXT: s_cmp_lg_u32 s6, 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s5, 1, 0
; GFX12-NEXT: s_cmp_lg_u32 s0, 0
; GFX12-NEXT: s_cselect_b32 s0, 1, 0
@@ -1154,23 +1186,28 @@ define void @icmp_p0_uniform(ptr inreg %a, ptr inreg %b, ptr addrspace(1) %p) {
; GFX12-NEXT: v_cmp_ge_u64_e64 s0, s[0:1], s[2:3]
; GFX12-NEXT: s_cselect_b32 s5, 1, 0
; GFX12-NEXT: s_cmp_lg_u32 s7, 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s7, 1, 0
; GFX12-NEXT: s_cmp_lg_u32 s8, 0
; GFX12-NEXT: s_cselect_b32 s1, 1, 0
; GFX12-NEXT: s_cmp_lg_u32 s0, 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s0, 1, 0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_cmp_lg_u32 s4, 0
; GFX12-NEXT: s_cselect_b32 s2, 1, 0
; GFX12-NEXT: s_cmp_lg_u32 s6, 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s3, 1, 0
; GFX12-NEXT: s_cmp_lg_u32 s5, 0
; GFX12-NEXT: s_cselect_b32 s4, 1, 0
; GFX12-NEXT: s_cmp_lg_u32 s7, 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s5, 1, 0
; GFX12-NEXT: s_cmp_lg_u32 s1, 0
; GFX12-NEXT: s_cselect_b32 s1, 1, 0
; GFX12-NEXT: s_cmp_lg_u32 s0, 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s0, 1, 0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_add_co_i32 s2, s2, s3
@@ -1355,23 +1392,28 @@ define void @icmp_p1_uniform(ptr addrspace(1) inreg %a, ptr addrspace(1) inreg %
; GFX12-NEXT: v_cmp_ge_u64_e64 s0, s[0:1], s[2:3]
; GFX12-NEXT: s_cselect_b32 s5, 1, 0
; GFX12-NEXT: s_cmp_lg_u32 s7, 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s7, 1, 0
; GFX12-NEXT: s_cmp_lg_u32 s8, 0
; GFX12-NEXT: s_cselect_b32 s1, 1, 0
; GFX12-NEXT: s_cmp_lg_u32 s0, 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s0, 1, 0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_cmp_lg_u32 s4, 0
; GFX12-NEXT: s_cselect_b32 s2, 1, 0
; GFX12-NEXT: s_cmp_lg_u32 s6, 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s3, 1, 0
; GFX12-NEXT: s_cmp_lg_u32 s5, 0
; GFX12-NEXT: s_cselect_b32 s4, 1, 0
; GFX12-NEXT: s_cmp_lg_u32 s7, 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s5, 1, 0
; GFX12-NEXT: s_cmp_lg_u32 s1, 0
; GFX12-NEXT: s_cselect_b32 s1, 1, 0
; GFX12-NEXT: s_cmp_lg_u32 s0, 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s0, 1, 0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_add_co_i32 s2, s2, s3
diff --git a/llvm/test/CodeGen/AMDGPU/GlobalISel/inst-select-copy-scc-vcc.ll b/llvm/test/CodeGen/AMDGPU/GlobalISel/inst-select-copy-scc-vcc.ll
index e524a6fef67d2c..efc71d5091e260 100644
--- a/llvm/test/CodeGen/AMDGPU/GlobalISel/inst-select-copy-scc-vcc.ll
+++ b/llvm/test/CodeGen/AMDGPU/GlobalISel/inst-select-copy-scc-vcc.ll
@@ -44,8 +44,8 @@ define amdgpu_kernel void @fcmp_uniform_select(float %a, i32 %b, i32 %c, ptr add
; GFX11-NEXT: s_waitcnt lgkmcnt(0)
; GFX11-NEXT: v_cmp_eq_f32_e64 s0, s0, 0
; GFX11-NEXT: s_cmp_lg_u32 s0, 0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_cselect_b32 s0, s1, s6
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: v_mov_b32_e32 v0, s0
; GFX11-NEXT: global_store_b32 v1, v0, s[2:3]
; GFX11-NEXT: s_endpgm
diff --git a/llvm/test/CodeGen/AMDGPU/GlobalISel/llvm.amdgcn.image.gather4.a16.dim.ll b/llvm/test/CodeGen/AMDGPU/GlobalISel/llvm.amdgcn.image.gather4.a16.dim.ll
index 0cd3a5d235ea60..a746c610292d34 100644
--- a/llvm/test/CodeGen/AMDGPU/GlobalISel/llvm.amdgcn.image.gather4.a16.dim.ll
+++ b/llvm/test/CodeGen/AMDGPU/GlobalISel/llvm.amdgcn.image.gather4.a16.dim.ll
@@ -80,6 +80,7 @@ define amdgpu_ps <4 x float> @gather4_2d(<8 x i32> inreg %rsrc, <4 x i32> inreg
; GFX11-FAKE16-NEXT: s_mov_b32 s14, exec_lo
; GFX11-FAKE16-NEXT: s_mov_b32 s0, s2
; GFX11-FAKE16-NEXT: s_wqm_b32 exec_lo, exec_lo
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_and_b32_e32 v0, 0xffff, v0
; GFX11-FAKE16-NEXT: s_mov_b32 s1, s3
; GFX11-FAKE16-NEXT: s_mov_b32 s2, s4
@@ -125,6 +126,7 @@ define amdgpu_ps <4 x float> @gather4_2d(<8 x i32> inreg %rsrc, <4 x i32> inreg
; GFX12-FAKE16-NEXT: s_mov_b32 s14, exec_lo
; GFX12-FAKE16-NEXT: s_mov_b32 s0, s2
; GFX12-FAKE16-NEXT: s_wqm_b32 exec_lo, exec_lo
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: v_and_b32_e32 v0, 0xffff, v0
; GFX12-FAKE16-NEXT: s_mov_b32 s1, s3
; GFX12-FAKE16-NEXT: s_mov_b32 s2, s4
@@ -225,6 +227,7 @@ define amdgpu_ps <4 x float> @gather4_cube(<8 x i32> inreg %rsrc, <4 x i32> inre
; GFX11-FAKE16-NEXT: s_mov_b32 s14, exec_lo
; GFX11-FAKE16-NEXT: s_mov_b32 s0, s2
; GFX11-FAKE16-NEXT: s_wqm_b32 exec_lo, exec_lo
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_and_b32_e32 v0, 0xffff, v0
; GFX11-FAKE16-NEXT: v_and_b32_e32 v2, 0xffff, v2
; GFX11-FAKE16-NEXT: s_mov_b32 s1, s3
@@ -272,6 +275,7 @@ define amdgpu_ps <4 x float> @gather4_cube(<8 x i32> inreg %rsrc, <4 x i32> inre
; GFX12-FAKE16-NEXT: s_mov_b32 s14, exec_lo
; GFX12-FAKE16-NEXT: s_mov_b32 s0, s2
; GFX12-FAKE16-NEXT: s_wqm_b32 exec_lo, exec_lo
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: v_and_b32_e32 v0, 0xffff, v0
; GFX12-FAKE16-NEXT: v_and_b32_e32 v2, 0xffff, v2
; GFX12-FAKE16-NEXT: s_mov_b32 s1, s3
@@ -374,6 +378,7 @@ define amdgpu_ps <4 x float> @gather4_2darray(<8 x i32> inreg %rsrc, <4 x i32> i
; GFX11-FAKE16-NEXT: s_mov_b32 s14, exec_lo
; GFX11-FAKE16-NEXT: s_mov_b32 s0, s2
; GFX11-FAKE16-NEXT: s_wqm_b32 exec_lo, exec_lo
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_and_b32_e32 v0, 0xffff, v0
; GFX11-FAKE16-NEXT: v_and_b32_e32 v2, 0xffff, v2
; GFX11-FAKE16-NEXT: s_mov_b32 s1, s3
@@ -421,6 +426,7 @@ define amdgpu_ps <4 x float> @gather4_2darray(<8 x i32> inreg %rsrc, <4 x i32> i
; GFX12-FAKE16-NEXT: s_mov_b32 s14, exec_lo
; GFX12-FAKE16-NEXT: s_mov_b32 s0, s2
; GFX12-FAKE16-NEXT: s_wqm_b32 exec_lo, exec_lo
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: v_and_b32_e32 v0, 0xffff, v0
; GFX12-FAKE16-NEXT: v_and_b32_e32 v2, 0xffff, v2
; GFX12-FAKE16-NEXT: s_mov_b32 s1, s3
@@ -519,6 +525,7 @@ define amdgpu_ps <4 x float> @gather4_c_2d(<8 x i32> inreg %rsrc, <4 x i32> inre
; GFX11-FAKE16-NEXT: s_mov_b32 s14, exec_lo
; GFX11-FAKE16-NEXT: s_mov_b32 s0, s2
; GFX11-FAKE16-NEXT: s_wqm_b32 exec_lo, exec_lo
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_and_b32_e32 v1, 0xffff, v1
; GFX11-FAKE16-NEXT: s_mov_b32 s1, s3
; GFX11-FAKE16-NEXT: s_mov_b32 s2, s4
@@ -564,6 +571,7 @@ define amdgpu_ps <4 x float> @gather4_c_2d(<8 x i32> inreg %rsrc, <4 x i32> inre
; GFX12-FAKE16-NEXT: s_mov_b32 s14, exec_lo
; GFX12-FAKE16-NEXT: s_mov_b32 s0, s2
; GFX12-FAKE16-NEXT: s_wqm_b32 exec_lo, exec_lo
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: v_and_b32_e32 v1, 0xffff, v1
; GFX12-FAKE16-NEXT: s_mov_b32 s1, s3
; GFX12-FAKE16-NEXT: s_mov_b32 s2, s4
@@ -664,6 +672,7 @@ define amdgpu_ps <4 x float> @gather4_cl_2d(<8 x i32> inreg %rsrc, <4 x i32> inr
; GFX11-FAKE16-NEXT: s_mov_b32 s14, exec_lo
; GFX11-FAKE16-NEXT: s_mov_b32 s0, s2
; GFX11-FAKE16-NEXT: s_wqm_b32 exec_lo, exec_lo
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_and_b32_e32 v0, 0xffff, v0
; GFX11-FAKE16-NEXT: v_and_b32_e32 v2, 0xffff, v2
; GFX11-FAKE16-NEXT: s_mov_b32 s1, s3
@@ -711,6 +720,7 @@ define amdgpu_ps <4 x float> @gather4_cl_2d(<8 x i32> inreg %rsrc, <4 x i32> inr
; GFX12-FAKE16-NEXT: s_mov_b32 s14, exec_lo
; GFX12-FAKE16-NEXT: s_mov_b32 s0, s2
; GFX12-FAKE16-NEXT: s_wqm_b32 exec_lo, exec_lo
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: v_and_b32_e32 v0, 0xffff, v0
; GFX12-FAKE16-NEXT: v_and_b32_e32 v2, 0xffff, v2
; GFX12-FAKE16-NEXT: s_mov_b32 s1, s3
@@ -813,6 +823,7 @@ define amdgpu_ps <4 x float> @gather4_c_cl_2d(<8 x i32> inreg %rsrc, <4 x i32> i
; GFX11-FAKE16-NEXT: s_mov_b32 s14, exec_lo
; GFX11-FAKE16-NEXT: s_mov_b32 s0, s2
; GFX11-FAKE16-NEXT: s_wqm_b32 exec_lo, exec_lo
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_and_b32_e32 v1, 0xffff, v1
; GFX11-FAKE16-NEXT: v_and_b32_e32 v3, 0xffff, v3
; GFX11-FAKE16-NEXT: s_mov_b32 s1, s3
@@ -860,6 +871,7 @@ define amdgpu_ps <4 x float> @gather4_c_cl_2d(<8 x i32> inreg %rsrc, <4 x i32> i
; GFX12-FAKE16-NEXT: s_mov_b32 s14, exec_lo
; GFX12-FAKE16-NEXT: s_mov_b32 s0, s2
; GFX12-FAKE16-NEXT: s_wqm_b32 exec_lo, exec_lo
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: v_and_b32_e32 v1, 0xffff, v1
; GFX12-FAKE16-NEXT: v_and_b32_e32 v3, 0xffff, v3
; GFX12-FAKE16-NEXT: s_mov_b32 s1, s3
@@ -962,6 +974,7 @@ define amdgpu_ps <4 x float> @gather4_b_2d(<8 x i32> inreg %rsrc, <4 x i32> inre
; GFX11-FAKE16-NEXT: s_mov_b32 s14, exec_lo
; GFX11-FAKE16-NEXT: s_mov_b32 s0, s2
; GFX11-FAKE16-NEXT: s_wqm_b32 exec_lo, exec_lo
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_and_b32_e32 v0, 0xffff, v0
; GFX11-FAKE16-NEXT: v_and_b32_e32 v1, 0xffff, v1
; GFX11-FAKE16-NEXT: s_mov_b32 s1, s3
@@ -1009,6 +1022,7 @@ define amdgpu_ps <4 x float> @gather4_b_2d(<8 x i32> inreg %rsrc, <4 x i32> inre
; GFX12-FAKE16-NEXT: s_mov_b32 s14, exec_lo
; GFX12-FAKE16-NEXT: s_mov_b32 s0, s2
; GFX12-FAKE16-NEXT: s_wqm_b32 exec_lo, exec_lo
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: v_and_b32_e32 v0, 0xffff, v0
; GFX12-FAKE16-NEXT: v_and_b32_e32 v1, 0xffff, v1
; GFX12-FAKE16-NEXT: s_mov_b32 s1, s3
@@ -1111,6 +1125,7 @@ define amdgpu_ps <4 x float> @gather4_c_b_2d(<8 x i32> inreg %rsrc, <4 x i32> in
; GFX11-FAKE16-NEXT: s_mov_b32 s14, exec_lo
; GFX11-FAKE16-NEXT: s_mov_b32 s0, s2
; GFX11-FAKE16-NEXT: s_wqm_b32 exec_lo, exec_lo
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_and_b32_e32 v0, 0xffff, v0
; GFX11-FAKE16-NEXT: v_and_b32_e32 v2, 0xffff, v2
; GFX11-FAKE16-NEXT: s_mov_b32 s1, s3
@@ -1158,6 +1173,7 @@ define amdgpu_ps <4 x float> @gather4_c_b_2d(<8 x i32> inreg %rsrc, <4 x i32> in
; GFX12-FAKE16-NEXT: s_mov_b32 s14, exec_lo
; GFX12-FAKE16-NEXT: s_mov_b32 s0, s2
; GFX12-FAKE16-NEXT: s_wqm_b32 exec_lo, exec_lo
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: v_and_b32_e32 v0, 0xffff, v0
; GFX12-FAKE16-NEXT: v_and_b32_e32 v2, 0xffff, v2
; GFX12-FAKE16-NEXT: s_mov_b32 s1, s3
@@ -1262,6 +1278,7 @@ define amdgpu_ps <4 x float> @gather4_b_cl_2d(<8 x i32> inreg %rsrc, <4 x i32> i
; GFX11-FAKE16-NEXT: s_mov_b32 s14, exec_lo
; GFX11-FAKE16-NEXT: s_mov_b32 s0, s2
; GFX11-FAKE16-NEXT: s_wqm_b32 exec_lo, exec_lo
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_and_b32_e32 v0, 0xffff, v0
; GFX11-FAKE16-NEXT: v_and_b32_e32 v1, 0xffff, v1
; GFX11-FAKE16-NEXT: v_and_b32_e32 v3, 0xffff, v3
@@ -1311,6 +1328,7 @@ define amdgpu_ps <4 x float> @gather4_b_cl_2d(<8 x i32> inreg %rsrc, <4 x i32> i
; GFX12-FAKE16-NEXT: s_mov_b32 s14, exec_lo
; GFX12-FAKE16-NEXT: s_mov_b32 s0, s2
; GFX12-FAKE16-NEXT: s_wqm_b32 exec_lo, exec_lo
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: v_and_b32_e32 v0, 0xffff, v0
; GFX12-FAKE16-NEXT: v_and_b32_e32 v1, 0xffff, v1
; GFX12-FAKE16-NEXT: v_and_b32_e32 v3, 0xffff, v3
@@ -1417,6 +1435,7 @@ define amdgpu_ps <4 x float> @gather4_c_b_cl_2d(<8 x i32> inreg %rsrc, <4 x i32>
; GFX11-FAKE16-NEXT: s_mov_b32 s14, exec_lo
; GFX11-FAKE16-NEXT: s_mov_b32 s0, s2
; GFX11-FAKE16-NEXT: s_wqm_b32 exec_lo, exec_lo
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_and_b32_e32 v0, 0xffff, v0
; GFX11-FAKE16-NEXT: v_and_b32_e32 v2, 0xffff, v2
; GFX11-FAKE16-NEXT: v_and_b32_e32 v4, 0xffff, v4
@@ -1466,6 +1485,7 @@ define amdgpu_ps <4 x float> @gather4_c_b_cl_2d(<8 x i32> inreg %rsrc, <4 x i32>
; GFX12-FAKE16-NEXT: s_mov_b32 s14, exec_lo
; GFX12-FAKE16-NEXT: s_mov_b32 s0, s2
; GFX12-FAKE16-NEXT: s_wqm_b32 exec_lo, exec_lo
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: v_and_b32_e32 v0, 0xffff, v0
; GFX12-FAKE16-NEXT: v_and_b32_e32 v2, 0xffff, v2
; GFX12-FAKE16-NEXT: v_and_b32_e32 v4, 0xffff, v4
diff --git a/llvm/test/CodeGen/AMDGPU/GlobalISel/llvm.amdgcn.interp.inreg.ll b/llvm/test/CodeGen/AMDGPU/GlobalISel/llvm.amdgcn.interp.inreg.ll
index 225ae801fc12b2..54dd7bc65ed5ce 100644
--- a/llvm/test/CodeGen/AMDGPU/GlobalISel/llvm.amdgcn.interp.inreg.ll
+++ b/llvm/test/CodeGen/AMDGPU/GlobalISel/llvm.amdgcn.interp.inreg.ll
@@ -13,13 +13,13 @@ define amdgpu_ps void @v_interp_f32(float inreg %i, float inreg %j, i32 inreg %m
; GFX11-NEXT: lds_param_load v0, attr0.y wait_vdst:15
; GFX11-NEXT: lds_param_load v1, attr1.x wait_vdst:15
; GFX11-NEXT: s_mov_b32 exec_lo, s3
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX11-NEXT: v_mov_b32_e32 v2, s0
; GFX11-NEXT: v_mov_b32_e32 v4, s1
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX11-NEXT: v_interp_p10_f32 v3, v0, v2, v0 wait_exp:1
; GFX11-NEXT: v_interp_p10_f32 v2, v1, v2, v1 wait_exp:0
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_interp_p2_f32 v5, v0, v4, v3 wait_exp:7
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_interp_p2_f32 v4, v1, v4, v5 wait_exp:7
; GFX11-NEXT: exp mrt0, v3, v2, v5, v4 done
; GFX11-NEXT: s_endpgm
@@ -32,13 +32,13 @@ define amdgpu_ps void @v_interp_f32(float inreg %i, float inreg %j, i32 inreg %m
; GFX12-NEXT: ds_param_load v0, attr0.y wait_va_vdst:15 wait_vm_vsrc:1
; GFX12-NEXT: ds_param_load v1, attr1.x wait_va_vdst:15 wait_vm_vsrc:1
; GFX12-NEXT: s_mov_b32 exec_lo, s3
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX12-NEXT: v_mov_b32_e32 v2, s0
; GFX12-NEXT: v_mov_b32_e32 v4, s1
-; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX12-NEXT: v_interp_p10_f32 v3, v0, v2, v0 wait_exp:1
; GFX12-NEXT: v_interp_p10_f32 v2, v1, v2, v1 wait_exp:0
+; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX12-NEXT: v_interp_p2_f32 v5, v0, v4, v3 wait_exp:7
-; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-NEXT: v_interp_p2_f32 v4, v1, v4, v5 wait_exp:7
; GFX12-NEXT: export mrt0, v3, v2, v5, v4 done
; GFX12-NEXT: s_endpgm
@@ -64,17 +64,17 @@ define amdgpu_ps void @v_interp_f32_many(float inreg %i, float inreg %j, i32 inr
; GFX11-NEXT: lds_param_load v2, attr2.x wait_vdst:15
; GFX11-NEXT: lds_param_load v3, attr3.x wait_vdst:15
; GFX11-NEXT: s_mov_b32 exec_lo, s3
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_dual_mov_b32 v4, s0 :: v_dual_mov_b32 v5, s1
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_3) | instid1(VALU_DEP_4)
; GFX11-NEXT: v_interp_p10_f32 v6, v0, v4, v0 wait_exp:3
; GFX11-NEXT: v_interp_p10_f32 v7, v1, v4, v1 wait_exp:2
; GFX11-NEXT: v_interp_p10_f32 v8, v2, v4, v2 wait_exp:1
; GFX11-NEXT: v_interp_p10_f32 v4, v3, v4, v3 wait_exp:0
-; GFX11-NEXT: v_interp_p2_f32 v6, v0, v5, v6 wait_exp:7
; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
+; GFX11-NEXT: v_interp_p2_f32 v6, v0, v5, v6 wait_exp:7
; GFX11-NEXT: v_interp_p2_f32 v7, v1, v5, v7 wait_exp:7
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
; GFX11-NEXT: v_interp_p2_f32 v8, v2, v5, v8 wait_exp:7
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_4)
; GFX11-NEXT: v_interp_p2_f32 v4, v3, v5, v4 wait_exp:7
; GFX11-NEXT: exp mrt0, v6, v7, v8, v4 done
; GFX11-NEXT: s_endpgm
@@ -89,17 +89,17 @@ define amdgpu_ps void @v_interp_f32_many(float inreg %i, float inreg %j, i32 inr
; GFX12-NEXT: ds_param_load v2, attr2.x wait_va_vdst:15 wait_vm_vsrc:1
; GFX12-NEXT: ds_param_load v3, attr3.x wait_va_vdst:15 wait_vm_vsrc:1
; GFX12-NEXT: s_mov_b32 exec_lo, s3
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX12-NEXT: v_dual_mov_b32 v4, s0 :: v_dual_mov_b32 v5, s1
-; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_3) | instid1(VALU_DEP_4)
; GFX12-NEXT: v_interp_p10_f32 v6, v0, v4, v0 wait_exp:3
; GFX12-NEXT: v_interp_p10_f32 v7, v1, v4, v1 wait_exp:2
; GFX12-NEXT: v_interp_p10_f32 v8, v2, v4, v2 wait_exp:1
; GFX12-NEXT: v_interp_p10_f32 v4, v3, v4, v3 wait_exp:0
-; GFX12-NEXT: v_interp_p2_f32 v6, v0, v5, v6 wait_exp:7
; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
+; GFX12-NEXT: v_interp_p2_f32 v6, v0, v5, v6 wait_exp:7
; GFX12-NEXT: v_interp_p2_f32 v7, v1, v5, v7 wait_exp:7
+; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
; GFX12-NEXT: v_interp_p2_f32 v8, v2, v5, v8 wait_exp:7
-; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_4)
; GFX12-NEXT: v_interp_p2_f32 v4, v3, v5, v4 wait_exp:7
; GFX12-NEXT: export mrt0, v6, v7, v8, v4 done
; GFX12-NEXT: s_endpgm
@@ -199,14 +199,15 @@ define amdgpu_ps half @v_interp_f16(float inreg %i, float inreg %j, i32 inreg %m
; GFX11-TRUE16-NEXT: s_mov_b32 m0, s2
; GFX11-TRUE16-NEXT: lds_param_load v1, attr0.x wait_vdst:15
; GFX11-TRUE16-NEXT: s_mov_b32 exec_lo, s3
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX11-TRUE16-NEXT: v_mov_b32_e32 v0, s0
; GFX11-TRUE16-NEXT: v_mov_b32_e32 v2, s1
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX11-TRUE16-NEXT: v_interp_p10_f16_f32 v3, v1.l, v0, v1.l wait_exp:0
; GFX11-TRUE16-NEXT: v_interp_p10_f16_f32 v4, v1.h, v0, v1.h wait_exp:7
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-TRUE16-NEXT: v_interp_p2_f16_f32 v0.l, v1.l, v2, v3 wait_exp:7
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_interp_p2_f16_f32 v0.h, v1.h, v2, v4 wait_exp:7
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_add_f16_e32 v0.l, v0.l, v0.h
; GFX11-TRUE16-NEXT: ; return to shader part epilog
;
@@ -217,14 +218,15 @@ define amdgpu_ps half @v_interp_f16(float inreg %i, float inreg %j, i32 inreg %m
; GFX11-FAKE16-NEXT: s_mov_b32 m0, s2
; GFX11-FAKE16-NEXT: lds_param_load v1, attr0.x wait_vdst:15
; GFX11-FAKE16-NEXT: s_mov_b32 exec_lo, s3
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v0, s0
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v2, s1
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX11-FAKE16-NEXT: v_interp_p10_f16_f32 v3, v1, v0, v1 wait_exp:0
; GFX11-FAKE16-NEXT: v_interp_p10_f16_f32 v0, v1, v0, v1 op_sel:[1,0,1,0] wait_exp:7
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-FAKE16-NEXT: v_interp_p2_f16_f32 v3, v1, v2, v3 wait_exp:7
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_interp_p2_f16_f32 v0, v1, v2, v0 op_sel:[1,0,0,0] wait_exp:7
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_add_f16_e32 v0, v3, v0
; GFX11-FAKE16-NEXT: ; return to shader part epilog
;
@@ -235,14 +237,15 @@ define amdgpu_ps half @v_interp_f16(float inreg %i, float inreg %j, i32 inreg %m
; GFX12-TRUE16-NEXT: s_mov_b32 m0, s2
; GFX12-TRUE16-NEXT: ds_param_load v1, attr0.x wait_va_vdst:15 wait_vm_vsrc:1
; GFX12-TRUE16-NEXT: s_mov_b32 exec_lo, s3
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX12-TRUE16-NEXT: v_mov_b32_e32 v0, s0
; GFX12-TRUE16-NEXT: v_mov_b32_e32 v2, s1
-; GFX12-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX12-TRUE16-NEXT: v_interp_p10_f16_f32 v3, v1.l, v0, v1.l wait_exp:0
; GFX12-TRUE16-NEXT: v_interp_p10_f16_f32 v4, v1.h, v0, v1.h wait_exp:7
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX12-TRUE16-NEXT: v_interp_p2_f16_f32 v0.l, v1.l, v2, v3 wait_exp:7
-; GFX12-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX12-TRUE16-NEXT: v_interp_p2_f16_f32 v0.h, v1.h, v2, v4 wait_exp:7
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-TRUE16-NEXT: v_add_f16_e32 v0.l, v0.l, v0.h
; GFX12-TRUE16-NEXT: ; return to shader part epilog
;
@@ -253,14 +256,15 @@ define amdgpu_ps half @v_interp_f16(float inreg %i, float inreg %j, i32 inreg %m
; GFX12-FAKE16-NEXT: s_mov_b32 m0, s2
; GFX12-FAKE16-NEXT: ds_param_load v1, attr0.x wait_va_vdst:15 wait_vm_vsrc:1
; GFX12-FAKE16-NEXT: s_mov_b32 exec_lo, s3
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX12-FAKE16-NEXT: v_mov_b32_e32 v0, s0
; GFX12-FAKE16-NEXT: v_mov_b32_e32 v2, s1
-; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX12-FAKE16-NEXT: v_interp_p10_f16_f32 v3, v1, v0, v1 wait_exp:0
; GFX12-FAKE16-NEXT: v_interp_p10_f16_f32 v0, v1, v0, v1 op_sel:[1,0,1,0] wait_exp:7
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX12-FAKE16-NEXT: v_interp_p2_f16_f32 v3, v1, v2, v3 wait_exp:7
-; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX12-FAKE16-NEXT: v_interp_p2_f16_f32 v0, v1, v2, v0 op_sel:[1,0,0,0] wait_exp:7
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-FAKE16-NEXT: v_add_f16_e32 v0, v3, v0
; GFX12-FAKE16-NEXT: ; return to shader part epilog
main_body:
@@ -281,14 +285,15 @@ define amdgpu_ps half @v_interp_rtz_f16(float inreg %i, float inreg %j, i32 inre
; GFX11-TRUE16-NEXT: s_mov_b32 m0, s2
; GFX11-TRUE16-NEXT: lds_param_load v1, attr0.x wait_vdst:15
; GFX11-TRUE16-NEXT: s_mov_b32 exec_lo, s3
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX11-TRUE16-NEXT: v_mov_b32_e32 v0, s0
; GFX11-TRUE16-NEXT: v_mov_b32_e32 v2, s1
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX11-TRUE16-NEXT: v_interp_p10_rtz_f16_f32 v3, v1.l, v0, v1.l wait_exp:0
; GFX11-TRUE16-NEXT: v_interp_p10_rtz_f16_f32 v4, v1.h, v0, v1.h wait_exp:7
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-TRUE16-NEXT: v_interp_p2_rtz_f16_f32 v0.l, v1.l, v2, v3 wait_exp:7
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_interp_p2_rtz_f16_f32 v0.h, v1.h, v2, v4 wait_exp:7
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_add_f16_e32 v0.l, v0.l, v0.h
; GFX11-TRUE16-NEXT: ; return to shader part epilog
;
@@ -299,14 +304,15 @@ define amdgpu_ps half @v_interp_rtz_f16(float inreg %i, float inreg %j, i32 inre
; GFX11-FAKE16-NEXT: s_mov_b32 m0, s2
; GFX11-FAKE16-NEXT: lds_param_load v1, attr0.x wait_vdst:15
; GFX11-FAKE16-NEXT: s_mov_b32 exec_lo, s3
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v0, s0
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v2, s1
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX11-FAKE16-NEXT: v_interp_p10_rtz_f16_f32 v3, v1, v0, v1 wait_exp:0
; GFX11-FAKE16-NEXT: v_interp_p10_rtz_f16_f32 v0, v1, v0, v1 op_sel:[1,0,1,0] wait_exp:7
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-FAKE16-NEXT: v_interp_p2_rtz_f16_f32 v3, v1, v2, v3 wait_exp:7
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_interp_p2_rtz_f16_f32 v0, v1, v2, v0 op_sel:[1,0,0,0] wait_exp:7
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_add_f16_e32 v0, v3, v0
; GFX11-FAKE16-NEXT: ; return to shader part epilog
;
@@ -317,14 +323,15 @@ define amdgpu_ps half @v_interp_rtz_f16(float inreg %i, float inreg %j, i32 inre
; GFX12-TRUE16-NEXT: s_mov_b32 m0, s2
; GFX12-TRUE16-NEXT: ds_param_load v1, attr0.x wait_va_vdst:15 wait_vm_vsrc:1
; GFX12-TRUE16-NEXT: s_mov_b32 exec_lo, s3
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX12-TRUE16-NEXT: v_mov_b32_e32 v0, s0
; GFX12-TRUE16-NEXT: v_mov_b32_e32 v2, s1
-; GFX12-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX12-TRUE16-NEXT: v_interp_p10_rtz_f16_f32 v3, v1.l, v0, v1.l wait_exp:0
; GFX12-TRUE16-NEXT: v_interp_p10_rtz_f16_f32 v4, v1.h, v0, v1.h wait_exp:7
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX12-TRUE16-NEXT: v_interp_p2_rtz_f16_f32 v0.l, v1.l, v2, v3 wait_exp:7
-; GFX12-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX12-TRUE16-NEXT: v_interp_p2_rtz_f16_f32 v0.h, v1.h, v2, v4 wait_exp:7
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-TRUE16-NEXT: v_add_f16_e32 v0.l, v0.l, v0.h
; GFX12-TRUE16-NEXT: ; return to shader part epilog
;
@@ -335,14 +342,15 @@ define amdgpu_ps half @v_interp_rtz_f16(float inreg %i, float inreg %j, i32 inre
; GFX12-FAKE16-NEXT: s_mov_b32 m0, s2
; GFX12-FAKE16-NEXT: ds_param_load v1, attr0.x wait_va_vdst:15 wait_vm_vsrc:1
; GFX12-FAKE16-NEXT: s_mov_b32 exec_lo, s3
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX12-FAKE16-NEXT: v_mov_b32_e32 v0, s0
; GFX12-FAKE16-NEXT: v_mov_b32_e32 v2, s1
-; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX12-FAKE16-NEXT: v_interp_p10_rtz_f16_f32 v3, v1, v0, v1 wait_exp:0
; GFX12-FAKE16-NEXT: v_interp_p10_rtz_f16_f32 v0, v1, v0, v1 op_sel:[1,0,1,0] wait_exp:7
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX12-FAKE16-NEXT: v_interp_p2_rtz_f16_f32 v3, v1, v2, v3 wait_exp:7
-; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX12-FAKE16-NEXT: v_interp_p2_rtz_f16_f32 v0, v1, v2, v0 op_sel:[1,0,0,0] wait_exp:7
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-FAKE16-NEXT: v_add_f16_e32 v0, v3, v0
; GFX12-FAKE16-NEXT: ; return to shader part epilog
main_body:
diff --git a/llvm/test/CodeGen/AMDGPU/GlobalISel/llvm.amdgcn.intersect_ray.ll b/llvm/test/CodeGen/AMDGPU/GlobalISel/llvm.amdgcn.intersect_ray.ll
index ee93f413566e54..8e2c019f69edd4 100644
--- a/llvm/test/CodeGen/AMDGPU/GlobalISel/llvm.amdgcn.intersect_ray.ll
+++ b/llvm/test/CodeGen/AMDGPU/GlobalISel/llvm.amdgcn.intersect_ray.ll
@@ -256,7 +256,7 @@ define amdgpu_ps <4 x float> @image_bvh_intersect_ray_vgpr_descr(i32 %node_ptr,
; GFX11-NEXT: v_cmp_eq_u64_e32 vcc_lo, s[4:5], v[11:12]
; GFX11-NEXT: v_cmp_eq_u64_e64 s0, s[6:7], v[13:14]
; GFX11-NEXT: s_and_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_saveexec_b32 s0, s0
; GFX11-NEXT: image_bvh_intersect_ray v[0:3], [v18, v19, v[15:17], v[5:7], v[8:10]], s[4:7]
; GFX11-NEXT: ; implicit-def: $vgpr11
@@ -377,7 +377,7 @@ define amdgpu_ps <4 x float> @image_bvh_intersect_ray_a16_vgpr_descr(i32 %node_p
; GFX11-TRUE16-NEXT: v_cmp_eq_u64_e32 vcc_lo, s[4:5], v[9:10]
; GFX11-TRUE16-NEXT: v_cmp_eq_u64_e64 s0, s[6:7], v[11:12]
; GFX11-TRUE16-NEXT: s_and_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_saveexec_b32 s0, s0
; GFX11-TRUE16-NEXT: image_bvh_intersect_ray v[0:3], [v16, v17, v[13:15], v[18:20]], s[4:7] a16
; GFX11-TRUE16-NEXT: ; implicit-def: $vgpr9
@@ -413,7 +413,7 @@ define amdgpu_ps <4 x float> @image_bvh_intersect_ray_a16_vgpr_descr(i32 %node_p
; GFX11-FAKE16-NEXT: v_cmp_eq_u64_e32 vcc_lo, s[4:5], v[9:10]
; GFX11-FAKE16-NEXT: v_cmp_eq_u64_e64 s0, s[6:7], v[11:12]
; GFX11-FAKE16-NEXT: s_and_b32 s0, vcc_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_saveexec_b32 s0, s0
; GFX11-FAKE16-NEXT: image_bvh_intersect_ray v[0:3], [v16, v17, v[13:15], v[4:6]], s[4:7] a16
; GFX11-FAKE16-NEXT: ; implicit-def: $vgpr9
@@ -523,7 +523,7 @@ define amdgpu_ps <4 x float> @image_bvh64_intersect_ray_vgpr_descr(i64 %node_ptr
; GFX11-NEXT: v_cmp_eq_u64_e32 vcc_lo, s[4:5], v[12:13]
; GFX11-NEXT: v_cmp_eq_u64_e64 s0, s[6:7], v[14:15]
; GFX11-NEXT: s_and_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_saveexec_b32 s0, s0
; GFX11-NEXT: image_bvh64_intersect_ray v[0:3], [v[19:20], v21, v[16:18], v[6:8], v[9:11]], s[4:7]
; GFX11-NEXT: ; implicit-def: $vgpr12
@@ -647,7 +647,7 @@ define amdgpu_ps <4 x float> @image_bvh64_intersect_ray_a16_vgpr_descr(i64 %node
; GFX11-TRUE16-NEXT: v_cmp_eq_u64_e32 vcc_lo, s[4:5], v[10:11]
; GFX11-TRUE16-NEXT: v_cmp_eq_u64_e64 s0, s[6:7], v[12:13]
; GFX11-TRUE16-NEXT: s_and_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_saveexec_b32 s0, s0
; GFX11-TRUE16-NEXT: image_bvh64_intersect_ray v[0:3], [v[17:18], v19, v[14:16], v[4:6]], s[4:7] a16
; GFX11-TRUE16-NEXT: ; implicit-def: $vgpr10
@@ -684,7 +684,7 @@ define amdgpu_ps <4 x float> @image_bvh64_intersect_ray_a16_vgpr_descr(i64 %node
; GFX11-FAKE16-NEXT: v_cmp_eq_u64_e32 vcc_lo, s[4:5], v[10:11]
; GFX11-FAKE16-NEXT: v_cmp_eq_u64_e64 s0, s[6:7], v[12:13]
; GFX11-FAKE16-NEXT: s_and_b32 s0, vcc_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_saveexec_b32 s0, s0
; GFX11-FAKE16-NEXT: image_bvh64_intersect_ray v[0:3], [v[17:18], v19, v[14:16], v[4:6]], s[4:7] a16
; GFX11-FAKE16-NEXT: ; implicit-def: $vgpr10
@@ -785,11 +785,10 @@ define amdgpu_kernel void @image_bvh_intersect_ray_nsa_reassign(ptr %p_node_ptr,
; GFX11-NEXT: v_mov_b32_e32 v0, s0
; GFX11-NEXT: s_mov_b32 s1, 1.0
; GFX11-NEXT: s_mov_b32 s0, 0
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_add_co_u32 v0, vcc_lo, v0, v4
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-NEXT: v_add_co_u32 v2, vcc_lo, v2, v4
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX11-NEXT: flat_load_b32 v9, v[0:1]
; GFX11-NEXT: flat_load_b32 v10, v[2:3]
@@ -893,7 +892,6 @@ define amdgpu_kernel void @image_bvh_intersect_ray_a16_nsa_reassign(ptr %p_node_
; GFX11-TRUE16-NEXT: s_mov_b32 s0, 0
; GFX11-TRUE16-NEXT: s_pack_ll_b32_b16 s10, s3, 0x4500
; GFX11-TRUE16-NEXT: v_add_co_u32 v0, vcc_lo, v0, v4
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-TRUE16-NEXT: v_add_co_u32 v2, vcc_lo, v2, v4
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
@@ -924,11 +922,10 @@ define amdgpu_kernel void @image_bvh_intersect_ray_a16_nsa_reassign(ptr %p_node_
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v0, s0
; GFX11-FAKE16-NEXT: s_mov_b32 s1, 1.0
; GFX11-FAKE16-NEXT: s_mov_b32 s0, 0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_add_co_u32 v0, vcc_lo, v0, v4
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-FAKE16-NEXT: v_add_co_u32 v2, vcc_lo, v2, v4
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX11-FAKE16-NEXT: flat_load_b32 v6, v[0:1]
; GFX11-FAKE16-NEXT: flat_load_b32 v7, v[2:3]
@@ -1042,7 +1039,7 @@ define amdgpu_kernel void @image_bvh64_intersect_ray_nsa_reassign(ptr %p_ray, <4
; GFX11-NEXT: v_dual_mov_b32 v1, s7 :: v_dual_lshlrev_b32 v2, 2, v0
; GFX11-NEXT: v_mov_b32_e32 v0, s6
; GFX11-NEXT: s_mov_b32 s6, 2.0
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_3)
; GFX11-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-NEXT: v_mov_b32_e32 v2, s6
@@ -1144,7 +1141,6 @@ define amdgpu_kernel void @image_bvh64_intersect_ray_a16_nsa_reassign(ptr %p_ray
; GFX11-TRUE16-NEXT: s_pack_ll_b32_b16 s10, s7, 0x4500
; GFX11-TRUE16-NEXT: v_mov_b32_e32 v3, s8
; GFX11-TRUE16-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-TRUE16-NEXT: v_dual_mov_b32 v6, 0xb36211c6 :: v_dual_mov_b32 v5, s10
; GFX11-TRUE16-NEXT: v_mov_b32_e32 v2, s6
@@ -1176,7 +1172,7 @@ define amdgpu_kernel void @image_bvh64_intersect_ray_a16_nsa_reassign(ptr %p_ray
; GFX11-FAKE16-NEXT: v_dual_mov_b32 v1, s7 :: v_dual_lshlrev_b32 v2, 2, v0
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v0, s6
; GFX11-FAKE16-NEXT: s_mov_b32 s6, 2.0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v2, s6
diff --git a/llvm/test/CodeGen/AMDGPU/GlobalISel/mubuf-global.ll b/llvm/test/CodeGen/AMDGPU/GlobalISel/mubuf-global.ll
index 3054c485eee11d..88ae3e635d87bb 100644
--- a/llvm/test/CodeGen/AMDGPU/GlobalISel/mubuf-global.ll
+++ b/llvm/test/CodeGen/AMDGPU/GlobalISel/mubuf-global.ll
@@ -235,7 +235,6 @@ define amdgpu_ps void @mubuf_store_vgpr_ptr_offset4294967296(ptr addrspace(1) %p
; GFX12-LABEL: mubuf_store_vgpr_ptr_offset4294967296:
; GFX12: ; %bb.0:
; GFX12-NEXT: v_add_co_u32 v0, vcc_lo, v0, 0
-; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-NEXT: v_add_co_ci_u32_e64 v1, null, 4, v1, vcc_lo
; GFX12-NEXT: v_mov_b32_e32 v2, 0
; GFX12-NEXT: global_store_b32 v[0:1], v2, off
@@ -269,7 +268,6 @@ define amdgpu_ps void @mubuf_store_vgpr_ptr_offset4294967297(ptr addrspace(1) %p
; GFX12-LABEL: mubuf_store_vgpr_ptr_offset4294967297:
; GFX12: ; %bb.0:
; GFX12-NEXT: v_add_co_u32 v0, vcc_lo, v0, 4
-; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-NEXT: v_add_co_ci_u32_e64 v1, null, 4, v1, vcc_lo
; GFX12-NEXT: v_mov_b32_e32 v2, 0
; GFX12-NEXT: global_store_b32 v[0:1], v2, off
@@ -381,7 +379,7 @@ define amdgpu_ps void @mubuf_store_vgpr_ptr_sgpr_offset(ptr addrspace(1) %ptr, i
; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_lshl_b64 s[0:1], s[2:3], 2
; GFX12-NEXT: v_dual_mov_b32 v3, s1 :: v_dual_mov_b32 v2, s0
-; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX12-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
; GFX12-NEXT: v_add_co_ci_u32_e64 v1, null, v1, v3, vcc_lo
; GFX12-NEXT: v_mov_b32_e32 v2, 0
@@ -419,7 +417,7 @@ define amdgpu_ps void @mubuf_store_vgpr_ptr_sgpr_offset_offset256(ptr addrspace(
; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_lshl_b64 s[0:1], s[2:3], 2
; GFX12-NEXT: v_dual_mov_b32 v3, s1 :: v_dual_mov_b32 v2, s0
-; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX12-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
; GFX12-NEXT: v_add_co_ci_u32_e64 v1, null, v1, v3, vcc_lo
; GFX12-NEXT: v_mov_b32_e32 v2, 0
@@ -458,7 +456,7 @@ define amdgpu_ps void @mubuf_store_vgpr_ptr_sgpr_offset256_offset(ptr addrspace(
; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_lshl_b64 s[0:1], s[2:3], 2
; GFX12-NEXT: v_dual_mov_b32 v3, s1 :: v_dual_mov_b32 v2, s0
-; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX12-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
; GFX12-NEXT: v_add_co_ci_u32_e64 v1, null, v1, v3, vcc_lo
; GFX12-NEXT: v_mov_b32_e32 v2, 0
@@ -502,7 +500,7 @@ define amdgpu_ps void @mubuf_store_sgpr_ptr_vgpr_offset(ptr addrspace(1) inreg %
; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX12-NEXT: v_lshlrev_b64_e32 v[0:1], 2, v[0:1]
; GFX12-NEXT: v_add_co_u32 v0, vcc_lo, v2, v0
-; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX12-NEXT: v_add_co_ci_u32_e64 v1, null, v3, v1, vcc_lo
; GFX12-NEXT: v_mov_b32_e32 v2, 0
; GFX12-NEXT: global_store_b32 v[0:1], v2, off
@@ -546,7 +544,7 @@ define amdgpu_ps void @mubuf_store_sgpr_ptr_vgpr_offset_offset4095(ptr addrspace
; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX12-NEXT: v_lshlrev_b64_e32 v[0:1], 2, v[0:1]
; GFX12-NEXT: v_add_co_u32 v0, vcc_lo, v2, v0
-; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX12-NEXT: v_add_co_ci_u32_e64 v1, null, v3, v1, vcc_lo
; GFX12-NEXT: v_mov_b32_e32 v2, 0
; GFX12-NEXT: global_store_b32 v[0:1], v2, off offset:16380
@@ -590,7 +588,7 @@ define amdgpu_ps void @mubuf_store_sgpr_ptr_offset4095_vgpr_offset(ptr addrspace
; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX12-NEXT: v_lshlrev_b64_e32 v[0:1], 2, v[0:1]
; GFX12-NEXT: v_add_co_u32 v0, vcc_lo, v2, v0
-; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX12-NEXT: v_add_co_ci_u32_e64 v1, null, v3, v1, vcc_lo
; GFX12-NEXT: v_mov_b32_e32 v2, 0
; GFX12-NEXT: global_store_b32 v[0:1], v2, off offset:16380
@@ -833,7 +831,6 @@ define amdgpu_ps float @mubuf_load_vgpr_ptr_offset4294967296(ptr addrspace(1) %p
; GFX12-LABEL: mubuf_load_vgpr_ptr_offset4294967296:
; GFX12: ; %bb.0:
; GFX12-NEXT: v_add_co_u32 v0, vcc_lo, v0, 0
-; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-NEXT: v_add_co_ci_u32_e64 v1, null, 4, v1, vcc_lo
; GFX12-NEXT: global_load_b32 v0, v[0:1], off scope:SCOPE_SYS
; GFX12-NEXT: s_wait_loadcnt 0x0
@@ -867,7 +864,6 @@ define amdgpu_ps float @mubuf_load_vgpr_ptr_offset4294967297(ptr addrspace(1) %p
; GFX12-LABEL: mubuf_load_vgpr_ptr_offset4294967297:
; GFX12: ; %bb.0:
; GFX12-NEXT: v_add_co_u32 v0, vcc_lo, v0, 4
-; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-NEXT: v_add_co_ci_u32_e64 v1, null, 4, v1, vcc_lo
; GFX12-NEXT: global_load_b32 v0, v[0:1], off scope:SCOPE_SYS
; GFX12-NEXT: s_wait_loadcnt 0x0
@@ -980,7 +976,7 @@ define amdgpu_ps float @mubuf_load_vgpr_ptr_sgpr_offset(ptr addrspace(1) %ptr, i
; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_lshl_b64 s[0:1], s[2:3], 2
; GFX12-NEXT: v_dual_mov_b32 v3, s1 :: v_dual_mov_b32 v2, s0
-; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX12-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
; GFX12-NEXT: v_add_co_ci_u32_e64 v1, null, v1, v3, vcc_lo
; GFX12-NEXT: global_load_b32 v0, v[0:1], off scope:SCOPE_SYS
@@ -1018,7 +1014,7 @@ define amdgpu_ps float @mubuf_load_vgpr_ptr_sgpr_offset_offset256(ptr addrspace(
; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_lshl_b64 s[0:1], s[2:3], 2
; GFX12-NEXT: v_dual_mov_b32 v3, s1 :: v_dual_mov_b32 v2, s0
-; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX12-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
; GFX12-NEXT: v_add_co_ci_u32_e64 v1, null, v1, v3, vcc_lo
; GFX12-NEXT: global_load_b32 v0, v[0:1], off offset:1024 scope:SCOPE_SYS
@@ -1057,7 +1053,7 @@ define amdgpu_ps float @mubuf_load_vgpr_ptr_sgpr_offset256_offset(ptr addrspace(
; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_lshl_b64 s[0:1], s[2:3], 2
; GFX12-NEXT: v_dual_mov_b32 v3, s1 :: v_dual_mov_b32 v2, s0
-; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX12-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
; GFX12-NEXT: v_add_co_ci_u32_e64 v1, null, v1, v3, vcc_lo
; GFX12-NEXT: global_load_b32 v0, v[0:1], off offset:1024 scope:SCOPE_SYS
@@ -1101,7 +1097,7 @@ define amdgpu_ps float @mubuf_load_sgpr_ptr_vgpr_offset(ptr addrspace(1) inreg %
; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX12-NEXT: v_lshlrev_b64_e32 v[0:1], 2, v[0:1]
; GFX12-NEXT: v_add_co_u32 v0, vcc_lo, v2, v0
-; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX12-NEXT: v_add_co_ci_u32_e64 v1, null, v3, v1, vcc_lo
; GFX12-NEXT: global_load_b32 v0, v[0:1], off scope:SCOPE_SYS
; GFX12-NEXT: s_wait_loadcnt 0x0
@@ -1145,7 +1141,7 @@ define amdgpu_ps float @mubuf_load_sgpr_ptr_vgpr_offset_offset4095(ptr addrspace
; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX12-NEXT: v_lshlrev_b64_e32 v[0:1], 2, v[0:1]
; GFX12-NEXT: v_add_co_u32 v0, vcc_lo, v2, v0
-; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX12-NEXT: v_add_co_ci_u32_e64 v1, null, v3, v1, vcc_lo
; GFX12-NEXT: global_load_b32 v0, v[0:1], off offset:16380 scope:SCOPE_SYS
; GFX12-NEXT: s_wait_loadcnt 0x0
@@ -1189,7 +1185,7 @@ define amdgpu_ps float @mubuf_load_sgpr_ptr_offset4095_vgpr_offset(ptr addrspace
; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX12-NEXT: v_lshlrev_b64_e32 v[0:1], 2, v[0:1]
; GFX12-NEXT: v_add_co_u32 v0, vcc_lo, v2, v0
-; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX12-NEXT: v_add_co_ci_u32_e64 v1, null, v3, v1, vcc_lo
; GFX12-NEXT: global_load_b32 v0, v[0:1], off offset:16380 scope:SCOPE_SYS
; GFX12-NEXT: s_wait_loadcnt 0x0
@@ -1360,7 +1356,6 @@ define amdgpu_ps float @mubuf_atomicrmw_vgpr_ptr_offset4294967296(ptr addrspace(
; GFX12-LABEL: mubuf_atomicrmw_vgpr_ptr_offset4294967296:
; GFX12: ; %bb.0:
; GFX12-NEXT: v_add_co_u32 v0, vcc_lo, v0, 0
-; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-NEXT: v_add_co_ci_u32_e64 v1, null, 4, v1, vcc_lo
; GFX12-NEXT: v_mov_b32_e32 v2, 2
; GFX12-NEXT: global_atomic_add_u32 v0, v[0:1], v2, off th:TH_ATOMIC_RETURN scope:SCOPE_DEV
@@ -1410,7 +1405,7 @@ define amdgpu_ps float @mubuf_atomicrmw_sgpr_ptr_vgpr_offset(ptr addrspace(1) in
; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX12-NEXT: v_lshlrev_b64_e32 v[0:1], 2, v[0:1]
; GFX12-NEXT: v_add_co_u32 v0, vcc_lo, v2, v0
-; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX12-NEXT: v_add_co_ci_u32_e64 v1, null, v3, v1, vcc_lo
; GFX12-NEXT: v_mov_b32_e32 v2, 2
; GFX12-NEXT: global_atomic_add_u32 v0, v[0:1], v2, off th:TH_ATOMIC_RETURN scope:SCOPE_DEV
@@ -1643,7 +1638,7 @@ define amdgpu_ps float @mubuf_cmpxchg_sgpr_ptr_vgpr_offset(ptr addrspace(1) inre
; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX12-NEXT: v_lshlrev_b64_e32 v[0:1], 2, v[0:1]
; GFX12-NEXT: v_add_co_u32 v0, vcc_lo, v4, v0
-; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX12-NEXT: v_add_co_ci_u32_e64 v1, null, v5, v1, vcc_lo
; GFX12-NEXT: global_atomic_cmpswap_b32 v0, v[0:1], v[2:3], off th:TH_ATOMIC_RETURN scope:SCOPE_DEV
; GFX12-NEXT: s_wait_loadcnt 0x0
diff --git a/llvm/test/CodeGen/AMDGPU/GlobalISel/mul-known-bits.i64.ll b/llvm/test/CodeGen/AMDGPU/GlobalISel/mul-known-bits.i64.ll
index d1621dbc267da9..98e381adc9f1b4 100644
--- a/llvm/test/CodeGen/AMDGPU/GlobalISel/mul-known-bits.i64.ll
+++ b/llvm/test/CodeGen/AMDGPU/GlobalISel/mul-known-bits.i64.ll
@@ -526,6 +526,7 @@ define amdgpu_kernel void @v_mul64_masked_before_and_in_branch(ptr addrspace(1)
; GFX11-NEXT: ; implicit-def: $vgpr0_vgpr1
; GFX11-NEXT: s_waitcnt vmcnt(1)
; GFX11-NEXT: v_cmpx_ge_u64_e32 0, v[2:3]
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s2, exec_lo, s2
; GFX11-NEXT: s_cbranch_execz .LBB10_2
; GFX11-NEXT: ; %bb.1: ; %else
@@ -545,6 +546,7 @@ define amdgpu_kernel void @v_mul64_masked_before_and_in_branch(ptr addrspace(1)
; GFX11-NEXT: v_mov_b32_e32 v0, 0
; GFX11-NEXT: .LBB10_4: ; %endif
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s2
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: v_mov_b32_e32 v2, 0
; GFX11-NEXT: global_store_b64 v2, v[0:1], s[0:1]
; GFX11-NEXT: s_endpgm
diff --git a/llvm/test/CodeGen/AMDGPU/GlobalISel/mul.ll b/llvm/test/CodeGen/AMDGPU/GlobalISel/mul.ll
index 46fb53085ebf05..9144df9b218819 100644
--- a/llvm/test/CodeGen/AMDGPU/GlobalISel/mul.ll
+++ b/llvm/test/CodeGen/AMDGPU/GlobalISel/mul.ll
@@ -2741,11 +2741,13 @@ define amdgpu_ps <8 x i32> @s_mul_i256(i256 inreg %num, i256 inreg %den) {
; GFX12-NEXT: s_mul_i32 s0, s0, s8
; GFX12-NEXT: s_add_co_ci_u32 s25, s25, 0
; GFX12-NEXT: s_cmp_lg_u32 s23, 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_add_co_ci_u32 s15, s25, s21
; GFX12-NEXT: s_add_co_ci_u32 s21, s22, s26
; GFX12-NEXT: s_cmp_lg_u32 s38, 0
; GFX12-NEXT: s_add_co_ci_u32 s1, s21, s1
; GFX12-NEXT: s_cmp_lg_u32 s37, 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_add_co_ci_u32 s1, s1, s2
; GFX12-NEXT: s_cmp_lg_u32 s36, 0
; GFX12-NEXT: s_mov_b32 s2, s17
@@ -2932,11 +2934,13 @@ define amdgpu_ps <8 x i32> @s_mul_i256(i256 inreg %num, i256 inreg %den) {
; GFX1250-NEXT: s_mul_i32 s0, s0, s8
; GFX1250-NEXT: s_add_co_ci_u32 s25, s25, 0
; GFX1250-NEXT: s_cmp_lg_u32 s23, 0
+; GFX1250-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
; GFX1250-NEXT: s_add_co_ci_u32 s15, s25, s21
; GFX1250-NEXT: s_add_co_ci_u32 s21, s22, s26
; GFX1250-NEXT: s_cmp_lg_u32 s38, 0
; GFX1250-NEXT: s_add_co_ci_u32 s1, s21, s1
; GFX1250-NEXT: s_cmp_lg_u32 s37, 0
+; GFX1250-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1250-NEXT: s_add_co_ci_u32 s1, s1, s2
; GFX1250-NEXT: s_cmp_lg_u32 s36, 0
; GFX1250-NEXT: s_mov_b32 s2, s17
@@ -3119,11 +3123,13 @@ define amdgpu_ps <8 x i32> @s_mul_i256(i256 inreg %num, i256 inreg %den) {
; GFX13-NEXT: s_mul_i32 s0, s0, s8
; GFX13-NEXT: s_add_co_ci_u32 s25, s25, 0
; GFX13-NEXT: s_cmp_lg_u32 s23, 0
+; GFX13-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
; GFX13-NEXT: s_add_co_ci_u32 s15, s25, s21
; GFX13-NEXT: s_add_co_ci_u32 s21, s22, s26
; GFX13-NEXT: s_cmp_lg_u32 s38, 0
; GFX13-NEXT: s_add_co_ci_u32 s1, s21, s1
; GFX13-NEXT: s_cmp_lg_u32 s37, 0
+; GFX13-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX13-NEXT: s_add_co_ci_u32 s1, s1, s2
; GFX13-NEXT: s_cmp_lg_u32 s36, 0
; GFX13-NEXT: s_mov_b32 s2, s17
@@ -3661,24 +3667,23 @@ define i256 @v_mul_i256(i256 %num, i256 %den) {
; GFX1250-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, s2
; GFX1250-NEXT: v_mad_co_u64_u32 v[14:15], s5, v5, v8, v[12:13]
; GFX1250-NEXT: v_mad_co_u64_u32 v[12:13], s2, v17, v8, v[20:21]
-; GFX1250-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_4)
; GFX1250-NEXT: v_add_co_ci_u32_e64 v3, s2, v3, v10, s2
; GFX1250-NEXT: v_add_co_ci_u32_e64 v4, s2, v6, v11, s2
-; GFX1250-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1250-NEXT: v_add_co_ci_u32_e64 v5, s2, v1, v14, s2
; GFX1250-NEXT: v_add_co_ci_u32_e64 v6, s2, v28, v15, s2
-; GFX1250-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1250-NEXT: v_add_co_ci_u32_e64 v1, null, v25, v22, s2
+; GFX1250-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1250-NEXT: v_add_co_ci_u32_e64 v1, null, v1, v9, s5
-; GFX1250-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX1250-NEXT: v_add_co_ci_u32_e64 v1, null, v1, v2, s4
; GFX1250-NEXT: v_mov_b32_e32 v2, v13
+; GFX1250-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1250-NEXT: v_add_co_ci_u32_e64 v1, null, v1, v31, s3
-; GFX1250-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1250-NEXT: v_add_co_ci_u32_e64 v1, null, v1, v30, s1
-; GFX1250-NEXT: v_add_co_ci_u32_e64 v1, null, v1, v29, vcc_lo
; GFX1250-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-NEXT: v_add_co_ci_u32_e64 v1, null, v1, v29, vcc_lo
; GFX1250-NEXT: v_add_co_ci_u32_e64 v1, null, v1, v24, s0
+; GFX1250-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1250-NEXT: v_mad_u32 v7, v7, v8, v1
; GFX1250-NEXT: v_mov_b32_e32 v1, v12
; GFX1250-NEXT: s_set_pc_i64 s[30:31]
@@ -3753,23 +3758,22 @@ define i256 @v_mul_i256(i256 %num, i256 %den) {
; GFX13-NEXT: v_mad_co_u64_u32 v[14:15], s5, v21, v8, v[10:11]
; GFX13-NEXT: v_add_co_ci_u32_e64 v10, null, 0, v1, s2
; GFX13-NEXT: v_mad_co_u64_u32 v[1:2], s2, v17, v8, v[5:6]
-; GFX13-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX13-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_4)
; GFX13-NEXT: v_add_co_ci_u32_e64 v3, s2, v3, v12, s2
; GFX13-NEXT: v_add_co_ci_u32_e64 v4, s2, v28, v13, s2
-; GFX13-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX13-NEXT: v_add_co_ci_u32_e64 v5, s2, v10, v14, s2
; GFX13-NEXT: v_add_co_ci_u32_e64 v6, s2, v25, v15, s2
-; GFX13-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX13-NEXT: v_add_co_ci_u32_e64 v7, null, v7, v20, s2
-; GFX13-NEXT: v_add_co_ci_u32_e64 v7, null, v7, v9, s5
; GFX13-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX13-NEXT: v_add_co_ci_u32_e64 v7, null, v7, v9, s5
; GFX13-NEXT: v_add_co_ci_u32_e64 v7, null, v7, v18, s4
-; GFX13-NEXT: v_add_co_ci_u32_e64 v7, null, v7, v30, s3
; GFX13-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX13-NEXT: v_add_co_ci_u32_e64 v7, null, v7, v30, s3
; GFX13-NEXT: v_add_co_ci_u32_e64 v7, null, v7, v29, s0
-; GFX13-NEXT: v_add_co_ci_u32_e64 v7, null, v7, v26, s1
; GFX13-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX13-NEXT: v_add_co_ci_u32_e64 v7, null, v7, v26, s1
; GFX13-NEXT: v_add_co_ci_u32_e64 v7, null, v7, v27, vcc_lo
+; GFX13-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX13-NEXT: v_mad_co_u64_u32 v[7:8], null, v22, v8, v[7:8]
; GFX13-NEXT: s_set_pc_i64 s[30:31]
%result = mul i256 %num, %den
diff --git a/llvm/test/CodeGen/AMDGPU/GlobalISel/regbanklegalize-amdgcn.s.buffer.load.ll b/llvm/test/CodeGen/AMDGPU/GlobalISel/regbanklegalize-amdgcn.s.buffer.load.ll
index 69852bfd4b8dd5..9b76362060a1a6 100644
--- a/llvm/test/CodeGen/AMDGPU/GlobalISel/regbanklegalize-amdgcn.s.buffer.load.ll
+++ b/llvm/test/CodeGen/AMDGPU/GlobalISel/regbanklegalize-amdgcn.s.buffer.load.ll
@@ -2,7 +2,7 @@
; RUN: llc -global-isel -mtriple=amdgpu10.10-mesa-mesa3d -o - %s | FileCheck %s -check-prefix=GFX10
; RUN: llc -global-isel -mtriple=amdgpu12.00-amd-amdhsa -o - %s | FileCheck %s -check-prefix=GFX12
; ----------------------------------------------------------------------------
-; Case 1: Uniform result — SGPR rsrc + SGPR offset
+; Case 1: Uniform result — SGPR rsrc + SGPR offset
; ----------------------------------------------------------------------------
; Case 1a: i32, SGPR rsrc, SGPR offset
@@ -493,9 +493,11 @@ define amdgpu_ps void @s_buffer_load_i32_vgpr_rsrc_sgpr_offset(<4 x i32> %rsrc,
; GFX12-NEXT: ; implicit-def: $vgpr4
; GFX12-NEXT: ; implicit-def: $vgpr2_vgpr3
; GFX12-NEXT: s_xor_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB12_1
; GFX12-NEXT: ; %bb.2:
; GFX12-NEXT: s_mov_b32 exec_lo, s1
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: v_mov_b32_e32 v0, 0
; GFX12-NEXT: s_wait_loadcnt 0x0
; GFX12-NEXT: global_store_b32 v0, v1, s[8:9]
@@ -560,9 +562,11 @@ define amdgpu_ps void @s_buffer_load_v2i32_vgpr_rsrc_sgpr_offset(<4 x i32> %rsrc
; GFX12-NEXT: ; implicit-def: $vgpr6
; GFX12-NEXT: ; implicit-def: $vgpr2_vgpr3
; GFX12-NEXT: s_xor_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB13_1
; GFX12-NEXT: ; %bb.2:
; GFX12-NEXT: s_mov_b32 exec_lo, s1
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: v_mov_b32_e32 v0, 0
; GFX12-NEXT: s_wait_loadcnt 0x0
; GFX12-NEXT: global_store_b64 v0, v[4:5], s[8:9]
@@ -627,9 +631,11 @@ define amdgpu_ps void @s_buffer_load_i96_vgpr_rsrc_sgpr_offset(<4 x i32> %rsrc,
; GFX12-NEXT: ; implicit-def: $vgpr7
; GFX12-NEXT: ; implicit-def: $vgpr2_vgpr3
; GFX12-NEXT: s_xor_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB14_1
; GFX12-NEXT: ; %bb.2:
; GFX12-NEXT: s_mov_b32 exec_lo, s1
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: v_mov_b32_e32 v0, 0
; GFX12-NEXT: s_wait_loadcnt 0x0
; GFX12-NEXT: global_store_b96 v0, v[4:6], s[8:9]
@@ -694,9 +700,11 @@ define amdgpu_ps void @s_buffer_load_v4i32_vgpr_rsrc_sgpr_offset(<4 x i32> %rsrc
; GFX12-NEXT: ; implicit-def: $vgpr8
; GFX12-NEXT: ; implicit-def: $vgpr2_vgpr3
; GFX12-NEXT: s_xor_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB15_1
; GFX12-NEXT: ; %bb.2:
; GFX12-NEXT: s_mov_b32 exec_lo, s1
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: v_mov_b32_e32 v0, 0
; GFX12-NEXT: s_wait_loadcnt 0x0
; GFX12-NEXT: global_store_b128 v0, v[4:7], s[8:9]
@@ -706,7 +714,7 @@ define amdgpu_ps void @s_buffer_load_v4i32_vgpr_rsrc_sgpr_offset(<4 x i32> %rsrc
ret void
}
-; Case 3e: <8 x i32> (256-bit), VGPR rsrc, SGPR offset → 2x MUBUF + waterfall
+; Case 3e: <8 x i32> (256-bit), VGPR rsrc, SGPR offset → 2x MUBUF + waterfall
define amdgpu_ps void @s_buffer_load_v8i32_vgpr_rsrc_sgpr_offset(<4 x i32> %rsrc, i32 inreg %soffset, ptr addrspace(1) inreg %out) {
; GFX10-LABEL: s_buffer_load_v8i32_vgpr_rsrc_sgpr_offset:
; GFX10: ; %bb.0:
@@ -767,9 +775,11 @@ define amdgpu_ps void @s_buffer_load_v8i32_vgpr_rsrc_sgpr_offset(<4 x i32> %rsrc
; GFX12-NEXT: ; implicit-def: $vgpr12
; GFX12-NEXT: ; implicit-def: $vgpr2_vgpr3
; GFX12-NEXT: s_xor_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB16_1
; GFX12-NEXT: ; %bb.2:
; GFX12-NEXT: s_mov_b32 exec_lo, s1
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: v_mov_b32_e32 v0, 0
; GFX12-NEXT: s_wait_loadcnt 0x1
; GFX12-NEXT: global_store_b128 v0, v[4:7], s[8:9]
@@ -828,7 +838,7 @@ define amdgpu_ps void @s_buffer_load_i32_vgpr_rsrc_vgpr_offset(<4 x i32> %rsrc,
; GFX12-NEXT: v_cmp_eq_u64_e32 vcc_lo, s[4:5], v[0:1]
; GFX12-NEXT: v_cmp_eq_u64_e64 s2, s[6:7], v[2:3]
; GFX12-NEXT: s_and_b32 s2, vcc_lo, s2
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_and_saveexec_b32 s2, s2
; GFX12-NEXT: buffer_load_b32 v1, v4, s[4:7], null offen
; GFX12-NEXT: ; implicit-def: $vgpr0
@@ -838,6 +848,7 @@ define amdgpu_ps void @s_buffer_load_i32_vgpr_rsrc_vgpr_offset(<4 x i32> %rsrc,
; GFX12-NEXT: s_cbranch_execnz .LBB17_1
; GFX12-NEXT: ; %bb.2:
; GFX12-NEXT: s_mov_b32 exec_lo, s3
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: v_mov_b32_e32 v0, 0
; GFX12-NEXT: s_wait_loadcnt 0x0
; GFX12-NEXT: global_store_b32 v0, v1, s[0:1]
@@ -888,7 +899,7 @@ define amdgpu_ps void @s_buffer_load_v2i32_vgpr_rsrc_vgpr_offset(<4 x i32> %rsrc
; GFX12-NEXT: v_cmp_eq_u64_e32 vcc_lo, s[4:5], v[0:1]
; GFX12-NEXT: v_cmp_eq_u64_e64 s2, s[6:7], v[2:3]
; GFX12-NEXT: s_and_b32 s2, vcc_lo, s2
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_3) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_and_saveexec_b32 s2, s2
; GFX12-NEXT: s_wait_loadcnt 0x0
; GFX12-NEXT: buffer_load_b64 v[5:6], v4, s[4:7], null offen
@@ -899,6 +910,7 @@ define amdgpu_ps void @s_buffer_load_v2i32_vgpr_rsrc_vgpr_offset(<4 x i32> %rsrc
; GFX12-NEXT: s_cbranch_execnz .LBB18_1
; GFX12-NEXT: ; %bb.2:
; GFX12-NEXT: s_mov_b32 exec_lo, s3
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: v_mov_b32_e32 v0, 0
; GFX12-NEXT: s_wait_loadcnt 0x0
; GFX12-NEXT: global_store_b64 v0, v[5:6], s[0:1]
@@ -949,7 +961,7 @@ define amdgpu_ps void @s_buffer_load_v4i32_vgpr_rsrc_vgpr_offset(<4 x i32> %rsrc
; GFX12-NEXT: v_cmp_eq_u64_e32 vcc_lo, s[4:5], v[0:1]
; GFX12-NEXT: v_cmp_eq_u64_e64 s2, s[6:7], v[2:3]
; GFX12-NEXT: s_and_b32 s2, vcc_lo, s2
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_3) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_and_saveexec_b32 s2, s2
; GFX12-NEXT: s_wait_loadcnt 0x0
; GFX12-NEXT: buffer_load_b128 v[5:8], v4, s[4:7], null offen
@@ -960,6 +972,7 @@ define amdgpu_ps void @s_buffer_load_v4i32_vgpr_rsrc_vgpr_offset(<4 x i32> %rsrc
; GFX12-NEXT: s_cbranch_execnz .LBB19_1
; GFX12-NEXT: ; %bb.2:
; GFX12-NEXT: s_mov_b32 exec_lo, s3
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: v_mov_b32_e32 v0, 0
; GFX12-NEXT: s_wait_loadcnt 0x0
; GFX12-NEXT: global_store_b128 v0, v[5:8], s[0:1]
@@ -969,7 +982,7 @@ define amdgpu_ps void @s_buffer_load_v4i32_vgpr_rsrc_vgpr_offset(<4 x i32> %rsrc
ret void
}
-; Case 4d: <8 x i32> (256-bit), VGPR rsrc, VGPR offset → 2x MUBUF + waterfall
+; Case 4d: <8 x i32> (256-bit), VGPR rsrc, VGPR offset → 2x MUBUF + waterfall
define amdgpu_ps void @s_buffer_load_v8i32_vgpr_rsrc_vgpr_offset(<4 x i32> %rsrc, i32 %offset, ptr addrspace(1) inreg %out) {
; GFX10-LABEL: s_buffer_load_v8i32_vgpr_rsrc_vgpr_offset:
; GFX10: ; %bb.0:
@@ -1024,9 +1037,11 @@ define amdgpu_ps void @s_buffer_load_v8i32_vgpr_rsrc_vgpr_offset(<4 x i32> %rsrc
; GFX12-NEXT: ; implicit-def: $vgpr4
; GFX12-NEXT: ; implicit-def: $vgpr2_vgpr3
; GFX12-NEXT: s_xor_b32 exec_lo, exec_lo, s2
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB20_1
; GFX12-NEXT: ; %bb.2:
; GFX12-NEXT: s_mov_b32 exec_lo, s3
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: v_mov_b32_e32 v0, 0
; GFX12-NEXT: s_wait_loadcnt 0x1
; GFX12-NEXT: global_store_b128 v0, v[5:8], s[0:1]
diff --git a/llvm/test/CodeGen/AMDGPU/GlobalISel/regbanklegalize-amdgcn.s.buffer.load.subdword.ll b/llvm/test/CodeGen/AMDGPU/GlobalISel/regbanklegalize-amdgcn.s.buffer.load.subdword.ll
index 42779b9803ece5..01705a1b11bce0 100644
--- a/llvm/test/CodeGen/AMDGPU/GlobalISel/regbanklegalize-amdgcn.s.buffer.load.subdword.ll
+++ b/llvm/test/CodeGen/AMDGPU/GlobalISel/regbanklegalize-amdgcn.s.buffer.load.subdword.ll
@@ -1,7 +1,7 @@
; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py UTC_ARGS: --version 6
; RUN: llc -global-isel -mtriple=amdgpu12.00-amd-amdhsa -o - %s | FileCheck %s -check-prefix=GFX12
; ----------------------------------------------------------------------------
-; Case 1: Sub-dword, uniform — SGPR rsrc + SGPR offset
+; Case 1: Sub-dword, uniform — SGPR rsrc + SGPR offset
; ----------------------------------------------------------------------------
; Case 1a: i8 signed, SGPR rsrc, SGPR offset
@@ -160,9 +160,11 @@ define amdgpu_ps void @s_buffer_load_u8_vgpr_rsrc_sgpr_offset(<4 x i32> %rsrc, i
; GFX12-NEXT: ; implicit-def: $vgpr4
; GFX12-NEXT: ; implicit-def: $vgpr2_vgpr3
; GFX12-NEXT: s_xor_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB8_1
; GFX12-NEXT: ; %bb.2:
; GFX12-NEXT: s_mov_b32 exec_lo, s1
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: v_mov_b32_e32 v0, 0
; GFX12-NEXT: s_wait_loadcnt 0x0
; GFX12-NEXT: global_store_b8 v0, v1, s[8:9]
@@ -198,9 +200,11 @@ define amdgpu_ps void @s_buffer_load_i8_vgpr_rsrc_sgpr_offset(<4 x i32> %rsrc, i
; GFX12-NEXT: ; implicit-def: $vgpr4
; GFX12-NEXT: ; implicit-def: $vgpr2_vgpr3
; GFX12-NEXT: s_xor_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB9_1
; GFX12-NEXT: ; %bb.2:
; GFX12-NEXT: s_mov_b32 exec_lo, s1
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: v_mov_b32_e32 v0, 0
; GFX12-NEXT: s_wait_loadcnt 0x0
; GFX12-NEXT: global_store_b8 v0, v1, s[8:9]
@@ -236,9 +240,11 @@ define amdgpu_ps void @s_buffer_load_u16_vgpr_rsrc_sgpr_offset(<4 x i32> %rsrc,
; GFX12-NEXT: ; implicit-def: $vgpr4
; GFX12-NEXT: ; implicit-def: $vgpr2_vgpr3
; GFX12-NEXT: s_xor_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB10_1
; GFX12-NEXT: ; %bb.2:
; GFX12-NEXT: s_mov_b32 exec_lo, s1
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: v_mov_b32_e32 v0, 0
; GFX12-NEXT: s_wait_loadcnt 0x0
; GFX12-NEXT: global_store_b16 v0, v1, s[8:9]
@@ -274,9 +280,11 @@ define amdgpu_ps void @s_buffer_load_i16_vgpr_rsrc_sgpr_offset(<4 x i32> %rsrc,
; GFX12-NEXT: ; implicit-def: $vgpr4
; GFX12-NEXT: ; implicit-def: $vgpr2_vgpr3
; GFX12-NEXT: s_xor_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB11_1
; GFX12-NEXT: ; %bb.2:
; GFX12-NEXT: s_mov_b32 exec_lo, s1
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: v_mov_b32_e32 v0, 0
; GFX12-NEXT: s_wait_loadcnt 0x0
; GFX12-NEXT: global_store_b16 v0, v1, s[8:9]
@@ -306,7 +314,7 @@ define amdgpu_ps void @s_buffer_load_u8_vgpr_rsrc_vgpr_offset(<4 x i32> %rsrc, i
; GFX12-NEXT: v_cmp_eq_u64_e32 vcc_lo, s[4:5], v[0:1]
; GFX12-NEXT: v_cmp_eq_u64_e64 s2, s[6:7], v[2:3]
; GFX12-NEXT: s_and_b32 s2, vcc_lo, s2
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_and_saveexec_b32 s2, s2
; GFX12-NEXT: buffer_load_u8 v1, v4, s[4:7], null offen
; GFX12-NEXT: ; implicit-def: $vgpr0
@@ -316,6 +324,7 @@ define amdgpu_ps void @s_buffer_load_u8_vgpr_rsrc_vgpr_offset(<4 x i32> %rsrc, i
; GFX12-NEXT: s_cbranch_execnz .LBB12_1
; GFX12-NEXT: ; %bb.2:
; GFX12-NEXT: s_mov_b32 exec_lo, s3
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: v_mov_b32_e32 v0, 0
; GFX12-NEXT: s_wait_loadcnt 0x0
; GFX12-NEXT: global_store_b8 v0, v1, s[0:1]
@@ -341,7 +350,7 @@ define amdgpu_ps void @s_buffer_load_i8_vgpr_rsrc_vgpr_offset(<4 x i32> %rsrc, i
; GFX12-NEXT: v_cmp_eq_u64_e32 vcc_lo, s[4:5], v[0:1]
; GFX12-NEXT: v_cmp_eq_u64_e64 s2, s[6:7], v[2:3]
; GFX12-NEXT: s_and_b32 s2, vcc_lo, s2
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_and_saveexec_b32 s2, s2
; GFX12-NEXT: buffer_load_u8 v1, v4, s[4:7], null offen
; GFX12-NEXT: ; implicit-def: $vgpr0
@@ -351,6 +360,7 @@ define amdgpu_ps void @s_buffer_load_i8_vgpr_rsrc_vgpr_offset(<4 x i32> %rsrc, i
; GFX12-NEXT: s_cbranch_execnz .LBB13_1
; GFX12-NEXT: ; %bb.2:
; GFX12-NEXT: s_mov_b32 exec_lo, s3
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: v_mov_b32_e32 v0, 0
; GFX12-NEXT: s_wait_loadcnt 0x0
; GFX12-NEXT: global_store_b8 v0, v1, s[0:1]
@@ -376,7 +386,7 @@ define amdgpu_ps void @s_buffer_load_u16_vgpr_rsrc_vgpr_offset(<4 x i32> %rsrc,
; GFX12-NEXT: v_cmp_eq_u64_e32 vcc_lo, s[4:5], v[0:1]
; GFX12-NEXT: v_cmp_eq_u64_e64 s2, s[6:7], v[2:3]
; GFX12-NEXT: s_and_b32 s2, vcc_lo, s2
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_and_saveexec_b32 s2, s2
; GFX12-NEXT: buffer_load_u16 v1, v4, s[4:7], null offen
; GFX12-NEXT: ; implicit-def: $vgpr0
@@ -386,6 +396,7 @@ define amdgpu_ps void @s_buffer_load_u16_vgpr_rsrc_vgpr_offset(<4 x i32> %rsrc,
; GFX12-NEXT: s_cbranch_execnz .LBB14_1
; GFX12-NEXT: ; %bb.2:
; GFX12-NEXT: s_mov_b32 exec_lo, s3
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: v_mov_b32_e32 v0, 0
; GFX12-NEXT: s_wait_loadcnt 0x0
; GFX12-NEXT: global_store_b16 v0, v1, s[0:1]
@@ -411,7 +422,7 @@ define amdgpu_ps void @s_buffer_load_i16_vgpr_rsrc_vgpr_offset(<4 x i32> %rsrc,
; GFX12-NEXT: v_cmp_eq_u64_e32 vcc_lo, s[4:5], v[0:1]
; GFX12-NEXT: v_cmp_eq_u64_e64 s2, s[6:7], v[2:3]
; GFX12-NEXT: s_and_b32 s2, vcc_lo, s2
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_and_saveexec_b32 s2, s2
; GFX12-NEXT: buffer_load_u16 v1, v4, s[4:7], null offen
; GFX12-NEXT: ; implicit-def: $vgpr0
@@ -421,6 +432,7 @@ define amdgpu_ps void @s_buffer_load_i16_vgpr_rsrc_vgpr_offset(<4 x i32> %rsrc,
; GFX12-NEXT: s_cbranch_execnz .LBB15_1
; GFX12-NEXT: ; %bb.2:
; GFX12-NEXT: s_mov_b32 exec_lo, s3
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: v_mov_b32_e32 v0, 0
; GFX12-NEXT: s_wait_loadcnt 0x0
; GFX12-NEXT: global_store_b16 v0, v1, s[0:1]
diff --git a/llvm/test/CodeGen/AMDGPU/add.ll b/llvm/test/CodeGen/AMDGPU/add.ll
index 3a8d9a0a30e1e1..9c7db752b4b694 100644
--- a/llvm/test/CodeGen/AMDGPU/add.ll
+++ b/llvm/test/CodeGen/AMDGPU/add.ll
@@ -1278,6 +1278,7 @@ define amdgpu_kernel void @add64_in_branch(ptr addrspace(1) %out, ptr addrspace(
; GFX11-NEXT: s_load_b256 s[0:7], s[4:5], 0x24
; GFX11-NEXT: s_waitcnt lgkmcnt(0)
; GFX11-NEXT: s_cmp_lg_u64 s[4:5], 0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc0 .LBB9_2
; GFX11-NEXT: ; %bb.1: ; %else
; GFX11-NEXT: s_add_u32 s4, s4, s6
@@ -1292,6 +1293,7 @@ define amdgpu_kernel void @add64_in_branch(ptr addrspace(1) %out, ptr addrspace(
; GFX11-NEXT: s_and_b32 s6, s6, exec_lo
; GFX11-NEXT: s_cselect_b32 s6, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s6, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB9_5
; GFX11-NEXT: ; %bb.4: ; %if
; GFX11-NEXT: s_load_b64 s[4:5], s[2:3], 0x0
@@ -1307,6 +1309,7 @@ define amdgpu_kernel void @add64_in_branch(ptr addrspace(1) %out, ptr addrspace(
; GFX12-NEXT: s_load_b256 s[0:7], s[4:5], 0x24
; GFX12-NEXT: s_wait_kmcnt 0x0
; GFX12-NEXT: s_cmp_lg_u64 s[4:5], 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_scc0 .LBB9_2
; GFX12-NEXT: ; %bb.1: ; %else
; GFX12-NEXT: s_add_nc_u64 s[4:5], s[4:5], s[6:7]
@@ -1320,6 +1323,7 @@ define amdgpu_kernel void @add64_in_branch(ptr addrspace(1) %out, ptr addrspace(
; GFX12-NEXT: s_and_b32 s6, s6, exec_lo
; GFX12-NEXT: s_cselect_b32 s6, 1, 0
; GFX12-NEXT: s_cmp_lg_u32 s6, 1
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_scc1 .LBB9_5
; GFX12-NEXT: ; %bb.4: ; %if
; GFX12-NEXT: s_load_b64 s[4:5], s[2:3], 0x0
diff --git a/llvm/test/CodeGen/AMDGPU/add_i1.ll b/llvm/test/CodeGen/AMDGPU/add_i1.ll
index bb200468b763d3..485fb2e6da189f 100644
--- a/llvm/test/CodeGen/AMDGPU/add_i1.ll
+++ b/llvm/test/CodeGen/AMDGPU/add_i1.ll
@@ -201,9 +201,10 @@ define amdgpu_kernel void @add_i1_cf(ptr addrspace(1) %out, ptr addrspace(1) %a,
; GFX11-NEXT: v_and_b32_e32 v0, 0x3ff, v0
; GFX11-NEXT: s_mov_b32 s7, exec_lo
; GFX11-NEXT: ; implicit-def: $sgpr6
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_cmpx_lt_u32_e32 15, v0
; GFX11-NEXT: s_xor_b32 s7, exec_lo, s7
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_cbranch_execz .LBB2_2
; GFX11-NEXT: ; %bb.1: ; %else
; GFX11-NEXT: v_mov_b32_e32 v0, 0
diff --git a/llvm/test/CodeGen/AMDGPU/add_u64.ll b/llvm/test/CodeGen/AMDGPU/add_u64.ll
index 4bc7663daaba23..9056df1c237b5a 100644
--- a/llvm/test/CodeGen/AMDGPU/add_u64.ll
+++ b/llvm/test/CodeGen/AMDGPU/add_u64.ll
@@ -7,7 +7,6 @@ define amdgpu_ps <2 x float> @test_add_u64_vv(i64 %a, i64 %b) {
; GFX12-LABEL: test_add_u64_vv:
; GFX12: ; %bb.0:
; GFX12-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
-; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-NEXT: v_add_co_ci_u32_e64 v1, null, v1, v3, vcc_lo
; GFX12-NEXT: ; return to shader part epilog
;
@@ -28,7 +27,6 @@ define amdgpu_ps <2 x float> @test_add_u64_vs(i64 %a, i64 inreg %b) {
; GFX12-LABEL: test_add_u64_vs:
; GFX12: ; %bb.0:
; GFX12-NEXT: v_add_co_u32 v0, vcc_lo, v0, s0
-; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-NEXT: v_add_co_ci_u32_e64 v1, null, s1, v1, vcc_lo
; GFX12-NEXT: ; return to shader part epilog
;
@@ -49,7 +47,6 @@ define amdgpu_ps <2 x float> @test_add_u64_sv(i64 inreg %a, i64 %b) {
; GFX12-LABEL: test_add_u64_sv:
; GFX12: ; %bb.0:
; GFX12-NEXT: v_add_co_u32 v0, vcc_lo, s0, v0
-; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-NEXT: v_add_co_ci_u32_e64 v1, null, s1, v1, vcc_lo
; GFX12-NEXT: ; return to shader part epilog
;
@@ -93,7 +90,6 @@ define amdgpu_ps <2 x float> @test_add_u64_v_inline_lit(i64 %a) {
; GFX12-LABEL: test_add_u64_v_inline_lit:
; GFX12: ; %bb.0:
; GFX12-NEXT: v_add_co_u32 v0, vcc_lo, v0, 5
-; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX12-NEXT: ; return to shader part epilog
;
@@ -114,7 +110,6 @@ define amdgpu_ps <2 x float> @test_add_u64_v_small_imm(i64 %a) {
; GFX12-LABEL: test_add_u64_v_small_imm:
; GFX12: ; %bb.0:
; GFX12-NEXT: v_add_co_u32 v0, vcc_lo, 0x1f4, v0
-; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX12-NEXT: ; return to shader part epilog
;
@@ -135,7 +130,6 @@ define amdgpu_ps <2 x float> @test_add_u64_v_64bit_imm(i64 %a) {
; GFX12-LABEL: test_add_u64_v_64bit_imm:
; GFX12: ; %bb.0:
; GFX12-NEXT: v_add_co_u32 v0, vcc_lo, 0x3b9ac9ff, v0
-; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-NEXT: v_add_co_ci_u32_e64 v1, null, 1, v1, vcc_lo
; GFX12-NEXT: ; return to shader part epilog
;
diff --git a/llvm/test/CodeGen/AMDGPU/addrspacecast-gas.ll b/llvm/test/CodeGen/AMDGPU/addrspacecast-gas.ll
index e1a9d849c87cd0..a6c671ee0ce825 100644
--- a/llvm/test/CodeGen/AMDGPU/addrspacecast-gas.ll
+++ b/llvm/test/CodeGen/AMDGPU/addrspacecast-gas.ll
@@ -14,13 +14,14 @@ define amdgpu_kernel void @use_private_to_flat_addrspacecast(ptr addrspace(5) %p
; GFX1250-SDAG-NEXT: s_load_b32 s0, s[4:5], 0x24 nv
; GFX1250-SDAG-NEXT: v_mbcnt_lo_u32_b32 v0, -1, 0
; GFX1250-SDAG-NEXT: s_wait_kmcnt 0x0
-; GFX1250-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_2) | instid1(VALU_DEP_1)
+; GFX1250-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX1250-SDAG-NEXT: v_dual_mov_b32 v0, s0 :: v_dual_lshlrev_b32 v1, 20, v0
; GFX1250-SDAG-NEXT: s_cmp_lg_u32 s0, -1
; GFX1250-SDAG-NEXT: s_cselect_b32 vcc_lo, -1, 0
+; GFX1250-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1250-SDAG-NEXT: v_add_nc_u64_e32 v[0:1], src_flat_scratch_base_lo, v[0:1]
-; GFX1250-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-SDAG-NEXT: v_dual_mov_b32 v2, 0 :: v_dual_cndmask_b32 v1, 0, v1
+; GFX1250-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX1250-SDAG-NEXT: v_cndmask_b32_e32 v0, 0, v0, vcc_lo
; GFX1250-SDAG-NEXT: flat_store_b32 v[0:1], v2 scope:SCOPE_SYS
; GFX1250-SDAG-NEXT: s_wait_storecnt 0x0
@@ -84,7 +85,7 @@ define amdgpu_kernel void @use_private_to_flat_addrspacecast_nonnull(ptr addrspa
; GFX1250-GISEL-NEXT: v_lshlrev_b32_e32 v2, 20, v2
; GFX1250-GISEL-NEXT: s_wait_kmcnt 0x0
; GFX1250-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, s0, v0
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, v2, v1, vcc_lo
; GFX1250-GISEL-NEXT: v_mov_b32_e32 v2, 0
; GFX1250-GISEL-NEXT: flat_store_b32 v[0:1], v2 scope:SCOPE_SYS
@@ -107,6 +108,7 @@ define amdgpu_kernel void @use_flat_to_private_addrspacecast(ptr %ptr) {
; GFX1250-NEXT: s_wait_kmcnt 0x0
; GFX1250-NEXT: s_sub_co_i32 s2, s0, src_flat_scratch_base_lo
; GFX1250-NEXT: s_cmp_lg_u64 s[0:1], 0
+; GFX1250-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1250-NEXT: s_cselect_b32 s0, s2, -1
; GFX1250-NEXT: scratch_store_b32 off, v0, s0 scope:SCOPE_SYS
; GFX1250-NEXT: s_wait_storecnt 0x0
diff --git a/llvm/test/CodeGen/AMDGPU/amd.endpgm.ll b/llvm/test/CodeGen/AMDGPU/amd.endpgm.ll
index cc3d08642bd773..dd65f32c1e1ad4 100644
--- a/llvm/test/CodeGen/AMDGPU/amd.endpgm.ll
+++ b/llvm/test/CodeGen/AMDGPU/amd.endpgm.ll
@@ -115,6 +115,7 @@ define amdgpu_kernel void @test2(ptr %p, i32 %x) {
; GFX11-SDAG-NEXT: s_load_b32 s0, s[4:5], 0x2c
; GFX11-SDAG-NEXT: s_waitcnt lgkmcnt(0)
; GFX11-SDAG-NEXT: s_cmp_lt_i32 s0, 1
+; GFX11-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-SDAG-NEXT: s_cbranch_scc0 .LBB2_2
; GFX11-SDAG-NEXT: ; %bb.1: ; %else
; GFX11-SDAG-NEXT: s_load_b64 s[2:3], s[4:5], 0x24
@@ -131,6 +132,7 @@ define amdgpu_kernel void @test2(ptr %p, i32 %x) {
; GFX11-GISEL-NEXT: s_load_b32 s0, s[4:5], 0x2c
; GFX11-GISEL-NEXT: s_waitcnt lgkmcnt(0)
; GFX11-GISEL-NEXT: s_cmp_le_i32 s0, 0
+; GFX11-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-GISEL-NEXT: s_cbranch_scc0 .LBB2_2
; GFX11-GISEL-NEXT: ; %bb.1: ; %else
; GFX11-GISEL-NEXT: s_load_b64 s[2:3], s[4:5], 0x24
diff --git a/llvm/test/CodeGen/AMDGPU/amdgcn-call-whole-wave.ll b/llvm/test/CodeGen/AMDGPU/amdgcn-call-whole-wave.ll
index 06d92cc3f05cff..fbcdd3ef169b73 100644
--- a/llvm/test/CodeGen/AMDGPU/amdgcn-call-whole-wave.ll
+++ b/llvm/test/CodeGen/AMDGPU/amdgcn-call-whole-wave.ll
@@ -288,6 +288,7 @@ define amdgpu_gfx_whole_wave i32 @tail_call_from_whole_wave(i1 %active, i32 %x,
; DAGISEL-NEXT: scratch_store_b32 off, v246, s32 offset:568
; DAGISEL-NEXT: scratch_store_b32 off, v247, s32 offset:572
; DAGISEL-NEXT: s_mov_b32 exec_lo, -1
+; DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; DAGISEL-NEXT: v_add_nc_u32_e32 v1, 13, v0
; DAGISEL-NEXT: s_mov_b32 s37, good_callee at abs32@hi
; DAGISEL-NEXT: s_mov_b32 s36, good_callee at abs32@lo
@@ -603,6 +604,7 @@ define amdgpu_gfx_whole_wave i32 @tail_call_from_whole_wave(i1 %active, i32 %x,
; GISEL-NEXT: scratch_store_b32 off, v246, s32 offset:568
; GISEL-NEXT: scratch_store_b32 off, v247, s32 offset:572
; GISEL-NEXT: s_mov_b32 exec_lo, -1
+; GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GISEL-NEXT: v_add_nc_u32_e32 v1, 13, v0
; GISEL-NEXT: s_mov_b32 s36, good_callee at abs32@lo
; GISEL-NEXT: s_mov_b32 s37, good_callee at abs32@hi
diff --git a/llvm/test/CodeGen/AMDGPU/amdgcn.bitcast.1024bit.ll b/llvm/test/CodeGen/AMDGPU/amdgcn.bitcast.1024bit.ll
index 7be6ad4575d694..7f984f7ea66d87 100644
--- a/llvm/test/CodeGen/AMDGPU/amdgcn.bitcast.1024bit.ll
+++ b/llvm/test/CodeGen/AMDGPU/amdgcn.bitcast.1024bit.ll
@@ -168,8 +168,9 @@ define <32 x float> @bitcast_v32i32_to_v32f32(<32 x i32> %a, i32 %b) #0 {
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: s_waitcnt vmcnt(1)
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v32
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB0_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -562,6 +563,7 @@ define inreg <32 x float> @bitcast_v32i32_to_v32f32_scalar(<32 x i32> inreg %a,
; GFX11-NEXT: s_and_b32 s42, s42, exec_lo
; GFX11-NEXT: s_cselect_b32 s42, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s42, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB1_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: s_add_i32 s4, s4, 3
@@ -792,8 +794,9 @@ define <32 x i32> @bitcast_v32f32_to_v32i32(<32 x float> %a, i32 %b) #0 {
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: s_waitcnt vmcnt(1)
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v32
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB2_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -1373,6 +1376,7 @@ define inreg <32 x i32> @bitcast_v32f32_to_v32i32_scalar(<32 x float> inreg %a,
; GFX11-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB3_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_add_f32_e64 v31, s67, 1.0
@@ -1624,8 +1628,9 @@ define <16 x i64> @bitcast_v32i32_to_v16i64(<32 x i32> %a, i32 %b) #0 {
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: s_waitcnt vmcnt(1)
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v32
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB4_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -2018,6 +2023,7 @@ define inreg <16 x i64> @bitcast_v32i32_to_v16i64_scalar(<32 x i32> inreg %a, i3
; GFX11-NEXT: s_and_b32 s42, s42, exec_lo
; GFX11-NEXT: s_cselect_b32 s42, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s42, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB5_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: s_add_i32 s4, s4, 3
@@ -2248,8 +2254,9 @@ define <32 x i32> @bitcast_v16i64_to_v32i32(<16 x i64> %a, i32 %b) #0 {
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: s_waitcnt vmcnt(1)
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v32
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB6_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -2257,42 +2264,34 @@ define <32 x i32> @bitcast_v16i64_to_v32i32(<16 x i64> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0)
; GFX11-NEXT: v_add_co_ci_u32_e64 v31, null, 0, v31, vcc_lo
; GFX11-NEXT: v_add_co_u32 v28, vcc_lo, v28, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v29, null, 0, v29, vcc_lo
; GFX11-NEXT: v_add_co_u32 v26, vcc_lo, v26, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v27, null, 0, v27, vcc_lo
; GFX11-NEXT: v_add_co_u32 v24, vcc_lo, v24, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v25, null, 0, v25, vcc_lo
; GFX11-NEXT: v_add_co_u32 v22, vcc_lo, v22, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v23, null, 0, v23, vcc_lo
; GFX11-NEXT: v_add_co_u32 v20, vcc_lo, v20, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v21, null, 0, v21, vcc_lo
; GFX11-NEXT: v_add_co_u32 v18, vcc_lo, v18, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v19, null, 0, v19, vcc_lo
; GFX11-NEXT: v_add_co_u32 v16, vcc_lo, v16, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v17, null, 0, v17, vcc_lo
; GFX11-NEXT: v_add_co_u32 v14, vcc_lo, v14, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v15, null, 0, v15, vcc_lo
; GFX11-NEXT: v_add_co_u32 v12, vcc_lo, v12, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v13, null, 0, v13, vcc_lo
; GFX11-NEXT: v_add_co_u32 v10, vcc_lo, v10, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v11, null, 0, v11, vcc_lo
; GFX11-NEXT: v_add_co_u32 v8, vcc_lo, v8, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v9, null, 0, v9, vcc_lo
; GFX11-NEXT: v_add_co_u32 v6, vcc_lo, v6, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v7, null, 0, v7, vcc_lo
; GFX11-NEXT: v_add_co_u32 v4, vcc_lo, v4, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v5, null, 0, v5, vcc_lo
; GFX11-NEXT: v_add_co_u32 v2, vcc_lo, v2, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX11-NEXT: v_add_co_u32 v0, vcc_lo, v0, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-NEXT: .LBB6_2: ; %end
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -2650,6 +2649,7 @@ define inreg <32 x i32> @bitcast_v16i64_to_v32i32_scalar(<16 x i64> inreg %a, i3
; GFX11-NEXT: s_and_b32 s42, s42, exec_lo
; GFX11-NEXT: s_cselect_b32 s42, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s42, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB7_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: s_add_u32 s5, s5, 3
@@ -2880,8 +2880,9 @@ define <16 x double> @bitcast_v32i32_to_v16f64(<32 x i32> %a, i32 %b) #0 {
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: s_waitcnt vmcnt(1)
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v32
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB8_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -3274,6 +3275,7 @@ define inreg <16 x double> @bitcast_v32i32_to_v16f64_scalar(<32 x i32> inreg %a,
; GFX11-NEXT: s_and_b32 s42, s42, exec_lo
; GFX11-NEXT: s_cselect_b32 s42, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s42, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB9_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: s_add_i32 s4, s4, 3
@@ -3456,8 +3458,9 @@ define <32 x i32> @bitcast_v16f64_to_v32i32(<16 x double> %a, i32 %b) #0 {
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: s_waitcnt vmcnt(1)
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v32
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB10_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -3989,6 +3992,7 @@ define inreg <32 x i32> @bitcast_v16f64_to_v32i32_scalar(<16 x double> inreg %a,
; GFX11-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB11_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_add_f64 v[30:31], s[66:67], 1.0
@@ -7417,8 +7421,9 @@ define <128 x i8> @bitcast_v32i32_to_v128i8(<32 x i32> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v75, 8, v1
; GFX11-TRUE16-NEXT: .LBB12_4: ; %end
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_perm_b32 v39, v74, v66, 0xc0c0004
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX11-TRUE16-NEXT: v_perm_b32 v1, v1, v75, 0xc0c0004
; GFX11-TRUE16-NEXT: v_perm_b32 v2, v2, v73, 0xc0c0004
; GFX11-TRUE16-NEXT: v_perm_b32 v55, v72, v63, 0xc0c0004
@@ -7932,8 +7937,9 @@ define <128 x i8> @bitcast_v32i32_to_v128i8(<32 x i32> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v75, 8, v1
; GFX11-FAKE16-NEXT: .LBB12_4: ; %end
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_perm_b32 v39, v74, v66, 0xc0c0004
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX11-FAKE16-NEXT: v_perm_b32 v1, v1, v75, 0xc0c0004
; GFX11-FAKE16-NEXT: v_perm_b32 v2, v2, v73, 0xc0c0004
; GFX11-FAKE16-NEXT: v_perm_b32 v55, v72, v63, 0xc0c0004
@@ -20802,6 +20808,7 @@ define inreg <32 x i32> @bitcast_v128i8_to_v32i32_scalar(<128 x i8> inreg %a, i3
; GFX11-TRUE16-NEXT: s_and_b32 s75, s75, exec_lo
; GFX11-TRUE16-NEXT: s_cselect_b32 s75, 1, 0
; GFX11-TRUE16-NEXT: s_cmp_lg_u32 s75, 1
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_cbranch_scc1 .LBB15_5
; GFX11-TRUE16-NEXT: ; %bb.4: ; %cmp.true
; GFX11-TRUE16-NEXT: v_mov_b32_e32 v10, 0xc0c0004
@@ -21381,6 +21388,7 @@ define inreg <32 x i32> @bitcast_v128i8_to_v32i32_scalar(<128 x i8> inreg %a, i3
; GFX11-FAKE16-NEXT: s_and_b32 s75, s75, exec_lo
; GFX11-FAKE16-NEXT: s_cselect_b32 s75, 1, 0
; GFX11-FAKE16-NEXT: s_cmp_lg_u32 s75, 1
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_cbranch_scc1 .LBB15_5
; GFX11-FAKE16-NEXT: ; %bb.4: ; %cmp.true
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v10, 0xc0c0004
@@ -22436,8 +22444,9 @@ define <64 x bfloat> @bitcast_v32i32_to_v64bf16(<32 x i32> %a, i32 %b) #0 {
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: s_waitcnt vmcnt(1)
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v32
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB16_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -23239,6 +23248,7 @@ define inreg <64 x bfloat> @bitcast_v32i32_to_v64bf16_scalar(<32 x i32> inreg %a
; GFX11-NEXT: s_and_b32 s42, s42, exec_lo
; GFX11-NEXT: s_cselect_b32 s42, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s42, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB17_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: s_add_i32 s27, s27, 3
@@ -25134,8 +25144,9 @@ define <32 x i32> @bitcast_v64bf16_to_v32i32(<64 x bfloat> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-TRUE16-NEXT: s_waitcnt vmcnt(1)
; GFX11-TRUE16-NEXT: v_cmpx_ne_u32_e32 0, v32
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-TRUE16-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-TRUE16-NEXT: s_cbranch_execz .LBB18_2
; GFX11-TRUE16-NEXT: ; %bb.1: ; %cmp.true
@@ -25703,8 +25714,9 @@ define <32 x i32> @bitcast_v64bf16_to_v32i32(<64 x bfloat> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-FAKE16-NEXT: s_waitcnt vmcnt(1)
; GFX11-FAKE16-NEXT: v_cmpx_ne_u32_e32 0, v32
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-FAKE16-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-FAKE16-NEXT: s_cbranch_execz .LBB18_2
; GFX11-FAKE16-NEXT: ; %bb.1: ; %cmp.true
@@ -28594,6 +28606,7 @@ define inreg <32 x i32> @bitcast_v64bf16_to_v32i32_scalar(<64 x bfloat> inreg %a
; GFX11-TRUE16-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-TRUE16-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-TRUE16-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_cbranch_scc1 .LBB19_4
; GFX11-TRUE16-NEXT: ; %bb.3: ; %cmp.true
; GFX11-TRUE16-NEXT: s_and_b32 s0, s51, 0xffff0000
@@ -29323,6 +29336,7 @@ define inreg <32 x i32> @bitcast_v64bf16_to_v32i32_scalar(<64 x bfloat> inreg %a
; GFX11-FAKE16-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-FAKE16-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-FAKE16-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_cbranch_scc1 .LBB19_4
; GFX11-FAKE16-NEXT: ; %bb.3: ; %cmp.true
; GFX11-FAKE16-NEXT: s_and_b32 s1, s51, 0xffff0000
@@ -30438,8 +30452,9 @@ define <64 x half> @bitcast_v32i32_to_v64f16(<32 x i32> %a, i32 %b) #0 {
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: s_waitcnt vmcnt(1)
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v32
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB20_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -31076,6 +31091,7 @@ define inreg <64 x half> @bitcast_v32i32_to_v64f16_scalar(<32 x i32> inreg %a, i
; GFX11-NEXT: s_and_b32 s42, s42, exec_lo
; GFX11-NEXT: s_cselect_b32 s42, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s42, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB21_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: s_add_i32 s27, s27, 3
@@ -32045,8 +32061,9 @@ define <32 x i32> @bitcast_v64f16_to_v32i32(<64 x half> %a, i32 %b) #0 {
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: s_waitcnt vmcnt(1)
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v32
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB22_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -33159,6 +33176,7 @@ define inreg <32 x i32> @bitcast_v64f16_to_v32i32_scalar(<64 x half> inreg %a, i
; GFX11-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB23_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_pk_add_f16 v15, 0x200, s51 op_sel_hi:[0,1]
@@ -33654,8 +33672,9 @@ define <64 x i16> @bitcast_v32i32_to_v64i16(<32 x i32> %a, i32 %b) #0 {
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: s_waitcnt vmcnt(1)
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v32
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB24_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -34292,6 +34311,7 @@ define inreg <64 x i16> @bitcast_v32i32_to_v64i16_scalar(<32 x i32> inreg %a, i3
; GFX11-NEXT: s_and_b32 s42, s42, exec_lo
; GFX11-NEXT: s_cselect_b32 s42, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s42, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB25_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: s_add_i32 s27, s27, 3
@@ -35091,8 +35111,9 @@ define <32 x i32> @bitcast_v64i16_to_v32i32(<64 x i16> %a, i32 %b) #0 {
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: s_waitcnt vmcnt(1)
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v32
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB26_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -36062,6 +36083,7 @@ define inreg <32 x i32> @bitcast_v64i16_to_v32i32_scalar(<64 x i16> inreg %a, i3
; GFX11-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB27_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_pk_add_u16 v15, s51, 3 op_sel_hi:[1,0]
@@ -36313,8 +36335,9 @@ define <16 x i64> @bitcast_v32f32_to_v16i64(<32 x float> %a, i32 %b) #0 {
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: s_waitcnt vmcnt(1)
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v32
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB28_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -36894,6 +36917,7 @@ define inreg <16 x i64> @bitcast_v32f32_to_v16i64_scalar(<32 x float> inreg %a,
; GFX11-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB29_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_add_f32_e64 v31, s67, 1.0
@@ -37145,8 +37169,9 @@ define <32 x float> @bitcast_v16i64_to_v32f32(<16 x i64> %a, i32 %b) #0 {
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: s_waitcnt vmcnt(1)
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v32
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB30_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -37154,42 +37179,34 @@ define <32 x float> @bitcast_v16i64_to_v32f32(<16 x i64> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0)
; GFX11-NEXT: v_add_co_ci_u32_e64 v31, null, 0, v31, vcc_lo
; GFX11-NEXT: v_add_co_u32 v28, vcc_lo, v28, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v29, null, 0, v29, vcc_lo
; GFX11-NEXT: v_add_co_u32 v26, vcc_lo, v26, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v27, null, 0, v27, vcc_lo
; GFX11-NEXT: v_add_co_u32 v24, vcc_lo, v24, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v25, null, 0, v25, vcc_lo
; GFX11-NEXT: v_add_co_u32 v22, vcc_lo, v22, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v23, null, 0, v23, vcc_lo
; GFX11-NEXT: v_add_co_u32 v20, vcc_lo, v20, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v21, null, 0, v21, vcc_lo
; GFX11-NEXT: v_add_co_u32 v18, vcc_lo, v18, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v19, null, 0, v19, vcc_lo
; GFX11-NEXT: v_add_co_u32 v16, vcc_lo, v16, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v17, null, 0, v17, vcc_lo
; GFX11-NEXT: v_add_co_u32 v14, vcc_lo, v14, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v15, null, 0, v15, vcc_lo
; GFX11-NEXT: v_add_co_u32 v12, vcc_lo, v12, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v13, null, 0, v13, vcc_lo
; GFX11-NEXT: v_add_co_u32 v10, vcc_lo, v10, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v11, null, 0, v11, vcc_lo
; GFX11-NEXT: v_add_co_u32 v8, vcc_lo, v8, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v9, null, 0, v9, vcc_lo
; GFX11-NEXT: v_add_co_u32 v6, vcc_lo, v6, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v7, null, 0, v7, vcc_lo
; GFX11-NEXT: v_add_co_u32 v4, vcc_lo, v4, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v5, null, 0, v5, vcc_lo
; GFX11-NEXT: v_add_co_u32 v2, vcc_lo, v2, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX11-NEXT: v_add_co_u32 v0, vcc_lo, v0, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-NEXT: .LBB30_2: ; %end
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -37547,6 +37564,7 @@ define inreg <32 x float> @bitcast_v16i64_to_v32f32_scalar(<16 x i64> inreg %a,
; GFX11-NEXT: s_and_b32 s42, s42, exec_lo
; GFX11-NEXT: s_cselect_b32 s42, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s42, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB31_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: s_add_u32 s5, s5, 3
@@ -37777,8 +37795,9 @@ define <16 x double> @bitcast_v32f32_to_v16f64(<32 x float> %a, i32 %b) #0 {
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: s_waitcnt vmcnt(1)
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v32
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB32_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -38358,6 +38377,7 @@ define inreg <16 x double> @bitcast_v32f32_to_v16f64_scalar(<32 x float> inreg %
; GFX11-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB33_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_add_f32_e64 v31, s67, 1.0
@@ -38561,8 +38581,9 @@ define <32 x float> @bitcast_v16f64_to_v32f32(<16 x double> %a, i32 %b) #0 {
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: s_waitcnt vmcnt(1)
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v32
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB34_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -39094,6 +39115,7 @@ define inreg <32 x float> @bitcast_v16f64_to_v32f32_scalar(<16 x double> inreg %
; GFX11-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB35_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_add_f64 v[30:31], s[66:67], 1.0
@@ -42505,8 +42527,9 @@ define <128 x i8> @bitcast_v32f32_to_v128i8(<32 x float> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v75, 8, v1
; GFX11-TRUE16-NEXT: .LBB36_4: ; %end
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_perm_b32 v39, v74, v66, 0xc0c0004
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX11-TRUE16-NEXT: v_perm_b32 v1, v1, v75, 0xc0c0004
; GFX11-TRUE16-NEXT: v_perm_b32 v2, v2, v73, 0xc0c0004
; GFX11-TRUE16-NEXT: v_perm_b32 v55, v72, v63, 0xc0c0004
@@ -43003,8 +43026,9 @@ define <128 x i8> @bitcast_v32f32_to_v128i8(<32 x float> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v75, 8, v1
; GFX11-FAKE16-NEXT: .LBB36_4: ; %end
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_perm_b32 v39, v74, v66, 0xc0c0004
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX11-FAKE16-NEXT: v_perm_b32 v1, v1, v75, 0xc0c0004
; GFX11-FAKE16-NEXT: v_perm_b32 v2, v2, v73, 0xc0c0004
; GFX11-FAKE16-NEXT: v_perm_b32 v55, v72, v63, 0xc0c0004
@@ -47266,6 +47290,7 @@ define inreg <128 x i8> @bitcast_v32f32_to_v128i8_scalar(<32 x float> inreg %a,
; GFX11-NEXT: s_and_b32 s43, s30, exec_lo
; GFX11-NEXT: s_cselect_b32 s43, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s43, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB37_5
; GFX11-NEXT: ; %bb.4: ; %cmp.true
; GFX11-NEXT: v_add_f32_e64 v12, s15, 1.0
@@ -56812,6 +56837,7 @@ define inreg <32 x float> @bitcast_v128i8_to_v32f32_scalar(<128 x i8> inreg %a,
; GFX11-TRUE16-NEXT: s_and_b32 s75, s75, exec_lo
; GFX11-TRUE16-NEXT: s_cselect_b32 s75, 1, 0
; GFX11-TRUE16-NEXT: s_cmp_lg_u32 s75, 1
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_cbranch_scc1 .LBB39_5
; GFX11-TRUE16-NEXT: ; %bb.4: ; %cmp.true
; GFX11-TRUE16-NEXT: v_mov_b32_e32 v10, 0xc0c0004
@@ -57391,6 +57417,7 @@ define inreg <32 x float> @bitcast_v128i8_to_v32f32_scalar(<128 x i8> inreg %a,
; GFX11-FAKE16-NEXT: s_and_b32 s75, s75, exec_lo
; GFX11-FAKE16-NEXT: s_cselect_b32 s75, 1, 0
; GFX11-FAKE16-NEXT: s_cmp_lg_u32 s75, 1
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_cbranch_scc1 .LBB39_5
; GFX11-FAKE16-NEXT: ; %bb.4: ; %cmp.true
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v10, 0xc0c0004
@@ -58446,8 +58473,9 @@ define <64 x bfloat> @bitcast_v32f32_to_v64bf16(<32 x float> %a, i32 %b) #0 {
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: s_waitcnt vmcnt(1)
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v32
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB40_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -59433,6 +59461,7 @@ define inreg <64 x bfloat> @bitcast_v32f32_to_v64bf16_scalar(<32 x float> inreg
; GFX11-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB41_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_add_f32_e64 v15, s27, 1.0
@@ -61342,8 +61371,9 @@ define <32 x float> @bitcast_v64bf16_to_v32f32(<64 x bfloat> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-TRUE16-NEXT: s_waitcnt vmcnt(1)
; GFX11-TRUE16-NEXT: v_cmpx_ne_u32_e32 0, v32
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-TRUE16-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-TRUE16-NEXT: s_cbranch_execz .LBB42_2
; GFX11-TRUE16-NEXT: ; %bb.1: ; %cmp.true
@@ -61911,8 +61941,9 @@ define <32 x float> @bitcast_v64bf16_to_v32f32(<64 x bfloat> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-FAKE16-NEXT: s_waitcnt vmcnt(1)
; GFX11-FAKE16-NEXT: v_cmpx_ne_u32_e32 0, v32
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-FAKE16-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-FAKE16-NEXT: s_cbranch_execz .LBB42_2
; GFX11-FAKE16-NEXT: ; %bb.1: ; %cmp.true
@@ -64802,6 +64833,7 @@ define inreg <32 x float> @bitcast_v64bf16_to_v32f32_scalar(<64 x bfloat> inreg
; GFX11-TRUE16-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-TRUE16-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-TRUE16-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_cbranch_scc1 .LBB43_4
; GFX11-TRUE16-NEXT: ; %bb.3: ; %cmp.true
; GFX11-TRUE16-NEXT: s_and_b32 s0, s51, 0xffff0000
@@ -65531,6 +65563,7 @@ define inreg <32 x float> @bitcast_v64bf16_to_v32f32_scalar(<64 x bfloat> inreg
; GFX11-FAKE16-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-FAKE16-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-FAKE16-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_cbranch_scc1 .LBB43_4
; GFX11-FAKE16-NEXT: ; %bb.3: ; %cmp.true
; GFX11-FAKE16-NEXT: s_and_b32 s1, s51, 0xffff0000
@@ -66646,8 +66679,9 @@ define <64 x half> @bitcast_v32f32_to_v64f16(<32 x float> %a, i32 %b) #0 {
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: s_waitcnt vmcnt(1)
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v32
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB44_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -67417,6 +67451,7 @@ define inreg <64 x half> @bitcast_v32f32_to_v64f16_scalar(<32 x float> inreg %a,
; GFX11-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB45_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_add_f32_e64 v15, s27, 1.0
@@ -68400,8 +68435,9 @@ define <32 x float> @bitcast_v64f16_to_v32f32(<64 x half> %a, i32 %b) #0 {
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: s_waitcnt vmcnt(1)
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v32
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB46_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -69514,6 +69550,7 @@ define inreg <32 x float> @bitcast_v64f16_to_v32f32_scalar(<64 x half> inreg %a,
; GFX11-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB47_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_pk_add_f16 v15, 0x200, s51 op_sel_hi:[0,1]
@@ -70009,8 +70046,9 @@ define <64 x i16> @bitcast_v32f32_to_v64i16(<32 x float> %a, i32 %b) #0 {
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: s_waitcnt vmcnt(1)
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v32
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB48_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -70780,6 +70818,7 @@ define inreg <64 x i16> @bitcast_v32f32_to_v64i16_scalar(<32 x float> inreg %a,
; GFX11-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB49_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_add_f32_e64 v15, s27, 1.0
@@ -71593,8 +71632,9 @@ define <32 x float> @bitcast_v64i16_to_v32f32(<64 x i16> %a, i32 %b) #0 {
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: s_waitcnt vmcnt(1)
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v32
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB50_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -72564,6 +72604,7 @@ define inreg <32 x float> @bitcast_v64i16_to_v32f32_scalar(<64 x i16> inreg %a,
; GFX11-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB51_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_pk_add_u16 v15, s51, 3 op_sel_hi:[1,0]
@@ -72815,48 +72856,41 @@ define <16 x double> @bitcast_v16i64_to_v16f64(<16 x i64> %a, i32 %b) #0 {
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: s_waitcnt vmcnt(1)
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v32
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB52_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_add_co_u32 v0, vcc_lo, v0, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-NEXT: v_add_co_u32 v2, vcc_lo, v2, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX11-NEXT: v_add_co_u32 v4, vcc_lo, v4, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v5, null, 0, v5, vcc_lo
; GFX11-NEXT: v_add_co_u32 v6, vcc_lo, v6, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v7, null, 0, v7, vcc_lo
; GFX11-NEXT: v_add_co_u32 v8, vcc_lo, v8, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v9, null, 0, v9, vcc_lo
; GFX11-NEXT: v_add_co_u32 v10, vcc_lo, v10, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v11, null, 0, v11, vcc_lo
; GFX11-NEXT: v_add_co_u32 v12, vcc_lo, v12, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v13, null, 0, v13, vcc_lo
; GFX11-NEXT: v_add_co_u32 v14, vcc_lo, v14, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v15, null, 0, v15, vcc_lo
; GFX11-NEXT: v_add_co_u32 v16, vcc_lo, v16, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v17, null, 0, v17, vcc_lo
; GFX11-NEXT: v_add_co_u32 v18, vcc_lo, v18, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v19, null, 0, v19, vcc_lo
; GFX11-NEXT: v_add_co_u32 v20, vcc_lo, v20, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v21, null, 0, v21, vcc_lo
; GFX11-NEXT: v_add_co_u32 v22, vcc_lo, v22, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v23, null, 0, v23, vcc_lo
; GFX11-NEXT: v_add_co_u32 v24, vcc_lo, v24, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v25, null, 0, v25, vcc_lo
; GFX11-NEXT: v_add_co_u32 v26, vcc_lo, v26, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v27, null, 0, v27, vcc_lo
; GFX11-NEXT: v_add_co_u32 v28, vcc_lo, v28, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v29, null, 0, v29, vcc_lo
; GFX11-NEXT: v_add_co_u32 v30, vcc_lo, v30, 3
; GFX11-NEXT: s_waitcnt vmcnt(0)
@@ -73217,6 +73251,7 @@ define inreg <16 x double> @bitcast_v16i64_to_v16f64_scalar(<16 x i64> inreg %a,
; GFX11-NEXT: s_and_b32 s42, s42, exec_lo
; GFX11-NEXT: s_cselect_b32 s42, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s42, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB53_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: s_add_u32 s0, s0, 3
@@ -73398,8 +73433,9 @@ define <16 x i64> @bitcast_v16f64_to_v16i64(<16 x double> %a, i32 %b) #0 {
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: s_waitcnt vmcnt(1)
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v32
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB54_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -73931,6 +73967,7 @@ define inreg <16 x i64> @bitcast_v16f64_to_v16i64_scalar(<16 x double> inreg %a,
; GFX11-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB55_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_add_f64 v[0:1], s[36:37], 1.0
@@ -77228,43 +77265,35 @@ define <128 x i8> @bitcast_v16i64_to_v128i8(<16 x i64> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: s_cbranch_execz .LBB56_4
; GFX11-TRUE16-NEXT: ; %bb.3: ; %cmp.true
; GFX11-TRUE16-NEXT: v_add_co_u32 v1, vcc_lo, v1, 3
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v2, null, 0, v2, vcc_lo
; GFX11-TRUE16-NEXT: v_add_co_u32 v3, vcc_lo, v3, 3
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v4, null, 0, v4, vcc_lo
; GFX11-TRUE16-NEXT: v_add_co_u32 v5, vcc_lo, v5, 3
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v6, null, 0, v6, vcc_lo
; GFX11-TRUE16-NEXT: v_add_co_u32 v7, vcc_lo, v7, 3
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v8, null, 0, v8, vcc_lo
; GFX11-TRUE16-NEXT: v_add_co_u32 v9, vcc_lo, v9, 3
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v10, null, 0, v10, vcc_lo
; GFX11-TRUE16-NEXT: v_add_co_u32 v11, vcc_lo, v11, 3
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v12, null, 0, v12, vcc_lo
; GFX11-TRUE16-NEXT: v_add_co_u32 v13, vcc_lo, v13, 3
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v14, null, 0, v14, vcc_lo
; GFX11-TRUE16-NEXT: v_add_co_u32 v15, vcc_lo, v15, 3
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v16, null, 0, v16, vcc_lo
; GFX11-TRUE16-NEXT: v_add_co_u32 v17, vcc_lo, v17, 3
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v18, null, 0, v18, vcc_lo
; GFX11-TRUE16-NEXT: v_add_co_u32 v19, vcc_lo, v19, 3
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v20, null, 0, v20, vcc_lo
; GFX11-TRUE16-NEXT: v_add_co_u32 v21, vcc_lo, v21, 3
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v22, null, 0, v22, vcc_lo
; GFX11-TRUE16-NEXT: v_add_co_u32 v25, vcc_lo, v25, 3
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v26, null, 0, v26, vcc_lo
; GFX11-TRUE16-NEXT: v_add_co_u32 v27, vcc_lo, v27, 3
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v28, null, 0, v28, vcc_lo
; GFX11-TRUE16-NEXT: v_add_co_u32 v29, vcc_lo, v29, 3
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v30, null, 0, v30, vcc_lo
; GFX11-TRUE16-NEXT: s_waitcnt vmcnt(0)
; GFX11-TRUE16-NEXT: v_add_co_u32 v31, vcc_lo, v31, 3
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v32, null, 0, v32, vcc_lo
; GFX11-TRUE16-NEXT: v_add_co_u32 v23, vcc_lo, v23, 3
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v24, null, 0, v24, vcc_lo
@@ -77367,8 +77396,9 @@ define <128 x i8> @bitcast_v16i64_to_v128i8(<16 x i64> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v75, 8, v1
; GFX11-TRUE16-NEXT: .LBB56_4: ; %end
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_perm_b32 v39, v74, v66, 0xc0c0004
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX11-TRUE16-NEXT: v_perm_b32 v1, v1, v75, 0xc0c0004
; GFX11-TRUE16-NEXT: v_perm_b32 v2, v2, v73, 0xc0c0004
; GFX11-TRUE16-NEXT: v_perm_b32 v55, v72, v63, 0xc0c0004
@@ -77751,43 +77781,35 @@ define <128 x i8> @bitcast_v16i64_to_v128i8(<16 x i64> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: s_cbranch_execz .LBB56_4
; GFX11-FAKE16-NEXT: ; %bb.3: ; %cmp.true
; GFX11-FAKE16-NEXT: v_add_co_u32 v1, vcc_lo, v1, 3
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v2, null, 0, v2, vcc_lo
; GFX11-FAKE16-NEXT: v_add_co_u32 v3, vcc_lo, v3, 3
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v4, null, 0, v4, vcc_lo
; GFX11-FAKE16-NEXT: v_add_co_u32 v5, vcc_lo, v5, 3
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v6, null, 0, v6, vcc_lo
; GFX11-FAKE16-NEXT: v_add_co_u32 v7, vcc_lo, v7, 3
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v8, null, 0, v8, vcc_lo
; GFX11-FAKE16-NEXT: v_add_co_u32 v9, vcc_lo, v9, 3
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v10, null, 0, v10, vcc_lo
; GFX11-FAKE16-NEXT: v_add_co_u32 v11, vcc_lo, v11, 3
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v12, null, 0, v12, vcc_lo
; GFX11-FAKE16-NEXT: v_add_co_u32 v13, vcc_lo, v13, 3
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v14, null, 0, v14, vcc_lo
; GFX11-FAKE16-NEXT: v_add_co_u32 v15, vcc_lo, v15, 3
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v16, null, 0, v16, vcc_lo
; GFX11-FAKE16-NEXT: v_add_co_u32 v17, vcc_lo, v17, 3
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v18, null, 0, v18, vcc_lo
; GFX11-FAKE16-NEXT: v_add_co_u32 v19, vcc_lo, v19, 3
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v20, null, 0, v20, vcc_lo
; GFX11-FAKE16-NEXT: v_add_co_u32 v21, vcc_lo, v21, 3
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v22, null, 0, v22, vcc_lo
; GFX11-FAKE16-NEXT: v_add_co_u32 v25, vcc_lo, v25, 3
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v26, null, 0, v26, vcc_lo
; GFX11-FAKE16-NEXT: v_add_co_u32 v27, vcc_lo, v27, 3
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v28, null, 0, v28, vcc_lo
; GFX11-FAKE16-NEXT: v_add_co_u32 v29, vcc_lo, v29, 3
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v30, null, 0, v30, vcc_lo
; GFX11-FAKE16-NEXT: s_waitcnt vmcnt(0)
; GFX11-FAKE16-NEXT: v_add_co_u32 v31, vcc_lo, v31, 3
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v32, null, 0, v32, vcc_lo
; GFX11-FAKE16-NEXT: v_add_co_u32 v23, vcc_lo, v23, 3
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v24, null, 0, v24, vcc_lo
@@ -77890,8 +77912,9 @@ define <128 x i8> @bitcast_v16i64_to_v128i8(<16 x i64> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v75, 8, v1
; GFX11-FAKE16-NEXT: .LBB56_4: ; %end
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_perm_b32 v39, v74, v66, 0xc0c0004
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX11-FAKE16-NEXT: v_perm_b32 v1, v1, v75, 0xc0c0004
; GFX11-FAKE16-NEXT: v_perm_b32 v2, v2, v73, 0xc0c0004
; GFX11-FAKE16-NEXT: v_perm_b32 v55, v72, v63, 0xc0c0004
@@ -90827,6 +90850,7 @@ define inreg <16 x i64> @bitcast_v128i8_to_v16i64_scalar(<128 x i8> inreg %a, i3
; GFX11-TRUE16-NEXT: s_and_b32 s75, s75, exec_lo
; GFX11-TRUE16-NEXT: s_cselect_b32 s75, 1, 0
; GFX11-TRUE16-NEXT: s_cmp_lg_u32 s75, 1
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_cbranch_scc1 .LBB59_5
; GFX11-TRUE16-NEXT: ; %bb.4: ; %cmp.true
; GFX11-TRUE16-NEXT: v_mov_b32_e32 v10, 0xc0c0004
@@ -91406,6 +91430,7 @@ define inreg <16 x i64> @bitcast_v128i8_to_v16i64_scalar(<128 x i8> inreg %a, i3
; GFX11-FAKE16-NEXT: s_and_b32 s75, s75, exec_lo
; GFX11-FAKE16-NEXT: s_cselect_b32 s75, 1, 0
; GFX11-FAKE16-NEXT: s_cmp_lg_u32 s75, 1
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_cbranch_scc1 .LBB59_5
; GFX11-FAKE16-NEXT: ; %bb.4: ; %cmp.true
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v10, 0xc0c0004
@@ -92461,28 +92486,25 @@ define <64 x bfloat> @bitcast_v16i64_to_v64bf16(<16 x i64> %a, i32 %b) #0 {
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: s_waitcnt vmcnt(1)
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v32
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB60_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_add_co_u32 v14, vcc_lo, v14, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v15, null, 0, v15, vcc_lo
; GFX11-NEXT: v_add_co_u32 v12, vcc_lo, v12, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v13, null, 0, v13, vcc_lo
; GFX11-NEXT: v_add_co_u32 v10, vcc_lo, v10, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v11, null, 0, v11, vcc_lo
; GFX11-NEXT: v_add_co_u32 v8, vcc_lo, v8, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v9, null, 0, v9, vcc_lo
; GFX11-NEXT: v_add_co_u32 v6, vcc_lo, v6, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v7, null, 0, v7, vcc_lo
; GFX11-NEXT: v_add_co_u32 v4, vcc_lo, v4, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v5, null, 0, v5, vcc_lo
; GFX11-NEXT: v_add_co_u32 v2, vcc_lo, v2, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX11-NEXT: v_add_co_u32 v0, vcc_lo, v0, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
@@ -92490,22 +92512,18 @@ define <64 x bfloat> @bitcast_v16i64_to_v64bf16(<16 x i64> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0)
; GFX11-NEXT: v_add_co_ci_u32_e64 v31, null, 0, v31, vcc_lo
; GFX11-NEXT: v_add_co_u32 v28, vcc_lo, v28, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v29, null, 0, v29, vcc_lo
; GFX11-NEXT: v_add_co_u32 v26, vcc_lo, v26, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v27, null, 0, v27, vcc_lo
; GFX11-NEXT: v_add_co_u32 v24, vcc_lo, v24, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v25, null, 0, v25, vcc_lo
; GFX11-NEXT: v_add_co_u32 v22, vcc_lo, v22, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v23, null, 0, v23, vcc_lo
; GFX11-NEXT: v_add_co_u32 v20, vcc_lo, v20, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v21, null, 0, v21, vcc_lo
; GFX11-NEXT: v_add_co_u32 v18, vcc_lo, v18, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v19, null, 0, v19, vcc_lo
; GFX11-NEXT: v_add_co_u32 v16, vcc_lo, v16, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v17, null, 0, v17, vcc_lo
; GFX11-NEXT: .LBB60_2: ; %end
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -93245,6 +93263,7 @@ define inreg <64 x bfloat> @bitcast_v16i64_to_v64bf16_scalar(<16 x i64> inreg %a
; GFX11-NEXT: s_and_b32 s42, s42, exec_lo
; GFX11-NEXT: s_cselect_b32 s42, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s42, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB61_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: s_add_u32 s26, s26, 3
@@ -95140,8 +95159,9 @@ define <16 x i64> @bitcast_v64bf16_to_v16i64(<64 x bfloat> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-TRUE16-NEXT: s_waitcnt vmcnt(1)
; GFX11-TRUE16-NEXT: v_cmpx_ne_u32_e32 0, v32
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-TRUE16-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-TRUE16-NEXT: s_cbranch_execz .LBB62_2
; GFX11-TRUE16-NEXT: ; %bb.1: ; %cmp.true
@@ -95709,8 +95729,9 @@ define <16 x i64> @bitcast_v64bf16_to_v16i64(<64 x bfloat> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-FAKE16-NEXT: s_waitcnt vmcnt(1)
; GFX11-FAKE16-NEXT: v_cmpx_ne_u32_e32 0, v32
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-FAKE16-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-FAKE16-NEXT: s_cbranch_execz .LBB62_2
; GFX11-FAKE16-NEXT: ; %bb.1: ; %cmp.true
@@ -98600,6 +98621,7 @@ define inreg <16 x i64> @bitcast_v64bf16_to_v16i64_scalar(<64 x bfloat> inreg %a
; GFX11-TRUE16-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-TRUE16-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-TRUE16-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_cbranch_scc1 .LBB63_4
; GFX11-TRUE16-NEXT: ; %bb.3: ; %cmp.true
; GFX11-TRUE16-NEXT: s_and_b32 s0, s51, 0xffff0000
@@ -99329,6 +99351,7 @@ define inreg <16 x i64> @bitcast_v64bf16_to_v16i64_scalar(<64 x bfloat> inreg %a
; GFX11-FAKE16-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-FAKE16-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-FAKE16-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_cbranch_scc1 .LBB63_4
; GFX11-FAKE16-NEXT: ; %bb.3: ; %cmp.true
; GFX11-FAKE16-NEXT: s_and_b32 s1, s51, 0xffff0000
@@ -100444,28 +100467,25 @@ define <64 x half> @bitcast_v16i64_to_v64f16(<16 x i64> %a, i32 %b) #0 {
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: s_waitcnt vmcnt(1)
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v32
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB64_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_add_co_u32 v14, vcc_lo, v14, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v15, null, 0, v15, vcc_lo
; GFX11-NEXT: v_add_co_u32 v12, vcc_lo, v12, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v13, null, 0, v13, vcc_lo
; GFX11-NEXT: v_add_co_u32 v10, vcc_lo, v10, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v11, null, 0, v11, vcc_lo
; GFX11-NEXT: v_add_co_u32 v8, vcc_lo, v8, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v9, null, 0, v9, vcc_lo
; GFX11-NEXT: v_add_co_u32 v6, vcc_lo, v6, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v7, null, 0, v7, vcc_lo
; GFX11-NEXT: v_add_co_u32 v4, vcc_lo, v4, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v5, null, 0, v5, vcc_lo
; GFX11-NEXT: v_add_co_u32 v2, vcc_lo, v2, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX11-NEXT: v_add_co_u32 v0, vcc_lo, v0, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
@@ -100473,22 +100493,18 @@ define <64 x half> @bitcast_v16i64_to_v64f16(<16 x i64> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0)
; GFX11-NEXT: v_add_co_ci_u32_e64 v31, null, 0, v31, vcc_lo
; GFX11-NEXT: v_add_co_u32 v28, vcc_lo, v28, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v29, null, 0, v29, vcc_lo
; GFX11-NEXT: v_add_co_u32 v26, vcc_lo, v26, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v27, null, 0, v27, vcc_lo
; GFX11-NEXT: v_add_co_u32 v24, vcc_lo, v24, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v25, null, 0, v25, vcc_lo
; GFX11-NEXT: v_add_co_u32 v22, vcc_lo, v22, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v23, null, 0, v23, vcc_lo
; GFX11-NEXT: v_add_co_u32 v20, vcc_lo, v20, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v21, null, 0, v21, vcc_lo
; GFX11-NEXT: v_add_co_u32 v18, vcc_lo, v18, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v19, null, 0, v19, vcc_lo
; GFX11-NEXT: v_add_co_u32 v16, vcc_lo, v16, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v17, null, 0, v17, vcc_lo
; GFX11-NEXT: .LBB64_2: ; %end
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -101090,6 +101106,7 @@ define inreg <64 x half> @bitcast_v16i64_to_v64f16_scalar(<16 x i64> inreg %a, i
; GFX11-NEXT: s_and_b32 s42, s42, exec_lo
; GFX11-NEXT: s_cselect_b32 s42, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s42, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB65_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: s_add_u32 s26, s26, 3
@@ -102059,8 +102076,9 @@ define <16 x i64> @bitcast_v64f16_to_v16i64(<64 x half> %a, i32 %b) #0 {
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: s_waitcnt vmcnt(1)
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v32
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB66_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -103173,6 +103191,7 @@ define inreg <16 x i64> @bitcast_v64f16_to_v16i64_scalar(<64 x half> inreg %a, i
; GFX11-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB67_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_pk_add_f16 v15, 0x200, s51 op_sel_hi:[0,1]
@@ -103668,28 +103687,25 @@ define <64 x i16> @bitcast_v16i64_to_v64i16(<16 x i64> %a, i32 %b) #0 {
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: s_waitcnt vmcnt(1)
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v32
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB68_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_add_co_u32 v14, vcc_lo, v14, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v15, null, 0, v15, vcc_lo
; GFX11-NEXT: v_add_co_u32 v12, vcc_lo, v12, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v13, null, 0, v13, vcc_lo
; GFX11-NEXT: v_add_co_u32 v10, vcc_lo, v10, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v11, null, 0, v11, vcc_lo
; GFX11-NEXT: v_add_co_u32 v8, vcc_lo, v8, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v9, null, 0, v9, vcc_lo
; GFX11-NEXT: v_add_co_u32 v6, vcc_lo, v6, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v7, null, 0, v7, vcc_lo
; GFX11-NEXT: v_add_co_u32 v4, vcc_lo, v4, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v5, null, 0, v5, vcc_lo
; GFX11-NEXT: v_add_co_u32 v2, vcc_lo, v2, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX11-NEXT: v_add_co_u32 v0, vcc_lo, v0, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
@@ -103697,22 +103713,18 @@ define <64 x i16> @bitcast_v16i64_to_v64i16(<16 x i64> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0)
; GFX11-NEXT: v_add_co_ci_u32_e64 v31, null, 0, v31, vcc_lo
; GFX11-NEXT: v_add_co_u32 v28, vcc_lo, v28, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v29, null, 0, v29, vcc_lo
; GFX11-NEXT: v_add_co_u32 v26, vcc_lo, v26, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v27, null, 0, v27, vcc_lo
; GFX11-NEXT: v_add_co_u32 v24, vcc_lo, v24, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v25, null, 0, v25, vcc_lo
; GFX11-NEXT: v_add_co_u32 v22, vcc_lo, v22, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v23, null, 0, v23, vcc_lo
; GFX11-NEXT: v_add_co_u32 v20, vcc_lo, v20, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v21, null, 0, v21, vcc_lo
; GFX11-NEXT: v_add_co_u32 v18, vcc_lo, v18, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v19, null, 0, v19, vcc_lo
; GFX11-NEXT: v_add_co_u32 v16, vcc_lo, v16, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v17, null, 0, v17, vcc_lo
; GFX11-NEXT: .LBB68_2: ; %end
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -104314,6 +104326,7 @@ define inreg <64 x i16> @bitcast_v16i64_to_v64i16_scalar(<16 x i64> inreg %a, i3
; GFX11-NEXT: s_and_b32 s42, s42, exec_lo
; GFX11-NEXT: s_cselect_b32 s42, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s42, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB69_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: s_add_u32 s26, s26, 3
@@ -105113,8 +105126,9 @@ define <16 x i64> @bitcast_v64i16_to_v16i64(<64 x i16> %a, i32 %b) #0 {
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: s_waitcnt vmcnt(1)
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v32
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB70_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -106084,6 +106098,7 @@ define inreg <16 x i64> @bitcast_v64i16_to_v16i64_scalar(<64 x i16> inreg %a, i3
; GFX11-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB71_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_pk_add_u16 v15, s51, 3 op_sel_hi:[1,0]
@@ -109467,8 +109482,9 @@ define <128 x i8> @bitcast_v16f64_to_v128i8(<16 x double> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v75, 8, v1
; GFX11-TRUE16-NEXT: .LBB72_4: ; %end
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_perm_b32 v39, v74, v66, 0xc0c0004
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX11-TRUE16-NEXT: v_perm_b32 v1, v1, v75, 0xc0c0004
; GFX11-TRUE16-NEXT: v_perm_b32 v2, v2, v73, 0xc0c0004
; GFX11-TRUE16-NEXT: v_perm_b32 v55, v72, v63, 0xc0c0004
@@ -109965,8 +109981,9 @@ define <128 x i8> @bitcast_v16f64_to_v128i8(<16 x double> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v75, 8, v1
; GFX11-FAKE16-NEXT: .LBB72_4: ; %end
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_perm_b32 v39, v74, v66, 0xc0c0004
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX11-FAKE16-NEXT: v_perm_b32 v1, v1, v75, 0xc0c0004
; GFX11-FAKE16-NEXT: v_perm_b32 v2, v2, v73, 0xc0c0004
; GFX11-FAKE16-NEXT: v_perm_b32 v55, v72, v63, 0xc0c0004
@@ -114322,6 +114339,7 @@ define inreg <128 x i8> @bitcast_v16f64_to_v128i8_scalar(<16 x double> inreg %a,
; GFX11-NEXT: s_and_b32 s43, s30, exec_lo
; GFX11-NEXT: s_cselect_b32 s43, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s43, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB73_5
; GFX11-NEXT: ; %bb.4: ; %cmp.true
; GFX11-NEXT: v_add_f64 v[21:22], s[22:23], 1.0
@@ -123853,6 +123871,7 @@ define inreg <16 x double> @bitcast_v128i8_to_v16f64_scalar(<128 x i8> inreg %a,
; GFX11-TRUE16-NEXT: s_and_b32 s75, s75, exec_lo
; GFX11-TRUE16-NEXT: s_cselect_b32 s75, 1, 0
; GFX11-TRUE16-NEXT: s_cmp_lg_u32 s75, 1
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_cbranch_scc1 .LBB75_5
; GFX11-TRUE16-NEXT: ; %bb.4: ; %cmp.true
; GFX11-TRUE16-NEXT: v_mov_b32_e32 v10, 0xc0c0004
@@ -124432,6 +124451,7 @@ define inreg <16 x double> @bitcast_v128i8_to_v16f64_scalar(<128 x i8> inreg %a,
; GFX11-FAKE16-NEXT: s_and_b32 s75, s75, exec_lo
; GFX11-FAKE16-NEXT: s_cselect_b32 s75, 1, 0
; GFX11-FAKE16-NEXT: s_cmp_lg_u32 s75, 1
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_cbranch_scc1 .LBB75_5
; GFX11-FAKE16-NEXT: ; %bb.4: ; %cmp.true
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v10, 0xc0c0004
@@ -125421,8 +125441,9 @@ define <64 x bfloat> @bitcast_v16f64_to_v64bf16(<16 x double> %a, i32 %b) #0 {
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: s_waitcnt vmcnt(1)
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v32
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB76_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -126359,6 +126380,7 @@ define inreg <64 x bfloat> @bitcast_v16f64_to_v64bf16_scalar(<16 x double> inreg
; GFX11-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB77_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_add_f64 v[14:15], s[26:27], 1.0
@@ -128252,8 +128274,9 @@ define <16 x double> @bitcast_v64bf16_to_v16f64(<64 x bfloat> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-TRUE16-NEXT: s_waitcnt vmcnt(1)
; GFX11-TRUE16-NEXT: v_cmpx_ne_u32_e32 0, v32
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-TRUE16-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-TRUE16-NEXT: s_cbranch_execz .LBB78_2
; GFX11-TRUE16-NEXT: ; %bb.1: ; %cmp.true
@@ -128821,8 +128844,9 @@ define <16 x double> @bitcast_v64bf16_to_v16f64(<64 x bfloat> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-FAKE16-NEXT: s_waitcnt vmcnt(1)
; GFX11-FAKE16-NEXT: v_cmpx_ne_u32_e32 0, v32
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-FAKE16-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-FAKE16-NEXT: s_cbranch_execz .LBB78_2
; GFX11-FAKE16-NEXT: ; %bb.1: ; %cmp.true
@@ -131712,6 +131736,7 @@ define inreg <16 x double> @bitcast_v64bf16_to_v16f64_scalar(<64 x bfloat> inreg
; GFX11-TRUE16-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-TRUE16-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-TRUE16-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_cbranch_scc1 .LBB79_4
; GFX11-TRUE16-NEXT: ; %bb.3: ; %cmp.true
; GFX11-TRUE16-NEXT: s_and_b32 s0, s51, 0xffff0000
@@ -132441,6 +132466,7 @@ define inreg <16 x double> @bitcast_v64bf16_to_v16f64_scalar(<64 x bfloat> inreg
; GFX11-FAKE16-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-FAKE16-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-FAKE16-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_cbranch_scc1 .LBB79_4
; GFX11-FAKE16-NEXT: ; %bb.3: ; %cmp.true
; GFX11-FAKE16-NEXT: s_and_b32 s1, s51, 0xffff0000
@@ -133509,8 +133535,9 @@ define <64 x half> @bitcast_v16f64_to_v64f16(<16 x double> %a, i32 %b) #0 {
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: s_waitcnt vmcnt(1)
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v32
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB80_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -134233,6 +134260,7 @@ define inreg <64 x half> @bitcast_v16f64_to_v64f16_scalar(<16 x double> inreg %a
; GFX11-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB81_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_add_f64 v[14:15], s[26:27], 1.0
@@ -135200,8 +135228,9 @@ define <16 x double> @bitcast_v64f16_to_v16f64(<64 x half> %a, i32 %b) #0 {
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: s_waitcnt vmcnt(1)
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v32
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB82_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -136314,6 +136343,7 @@ define inreg <16 x double> @bitcast_v64f16_to_v16f64_scalar(<64 x half> inreg %a
; GFX11-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB83_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_pk_add_f16 v15, 0x200, s51 op_sel_hi:[0,1]
@@ -136762,8 +136792,9 @@ define <64 x i16> @bitcast_v16f64_to_v64i16(<16 x double> %a, i32 %b) #0 {
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: s_waitcnt vmcnt(1)
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v32
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB84_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -137486,6 +137517,7 @@ define inreg <64 x i16> @bitcast_v16f64_to_v64i16_scalar(<16 x double> inreg %a,
; GFX11-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB85_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_add_f64 v[14:15], s[26:27], 1.0
@@ -138283,8 +138315,9 @@ define <16 x double> @bitcast_v64i16_to_v16f64(<64 x i16> %a, i32 %b) #0 {
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: s_waitcnt vmcnt(1)
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v32
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB86_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -139254,6 +139287,7 @@ define inreg <16 x double> @bitcast_v64i16_to_v16f64_scalar(<64 x i16> inreg %a,
; GFX11-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB87_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_pk_add_u16 v15, s51, 3 op_sel_hi:[1,0]
@@ -149258,6 +149292,7 @@ define inreg <64 x bfloat> @bitcast_v128i8_to_v64bf16_scalar(<128 x i8> inreg %a
; GFX11-TRUE16-NEXT: s_and_b32 s75, s75, exec_lo
; GFX11-TRUE16-NEXT: s_cselect_b32 s75, 1, 0
; GFX11-TRUE16-NEXT: s_cmp_lg_u32 s75, 1
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_cbranch_scc1 .LBB89_5
; GFX11-TRUE16-NEXT: ; %bb.4: ; %cmp.true
; GFX11-TRUE16-NEXT: v_add_nc_u32_e32 v0, 3, v180
@@ -149766,6 +149801,7 @@ define inreg <64 x bfloat> @bitcast_v128i8_to_v64bf16_scalar(<128 x i8> inreg %a
; GFX11-FAKE16-NEXT: s_and_b32 s75, s75, exec_lo
; GFX11-FAKE16-NEXT: s_cselect_b32 s75, 1, 0
; GFX11-FAKE16-NEXT: s_cmp_lg_u32 s75, 1
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_cbranch_scc1 .LBB89_5
; GFX11-FAKE16-NEXT: ; %bb.4: ; %cmp.true
; GFX11-FAKE16-NEXT: v_add_nc_u32_e32 v0, 3, v182
@@ -155991,6 +156027,7 @@ define <128 x i8> @bitcast_v64bf16_to_v128i8(<64 x bfloat> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v144, 8, v15
; GFX11-TRUE16-NEXT: .LBB90_4: ; %end
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_perm_b32 v35, v112, v67, 0xc0c0004
; GFX11-TRUE16-NEXT: v_perm_b32 v1, v1, v63, 0xc0c0004
; GFX11-TRUE16-NEXT: v_perm_b32 v2, v2, v61, 0xc0c0004
@@ -157000,6 +157037,7 @@ define <128 x i8> @bitcast_v64bf16_to_v128i8(<64 x bfloat> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v67, 16, v72
; GFX11-FAKE16-NEXT: .LBB90_4: ; %end
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_perm_b32 v39, v150, v64, 0xc0c0004
; GFX11-FAKE16-NEXT: v_perm_b32 v1, v1, v60, 0xc0c0004
; GFX11-FAKE16-NEXT: v_perm_b32 v2, v2, v59, 0xc0c0004
@@ -163091,6 +163129,7 @@ define inreg <128 x i8> @bitcast_v64bf16_to_v128i8_scalar(<64 x bfloat> inreg %a
; GFX11-TRUE16-NEXT: s_and_b32 s43, s104, exec_lo
; GFX11-TRUE16-NEXT: s_cselect_b32 s43, 1, 0
; GFX11-TRUE16-NEXT: s_cmp_lg_u32 s43, 1
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_cbranch_scc1 .LBB91_5
; GFX11-TRUE16-NEXT: ; %bb.4: ; %cmp.true
; GFX11-TRUE16-NEXT: s_lshl_b32 s42, s29, 16
@@ -164546,6 +164585,7 @@ define inreg <128 x i8> @bitcast_v64bf16_to_v128i8_scalar(<64 x bfloat> inreg %a
; GFX11-FAKE16-NEXT: s_and_b32 s43, s104, exec_lo
; GFX11-FAKE16-NEXT: s_cselect_b32 s43, 1, 0
; GFX11-FAKE16-NEXT: s_cmp_lg_u32 s43, 1
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_cbranch_scc1 .LBB91_5
; GFX11-FAKE16-NEXT: ; %bb.4: ; %cmp.true
; GFX11-FAKE16-NEXT: s_lshl_b32 s42, s29, 16
@@ -175552,6 +175592,7 @@ define inreg <64 x half> @bitcast_v128i8_to_v64f16_scalar(<128 x i8> inreg %a, i
; GFX11-TRUE16-NEXT: s_and_b32 s75, s75, exec_lo
; GFX11-TRUE16-NEXT: s_cselect_b32 s75, 1, 0
; GFX11-TRUE16-NEXT: s_cmp_lg_u32 s75, 1
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_cbranch_scc1 .LBB93_5
; GFX11-TRUE16-NEXT: ; %bb.4: ; %cmp.true
; GFX11-TRUE16-NEXT: v_add_nc_u32_e32 v0, 3, v180
@@ -176060,6 +176101,7 @@ define inreg <64 x half> @bitcast_v128i8_to_v64f16_scalar(<128 x i8> inreg %a, i
; GFX11-FAKE16-NEXT: s_and_b32 s75, s75, exec_lo
; GFX11-FAKE16-NEXT: s_cselect_b32 s75, 1, 0
; GFX11-FAKE16-NEXT: s_cmp_lg_u32 s75, 1
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_cbranch_scc1 .LBB93_5
; GFX11-FAKE16-NEXT: ; %bb.4: ; %cmp.true
; GFX11-FAKE16-NEXT: v_add_nc_u32_e32 v0, 3, v182
@@ -180558,6 +180600,7 @@ define <128 x i8> @bitcast_v64f16_to_v128i8(<64 x half> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v77, 8, v17
; GFX11-TRUE16-NEXT: .LBB94_4: ; %end
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_perm_b32 v35, v45, v67, 0xc0c0004
; GFX11-TRUE16-NEXT: v_perm_b32 v1, v1, v46, 0xc0c0004
; GFX11-TRUE16-NEXT: v_perm_b32 v2, v2, v43, 0xc0c0004
@@ -181077,6 +181120,7 @@ define <128 x i8> @bitcast_v64f16_to_v128i8(<64 x half> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v76, 8, v17
; GFX11-FAKE16-NEXT: .LBB94_4: ; %end
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_perm_b32 v39, v161, v64, 0xc0c0004
; GFX11-FAKE16-NEXT: v_perm_b32 v1, v1, v162, 0xc0c0004
; GFX11-FAKE16-NEXT: v_perm_b32 v2, v2, v151, 0xc0c0004
@@ -185840,6 +185884,7 @@ define inreg <128 x i8> @bitcast_v64f16_to_v128i8_scalar(<64 x half> inreg %a, i
; GFX11-NEXT: s_and_b32 s43, s100, exec_lo
; GFX11-NEXT: s_cselect_b32 s43, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s43, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB95_5
; GFX11-NEXT: ; %bb.4: ; %cmp.true
; GFX11-NEXT: v_pk_add_f16 v28, 0x200, s19 op_sel_hi:[0,1]
@@ -196257,6 +196302,7 @@ define inreg <64 x i16> @bitcast_v128i8_to_v64i16_scalar(<128 x i8> inreg %a, i3
; GFX11-TRUE16-NEXT: s_and_b32 s75, s75, exec_lo
; GFX11-TRUE16-NEXT: s_cselect_b32 s75, 1, 0
; GFX11-TRUE16-NEXT: s_cmp_lg_u32 s75, 1
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_cbranch_scc1 .LBB97_5
; GFX11-TRUE16-NEXT: ; %bb.4: ; %cmp.true
; GFX11-TRUE16-NEXT: v_add_nc_u32_e32 v0, 3, v180
@@ -196765,6 +196811,7 @@ define inreg <64 x i16> @bitcast_v128i8_to_v64i16_scalar(<128 x i8> inreg %a, i3
; GFX11-FAKE16-NEXT: s_and_b32 s75, s75, exec_lo
; GFX11-FAKE16-NEXT: s_cselect_b32 s75, 1, 0
; GFX11-FAKE16-NEXT: s_cmp_lg_u32 s75, 1
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_cbranch_scc1 .LBB97_5
; GFX11-FAKE16-NEXT: ; %bb.4: ; %cmp.true
; GFX11-FAKE16-NEXT: v_add_nc_u32_e32 v0, 3, v182
@@ -201371,6 +201418,7 @@ define <128 x i8> @bitcast_v64i16_to_v128i8(<64 x i16> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v77, 8, v17
; GFX11-TRUE16-NEXT: .LBB98_4: ; %end
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_perm_b32 v35, v45, v67, 0xc0c0004
; GFX11-TRUE16-NEXT: v_perm_b32 v1, v1, v46, 0xc0c0004
; GFX11-TRUE16-NEXT: v_perm_b32 v2, v2, v43, 0xc0c0004
@@ -201890,6 +201938,7 @@ define <128 x i8> @bitcast_v64i16_to_v128i8(<64 x i16> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v76, 8, v17
; GFX11-FAKE16-NEXT: .LBB98_4: ; %end
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_perm_b32 v39, v161, v64, 0xc0c0004
; GFX11-FAKE16-NEXT: v_perm_b32 v1, v1, v162, 0xc0c0004
; GFX11-FAKE16-NEXT: v_perm_b32 v2, v2, v151, 0xc0c0004
@@ -206266,6 +206315,7 @@ define inreg <128 x i8> @bitcast_v64i16_to_v128i8_scalar(<64 x i16> inreg %a, i3
; GFX11-NEXT: s_and_b32 s43, s100, exec_lo
; GFX11-NEXT: s_cselect_b32 s43, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s43, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB99_5
; GFX11-NEXT: ; %bb.4: ; %cmp.true
; GFX11-NEXT: v_pk_add_u16 v28, s19, 3 op_sel_hi:[1,0]
@@ -209110,8 +209160,9 @@ define <64 x half> @bitcast_v64bf16_to_v64f16(<64 x bfloat> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-TRUE16-NEXT: s_waitcnt vmcnt(1)
; GFX11-TRUE16-NEXT: v_cmpx_ne_u32_e32 0, v32
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-TRUE16-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-TRUE16-NEXT: s_cbranch_execz .LBB100_2
; GFX11-TRUE16-NEXT: ; %bb.1: ; %cmp.true
@@ -209674,8 +209725,9 @@ define <64 x half> @bitcast_v64bf16_to_v64f16(<64 x bfloat> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-FAKE16-NEXT: s_waitcnt vmcnt(1)
; GFX11-FAKE16-NEXT: v_cmpx_ne_u32_e32 0, v32
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-FAKE16-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-FAKE16-NEXT: s_cbranch_execz .LBB100_2
; GFX11-FAKE16-NEXT: ; %bb.1: ; %cmp.true
@@ -213236,6 +213288,7 @@ define inreg <64 x half> @bitcast_v64bf16_to_v64f16_scalar(<64 x bfloat> inreg %
; GFX11-TRUE16-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-TRUE16-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-TRUE16-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_cbranch_scc1 .LBB101_4
; GFX11-TRUE16-NEXT: ; %bb.3: ; %cmp.true
; GFX11-TRUE16-NEXT: s_and_b32 s0, s36, 0xffff0000
@@ -213929,6 +213982,7 @@ define inreg <64 x half> @bitcast_v64bf16_to_v64f16_scalar(<64 x bfloat> inreg %
; GFX11-FAKE16-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-FAKE16-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-FAKE16-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_cbranch_scc1 .LBB101_4
; GFX11-FAKE16-NEXT: ; %bb.3: ; %cmp.true
; GFX11-FAKE16-NEXT: s_and_b32 s0, s36, 0xffff0000
@@ -215905,8 +215959,9 @@ define <64 x bfloat> @bitcast_v64f16_to_v64bf16(<64 x half> %a, i32 %b) #0 {
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: s_waitcnt vmcnt(1)
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v32
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB102_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -217362,6 +217417,7 @@ define inreg <64 x bfloat> @bitcast_v64f16_to_v64bf16_scalar(<64 x half> inreg %
; GFX11-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB103_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_pk_add_f16 v15, 0x200, s27 op_sel_hi:[0,1]
@@ -219751,8 +219807,9 @@ define <64 x i16> @bitcast_v64bf16_to_v64i16(<64 x bfloat> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-TRUE16-NEXT: s_waitcnt vmcnt(1)
; GFX11-TRUE16-NEXT: v_cmpx_ne_u32_e32 0, v32
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-TRUE16-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-TRUE16-NEXT: s_cbranch_execz .LBB104_2
; GFX11-TRUE16-NEXT: ; %bb.1: ; %cmp.true
@@ -220354,8 +220411,9 @@ define <64 x i16> @bitcast_v64bf16_to_v64i16(<64 x bfloat> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-FAKE16-NEXT: s_waitcnt vmcnt(1)
; GFX11-FAKE16-NEXT: v_cmpx_ne_u32_e32 0, v32
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-FAKE16-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-FAKE16-NEXT: s_cbranch_execz .LBB104_2
; GFX11-FAKE16-NEXT: ; %bb.1: ; %cmp.true
@@ -223502,6 +223560,7 @@ define inreg <64 x i16> @bitcast_v64bf16_to_v64i16_scalar(<64 x bfloat> inreg %a
; GFX11-TRUE16-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-TRUE16-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-TRUE16-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_cbranch_scc1 .LBB105_4
; GFX11-TRUE16-NEXT: ; %bb.3: ; %cmp.true
; GFX11-TRUE16-NEXT: s_and_b32 s0, s36, 0xffff0000
@@ -224120,6 +224179,7 @@ define inreg <64 x i16> @bitcast_v64bf16_to_v64i16_scalar(<64 x bfloat> inreg %a
; GFX11-FAKE16-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-FAKE16-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-FAKE16-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_cbranch_scc1 .LBB105_4
; GFX11-FAKE16-NEXT: ; %bb.3: ; %cmp.true
; GFX11-FAKE16-NEXT: s_lshl_b32 s1, s36, 16
@@ -225677,8 +225737,9 @@ define <64 x bfloat> @bitcast_v64i16_to_v64bf16(<64 x i16> %a, i32 %b) #0 {
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: s_waitcnt vmcnt(1)
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v32
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB106_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -226931,6 +226992,7 @@ define inreg <64 x bfloat> @bitcast_v64i16_to_v64bf16_scalar(<64 x i16> inreg %a
; GFX11-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB107_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_pk_add_u16 v15, s27, 3 op_sel_hi:[1,0]
@@ -227670,8 +227732,9 @@ define <64 x i16> @bitcast_v64f16_to_v64i16(<64 x half> %a, i32 %b) #0 {
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: s_waitcnt vmcnt(1)
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v32
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB108_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -228825,6 +228888,7 @@ define inreg <64 x i16> @bitcast_v64f16_to_v64i16_scalar(<64 x half> inreg %a, i
; GFX11-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB109_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_pk_add_f16 v15, 0x200, s27 op_sel_hi:[0,1]
@@ -230034,8 +230098,9 @@ define <64 x half> @bitcast_v64i16_to_v64f16(<64 x i16> %a, i32 %b) #0 {
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: s_waitcnt vmcnt(1)
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v32
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB110_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -231329,6 +231394,7 @@ define inreg <64 x half> @bitcast_v64i16_to_v64f16_scalar(<64 x i16> inreg %a, i
; GFX11-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB111_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_pk_add_u16 v15, s27, 3 op_sel_hi:[1,0]
diff --git a/llvm/test/CodeGen/AMDGPU/amdgcn.bitcast.128bit.ll b/llvm/test/CodeGen/AMDGPU/amdgcn.bitcast.128bit.ll
index 17fa6366bbca10..787e0dfbed3de9 100644
--- a/llvm/test/CodeGen/AMDGPU/amdgcn.bitcast.128bit.ll
+++ b/llvm/test/CodeGen/AMDGPU/amdgcn.bitcast.128bit.ll
@@ -60,8 +60,9 @@ define <4 x float> @bitcast_v4i32_to_v4f32(<4 x i32> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v4
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_add_nc_u32_e32 v3, 3, v3
@@ -183,6 +184,7 @@ define inreg <4 x float> @bitcast_v4i32_to_v4f32_scalar(<4 x i32> inreg %a, i32
; GFX11-NEXT: s_and_b32 s4, s4, exec_lo
; GFX11-NEXT: s_cselect_b32 s4, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s4, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB1_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: s_add_i32 s3, s3, 3
@@ -265,8 +267,9 @@ define <4 x i32> @bitcast_v4f32_to_v4i32(<4 x float> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v4
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_dual_add_f32 v3, 1.0, v3 :: v_dual_add_f32 v2, 1.0, v2
@@ -389,6 +392,7 @@ define inreg <4 x i32> @bitcast_v4f32_to_v4i32_scalar(<4 x float> inreg %a, i32
; GFX11-NEXT: s_and_b32 s4, s4, exec_lo
; GFX11-NEXT: s_cselect_b32 s4, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s4, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB3_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_add_f32_e64 v3, s3, 1.0
@@ -471,8 +475,9 @@ define <2 x i64> @bitcast_v4i32_to_v2i64(<4 x i32> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v4
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_add_nc_u32_e32 v3, 3, v3
@@ -594,6 +599,7 @@ define inreg <2 x i64> @bitcast_v4i32_to_v2i64_scalar(<4 x i32> inreg %a, i32 in
; GFX11-NEXT: s_and_b32 s4, s4, exec_lo
; GFX11-NEXT: s_cselect_b32 s4, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s4, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB5_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: s_add_i32 s3, s3, 3
@@ -676,12 +682,12 @@ define <4 x i32> @bitcast_v2i64_to_v4i32(<2 x i64> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v4
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_add_co_u32 v2, vcc_lo, v2, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX11-NEXT: v_add_co_u32 v0, vcc_lo, v0, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
@@ -800,6 +806,7 @@ define inreg <4 x i32> @bitcast_v2i64_to_v4i32_scalar(<2 x i64> inreg %a, i32 in
; GFX11-NEXT: s_and_b32 s4, s4, exec_lo
; GFX11-NEXT: s_cselect_b32 s4, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s4, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB7_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: s_add_u32 s2, s2, 3
@@ -882,8 +889,9 @@ define <2 x double> @bitcast_v4i32_to_v2f64(<4 x i32> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v4
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_add_nc_u32_e32 v3, 3, v3
@@ -1005,6 +1013,7 @@ define inreg <2 x double> @bitcast_v4i32_to_v2f64_scalar(<4 x i32> inreg %a, i32
; GFX11-NEXT: s_and_b32 s4, s4, exec_lo
; GFX11-NEXT: s_cselect_b32 s4, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s4, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB9_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: s_add_i32 s3, s3, 3
@@ -1083,8 +1092,9 @@ define <4 x i32> @bitcast_v2f64_to_v4i32(<2 x double> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v4
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB10_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -1202,6 +1212,7 @@ define inreg <4 x i32> @bitcast_v2f64_to_v4i32_scalar(<2 x double> inreg %a, i32
; GFX11-NEXT: s_and_b32 s4, s4, exec_lo
; GFX11-NEXT: s_cselect_b32 s4, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s4, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB11_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_add_f64 v[2:3], s[2:3], 1.0
@@ -1308,8 +1319,9 @@ define <8 x i16> @bitcast_v4i32_to_v8i16(<4 x i32> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v4
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_add_nc_u32_e32 v3, 3, v3
@@ -1455,6 +1467,7 @@ define inreg <8 x i16> @bitcast_v4i32_to_v8i16_scalar(<4 x i32> inreg %a, i32 in
; GFX11-NEXT: s_and_b32 s4, s4, exec_lo
; GFX11-NEXT: s_cselect_b32 s4, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s4, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB13_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: s_add_i32 s3, s3, 3
@@ -1596,8 +1609,9 @@ define <4 x i32> @bitcast_v8i16_to_v4i32(<8 x i16> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v4
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_pk_add_u16 v3, v3, 3 op_sel_hi:[1,0]
@@ -1769,6 +1783,7 @@ define inreg <4 x i32> @bitcast_v8i16_to_v4i32_scalar(<8 x i16> inreg %a, i32 in
; GFX11-NEXT: s_and_b32 s4, s4, exec_lo
; GFX11-NEXT: s_cselect_b32 s4, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s4, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB15_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_pk_add_u16 v3, s3, 3 op_sel_hi:[1,0]
@@ -1877,8 +1892,9 @@ define <8 x half> @bitcast_v4i32_to_v8f16(<4 x i32> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v4
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_add_nc_u32_e32 v3, 3, v3
@@ -2024,6 +2040,7 @@ define inreg <8 x half> @bitcast_v4i32_to_v8f16_scalar(<4 x i32> inreg %a, i32 i
; GFX11-NEXT: s_and_b32 s4, s4, exec_lo
; GFX11-NEXT: s_cselect_b32 s4, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s4, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB17_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: s_add_i32 s3, s3, 3
@@ -2182,8 +2199,9 @@ define <4 x i32> @bitcast_v8f16_to_v4i32(<8 x half> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v4
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_pk_add_f16 v3, 0x200, v3 op_sel_hi:[0,1]
@@ -2371,6 +2389,7 @@ define inreg <4 x i32> @bitcast_v8f16_to_v4i32_scalar(<8 x half> inreg %a, i32 i
; GFX11-NEXT: s_and_b32 s4, s4, exec_lo
; GFX11-NEXT: s_cselect_b32 s4, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s4, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB19_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_pk_add_f16 v3, 0x200, s3 op_sel_hi:[0,1]
@@ -2500,8 +2519,9 @@ define <8 x bfloat> @bitcast_v4i32_to_v8bf16(<4 x i32> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v4
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_add_nc_u32_e32 v3, 3, v3
@@ -2659,6 +2679,7 @@ define inreg <8 x bfloat> @bitcast_v4i32_to_v8bf16_scalar(<4 x i32> inreg %a, i3
; GFX11-NEXT: s_and_b32 s4, s4, exec_lo
; GFX11-NEXT: s_cselect_b32 s4, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s4, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB21_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: s_add_i32 s3, s3, 3
@@ -2931,8 +2952,9 @@ define <4 x i32> @bitcast_v8bf16_to_v4i32(<8 x bfloat> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-TRUE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-TRUE16-NEXT: v_cmpx_ne_u32_e32 0, v4
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-TRUE16-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-TRUE16-NEXT: s_cbranch_execz .LBB22_2
; GFX11-TRUE16-NEXT: ; %bb.1: ; %cmp.true
@@ -3024,8 +3046,9 @@ define <4 x i32> @bitcast_v8bf16_to_v4i32(<8 x bfloat> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-FAKE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-FAKE16-NEXT: v_cmpx_ne_u32_e32 0, v4
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-FAKE16-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-FAKE16-NEXT: s_cbranch_execz .LBB22_2
; GFX11-FAKE16-NEXT: ; %bb.1: ; %cmp.true
@@ -3395,6 +3418,7 @@ define inreg <4 x i32> @bitcast_v8bf16_to_v4i32_scalar(<8 x bfloat> inreg %a, i3
; GFX11-TRUE16-NEXT: s_and_b32 s4, s4, exec_lo
; GFX11-TRUE16-NEXT: s_cselect_b32 s4, 1, 0
; GFX11-TRUE16-NEXT: s_cmp_lg_u32 s4, 1
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_cbranch_scc1 .LBB23_4
; GFX11-TRUE16-NEXT: ; %bb.3: ; %cmp.true
; GFX11-TRUE16-NEXT: s_pack_lh_b32_b16 s4, 0, s3
@@ -3496,6 +3520,7 @@ define inreg <4 x i32> @bitcast_v8bf16_to_v4i32_scalar(<8 x bfloat> inreg %a, i3
; GFX11-FAKE16-NEXT: s_and_b32 s4, s4, exec_lo
; GFX11-FAKE16-NEXT: s_cselect_b32 s4, 1, 0
; GFX11-FAKE16-NEXT: s_cmp_lg_u32 s4, 1
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_cbranch_scc1 .LBB23_4
; GFX11-FAKE16-NEXT: ; %bb.3: ; %cmp.true
; GFX11-FAKE16-NEXT: s_lshl_b32 s4, s3, 16
@@ -3813,6 +3838,7 @@ define <16 x i8> @bitcast_v4i32_to_v16i8(<4 x i32> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: ; implicit-def: $vgpr14_lo16
; GFX11-TRUE16-NEXT: ; implicit-def: $vgpr15_lo16
; GFX11-TRUE16-NEXT: v_cmpx_ne_u32_e32 0, v4
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execz .LBB24_2
; GFX11-TRUE16-NEXT: ; %bb.1: ; %cmp.false
@@ -3852,6 +3878,7 @@ define <16 x i8> @bitcast_v4i32_to_v16i8(<4 x i32> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v1, 8, v16
; GFX11-TRUE16-NEXT: .LBB24_4: ; %end
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v0.l, v16.l
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v4.l, v17.l
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v8.l, v11.l
@@ -3877,6 +3904,7 @@ define <16 x i8> @bitcast_v4i32_to_v16i8(<4 x i32> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: ; implicit-def: $vgpr14
; GFX11-FAKE16-NEXT: ; implicit-def: $vgpr15
; GFX11-FAKE16-NEXT: v_cmpx_ne_u32_e32 0, v4
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX11-FAKE16-NEXT: s_cbranch_execz .LBB24_2
; GFX11-FAKE16-NEXT: ; %bb.1: ; %cmp.false
@@ -3916,6 +3944,7 @@ define <16 x i8> @bitcast_v4i32_to_v16i8(<4 x i32> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v1, 8, v18
; GFX11-FAKE16-NEXT: .LBB24_4: ; %end
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v0, v18
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v4, v19
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v8, v16
@@ -4203,6 +4232,7 @@ define inreg <16 x i8> @bitcast_v4i32_to_v16i8_scalar(<4 x i32> inreg %a, i32 in
; GFX11-NEXT: s_and_b32 s5, s18, exec_lo
; GFX11-NEXT: s_cselect_b32 s5, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s5, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB25_5
; GFX11-NEXT: ; %bb.4: ; %cmp.true
; GFX11-NEXT: s_add_i32 s1, s1, 3
@@ -4566,6 +4596,7 @@ define <4 x i32> @bitcast_v16i8_to_v4i32(<16 x i8> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-TRUE16-NEXT: ; implicit-def: $vgpr0_vgpr1_vgpr2_vgpr3
; GFX11-TRUE16-NEXT: v_cmpx_ne_u32_e32 0, v16
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB26_3
; GFX11-TRUE16-NEXT: ; %bb.1: ; %Flow
@@ -4658,6 +4689,7 @@ define <4 x i32> @bitcast_v16i8_to_v4i32(<16 x i8> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-FAKE16-NEXT: ; implicit-def: $vgpr0_vgpr1_vgpr2_vgpr3
; GFX11-FAKE16-NEXT: v_cmpx_ne_u32_e32 0, v16
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB26_3
; GFX11-FAKE16-NEXT: ; %bb.1: ; %Flow
@@ -5095,6 +5127,7 @@ define inreg <4 x i32> @bitcast_v16i8_to_v4i32_scalar(<16 x i8> inreg %a, i32 in
; GFX11-NEXT: s_and_b32 s4, s4, exec_lo
; GFX11-NEXT: s_cselect_b32 s4, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s4, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB27_5
; GFX11-NEXT: ; %bb.4: ; %cmp.true
; GFX11-NEXT: v_mov_b32_e32 v0, 0xc0c0004
@@ -5208,8 +5241,9 @@ define <2 x i64> @bitcast_v4f32_to_v2i64(<4 x float> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v4
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_dual_add_f32 v3, 1.0, v3 :: v_dual_add_f32 v2, 1.0, v2
@@ -5332,6 +5366,7 @@ define inreg <2 x i64> @bitcast_v4f32_to_v2i64_scalar(<4 x float> inreg %a, i32
; GFX11-NEXT: s_and_b32 s4, s4, exec_lo
; GFX11-NEXT: s_cselect_b32 s4, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s4, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB29_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_add_f32_e64 v3, s3, 1.0
@@ -5414,12 +5449,12 @@ define <4 x float> @bitcast_v2i64_to_v4f32(<2 x i64> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v4
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_add_co_u32 v2, vcc_lo, v2, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX11-NEXT: v_add_co_u32 v0, vcc_lo, v0, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
@@ -5538,6 +5573,7 @@ define inreg <4 x float> @bitcast_v2i64_to_v4f32_scalar(<2 x i64> inreg %a, i32
; GFX11-NEXT: s_and_b32 s4, s4, exec_lo
; GFX11-NEXT: s_cselect_b32 s4, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s4, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB31_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: s_add_u32 s2, s2, 3
@@ -5620,8 +5656,9 @@ define <2 x double> @bitcast_v4f32_to_v2f64(<4 x float> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v4
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_dual_add_f32 v3, 1.0, v3 :: v_dual_add_f32 v2, 1.0, v2
@@ -5744,6 +5781,7 @@ define inreg <2 x double> @bitcast_v4f32_to_v2f64_scalar(<4 x float> inreg %a, i
; GFX11-NEXT: s_and_b32 s4, s4, exec_lo
; GFX11-NEXT: s_cselect_b32 s4, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s4, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB33_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_add_f32_e64 v3, s3, 1.0
@@ -5822,8 +5860,9 @@ define <4 x float> @bitcast_v2f64_to_v4f32(<2 x double> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v4
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB34_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -5941,6 +5980,7 @@ define inreg <4 x float> @bitcast_v2f64_to_v4f32_scalar(<2 x double> inreg %a, i
; GFX11-NEXT: s_and_b32 s4, s4, exec_lo
; GFX11-NEXT: s_cselect_b32 s4, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s4, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB35_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_add_f64 v[2:3], s[2:3], 1.0
@@ -6047,8 +6087,9 @@ define <8 x i16> @bitcast_v4f32_to_v8i16(<4 x float> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v4
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_dual_add_f32 v3, 1.0, v3 :: v_dual_add_f32 v2, 1.0, v2
@@ -6200,6 +6241,7 @@ define inreg <8 x i16> @bitcast_v4f32_to_v8i16_scalar(<4 x float> inreg %a, i32
; GFX11-NEXT: s_and_b32 s4, s4, exec_lo
; GFX11-NEXT: s_cselect_b32 s4, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s4, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB37_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_add_f32_e64 v3, s3, 1.0
@@ -6341,8 +6383,9 @@ define <4 x float> @bitcast_v8i16_to_v4f32(<8 x i16> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v4
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_pk_add_u16 v3, v3, 3 op_sel_hi:[1,0]
@@ -6514,6 +6557,7 @@ define inreg <4 x float> @bitcast_v8i16_to_v4f32_scalar(<8 x i16> inreg %a, i32
; GFX11-NEXT: s_and_b32 s4, s4, exec_lo
; GFX11-NEXT: s_cselect_b32 s4, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s4, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB39_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_pk_add_u16 v3, s3, 3 op_sel_hi:[1,0]
@@ -6622,8 +6666,9 @@ define <8 x half> @bitcast_v4f32_to_v8f16(<4 x float> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v4
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_dual_add_f32 v3, 1.0, v3 :: v_dual_add_f32 v2, 1.0, v2
@@ -6775,6 +6820,7 @@ define inreg <8 x half> @bitcast_v4f32_to_v8f16_scalar(<4 x float> inreg %a, i32
; GFX11-NEXT: s_and_b32 s4, s4, exec_lo
; GFX11-NEXT: s_cselect_b32 s4, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s4, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB41_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_add_f32_e64 v3, s3, 1.0
@@ -6933,8 +6979,9 @@ define <4 x float> @bitcast_v8f16_to_v4f32(<8 x half> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v4
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_pk_add_f16 v3, 0x200, v3 op_sel_hi:[0,1]
@@ -7122,6 +7169,7 @@ define inreg <4 x float> @bitcast_v8f16_to_v4f32_scalar(<8 x half> inreg %a, i32
; GFX11-NEXT: s_and_b32 s4, s4, exec_lo
; GFX11-NEXT: s_cselect_b32 s4, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s4, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB43_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_pk_add_f16 v3, 0x200, s3 op_sel_hi:[0,1]
@@ -7251,8 +7299,9 @@ define <8 x bfloat> @bitcast_v4f32_to_v8bf16(<4 x float> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v4
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_dual_add_f32 v3, 1.0, v3 :: v_dual_add_f32 v2, 1.0, v2
@@ -7420,6 +7469,7 @@ define inreg <8 x bfloat> @bitcast_v4f32_to_v8bf16_scalar(<4 x float> inreg %a,
; GFX11-NEXT: s_and_b32 s4, s4, exec_lo
; GFX11-NEXT: s_cselect_b32 s4, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s4, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB45_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_add_f32_e64 v3, s3, 1.0
@@ -7692,8 +7742,9 @@ define <4 x float> @bitcast_v8bf16_to_v4f32(<8 x bfloat> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-TRUE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-TRUE16-NEXT: v_cmpx_ne_u32_e32 0, v4
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-TRUE16-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-TRUE16-NEXT: s_cbranch_execz .LBB46_2
; GFX11-TRUE16-NEXT: ; %bb.1: ; %cmp.true
@@ -7785,8 +7836,9 @@ define <4 x float> @bitcast_v8bf16_to_v4f32(<8 x bfloat> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-FAKE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-FAKE16-NEXT: v_cmpx_ne_u32_e32 0, v4
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-FAKE16-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-FAKE16-NEXT: s_cbranch_execz .LBB46_2
; GFX11-FAKE16-NEXT: ; %bb.1: ; %cmp.true
@@ -8156,6 +8208,7 @@ define inreg <4 x float> @bitcast_v8bf16_to_v4f32_scalar(<8 x bfloat> inreg %a,
; GFX11-TRUE16-NEXT: s_and_b32 s4, s4, exec_lo
; GFX11-TRUE16-NEXT: s_cselect_b32 s4, 1, 0
; GFX11-TRUE16-NEXT: s_cmp_lg_u32 s4, 1
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_cbranch_scc1 .LBB47_4
; GFX11-TRUE16-NEXT: ; %bb.3: ; %cmp.true
; GFX11-TRUE16-NEXT: s_pack_lh_b32_b16 s4, 0, s3
@@ -8257,6 +8310,7 @@ define inreg <4 x float> @bitcast_v8bf16_to_v4f32_scalar(<8 x bfloat> inreg %a,
; GFX11-FAKE16-NEXT: s_and_b32 s4, s4, exec_lo
; GFX11-FAKE16-NEXT: s_cselect_b32 s4, 1, 0
; GFX11-FAKE16-NEXT: s_cmp_lg_u32 s4, 1
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_cbranch_scc1 .LBB47_4
; GFX11-FAKE16-NEXT: ; %bb.3: ; %cmp.true
; GFX11-FAKE16-NEXT: s_lshl_b32 s4, s3, 16
@@ -8574,6 +8628,7 @@ define <16 x i8> @bitcast_v4f32_to_v16i8(<4 x float> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: ; implicit-def: $vgpr14_lo16
; GFX11-TRUE16-NEXT: ; implicit-def: $vgpr15_lo16
; GFX11-TRUE16-NEXT: v_cmpx_ne_u32_e32 0, v4
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execz .LBB48_2
; GFX11-TRUE16-NEXT: ; %bb.1: ; %cmp.false
@@ -8611,6 +8666,7 @@ define <16 x i8> @bitcast_v4f32_to_v16i8(<4 x float> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v1, 8, v16
; GFX11-TRUE16-NEXT: .LBB48_4: ; %end
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v0.l, v16.l
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v4.l, v17.l
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v8.l, v11.l
@@ -8636,6 +8692,7 @@ define <16 x i8> @bitcast_v4f32_to_v16i8(<4 x float> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: ; implicit-def: $vgpr14
; GFX11-FAKE16-NEXT: ; implicit-def: $vgpr15
; GFX11-FAKE16-NEXT: v_cmpx_ne_u32_e32 0, v4
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX11-FAKE16-NEXT: s_cbranch_execz .LBB48_2
; GFX11-FAKE16-NEXT: ; %bb.1: ; %cmp.false
@@ -8673,6 +8730,7 @@ define <16 x i8> @bitcast_v4f32_to_v16i8(<4 x float> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v1, 8, v18
; GFX11-FAKE16-NEXT: .LBB48_4: ; %end
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v0, v18
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v4, v19
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v8, v16
@@ -8980,6 +9038,7 @@ define inreg <16 x i8> @bitcast_v4f32_to_v16i8_scalar(<4 x float> inreg %a, i32
; GFX11-NEXT: s_and_b32 s5, s8, exec_lo
; GFX11-NEXT: s_cselect_b32 s5, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s5, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB49_5
; GFX11-NEXT: ; %bb.4: ; %cmp.true
; GFX11-NEXT: v_add_f32_e64 v19, s1, 1.0
@@ -9352,6 +9411,7 @@ define <4 x float> @bitcast_v16i8_to_v4f32(<16 x i8> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-TRUE16-NEXT: ; implicit-def: $vgpr0_vgpr1_vgpr2_vgpr3
; GFX11-TRUE16-NEXT: v_cmpx_ne_u32_e32 0, v16
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB50_3
; GFX11-TRUE16-NEXT: ; %bb.1: ; %Flow
@@ -9444,6 +9504,7 @@ define <4 x float> @bitcast_v16i8_to_v4f32(<16 x i8> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-FAKE16-NEXT: ; implicit-def: $vgpr0_vgpr1_vgpr2_vgpr3
; GFX11-FAKE16-NEXT: v_cmpx_ne_u32_e32 0, v16
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB50_3
; GFX11-FAKE16-NEXT: ; %bb.1: ; %Flow
@@ -9881,6 +9942,7 @@ define inreg <4 x float> @bitcast_v16i8_to_v4f32_scalar(<16 x i8> inreg %a, i32
; GFX11-NEXT: s_and_b32 s4, s4, exec_lo
; GFX11-NEXT: s_cselect_b32 s4, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s4, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB51_5
; GFX11-NEXT: ; %bb.4: ; %cmp.true
; GFX11-NEXT: v_mov_b32_e32 v0, 0xc0c0004
@@ -9994,12 +10056,12 @@ define <2 x double> @bitcast_v2i64_to_v2f64(<2 x i64> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v4
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_add_co_u32 v0, vcc_lo, v0, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-NEXT: v_add_co_u32 v2, vcc_lo, v2, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
@@ -10118,6 +10180,7 @@ define inreg <2 x double> @bitcast_v2i64_to_v2f64_scalar(<2 x i64> inreg %a, i32
; GFX11-NEXT: s_and_b32 s4, s4, exec_lo
; GFX11-NEXT: s_cselect_b32 s4, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s4, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB53_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: s_add_u32 s0, s0, 3
@@ -10195,8 +10258,9 @@ define <2 x i64> @bitcast_v2f64_to_v2i64(<2 x double> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v4
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB54_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -10314,6 +10378,7 @@ define inreg <2 x i64> @bitcast_v2f64_to_v2i64_scalar(<2 x double> inreg %a, i32
; GFX11-NEXT: s_and_b32 s4, s4, exec_lo
; GFX11-NEXT: s_cselect_b32 s4, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s4, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB55_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_add_f64 v[0:1], s[0:1], 1.0
@@ -10420,12 +10485,12 @@ define <8 x i16> @bitcast_v2i64_to_v8i16(<2 x i64> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v4
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_add_co_u32 v2, vcc_lo, v2, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX11-NEXT: v_add_co_u32 v0, vcc_lo, v0, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
@@ -10568,6 +10633,7 @@ define inreg <8 x i16> @bitcast_v2i64_to_v8i16_scalar(<2 x i64> inreg %a, i32 in
; GFX11-NEXT: s_and_b32 s4, s4, exec_lo
; GFX11-NEXT: s_cselect_b32 s4, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s4, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB57_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: s_add_u32 s2, s2, 3
@@ -10709,8 +10775,9 @@ define <2 x i64> @bitcast_v8i16_to_v2i64(<8 x i16> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v4
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_pk_add_u16 v3, v3, 3 op_sel_hi:[1,0]
@@ -10882,6 +10949,7 @@ define inreg <2 x i64> @bitcast_v8i16_to_v2i64_scalar(<8 x i16> inreg %a, i32 in
; GFX11-NEXT: s_and_b32 s4, s4, exec_lo
; GFX11-NEXT: s_cselect_b32 s4, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s4, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB59_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_pk_add_u16 v3, s3, 3 op_sel_hi:[1,0]
@@ -10990,12 +11058,12 @@ define <8 x half> @bitcast_v2i64_to_v8f16(<2 x i64> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v4
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_add_co_u32 v2, vcc_lo, v2, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX11-NEXT: v_add_co_u32 v0, vcc_lo, v0, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
@@ -11138,6 +11206,7 @@ define inreg <8 x half> @bitcast_v2i64_to_v8f16_scalar(<2 x i64> inreg %a, i32 i
; GFX11-NEXT: s_and_b32 s4, s4, exec_lo
; GFX11-NEXT: s_cselect_b32 s4, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s4, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB61_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: s_add_u32 s2, s2, 3
@@ -11296,8 +11365,9 @@ define <2 x i64> @bitcast_v8f16_to_v2i64(<8 x half> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v4
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_pk_add_f16 v3, 0x200, v3 op_sel_hi:[0,1]
@@ -11485,6 +11555,7 @@ define inreg <2 x i64> @bitcast_v8f16_to_v2i64_scalar(<8 x half> inreg %a, i32 i
; GFX11-NEXT: s_and_b32 s4, s4, exec_lo
; GFX11-NEXT: s_cselect_b32 s4, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s4, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB63_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_pk_add_f16 v3, 0x200, s3 op_sel_hi:[0,1]
@@ -11614,12 +11685,12 @@ define <8 x bfloat> @bitcast_v2i64_to_v8bf16(<2 x i64> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v4
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_add_co_u32 v2, vcc_lo, v2, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX11-NEXT: v_add_co_u32 v0, vcc_lo, v0, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
@@ -11774,6 +11845,7 @@ define inreg <8 x bfloat> @bitcast_v2i64_to_v8bf16_scalar(<2 x i64> inreg %a, i3
; GFX11-NEXT: s_and_b32 s4, s4, exec_lo
; GFX11-NEXT: s_cselect_b32 s4, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s4, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB65_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: s_add_u32 s2, s2, 3
@@ -12046,8 +12118,9 @@ define <2 x i64> @bitcast_v8bf16_to_v2i64(<8 x bfloat> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-TRUE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-TRUE16-NEXT: v_cmpx_ne_u32_e32 0, v4
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-TRUE16-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-TRUE16-NEXT: s_cbranch_execz .LBB66_2
; GFX11-TRUE16-NEXT: ; %bb.1: ; %cmp.true
@@ -12139,8 +12212,9 @@ define <2 x i64> @bitcast_v8bf16_to_v2i64(<8 x bfloat> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-FAKE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-FAKE16-NEXT: v_cmpx_ne_u32_e32 0, v4
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-FAKE16-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-FAKE16-NEXT: s_cbranch_execz .LBB66_2
; GFX11-FAKE16-NEXT: ; %bb.1: ; %cmp.true
@@ -12510,6 +12584,7 @@ define inreg <2 x i64> @bitcast_v8bf16_to_v2i64_scalar(<8 x bfloat> inreg %a, i3
; GFX11-TRUE16-NEXT: s_and_b32 s4, s4, exec_lo
; GFX11-TRUE16-NEXT: s_cselect_b32 s4, 1, 0
; GFX11-TRUE16-NEXT: s_cmp_lg_u32 s4, 1
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_cbranch_scc1 .LBB67_4
; GFX11-TRUE16-NEXT: ; %bb.3: ; %cmp.true
; GFX11-TRUE16-NEXT: s_pack_lh_b32_b16 s4, 0, s3
@@ -12611,6 +12686,7 @@ define inreg <2 x i64> @bitcast_v8bf16_to_v2i64_scalar(<8 x bfloat> inreg %a, i3
; GFX11-FAKE16-NEXT: s_and_b32 s4, s4, exec_lo
; GFX11-FAKE16-NEXT: s_cselect_b32 s4, 1, 0
; GFX11-FAKE16-NEXT: s_cmp_lg_u32 s4, 1
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_cbranch_scc1 .LBB67_4
; GFX11-FAKE16-NEXT: ; %bb.3: ; %cmp.true
; GFX11-FAKE16-NEXT: s_lshl_b32 s4, s3, 16
@@ -12928,6 +13004,7 @@ define <16 x i8> @bitcast_v2i64_to_v16i8(<2 x i64> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: ; implicit-def: $vgpr14_lo16
; GFX11-TRUE16-NEXT: ; implicit-def: $vgpr15_lo16
; GFX11-TRUE16-NEXT: v_cmpx_ne_u32_e32 0, v4
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execz .LBB68_2
; GFX11-TRUE16-NEXT: ; %bb.1: ; %cmp.false
@@ -12948,7 +13025,6 @@ define <16 x i8> @bitcast_v2i64_to_v16i8(<2 x i64> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: s_cbranch_execz .LBB68_4
; GFX11-TRUE16-NEXT: ; %bb.3: ; %cmp.true
; GFX11-TRUE16-NEXT: v_add_co_u32 v11, vcc_lo, v11, 3
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v12, null, 0, v12, vcc_lo
; GFX11-TRUE16-NEXT: v_add_co_u32 v16, vcc_lo, v16, 3
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v17, null, 0, v17, vcc_lo
@@ -12967,6 +13043,7 @@ define <16 x i8> @bitcast_v2i64_to_v16i8(<2 x i64> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v1, 8, v16
; GFX11-TRUE16-NEXT: .LBB68_4: ; %end
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v0.l, v16.l
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v4.l, v17.l
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v8.l, v11.l
@@ -12992,6 +13069,7 @@ define <16 x i8> @bitcast_v2i64_to_v16i8(<2 x i64> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: ; implicit-def: $vgpr14
; GFX11-FAKE16-NEXT: ; implicit-def: $vgpr15
; GFX11-FAKE16-NEXT: v_cmpx_ne_u32_e32 0, v4
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX11-FAKE16-NEXT: s_cbranch_execz .LBB68_2
; GFX11-FAKE16-NEXT: ; %bb.1: ; %cmp.false
@@ -13012,7 +13090,6 @@ define <16 x i8> @bitcast_v2i64_to_v16i8(<2 x i64> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: s_cbranch_execz .LBB68_4
; GFX11-FAKE16-NEXT: ; %bb.3: ; %cmp.true
; GFX11-FAKE16-NEXT: v_add_co_u32 v16, vcc_lo, v16, 3
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v17, null, 0, v17, vcc_lo
; GFX11-FAKE16-NEXT: v_add_co_u32 v18, vcc_lo, v18, 3
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v19, null, 0, v19, vcc_lo
@@ -13031,6 +13108,7 @@ define <16 x i8> @bitcast_v2i64_to_v16i8(<2 x i64> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v1, 8, v18
; GFX11-FAKE16-NEXT: .LBB68_4: ; %end
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v0, v18
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v4, v19
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v8, v16
@@ -13318,6 +13396,7 @@ define inreg <16 x i8> @bitcast_v2i64_to_v16i8_scalar(<2 x i64> inreg %a, i32 in
; GFX11-NEXT: s_and_b32 s5, s18, exec_lo
; GFX11-NEXT: s_cselect_b32 s5, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s5, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB69_5
; GFX11-NEXT: ; %bb.4: ; %cmp.true
; GFX11-NEXT: s_add_u32 s0, s0, 3
@@ -13681,6 +13760,7 @@ define <2 x i64> @bitcast_v16i8_to_v2i64(<16 x i8> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-TRUE16-NEXT: ; implicit-def: $vgpr0_vgpr1_vgpr2_vgpr3
; GFX11-TRUE16-NEXT: v_cmpx_ne_u32_e32 0, v16
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB70_3
; GFX11-TRUE16-NEXT: ; %bb.1: ; %Flow
@@ -13773,6 +13853,7 @@ define <2 x i64> @bitcast_v16i8_to_v2i64(<16 x i8> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-FAKE16-NEXT: ; implicit-def: $vgpr0_vgpr1_vgpr2_vgpr3
; GFX11-FAKE16-NEXT: v_cmpx_ne_u32_e32 0, v16
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB70_3
; GFX11-FAKE16-NEXT: ; %bb.1: ; %Flow
@@ -14210,6 +14291,7 @@ define inreg <2 x i64> @bitcast_v16i8_to_v2i64_scalar(<16 x i8> inreg %a, i32 in
; GFX11-NEXT: s_and_b32 s4, s4, exec_lo
; GFX11-NEXT: s_cselect_b32 s4, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s4, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB71_5
; GFX11-NEXT: ; %bb.4: ; %cmp.true
; GFX11-NEXT: v_mov_b32_e32 v0, 0xc0c0004
@@ -14345,8 +14427,9 @@ define <8 x i16> @bitcast_v2f64_to_v8i16(<2 x double> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v4
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB72_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -14493,6 +14576,7 @@ define inreg <8 x i16> @bitcast_v2f64_to_v8i16_scalar(<2 x double> inreg %a, i32
; GFX11-NEXT: s_and_b32 s4, s4, exec_lo
; GFX11-NEXT: s_cselect_b32 s4, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s4, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB73_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_add_f64 v[2:3], s[2:3], 1.0
@@ -14632,8 +14716,9 @@ define <2 x double> @bitcast_v8i16_to_v2f64(<8 x i16> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v4
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_pk_add_u16 v3, v3, 3 op_sel_hi:[1,0]
@@ -14805,6 +14890,7 @@ define inreg <2 x double> @bitcast_v8i16_to_v2f64_scalar(<8 x i16> inreg %a, i32
; GFX11-NEXT: s_and_b32 s4, s4, exec_lo
; GFX11-NEXT: s_cselect_b32 s4, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s4, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB75_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_pk_add_u16 v3, s3, 3 op_sel_hi:[1,0]
@@ -14909,8 +14995,9 @@ define <8 x half> @bitcast_v2f64_to_v8f16(<2 x double> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v4
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB76_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -15057,6 +15144,7 @@ define inreg <8 x half> @bitcast_v2f64_to_v8f16_scalar(<2 x double> inreg %a, i3
; GFX11-NEXT: s_and_b32 s4, s4, exec_lo
; GFX11-NEXT: s_cselect_b32 s4, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s4, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB77_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_add_f64 v[2:3], s[2:3], 1.0
@@ -15213,8 +15301,9 @@ define <2 x double> @bitcast_v8f16_to_v2f64(<8 x half> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v4
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_pk_add_f16 v3, 0x200, v3 op_sel_hi:[0,1]
@@ -15402,6 +15491,7 @@ define inreg <2 x double> @bitcast_v8f16_to_v2f64_scalar(<8 x half> inreg %a, i3
; GFX11-NEXT: s_and_b32 s4, s4, exec_lo
; GFX11-NEXT: s_cselect_b32 s4, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s4, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB79_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_pk_add_f16 v3, 0x200, s3 op_sel_hi:[0,1]
@@ -15525,8 +15615,9 @@ define <8 x bfloat> @bitcast_v2f64_to_v8bf16(<2 x double> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v4
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB80_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -15689,6 +15780,7 @@ define inreg <8 x bfloat> @bitcast_v2f64_to_v8bf16_scalar(<2 x double> inreg %a,
; GFX11-NEXT: s_and_b32 s4, s4, exec_lo
; GFX11-NEXT: s_cselect_b32 s4, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s4, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB81_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_add_f64 v[2:3], s[2:3], 1.0
@@ -15959,8 +16051,9 @@ define <2 x double> @bitcast_v8bf16_to_v2f64(<8 x bfloat> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-TRUE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-TRUE16-NEXT: v_cmpx_ne_u32_e32 0, v4
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-TRUE16-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-TRUE16-NEXT: s_cbranch_execz .LBB82_2
; GFX11-TRUE16-NEXT: ; %bb.1: ; %cmp.true
@@ -16052,8 +16145,9 @@ define <2 x double> @bitcast_v8bf16_to_v2f64(<8 x bfloat> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-FAKE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-FAKE16-NEXT: v_cmpx_ne_u32_e32 0, v4
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-FAKE16-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-FAKE16-NEXT: s_cbranch_execz .LBB82_2
; GFX11-FAKE16-NEXT: ; %bb.1: ; %cmp.true
@@ -16423,6 +16517,7 @@ define inreg <2 x double> @bitcast_v8bf16_to_v2f64_scalar(<8 x bfloat> inreg %a,
; GFX11-TRUE16-NEXT: s_and_b32 s4, s4, exec_lo
; GFX11-TRUE16-NEXT: s_cselect_b32 s4, 1, 0
; GFX11-TRUE16-NEXT: s_cmp_lg_u32 s4, 1
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_cbranch_scc1 .LBB83_4
; GFX11-TRUE16-NEXT: ; %bb.3: ; %cmp.true
; GFX11-TRUE16-NEXT: s_pack_lh_b32_b16 s4, 0, s3
@@ -16524,6 +16619,7 @@ define inreg <2 x double> @bitcast_v8bf16_to_v2f64_scalar(<8 x bfloat> inreg %a,
; GFX11-FAKE16-NEXT: s_and_b32 s4, s4, exec_lo
; GFX11-FAKE16-NEXT: s_cselect_b32 s4, 1, 0
; GFX11-FAKE16-NEXT: s_cmp_lg_u32 s4, 1
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_cbranch_scc1 .LBB83_4
; GFX11-FAKE16-NEXT: ; %bb.3: ; %cmp.true
; GFX11-FAKE16-NEXT: s_lshl_b32 s4, s3, 16
@@ -16839,6 +16935,7 @@ define <16 x i8> @bitcast_v2f64_to_v16i8(<2 x double> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: ; implicit-def: $vgpr14_lo16
; GFX11-TRUE16-NEXT: ; implicit-def: $vgpr15_lo16
; GFX11-TRUE16-NEXT: v_cmpx_ne_u32_e32 0, v4
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execz .LBB84_2
; GFX11-TRUE16-NEXT: ; %bb.1: ; %cmp.false
@@ -16875,6 +16972,7 @@ define <16 x i8> @bitcast_v2f64_to_v16i8(<2 x double> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v1, 8, v16
; GFX11-TRUE16-NEXT: .LBB84_4: ; %end
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v0.l, v16.l
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v4.l, v17.l
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v8.l, v11.l
@@ -16900,6 +16998,7 @@ define <16 x i8> @bitcast_v2f64_to_v16i8(<2 x double> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: ; implicit-def: $vgpr14
; GFX11-FAKE16-NEXT: ; implicit-def: $vgpr15
; GFX11-FAKE16-NEXT: v_cmpx_ne_u32_e32 0, v4
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX11-FAKE16-NEXT: s_cbranch_execz .LBB84_2
; GFX11-FAKE16-NEXT: ; %bb.1: ; %cmp.false
@@ -16936,6 +17035,7 @@ define <16 x i8> @bitcast_v2f64_to_v16i8(<2 x double> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v1, 8, v18
; GFX11-FAKE16-NEXT: .LBB84_4: ; %end
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v0, v18
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v4, v19
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v8, v16
@@ -17237,6 +17337,7 @@ define inreg <16 x i8> @bitcast_v2f64_to_v16i8_scalar(<2 x double> inreg %a, i32
; GFX11-NEXT: s_and_b32 s5, s8, exec_lo
; GFX11-NEXT: s_cselect_b32 s5, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s5, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB85_5
; GFX11-NEXT: ; %bb.4: ; %cmp.true
; GFX11-NEXT: v_add_f64 v[16:17], s[2:3], 1.0
@@ -17606,6 +17707,7 @@ define <2 x double> @bitcast_v16i8_to_v2f64(<16 x i8> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-TRUE16-NEXT: ; implicit-def: $vgpr0_vgpr1_vgpr2_vgpr3
; GFX11-TRUE16-NEXT: v_cmpx_ne_u32_e32 0, v16
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB86_3
; GFX11-TRUE16-NEXT: ; %bb.1: ; %Flow
@@ -17698,6 +17800,7 @@ define <2 x double> @bitcast_v16i8_to_v2f64(<16 x i8> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-FAKE16-NEXT: ; implicit-def: $vgpr0_vgpr1_vgpr2_vgpr3
; GFX11-FAKE16-NEXT: v_cmpx_ne_u32_e32 0, v16
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB86_3
; GFX11-FAKE16-NEXT: ; %bb.1: ; %Flow
@@ -18135,6 +18238,7 @@ define inreg <2 x double> @bitcast_v16i8_to_v2f64_scalar(<16 x i8> inreg %a, i32
; GFX11-NEXT: s_and_b32 s4, s4, exec_lo
; GFX11-NEXT: s_cselect_b32 s4, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s4, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB87_5
; GFX11-NEXT: ; %bb.4: ; %cmp.true
; GFX11-NEXT: v_mov_b32_e32 v0, 0xc0c0004
@@ -18323,8 +18427,9 @@ define <8 x half> @bitcast_v8i16_to_v8f16(<8 x i16> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v4
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_pk_add_u16 v3, v3, 3 op_sel_hi:[1,0]
@@ -18519,6 +18624,7 @@ define inreg <8 x half> @bitcast_v8i16_to_v8f16_scalar(<8 x i16> inreg %a, i32 i
; GFX11-NEXT: s_and_b32 s4, s4, exec_lo
; GFX11-NEXT: s_cselect_b32 s4, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s4, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB89_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_pk_add_u16 v3, s3, 3 op_sel_hi:[1,0]
@@ -18659,8 +18765,9 @@ define <8 x i16> @bitcast_v8f16_to_v8i16(<8 x half> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v4
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_pk_add_f16 v3, 0x200, v3 op_sel_hi:[0,1]
@@ -18854,6 +18961,7 @@ define inreg <8 x i16> @bitcast_v8f16_to_v8i16_scalar(<8 x half> inreg %a, i32 i
; GFX11-NEXT: s_and_b32 s4, s4, exec_lo
; GFX11-NEXT: s_cselect_b32 s4, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s4, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB91_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_pk_add_f16 v3, 0x200, s3 op_sel_hi:[0,1]
@@ -19006,8 +19114,9 @@ define <8 x bfloat> @bitcast_v8i16_to_v8bf16(<8 x i16> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v4
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_pk_add_u16 v3, v3, 3 op_sel_hi:[1,0]
@@ -19202,6 +19311,7 @@ define inreg <8 x bfloat> @bitcast_v8i16_to_v8bf16_scalar(<8 x i16> inreg %a, i3
; GFX11-NEXT: s_and_b32 s4, s4, exec_lo
; GFX11-NEXT: s_cselect_b32 s4, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s4, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB93_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_pk_add_u16 v3, s3, 3 op_sel_hi:[1,0]
@@ -19492,8 +19602,9 @@ define <8 x i16> @bitcast_v8bf16_to_v8i16(<8 x bfloat> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-TRUE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-TRUE16-NEXT: v_cmpx_ne_u32_e32 0, v4
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-TRUE16-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-TRUE16-NEXT: s_cbranch_execz .LBB94_2
; GFX11-TRUE16-NEXT: ; %bb.1: ; %cmp.true
@@ -19589,8 +19700,9 @@ define <8 x i16> @bitcast_v8bf16_to_v8i16(<8 x bfloat> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-FAKE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-FAKE16-NEXT: v_cmpx_ne_u32_e32 0, v4
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-FAKE16-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-FAKE16-NEXT: s_cbranch_execz .LBB94_2
; GFX11-FAKE16-NEXT: ; %bb.1: ; %cmp.true
@@ -19980,6 +20092,7 @@ define inreg <8 x i16> @bitcast_v8bf16_to_v8i16_scalar(<8 x bfloat> inreg %a, i3
; GFX11-TRUE16-NEXT: s_and_b32 s4, s4, exec_lo
; GFX11-TRUE16-NEXT: s_cselect_b32 s4, 1, 0
; GFX11-TRUE16-NEXT: s_cmp_lg_u32 s4, 1
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_cbranch_scc1 .LBB95_4
; GFX11-TRUE16-NEXT: ; %bb.3: ; %cmp.true
; GFX11-TRUE16-NEXT: s_pack_lh_b32_b16 s4, 0, s0
@@ -20070,6 +20183,7 @@ define inreg <8 x i16> @bitcast_v8bf16_to_v8i16_scalar(<8 x bfloat> inreg %a, i3
; GFX11-FAKE16-NEXT: s_and_b32 s4, s4, exec_lo
; GFX11-FAKE16-NEXT: s_cselect_b32 s4, 1, 0
; GFX11-FAKE16-NEXT: s_cmp_lg_u32 s4, 1
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_cbranch_scc1 .LBB95_4
; GFX11-FAKE16-NEXT: ; %bb.3: ; %cmp.true
; GFX11-FAKE16-NEXT: s_pack_lh_b32_b16 s4, 0, s0
@@ -20437,6 +20551,7 @@ define <16 x i8> @bitcast_v8i16_to_v16i8(<8 x i16> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: ; implicit-def: $vgpr14_lo16
; GFX11-TRUE16-NEXT: ; implicit-def: $vgpr15_lo16
; GFX11-TRUE16-NEXT: v_cmpx_ne_u32_e32 0, v4
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execz .LBB96_2
; GFX11-TRUE16-NEXT: ; %bb.1: ; %cmp.false
@@ -20476,6 +20591,7 @@ define <16 x i8> @bitcast_v8i16_to_v16i8(<8 x i16> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v1, 8, v16
; GFX11-TRUE16-NEXT: .LBB96_4: ; %end
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v0.l, v16.l
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v4.l, v17.l
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v8.l, v11.l
@@ -20501,6 +20617,7 @@ define <16 x i8> @bitcast_v8i16_to_v16i8(<8 x i16> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: ; implicit-def: $vgpr14
; GFX11-FAKE16-NEXT: ; implicit-def: $vgpr15
; GFX11-FAKE16-NEXT: v_cmpx_ne_u32_e32 0, v4
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX11-FAKE16-NEXT: s_cbranch_execz .LBB96_2
; GFX11-FAKE16-NEXT: ; %bb.1: ; %cmp.false
@@ -20540,6 +20657,7 @@ define <16 x i8> @bitcast_v8i16_to_v16i8(<8 x i16> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v1, 8, v18
; GFX11-FAKE16-NEXT: .LBB96_4: ; %end
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v0, v18
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v4, v19
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v8, v16
@@ -20879,6 +20997,7 @@ define inreg <16 x i8> @bitcast_v8i16_to_v16i8_scalar(<8 x i16> inreg %a, i32 in
; GFX11-NEXT: s_and_b32 s5, s8, exec_lo
; GFX11-NEXT: s_cselect_b32 s5, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s5, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB97_5
; GFX11-NEXT: ; %bb.4: ; %cmp.true
; GFX11-NEXT: v_pk_add_u16 v19, s1, 3 op_sel_hi:[1,0]
@@ -21272,6 +21391,7 @@ define <8 x i16> @bitcast_v16i8_to_v8i16(<16 x i8> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-TRUE16-NEXT: ; implicit-def: $vgpr0_vgpr1_vgpr2_vgpr3
; GFX11-TRUE16-NEXT: v_cmpx_ne_u32_e32 0, v16
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB98_3
; GFX11-TRUE16-NEXT: ; %bb.1: ; %Flow
@@ -21364,6 +21484,7 @@ define <8 x i16> @bitcast_v16i8_to_v8i16(<16 x i8> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-FAKE16-NEXT: ; implicit-def: $vgpr0_vgpr1_vgpr2_vgpr3
; GFX11-FAKE16-NEXT: v_cmpx_ne_u32_e32 0, v16
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB98_3
; GFX11-FAKE16-NEXT: ; %bb.1: ; %Flow
@@ -21828,6 +21949,7 @@ define inreg <8 x i16> @bitcast_v16i8_to_v8i16_scalar(<16 x i8> inreg %a, i32 in
; GFX11-NEXT: s_and_b32 s4, s4, exec_lo
; GFX11-NEXT: s_cselect_b32 s4, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s4, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB99_5
; GFX11-NEXT: ; %bb.4: ; %cmp.true
; GFX11-NEXT: v_mov_b32_e32 v0, 0xc0c0004
@@ -22027,8 +22149,9 @@ define <8 x bfloat> @bitcast_v8f16_to_v8bf16(<8 x half> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v4
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_pk_add_f16 v3, 0x200, v3 op_sel_hi:[0,1]
@@ -22240,6 +22363,7 @@ define inreg <8 x bfloat> @bitcast_v8f16_to_v8bf16_scalar(<8 x half> inreg %a, i
; GFX11-NEXT: s_and_b32 s4, s4, exec_lo
; GFX11-NEXT: s_cselect_b32 s4, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s4, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB101_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_pk_add_f16 v3, 0x200, s3 op_sel_hi:[0,1]
@@ -22535,8 +22659,9 @@ define <8 x half> @bitcast_v8bf16_to_v8f16(<8 x bfloat> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-TRUE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-TRUE16-NEXT: v_cmpx_ne_u32_e32 0, v4
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-TRUE16-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-TRUE16-NEXT: s_cbranch_execz .LBB102_2
; GFX11-TRUE16-NEXT: ; %bb.1: ; %cmp.true
@@ -22631,8 +22756,9 @@ define <8 x half> @bitcast_v8bf16_to_v8f16(<8 x bfloat> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-FAKE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-FAKE16-NEXT: v_cmpx_ne_u32_e32 0, v4
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-FAKE16-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-FAKE16-NEXT: s_cbranch_execz .LBB102_2
; GFX11-FAKE16-NEXT: ; %bb.1: ; %cmp.true
@@ -23027,6 +23153,7 @@ define inreg <8 x half> @bitcast_v8bf16_to_v8f16_scalar(<8 x bfloat> inreg %a, i
; GFX11-TRUE16-NEXT: s_and_b32 s4, s4, exec_lo
; GFX11-TRUE16-NEXT: s_cselect_b32 s4, 1, 0
; GFX11-TRUE16-NEXT: s_cmp_lg_u32 s4, 1
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_cbranch_scc1 .LBB103_4
; GFX11-TRUE16-NEXT: ; %bb.3: ; %cmp.true
; GFX11-TRUE16-NEXT: s_pack_lh_b32_b16 s4, 0, s0
@@ -23128,6 +23255,7 @@ define inreg <8 x half> @bitcast_v8bf16_to_v8f16_scalar(<8 x bfloat> inreg %a, i
; GFX11-FAKE16-NEXT: s_and_b32 s4, s4, exec_lo
; GFX11-FAKE16-NEXT: s_cselect_b32 s4, 1, 0
; GFX11-FAKE16-NEXT: s_cmp_lg_u32 s4, 1
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_cbranch_scc1 .LBB103_4
; GFX11-FAKE16-NEXT: ; %bb.3: ; %cmp.true
; GFX11-FAKE16-NEXT: s_pack_lh_b32_b16 s4, 0, s0
@@ -23507,6 +23635,7 @@ define <16 x i8> @bitcast_v8f16_to_v16i8(<8 x half> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: ; implicit-def: $vgpr14_lo16
; GFX11-TRUE16-NEXT: ; implicit-def: $vgpr15_lo16
; GFX11-TRUE16-NEXT: v_cmpx_ne_u32_e32 0, v4
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execz .LBB104_2
; GFX11-TRUE16-NEXT: ; %bb.1: ; %cmp.false
@@ -23546,6 +23675,7 @@ define <16 x i8> @bitcast_v8f16_to_v16i8(<8 x half> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v1, 8, v16
; GFX11-TRUE16-NEXT: .LBB104_4: ; %end
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v0.l, v16.l
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v4.l, v17.l
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v8.l, v11.l
@@ -23571,6 +23701,7 @@ define <16 x i8> @bitcast_v8f16_to_v16i8(<8 x half> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: ; implicit-def: $vgpr14
; GFX11-FAKE16-NEXT: ; implicit-def: $vgpr15
; GFX11-FAKE16-NEXT: v_cmpx_ne_u32_e32 0, v4
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX11-FAKE16-NEXT: s_cbranch_execz .LBB104_2
; GFX11-FAKE16-NEXT: ; %bb.1: ; %cmp.false
@@ -23610,6 +23741,7 @@ define <16 x i8> @bitcast_v8f16_to_v16i8(<8 x half> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v1, 8, v18
; GFX11-FAKE16-NEXT: .LBB104_4: ; %end
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v0, v18
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v4, v19
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v8, v16
@@ -23969,6 +24101,7 @@ define inreg <16 x i8> @bitcast_v8f16_to_v16i8_scalar(<8 x half> inreg %a, i32 i
; GFX11-NEXT: s_and_b32 s5, s8, exec_lo
; GFX11-NEXT: s_cselect_b32 s5, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s5, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB105_5
; GFX11-NEXT: ; %bb.4: ; %cmp.true
; GFX11-NEXT: v_pk_add_f16 v19, 0x200, s1 op_sel_hi:[0,1]
@@ -24362,6 +24495,7 @@ define <8 x half> @bitcast_v16i8_to_v8f16(<16 x i8> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-TRUE16-NEXT: ; implicit-def: $vgpr0_vgpr1_vgpr2_vgpr3
; GFX11-TRUE16-NEXT: v_cmpx_ne_u32_e32 0, v16
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB106_3
; GFX11-TRUE16-NEXT: ; %bb.1: ; %Flow
@@ -24454,6 +24588,7 @@ define <8 x half> @bitcast_v16i8_to_v8f16(<16 x i8> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-FAKE16-NEXT: ; implicit-def: $vgpr0_vgpr1_vgpr2_vgpr3
; GFX11-FAKE16-NEXT: v_cmpx_ne_u32_e32 0, v16
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB106_3
; GFX11-FAKE16-NEXT: ; %bb.1: ; %Flow
@@ -24918,6 +25053,7 @@ define inreg <8 x half> @bitcast_v16i8_to_v8f16_scalar(<16 x i8> inreg %a, i32 i
; GFX11-NEXT: s_and_b32 s4, s4, exec_lo
; GFX11-NEXT: s_cselect_b32 s4, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s4, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB107_5
; GFX11-NEXT: ; %bb.4: ; %cmp.true
; GFX11-NEXT: v_mov_b32_e32 v0, 0xc0c0004
@@ -25368,6 +25504,7 @@ define <16 x i8> @bitcast_v8bf16_to_v16i8(<8 x bfloat> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: ; implicit-def: $vgpr14_lo16
; GFX11-TRUE16-NEXT: ; implicit-def: $vgpr15_lo16
; GFX11-TRUE16-NEXT: v_cmpx_ne_u32_e32 0, v4
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execz .LBB108_2
; GFX11-TRUE16-NEXT: ; %bb.1: ; %cmp.false
@@ -25476,6 +25613,7 @@ define <16 x i8> @bitcast_v8bf16_to_v16i8(<8 x bfloat> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v9, 8, v11
; GFX11-TRUE16-NEXT: .LBB108_4: ; %end
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v0.l, v16.l
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v4.l, v17.l
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v8.l, v11.l
@@ -25501,6 +25639,7 @@ define <16 x i8> @bitcast_v8bf16_to_v16i8(<8 x bfloat> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: ; implicit-def: $vgpr14
; GFX11-FAKE16-NEXT: ; implicit-def: $vgpr15
; GFX11-FAKE16-NEXT: v_cmpx_ne_u32_e32 0, v4
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX11-FAKE16-NEXT: s_cbranch_execz .LBB108_2
; GFX11-FAKE16-NEXT: ; %bb.1: ; %cmp.false
@@ -25609,6 +25748,7 @@ define <16 x i8> @bitcast_v8bf16_to_v16i8(<8 x bfloat> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v1, 8, v0
; GFX11-FAKE16-NEXT: .LBB108_4: ; %end
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v0, v18
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v4, v19
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v8, v16
@@ -26082,6 +26222,7 @@ define inreg <16 x i8> @bitcast_v8bf16_to_v16i8_scalar(<8 x bfloat> inreg %a, i3
; GFX11-TRUE16-NEXT: s_and_b32 s5, s8, exec_lo
; GFX11-TRUE16-NEXT: s_cselect_b32 s5, 1, 0
; GFX11-TRUE16-NEXT: s_cmp_lg_u32 s5, 1
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_cbranch_scc1 .LBB109_5
; GFX11-TRUE16-NEXT: ; %bb.4: ; %cmp.true
; GFX11-TRUE16-NEXT: s_lshl_b32 s4, s1, 16
@@ -26238,6 +26379,7 @@ define inreg <16 x i8> @bitcast_v8bf16_to_v16i8_scalar(<8 x bfloat> inreg %a, i3
; GFX11-FAKE16-NEXT: s_and_b32 s5, s8, exec_lo
; GFX11-FAKE16-NEXT: s_cselect_b32 s5, 1, 0
; GFX11-FAKE16-NEXT: s_cmp_lg_u32 s5, 1
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_cbranch_scc1 .LBB109_5
; GFX11-FAKE16-NEXT: ; %bb.4: ; %cmp.true
; GFX11-FAKE16-NEXT: s_lshl_b32 s4, s1, 16
@@ -26707,6 +26849,7 @@ define <8 x bfloat> @bitcast_v16i8_to_v8bf16(<16 x i8> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-TRUE16-NEXT: ; implicit-def: $vgpr0_vgpr1_vgpr2_vgpr3
; GFX11-TRUE16-NEXT: v_cmpx_ne_u32_e32 0, v16
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB110_3
; GFX11-TRUE16-NEXT: ; %bb.1: ; %Flow
@@ -26799,6 +26942,7 @@ define <8 x bfloat> @bitcast_v16i8_to_v8bf16(<16 x i8> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-FAKE16-NEXT: ; implicit-def: $vgpr0_vgpr1_vgpr2_vgpr3
; GFX11-FAKE16-NEXT: v_cmpx_ne_u32_e32 0, v16
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB110_3
; GFX11-FAKE16-NEXT: ; %bb.1: ; %Flow
@@ -27259,6 +27403,7 @@ define inreg <8 x bfloat> @bitcast_v16i8_to_v8bf16_scalar(<16 x i8> inreg %a, i3
; GFX11-NEXT: s_and_b32 s4, s4, exec_lo
; GFX11-NEXT: s_cselect_b32 s4, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s4, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB111_5
; GFX11-NEXT: ; %bb.4: ; %cmp.true
; GFX11-NEXT: v_mov_b32_e32 v0, 0xc0c0004
diff --git a/llvm/test/CodeGen/AMDGPU/amdgcn.bitcast.160bit.ll b/llvm/test/CodeGen/AMDGPU/amdgcn.bitcast.160bit.ll
index 50037fb8e0a3ab..572b737b6183c2 100644
--- a/llvm/test/CodeGen/AMDGPU/amdgcn.bitcast.160bit.ll
+++ b/llvm/test/CodeGen/AMDGPU/amdgcn.bitcast.160bit.ll
@@ -63,8 +63,9 @@ define <5 x float> @bitcast_v5i32_to_v5f32(<5 x i32> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v5
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_add_nc_u32_e32 v4, 3, v4
@@ -193,6 +194,7 @@ define inreg <5 x float> @bitcast_v5i32_to_v5f32_scalar(<5 x i32> inreg %a, i32
; GFX11-NEXT: s_and_b32 s4, s4, exec_lo
; GFX11-NEXT: s_cselect_b32 s4, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s4, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB1_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: s_add_i32 s16, s16, 3
@@ -280,8 +282,9 @@ define <5 x i32> @bitcast_v5f32_to_v5i32(<5 x float> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v5
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_dual_add_f32 v4, 1.0, v4 :: v_dual_add_f32 v3, 1.0, v3
@@ -412,6 +415,7 @@ define inreg <5 x i32> @bitcast_v5f32_to_v5i32_scalar(<5 x float> inreg %a, i32
; GFX11-NEXT: s_and_b32 s5, s5, exec_lo
; GFX11-NEXT: s_cselect_b32 s5, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s5, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB3_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_add_f32_e64 v4, s4, 1.0
@@ -532,8 +536,9 @@ define <10 x i16> @bitcast_v5i32_to_v10i16(<5 x i32> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v5
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_add_nc_u32_e32 v4, 3, v4
@@ -692,6 +697,7 @@ define inreg <10 x i16> @bitcast_v5i32_to_v10i16_scalar(<5 x i32> inreg %a, i32
; GFX11-NEXT: s_and_b32 s4, s4, exec_lo
; GFX11-NEXT: s_cselect_b32 s4, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s4, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB5_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: s_add_i32 s16, s16, 3
@@ -851,8 +857,9 @@ define <5 x i32> @bitcast_v10i16_to_v5i32(<10 x i16> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v5
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_pk_add_u16 v4, v4, 3 op_sel_hi:[1,0]
@@ -1044,6 +1051,7 @@ define inreg <5 x i32> @bitcast_v10i16_to_v5i32_scalar(<10 x i16> inreg %a, i32
; GFX11-NEXT: s_and_b32 s5, s5, exec_lo
; GFX11-NEXT: s_cselect_b32 s5, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s5, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB7_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_pk_add_u16 v4, s4, 3 op_sel_hi:[1,0]
@@ -1164,8 +1172,9 @@ define <10 x half> @bitcast_v5i32_to_v10f16(<5 x i32> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v5
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_add_nc_u32_e32 v4, 3, v4
@@ -1324,6 +1333,7 @@ define inreg <10 x half> @bitcast_v5i32_to_v10f16_scalar(<5 x i32> inreg %a, i32
; GFX11-NEXT: s_and_b32 s4, s4, exec_lo
; GFX11-NEXT: s_cselect_b32 s4, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s4, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB9_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: s_add_i32 s16, s16, 3
@@ -1503,8 +1513,9 @@ define <5 x i32> @bitcast_v10f16_to_v5i32(<10 x half> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v5
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_pk_add_f16 v4, 0x200, v4 op_sel_hi:[0,1]
@@ -1715,6 +1726,7 @@ define inreg <5 x i32> @bitcast_v10f16_to_v5i32_scalar(<10 x half> inreg %a, i32
; GFX11-NEXT: s_and_b32 s5, s5, exec_lo
; GFX11-NEXT: s_cselect_b32 s5, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s5, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB11_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_pk_add_f16 v4, 0x200, s4 op_sel_hi:[0,1]
@@ -1835,8 +1847,9 @@ define <10 x i16> @bitcast_v5f32_to_v10i16(<5 x float> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v5
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_dual_add_f32 v4, 1.0, v4 :: v_dual_add_f32 v3, 1.0, v3
@@ -2009,6 +2022,7 @@ define inreg <10 x i16> @bitcast_v5f32_to_v10i16_scalar(<5 x float> inreg %a, i3
; GFX11-NEXT: s_and_b32 s8, s8, exec_lo
; GFX11-NEXT: s_cselect_b32 s8, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s8, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB13_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_add_f32_e64 v4, s4, 1.0
@@ -2169,8 +2183,9 @@ define <5 x float> @bitcast_v10i16_to_v5f32(<10 x i16> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v5
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_pk_add_u16 v4, v4, 3 op_sel_hi:[1,0]
@@ -2362,6 +2377,7 @@ define inreg <5 x float> @bitcast_v10i16_to_v5f32_scalar(<10 x i16> inreg %a, i3
; GFX11-NEXT: s_and_b32 s5, s5, exec_lo
; GFX11-NEXT: s_cselect_b32 s5, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s5, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB15_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_pk_add_u16 v4, s4, 3 op_sel_hi:[1,0]
@@ -2482,8 +2498,9 @@ define <10 x half> @bitcast_v5f32_to_v10f16(<5 x float> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v5
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_dual_add_f32 v4, 1.0, v4 :: v_dual_add_f32 v3, 1.0, v3
@@ -2656,6 +2673,7 @@ define inreg <10 x half> @bitcast_v5f32_to_v10f16_scalar(<5 x float> inreg %a, i
; GFX11-NEXT: s_and_b32 s8, s8, exec_lo
; GFX11-NEXT: s_cselect_b32 s8, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s8, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB17_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_add_f32_e64 v4, s4, 1.0
@@ -2836,8 +2854,9 @@ define <5 x float> @bitcast_v10f16_to_v5f32(<10 x half> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v5
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_pk_add_f16 v4, 0x200, v4 op_sel_hi:[0,1]
@@ -3048,6 +3067,7 @@ define inreg <5 x float> @bitcast_v10f16_to_v5f32_scalar(<10 x half> inreg %a, i
; GFX11-NEXT: s_and_b32 s5, s5, exec_lo
; GFX11-NEXT: s_cselect_b32 s5, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s5, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB19_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_pk_add_f16 v4, 0x200, s4 op_sel_hi:[0,1]
@@ -3226,8 +3246,9 @@ define <10 x half> @bitcast_v10i16_to_v10f16(<10 x i16> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v5
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_pk_add_u16 v4, v4, 3 op_sel_hi:[1,0]
@@ -3450,6 +3471,7 @@ define inreg <10 x half> @bitcast_v10i16_to_v10f16_scalar(<10 x i16> inreg %a, i
; GFX11-NEXT: s_and_b32 s8, s8, exec_lo
; GFX11-NEXT: s_cselect_b32 s8, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s8, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB21_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_pk_add_u16 v4, s4, 3 op_sel_hi:[1,0]
@@ -3609,8 +3631,9 @@ define <10 x i16> @bitcast_v10f16_to_v10i16(<10 x half> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v5
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_pk_add_f16 v4, 0x200, v4 op_sel_hi:[0,1]
@@ -3834,6 +3857,7 @@ define inreg <10 x i16> @bitcast_v10f16_to_v10i16_scalar(<10 x half> inreg %a, i
; GFX11-NEXT: s_and_b32 s8, s8, exec_lo
; GFX11-NEXT: s_cselect_b32 s8, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s8, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB23_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_pk_add_f16 v4, 0x200, s4 op_sel_hi:[0,1]
diff --git a/llvm/test/CodeGen/AMDGPU/amdgcn.bitcast.16bit.ll b/llvm/test/CodeGen/AMDGPU/amdgcn.bitcast.16bit.ll
index c7cd6ebeca16e0..aa87b158943f08 100644
--- a/llvm/test/CodeGen/AMDGPU/amdgcn.bitcast.16bit.ll
+++ b/llvm/test/CodeGen/AMDGPU/amdgcn.bitcast.16bit.ll
@@ -65,7 +65,7 @@ define half @bitcast_i16_to_f16(i16 %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: v_add_nc_u16 v1.l, v0.l, 3
; GFX11-TRUE16-NEXT: .LBB0_4: ; %end
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v0.l, v1.l
; GFX11-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -74,8 +74,9 @@ define half @bitcast_i16_to_f16(i16 %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-FAKE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-FAKE16-NEXT: v_cmpx_ne_u32_e32 0, v1
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-FAKE16-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-FAKE16-NEXT: ; %bb.1: ; %cmp.true
; GFX11-FAKE16-NEXT: v_add_nc_u16 v0, v0, 3
@@ -177,6 +178,7 @@ define inreg half @bitcast_i16_to_f16_scalar(i16 inreg %a, i32 inreg %b) #0 {
; GFX11-NEXT: s_and_b32 s1, s1, exec_lo
; GFX11-NEXT: s_cselect_b32 s1, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s1, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB1_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: s_add_i32 s0, s0, 3
@@ -273,7 +275,7 @@ define i16 @bitcast_f16_to_i16(half %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: v_add_f16_e32 v1.l, 0x200, v0.l
; GFX11-TRUE16-NEXT: .LBB2_4: ; %end
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v0.l, v1.l
; GFX11-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -282,8 +284,9 @@ define i16 @bitcast_f16_to_i16(half %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-FAKE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-FAKE16-NEXT: v_cmpx_ne_u32_e32 0, v1
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-FAKE16-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-FAKE16-NEXT: ; %bb.1: ; %cmp.true
; GFX11-FAKE16-NEXT: v_add_f16_e32 v0, 0x200, v0
@@ -393,6 +396,7 @@ define inreg i16 @bitcast_f16_to_i16_scalar(half inreg %a, i32 inreg %b) #0 {
; GFX11-TRUE16-NEXT: s_and_b32 s1, s1, exec_lo
; GFX11-TRUE16-NEXT: s_cselect_b32 s1, 1, 0
; GFX11-TRUE16-NEXT: s_cmp_lg_u32 s1, 1
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_cbranch_scc1 .LBB3_4
; GFX11-TRUE16-NEXT: ; %bb.3: ; %cmp.true
; GFX11-TRUE16-NEXT: v_add_f16_e64 v0.l, 0x200, s0
@@ -414,6 +418,7 @@ define inreg i16 @bitcast_f16_to_i16_scalar(half inreg %a, i32 inreg %b) #0 {
; GFX11-FAKE16-NEXT: s_and_b32 s1, s1, exec_lo
; GFX11-FAKE16-NEXT: s_cselect_b32 s1, 1, 0
; GFX11-FAKE16-NEXT: s_cmp_lg_u32 s1, 1
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_cbranch_scc1 .LBB3_4
; GFX11-FAKE16-NEXT: ; %bb.3: ; %cmp.true
; GFX11-FAKE16-NEXT: v_add_f16_e64 v0, 0x200, s0
@@ -500,7 +505,7 @@ define bfloat @bitcast_i16_to_bf16(i16 %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: v_add_nc_u16 v1.l, v0.l, 3
; GFX11-TRUE16-NEXT: .LBB4_4: ; %end
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v0.l, v1.l
; GFX11-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -509,8 +514,9 @@ define bfloat @bitcast_i16_to_bf16(i16 %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-FAKE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-FAKE16-NEXT: v_cmpx_ne_u32_e32 0, v1
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-FAKE16-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-FAKE16-NEXT: ; %bb.1: ; %cmp.true
; GFX11-FAKE16-NEXT: v_add_nc_u16 v0, v0, 3
@@ -616,6 +622,7 @@ define inreg bfloat @bitcast_i16_to_bf16_scalar(i16 inreg %a, i32 inreg %b) #0 {
; GFX11-NEXT: s_and_b32 s1, s1, exec_lo
; GFX11-NEXT: s_cselect_b32 s1, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s1, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB5_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: s_add_i32 s0, s0, 3
@@ -718,6 +725,7 @@ define i16 @bitcast_bf16_to_i16(bfloat %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-TRUE16-NEXT: ; implicit-def: $vgpr0_lo16
; GFX11-TRUE16-NEXT: v_cmpx_ne_u32_e32 0, v1
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execz .LBB6_2
; GFX11-TRUE16-NEXT: ; %bb.1: ; %cmp.false
@@ -750,8 +758,9 @@ define i16 @bitcast_bf16_to_i16(bfloat %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-FAKE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-FAKE16-NEXT: v_cmpx_ne_u32_e32 0, v1
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-FAKE16-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-FAKE16-NEXT: s_cbranch_execz .LBB6_2
; GFX11-FAKE16-NEXT: ; %bb.1: ; %cmp.true
@@ -888,6 +897,7 @@ define inreg i16 @bitcast_bf16_to_i16_scalar(bfloat inreg %a, i32 inreg %b) #0 {
; GFX11-NEXT: s_and_b32 s1, s1, exec_lo
; GFX11-NEXT: s_cselect_b32 s1, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s1, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB7_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: s_lshl_b32 s0, s0, 16
@@ -992,7 +1002,7 @@ define bfloat @bitcast_f16_to_bf16(half %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: v_add_f16_e32 v1.l, 0x200, v0.l
; GFX11-TRUE16-NEXT: .LBB8_4: ; %end
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v0.l, v1.l
; GFX11-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -1001,8 +1011,9 @@ define bfloat @bitcast_f16_to_bf16(half %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-FAKE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-FAKE16-NEXT: v_cmpx_ne_u32_e32 0, v1
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-FAKE16-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-FAKE16-NEXT: ; %bb.1: ; %cmp.true
; GFX11-FAKE16-NEXT: v_add_f16_e32 v0, 0x200, v0
@@ -1116,6 +1127,7 @@ define inreg bfloat @bitcast_f16_to_bf16_scalar(half inreg %a, i32 inreg %b) #0
; GFX11-TRUE16-NEXT: s_and_b32 s1, s1, exec_lo
; GFX11-TRUE16-NEXT: s_cselect_b32 s1, 1, 0
; GFX11-TRUE16-NEXT: s_cmp_lg_u32 s1, 1
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_cbranch_scc1 .LBB9_4
; GFX11-TRUE16-NEXT: ; %bb.3: ; %cmp.true
; GFX11-TRUE16-NEXT: v_add_f16_e64 v0.l, 0x200, s0
@@ -1137,6 +1149,7 @@ define inreg bfloat @bitcast_f16_to_bf16_scalar(half inreg %a, i32 inreg %b) #0
; GFX11-FAKE16-NEXT: s_and_b32 s1, s1, exec_lo
; GFX11-FAKE16-NEXT: s_cselect_b32 s1, 1, 0
; GFX11-FAKE16-NEXT: s_cmp_lg_u32 s1, 1
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_cbranch_scc1 .LBB9_4
; GFX11-FAKE16-NEXT: ; %bb.3: ; %cmp.true
; GFX11-FAKE16-NEXT: v_add_f16_e64 v0, 0x200, s0
@@ -1239,6 +1252,7 @@ define half @bitcast_bf16_to_f16(bfloat %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-TRUE16-NEXT: ; implicit-def: $vgpr0_lo16
; GFX11-TRUE16-NEXT: v_cmpx_ne_u32_e32 0, v1
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execz .LBB10_2
; GFX11-TRUE16-NEXT: ; %bb.1: ; %cmp.false
@@ -1271,8 +1285,9 @@ define half @bitcast_bf16_to_f16(bfloat %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-FAKE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-FAKE16-NEXT: v_cmpx_ne_u32_e32 0, v1
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-FAKE16-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-FAKE16-NEXT: s_cbranch_execz .LBB10_2
; GFX11-FAKE16-NEXT: ; %bb.1: ; %cmp.true
@@ -1409,6 +1424,7 @@ define inreg half @bitcast_bf16_to_f16_scalar(bfloat inreg %a, i32 inreg %b) #0
; GFX11-NEXT: s_and_b32 s1, s1, exec_lo
; GFX11-NEXT: s_cselect_b32 s1, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s1, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB11_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: s_lshl_b32 s0, s0, 16
diff --git a/llvm/test/CodeGen/AMDGPU/amdgcn.bitcast.192bit.ll b/llvm/test/CodeGen/AMDGPU/amdgcn.bitcast.192bit.ll
index b017b8ea65af55..be1c0b23f77ba2 100644
--- a/llvm/test/CodeGen/AMDGPU/amdgcn.bitcast.192bit.ll
+++ b/llvm/test/CodeGen/AMDGPU/amdgcn.bitcast.192bit.ll
@@ -66,8 +66,9 @@ define <6 x float> @bitcast_v6i32_to_v6f32(<6 x i32> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v6
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_add_nc_u32_e32 v5, 3, v5
@@ -203,6 +204,7 @@ define inreg <6 x float> @bitcast_v6i32_to_v6f32_scalar(<6 x i32> inreg %a, i32
; GFX11-NEXT: s_and_b32 s4, s4, exec_lo
; GFX11-NEXT: s_cselect_b32 s4, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s4, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB1_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: s_add_i32 s17, s17, 3
@@ -294,8 +296,9 @@ define <6 x i32> @bitcast_v6f32_to_v6i32(<6 x float> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v6
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_dual_add_f32 v5, 1.0, v5 :: v_dual_add_f32 v4, 1.0, v4
@@ -433,6 +436,7 @@ define inreg <6 x i32> @bitcast_v6f32_to_v6i32_scalar(<6 x float> inreg %a, i32
; GFX11-NEXT: s_and_b32 s6, s6, exec_lo
; GFX11-NEXT: s_cselect_b32 s6, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s6, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB3_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_add_f32_e64 v5, s5, 1.0
@@ -524,8 +528,9 @@ define <3 x i64> @bitcast_v6i32_to_v3i64(<6 x i32> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v6
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_add_nc_u32_e32 v5, 3, v5
@@ -661,6 +666,7 @@ define inreg <3 x i64> @bitcast_v6i32_to_v3i64_scalar(<6 x i32> inreg %a, i32 in
; GFX11-NEXT: s_and_b32 s4, s4, exec_lo
; GFX11-NEXT: s_cselect_b32 s4, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s4, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB5_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: s_add_i32 s17, s17, 3
@@ -752,17 +758,16 @@ define <6 x i32> @bitcast_v3i64_to_v6i32(<3 x i64> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v6
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_add_co_u32 v4, vcc_lo, v4, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v5, null, 0, v5, vcc_lo
; GFX11-NEXT: v_add_co_u32 v2, vcc_lo, v2, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX11-NEXT: v_add_co_u32 v0, vcc_lo, v0, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-NEXT: ; %bb.2: ; %end
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -891,6 +896,7 @@ define inreg <6 x i32> @bitcast_v3i64_to_v6i32_scalar(<3 x i64> inreg %a, i32 in
; GFX11-NEXT: s_and_b32 s4, s4, exec_lo
; GFX11-NEXT: s_cselect_b32 s4, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s4, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB7_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: s_add_u32 s16, s16, 3
@@ -982,8 +988,9 @@ define <3 x double> @bitcast_v6i32_to_v3f64(<6 x i32> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v6
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_add_nc_u32_e32 v5, 3, v5
@@ -1119,6 +1126,7 @@ define inreg <3 x double> @bitcast_v6i32_to_v3f64_scalar(<6 x i32> inreg %a, i32
; GFX11-NEXT: s_and_b32 s4, s4, exec_lo
; GFX11-NEXT: s_cselect_b32 s4, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s4, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB9_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: s_add_i32 s17, s17, 3
@@ -1203,8 +1211,9 @@ define <6 x i32> @bitcast_v3f64_to_v6i32(<3 x double> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v6
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB10_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -1334,6 +1343,7 @@ define inreg <6 x i32> @bitcast_v3f64_to_v6i32_scalar(<3 x double> inreg %a, i32
; GFX11-NEXT: s_and_b32 s6, s6, exec_lo
; GFX11-NEXT: s_cselect_b32 s6, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s6, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB11_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_add_f64 v[4:5], s[4:5], 1.0
@@ -1461,8 +1471,9 @@ define <12 x i16> @bitcast_v6i32_to_v12i16(<6 x i32> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v6
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_add_nc_u32_e32 v5, 3, v5
@@ -1634,6 +1645,7 @@ define inreg <12 x i16> @bitcast_v6i32_to_v12i16_scalar(<6 x i32> inreg %a, i32
; GFX11-NEXT: s_and_b32 s4, s4, exec_lo
; GFX11-NEXT: s_cselect_b32 s4, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s4, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB13_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: s_add_i32 s17, s17, 3
@@ -1809,8 +1821,9 @@ define <6 x i32> @bitcast_v12i16_to_v6i32(<12 x i16> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v6
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_pk_add_u16 v5, v5, 3 op_sel_hi:[1,0]
@@ -2022,6 +2035,7 @@ define inreg <6 x i32> @bitcast_v12i16_to_v6i32_scalar(<12 x i16> inreg %a, i32
; GFX11-NEXT: s_and_b32 s6, s6, exec_lo
; GFX11-NEXT: s_cselect_b32 s6, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s6, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB15_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_pk_add_u16 v5, s5, 3 op_sel_hi:[1,0]
@@ -2152,8 +2166,9 @@ define <12 x half> @bitcast_v6i32_to_v12f16(<6 x i32> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v6
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_add_nc_u32_e32 v5, 3, v5
@@ -2325,6 +2340,7 @@ define inreg <12 x half> @bitcast_v6i32_to_v12f16_scalar(<6 x i32> inreg %a, i32
; GFX11-NEXT: s_and_b32 s4, s4, exec_lo
; GFX11-NEXT: s_cselect_b32 s4, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s4, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB17_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: s_add_i32 s17, s17, 3
@@ -2524,8 +2540,9 @@ define <6 x i32> @bitcast_v12f16_to_v6i32(<12 x half> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v6
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_pk_add_f16 v5, 0x200, v5 op_sel_hi:[0,1]
@@ -2759,6 +2776,7 @@ define inreg <6 x i32> @bitcast_v12f16_to_v6i32_scalar(<12 x half> inreg %a, i32
; GFX11-NEXT: s_and_b32 s6, s6, exec_lo
; GFX11-NEXT: s_cselect_b32 s6, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s6, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB19_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_pk_add_f16 v5, 0x200, s5 op_sel_hi:[0,1]
@@ -2850,8 +2868,9 @@ define <3 x i64> @bitcast_v6f32_to_v3i64(<6 x float> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v6
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_dual_add_f32 v5, 1.0, v5 :: v_dual_add_f32 v4, 1.0, v4
@@ -2989,6 +3008,7 @@ define inreg <3 x i64> @bitcast_v6f32_to_v3i64_scalar(<6 x float> inreg %a, i32
; GFX11-NEXT: s_and_b32 s6, s6, exec_lo
; GFX11-NEXT: s_cselect_b32 s6, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s6, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB21_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_add_f32_e64 v5, s5, 1.0
@@ -3080,17 +3100,16 @@ define <6 x float> @bitcast_v3i64_to_v6f32(<3 x i64> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v6
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_add_co_u32 v4, vcc_lo, v4, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v5, null, 0, v5, vcc_lo
; GFX11-NEXT: v_add_co_u32 v2, vcc_lo, v2, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX11-NEXT: v_add_co_u32 v0, vcc_lo, v0, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-NEXT: ; %bb.2: ; %end
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -3219,6 +3238,7 @@ define inreg <6 x float> @bitcast_v3i64_to_v6f32_scalar(<3 x i64> inreg %a, i32
; GFX11-NEXT: s_and_b32 s4, s4, exec_lo
; GFX11-NEXT: s_cselect_b32 s4, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s4, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB23_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: s_add_u32 s16, s16, 3
@@ -3310,8 +3330,9 @@ define <3 x double> @bitcast_v6f32_to_v3f64(<6 x float> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v6
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_dual_add_f32 v5, 1.0, v5 :: v_dual_add_f32 v4, 1.0, v4
@@ -3449,6 +3470,7 @@ define inreg <3 x double> @bitcast_v6f32_to_v3f64_scalar(<6 x float> inreg %a, i
; GFX11-NEXT: s_and_b32 s6, s6, exec_lo
; GFX11-NEXT: s_cselect_b32 s6, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s6, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB25_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_add_f32_e64 v5, s5, 1.0
@@ -3533,8 +3555,9 @@ define <6 x float> @bitcast_v3f64_to_v6f32(<3 x double> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v6
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB26_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -3664,6 +3687,7 @@ define inreg <6 x float> @bitcast_v3f64_to_v6f32_scalar(<3 x double> inreg %a, i
; GFX11-NEXT: s_and_b32 s6, s6, exec_lo
; GFX11-NEXT: s_cselect_b32 s6, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s6, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB27_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_add_f64 v[4:5], s[4:5], 1.0
@@ -3791,8 +3815,9 @@ define <12 x i16> @bitcast_v6f32_to_v12i16(<6 x float> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v6
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_dual_add_f32 v5, 1.0, v5 :: v_dual_add_f32 v4, 1.0, v4
@@ -3977,6 +4002,7 @@ define inreg <12 x i16> @bitcast_v6f32_to_v12i16_scalar(<6 x float> inreg %a, i3
; GFX11-NEXT: s_and_b32 s8, s8, exec_lo
; GFX11-NEXT: s_cselect_b32 s8, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s8, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB29_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_add_f32_e64 v5, s5, 1.0
@@ -4153,8 +4179,9 @@ define <6 x float> @bitcast_v12i16_to_v6f32(<12 x i16> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v6
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_pk_add_u16 v5, v5, 3 op_sel_hi:[1,0]
@@ -4366,6 +4393,7 @@ define inreg <6 x float> @bitcast_v12i16_to_v6f32_scalar(<12 x i16> inreg %a, i3
; GFX11-NEXT: s_and_b32 s6, s6, exec_lo
; GFX11-NEXT: s_cselect_b32 s6, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s6, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB31_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_pk_add_u16 v5, s5, 3 op_sel_hi:[1,0]
@@ -4496,8 +4524,9 @@ define <12 x half> @bitcast_v6f32_to_v12f16(<6 x float> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v6
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_dual_add_f32 v5, 1.0, v5 :: v_dual_add_f32 v4, 1.0, v4
@@ -4682,6 +4711,7 @@ define inreg <12 x half> @bitcast_v6f32_to_v12f16_scalar(<6 x float> inreg %a, i
; GFX11-NEXT: s_and_b32 s8, s8, exec_lo
; GFX11-NEXT: s_cselect_b32 s8, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s8, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB33_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_add_f32_e64 v5, s5, 1.0
@@ -4882,8 +4912,9 @@ define <6 x float> @bitcast_v12f16_to_v6f32(<12 x half> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v6
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_pk_add_f16 v5, 0x200, v5 op_sel_hi:[0,1]
@@ -5117,6 +5148,7 @@ define inreg <6 x float> @bitcast_v12f16_to_v6f32_scalar(<12 x half> inreg %a, i
; GFX11-NEXT: s_and_b32 s6, s6, exec_lo
; GFX11-NEXT: s_cselect_b32 s6, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s6, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB35_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_pk_add_f16 v5, 0x200, s5 op_sel_hi:[0,1]
@@ -5208,17 +5240,16 @@ define <3 x double> @bitcast_v3i64_to_v3f64(<3 x i64> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v6
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_add_co_u32 v0, vcc_lo, v0, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-NEXT: v_add_co_u32 v2, vcc_lo, v2, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX11-NEXT: v_add_co_u32 v4, vcc_lo, v4, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v5, null, 0, v5, vcc_lo
; GFX11-NEXT: ; %bb.2: ; %end
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -5347,6 +5378,7 @@ define inreg <3 x double> @bitcast_v3i64_to_v3f64_scalar(<3 x i64> inreg %a, i32
; GFX11-NEXT: s_and_b32 s4, s4, exec_lo
; GFX11-NEXT: s_cselect_b32 s4, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s4, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB37_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: s_add_u32 s0, s0, 3
@@ -5430,8 +5462,9 @@ define <3 x i64> @bitcast_v3f64_to_v3i64(<3 x double> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v6
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB38_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -5561,6 +5594,7 @@ define inreg <3 x i64> @bitcast_v3f64_to_v3i64_scalar(<3 x double> inreg %a, i32
; GFX11-NEXT: s_and_b32 s6, s6, exec_lo
; GFX11-NEXT: s_cselect_b32 s6, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s6, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB39_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_add_f64 v[0:1], s[0:1], 1.0
@@ -5688,17 +5722,16 @@ define <12 x i16> @bitcast_v3i64_to_v12i16(<3 x i64> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v6
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_add_co_u32 v4, vcc_lo, v4, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v5, null, 0, v5, vcc_lo
; GFX11-NEXT: v_add_co_u32 v2, vcc_lo, v2, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX11-NEXT: v_add_co_u32 v0, vcc_lo, v0, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-NEXT: ; %bb.2: ; %end
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -5863,6 +5896,7 @@ define inreg <12 x i16> @bitcast_v3i64_to_v12i16_scalar(<3 x i64> inreg %a, i32
; GFX11-NEXT: s_and_b32 s4, s4, exec_lo
; GFX11-NEXT: s_cselect_b32 s4, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s4, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB41_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: s_add_u32 s16, s16, 3
@@ -6038,8 +6072,9 @@ define <3 x i64> @bitcast_v12i16_to_v3i64(<12 x i16> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v6
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_pk_add_u16 v5, v5, 3 op_sel_hi:[1,0]
@@ -6251,6 +6286,7 @@ define inreg <3 x i64> @bitcast_v12i16_to_v3i64_scalar(<12 x i16> inreg %a, i32
; GFX11-NEXT: s_and_b32 s6, s6, exec_lo
; GFX11-NEXT: s_cselect_b32 s6, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s6, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB43_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_pk_add_u16 v5, s5, 3 op_sel_hi:[1,0]
@@ -6381,17 +6417,16 @@ define <12 x half> @bitcast_v3i64_to_v12f16(<3 x i64> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v6
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_add_co_u32 v4, vcc_lo, v4, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v5, null, 0, v5, vcc_lo
; GFX11-NEXT: v_add_co_u32 v2, vcc_lo, v2, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX11-NEXT: v_add_co_u32 v0, vcc_lo, v0, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-NEXT: ; %bb.2: ; %end
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -6556,6 +6591,7 @@ define inreg <12 x half> @bitcast_v3i64_to_v12f16_scalar(<3 x i64> inreg %a, i32
; GFX11-NEXT: s_and_b32 s4, s4, exec_lo
; GFX11-NEXT: s_cselect_b32 s4, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s4, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB45_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: s_add_u32 s16, s16, 3
@@ -6755,8 +6791,9 @@ define <3 x i64> @bitcast_v12f16_to_v3i64(<12 x half> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v6
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_pk_add_f16 v5, 0x200, v5 op_sel_hi:[0,1]
@@ -6990,6 +7027,7 @@ define inreg <3 x i64> @bitcast_v12f16_to_v3i64_scalar(<12 x half> inreg %a, i32
; GFX11-NEXT: s_and_b32 s6, s6, exec_lo
; GFX11-NEXT: s_cselect_b32 s6, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s6, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB47_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_pk_add_f16 v5, 0x200, s5 op_sel_hi:[0,1]
@@ -7113,8 +7151,9 @@ define <12 x i16> @bitcast_v3f64_to_v12i16(<3 x double> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v6
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB48_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -7291,6 +7330,7 @@ define inreg <12 x i16> @bitcast_v3f64_to_v12i16_scalar(<3 x double> inreg %a, i
; GFX11-NEXT: s_and_b32 s8, s8, exec_lo
; GFX11-NEXT: s_cselect_b32 s8, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s8, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB49_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_add_f64 v[4:5], s[4:5], 1.0
@@ -7464,8 +7504,9 @@ define <3 x double> @bitcast_v12i16_to_v3f64(<12 x i16> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v6
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_pk_add_u16 v5, v5, 3 op_sel_hi:[1,0]
@@ -7677,6 +7718,7 @@ define inreg <3 x double> @bitcast_v12i16_to_v3f64_scalar(<12 x i16> inreg %a, i
; GFX11-NEXT: s_and_b32 s6, s6, exec_lo
; GFX11-NEXT: s_cselect_b32 s6, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s6, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB51_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_pk_add_u16 v5, s5, 3 op_sel_hi:[1,0]
@@ -7800,8 +7842,9 @@ define <12 x half> @bitcast_v3f64_to_v12f16(<3 x double> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v6
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB52_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -7978,6 +8021,7 @@ define inreg <12 x half> @bitcast_v3f64_to_v12f16_scalar(<3 x double> inreg %a,
; GFX11-NEXT: s_and_b32 s8, s8, exec_lo
; GFX11-NEXT: s_cselect_b32 s8, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s8, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB53_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_add_f64 v[4:5], s[4:5], 1.0
@@ -8175,8 +8219,9 @@ define <3 x double> @bitcast_v12f16_to_v3f64(<12 x half> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v6
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_pk_add_f16 v5, 0x200, v5 op_sel_hi:[0,1]
@@ -8410,6 +8455,7 @@ define inreg <3 x double> @bitcast_v12f16_to_v3f64_scalar(<12 x half> inreg %a,
; GFX11-NEXT: s_and_b32 s6, s6, exec_lo
; GFX11-NEXT: s_cselect_b32 s6, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s6, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB55_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_pk_add_f16 v5, 0x200, s5 op_sel_hi:[0,1]
@@ -8610,8 +8656,9 @@ define <12 x half> @bitcast_v12i16_to_v12f16(<12 x i16> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v6
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_pk_add_u16 v5, v5, 3 op_sel_hi:[1,0]
@@ -8860,6 +8907,7 @@ define inreg <12 x half> @bitcast_v12i16_to_v12f16_scalar(<12 x i16> inreg %a, i
; GFX11-NEXT: s_and_b32 s8, s8, exec_lo
; GFX11-NEXT: s_cselect_b32 s8, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s8, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB57_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_pk_add_u16 v5, s5, 3 op_sel_hi:[1,0]
@@ -9037,8 +9085,9 @@ define <12 x i16> @bitcast_v12f16_to_v12i16(<12 x half> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v6
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_pk_add_f16 v5, 0x200, v5 op_sel_hi:[0,1]
@@ -9285,6 +9334,7 @@ define inreg <12 x i16> @bitcast_v12f16_to_v12i16_scalar(<12 x half> inreg %a, i
; GFX11-NEXT: s_and_b32 s8, s8, exec_lo
; GFX11-NEXT: s_cselect_b32 s8, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s8, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB59_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_pk_add_f16 v5, 0x200, s5 op_sel_hi:[0,1]
diff --git a/llvm/test/CodeGen/AMDGPU/amdgcn.bitcast.224bit.ll b/llvm/test/CodeGen/AMDGPU/amdgcn.bitcast.224bit.ll
index 95793b95e155e0..4442b09e125bfb 100644
--- a/llvm/test/CodeGen/AMDGPU/amdgcn.bitcast.224bit.ll
+++ b/llvm/test/CodeGen/AMDGPU/amdgcn.bitcast.224bit.ll
@@ -69,8 +69,9 @@ define <7 x float> @bitcast_v7i32_to_v7f32(<7 x i32> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v7
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB0_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -214,6 +215,7 @@ define inreg <7 x float> @bitcast_v7i32_to_v7f32_scalar(<7 x i32> inreg %a, i32
; GFX11-NEXT: s_and_b32 s4, s4, exec_lo
; GFX11-NEXT: s_cselect_b32 s4, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s4, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB1_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: s_add_i32 s18, s18, 3
@@ -310,8 +312,9 @@ define <7 x i32> @bitcast_v7f32_to_v7i32(<7 x float> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v7
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_dual_add_f32 v6, 1.0, v6 :: v_dual_add_f32 v5, 1.0, v5
@@ -457,6 +460,7 @@ define inreg <7 x i32> @bitcast_v7f32_to_v7i32_scalar(<7 x float> inreg %a, i32
; GFX11-NEXT: s_and_b32 s7, s7, exec_lo
; GFX11-NEXT: s_cselect_b32 s7, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s7, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB3_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_add_f32_e64 v6, s6, 1.0
@@ -598,8 +602,9 @@ define <14 x i16> @bitcast_v7i32_to_v14i16(<7 x i32> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v7
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB4_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -785,6 +790,7 @@ define inreg <14 x i16> @bitcast_v7i32_to_v14i16_scalar(<7 x i32> inreg %a, i32
; GFX11-NEXT: s_and_b32 s4, s4, exec_lo
; GFX11-NEXT: s_cselect_b32 s4, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s4, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB5_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: s_add_i32 s18, s18, 3
@@ -977,8 +983,9 @@ define <7 x i32> @bitcast_v14i16_to_v7i32(<14 x i16> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v7
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB6_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -1211,6 +1218,7 @@ define inreg <7 x i32> @bitcast_v14i16_to_v7i32_scalar(<14 x i16> inreg %a, i32
; GFX11-NEXT: s_and_b32 s7, s7, exec_lo
; GFX11-NEXT: s_cselect_b32 s7, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s7, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB7_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_pk_add_u16 v6, s6, 3 op_sel_hi:[1,0]
@@ -1352,8 +1360,9 @@ define <14 x half> @bitcast_v7i32_to_v14f16(<7 x i32> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v7
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB8_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -1539,6 +1548,7 @@ define inreg <14 x half> @bitcast_v7i32_to_v14f16_scalar(<7 x i32> inreg %a, i32
; GFX11-NEXT: s_and_b32 s4, s4, exec_lo
; GFX11-NEXT: s_cselect_b32 s4, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s4, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB9_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: s_add_i32 s18, s18, 3
@@ -1759,8 +1769,9 @@ define <7 x i32> @bitcast_v14f16_to_v7i32(<14 x half> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v7
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB10_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -2018,6 +2029,7 @@ define inreg <7 x i32> @bitcast_v14f16_to_v7i32_scalar(<14 x half> inreg %a, i32
; GFX11-NEXT: s_and_b32 s7, s7, exec_lo
; GFX11-NEXT: s_cselect_b32 s7, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s7, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB11_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_pk_add_f16 v6, 0x200, s6 op_sel_hi:[0,1]
@@ -2159,8 +2171,9 @@ define <14 x i16> @bitcast_v7f32_to_v14i16(<7 x float> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v7
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_dual_add_f32 v6, 1.0, v6 :: v_dual_add_f32 v5, 1.0, v5
@@ -2358,6 +2371,7 @@ define inreg <14 x i16> @bitcast_v7f32_to_v14i16_scalar(<7 x float> inreg %a, i3
; GFX11-NEXT: s_and_b32 s8, s8, exec_lo
; GFX11-NEXT: s_cselect_b32 s8, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s8, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB13_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_add_f32_e64 v6, s6, 1.0
@@ -2550,8 +2564,9 @@ define <7 x float> @bitcast_v14i16_to_v7f32(<14 x i16> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v7
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB14_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -2784,6 +2799,7 @@ define inreg <7 x float> @bitcast_v14i16_to_v7f32_scalar(<14 x i16> inreg %a, i3
; GFX11-NEXT: s_and_b32 s7, s7, exec_lo
; GFX11-NEXT: s_cselect_b32 s7, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s7, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB15_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_pk_add_u16 v6, s6, 3 op_sel_hi:[1,0]
@@ -2925,8 +2941,9 @@ define <14 x half> @bitcast_v7f32_to_v14f16(<7 x float> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v7
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_dual_add_f32 v6, 1.0, v6 :: v_dual_add_f32 v5, 1.0, v5
@@ -3124,6 +3141,7 @@ define inreg <14 x half> @bitcast_v7f32_to_v14f16_scalar(<7 x float> inreg %a, i
; GFX11-NEXT: s_and_b32 s8, s8, exec_lo
; GFX11-NEXT: s_cselect_b32 s8, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s8, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB17_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_add_f32_e64 v6, s6, 1.0
@@ -3344,8 +3362,9 @@ define <7 x float> @bitcast_v14f16_to_v7f32(<14 x half> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v7
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB18_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -3603,6 +3622,7 @@ define inreg <7 x float> @bitcast_v14f16_to_v7f32_scalar(<14 x half> inreg %a, i
; GFX11-NEXT: s_and_b32 s7, s7, exec_lo
; GFX11-NEXT: s_cselect_b32 s7, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s7, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB19_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_pk_add_f16 v6, 0x200, s6 op_sel_hi:[0,1]
@@ -3824,8 +3844,9 @@ define <14 x half> @bitcast_v14i16_to_v14f16(<14 x i16> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v7
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB20_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -4099,6 +4120,7 @@ define inreg <14 x half> @bitcast_v14i16_to_v14f16_scalar(<14 x i16> inreg %a, i
; GFX11-NEXT: s_and_b32 s8, s8, exec_lo
; GFX11-NEXT: s_cselect_b32 s8, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s8, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB21_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_pk_add_u16 v6, s6, 3 op_sel_hi:[1,0]
@@ -4293,8 +4315,9 @@ define <14 x i16> @bitcast_v14f16_to_v14i16(<14 x half> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v7
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB22_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -4564,6 +4587,7 @@ define inreg <14 x i16> @bitcast_v14f16_to_v14i16_scalar(<14 x half> inreg %a, i
; GFX11-NEXT: s_and_b32 s8, s8, exec_lo
; GFX11-NEXT: s_cselect_b32 s8, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s8, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB23_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_pk_add_f16 v6, 0x200, s6 op_sel_hi:[0,1]
diff --git a/llvm/test/CodeGen/AMDGPU/amdgcn.bitcast.256bit.ll b/llvm/test/CodeGen/AMDGPU/amdgcn.bitcast.256bit.ll
index b18d1409ad0942..fb98d465c3624f 100644
--- a/llvm/test/CodeGen/AMDGPU/amdgcn.bitcast.256bit.ll
+++ b/llvm/test/CodeGen/AMDGPU/amdgcn.bitcast.256bit.ll
@@ -72,8 +72,9 @@ define <8 x float> @bitcast_v8i32_to_v8f32(<8 x i32> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v8
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB0_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -224,6 +225,7 @@ define inreg <8 x float> @bitcast_v8i32_to_v8f32_scalar(<8 x i32> inreg %a, i32
; GFX11-NEXT: s_and_b32 s4, s4, exec_lo
; GFX11-NEXT: s_cselect_b32 s4, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s4, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB1_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: s_add_i32 s19, s19, 3
@@ -324,8 +326,9 @@ define <8 x i32> @bitcast_v8f32_to_v8i32(<8 x float> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v8
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_dual_add_f32 v7, 1.0, v7 :: v_dual_add_f32 v6, 1.0, v6
@@ -478,6 +481,7 @@ define inreg <8 x i32> @bitcast_v8f32_to_v8i32_scalar(<8 x float> inreg %a, i32
; GFX11-NEXT: s_and_b32 s8, s8, exec_lo
; GFX11-NEXT: s_cselect_b32 s8, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s8, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB3_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_add_f32_e64 v7, s7, 1.0
@@ -578,8 +582,9 @@ define <4 x i64> @bitcast_v8i32_to_v4i64(<8 x i32> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v8
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB4_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -730,6 +735,7 @@ define inreg <4 x i64> @bitcast_v8i32_to_v4i64_scalar(<8 x i32> inreg %a, i32 in
; GFX11-NEXT: s_and_b32 s4, s4, exec_lo
; GFX11-NEXT: s_cselect_b32 s4, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s4, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB5_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: s_add_i32 s19, s19, 3
@@ -830,18 +836,17 @@ define <8 x i32> @bitcast_v4i64_to_v8i32(<4 x i64> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v8
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB6_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_add_co_u32 v6, vcc_lo, v6, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v7, null, 0, v7, vcc_lo
; GFX11-NEXT: v_add_co_u32 v4, vcc_lo, v4, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v5, null, 0, v5, vcc_lo
; GFX11-NEXT: v_add_co_u32 v2, vcc_lo, v2, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX11-NEXT: v_add_co_u32 v0, vcc_lo, v0, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
@@ -984,6 +989,7 @@ define inreg <8 x i32> @bitcast_v4i64_to_v8i32_scalar(<4 x i64> inreg %a, i32 in
; GFX11-NEXT: s_and_b32 s4, s4, exec_lo
; GFX11-NEXT: s_cselect_b32 s4, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s4, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB7_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: s_add_u32 s18, s18, 3
@@ -1084,8 +1090,9 @@ define <4 x double> @bitcast_v8i32_to_v4f64(<8 x i32> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v8
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB8_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -1236,6 +1243,7 @@ define inreg <4 x double> @bitcast_v8i32_to_v4f64_scalar(<8 x i32> inreg %a, i32
; GFX11-NEXT: s_and_b32 s4, s4, exec_lo
; GFX11-NEXT: s_cselect_b32 s4, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s4, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB9_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: s_add_i32 s19, s19, 3
@@ -1326,8 +1334,9 @@ define <8 x i32> @bitcast_v4f64_to_v8i32(<4 x double> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v8
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB10_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -1469,6 +1478,7 @@ define inreg <8 x i32> @bitcast_v4f64_to_v8i32_scalar(<4 x double> inreg %a, i32
; GFX11-NEXT: s_and_b32 s8, s8, exec_lo
; GFX11-NEXT: s_cselect_b32 s8, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s8, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB11_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_add_f64 v[6:7], s[6:7], 1.0
@@ -1616,8 +1626,9 @@ define <16 x i16> @bitcast_v8i32_to_v16i16(<8 x i32> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v8
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB12_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -1816,6 +1827,7 @@ define inreg <16 x i16> @bitcast_v8i32_to_v16i16_scalar(<8 x i32> inreg %a, i32
; GFX11-NEXT: s_and_b32 s4, s4, exec_lo
; GFX11-NEXT: s_cselect_b32 s4, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s4, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB13_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: s_add_i32 s19, s19, 3
@@ -2024,8 +2036,9 @@ define <8 x i32> @bitcast_v16i16_to_v8i32(<16 x i16> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v8
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB14_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -2278,6 +2291,7 @@ define inreg <8 x i32> @bitcast_v16i16_to_v8i32_scalar(<16 x i16> inreg %a, i32
; GFX11-NEXT: s_and_b32 s8, s8, exec_lo
; GFX11-NEXT: s_cselect_b32 s8, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s8, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB15_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_pk_add_u16 v7, s7, 3 op_sel_hi:[1,0]
@@ -2429,8 +2443,9 @@ define <16 x half> @bitcast_v8i32_to_v16f16(<8 x i32> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v8
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB16_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -2629,6 +2644,7 @@ define inreg <16 x half> @bitcast_v8i32_to_v16f16_scalar(<8 x i32> inreg %a, i32
; GFX11-NEXT: s_and_b32 s4, s4, exec_lo
; GFX11-NEXT: s_cselect_b32 s4, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s4, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB17_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: s_add_i32 s19, s19, 3
@@ -2869,8 +2885,9 @@ define <8 x i32> @bitcast_v16f16_to_v8i32(<16 x half> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v8
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB18_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -3151,6 +3168,7 @@ define inreg <8 x i32> @bitcast_v16f16_to_v8i32_scalar(<16 x half> inreg %a, i32
; GFX11-NEXT: s_and_b32 s8, s8, exec_lo
; GFX11-NEXT: s_cselect_b32 s8, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s8, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB19_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_pk_add_f16 v7, 0x200, s7 op_sel_hi:[0,1]
@@ -3343,8 +3361,9 @@ define <16 x bfloat> @bitcast_v8i32_to_v16bf16(<8 x i32> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v8
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB20_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -3567,6 +3586,7 @@ define inreg <16 x bfloat> @bitcast_v8i32_to_v16bf16_scalar(<8 x i32> inreg %a,
; GFX11-NEXT: s_and_b32 s4, s4, exec_lo
; GFX11-NEXT: s_cselect_b32 s4, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s4, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB21_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: s_add_i32 s19, s19, 3
@@ -4033,8 +4053,9 @@ define <8 x i32> @bitcast_v16bf16_to_v8i32(<16 x bfloat> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-TRUE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-TRUE16-NEXT: v_cmpx_ne_u32_e32 0, v8
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-TRUE16-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-TRUE16-NEXT: s_cbranch_execz .LBB22_2
; GFX11-TRUE16-NEXT: ; %bb.1: ; %cmp.true
@@ -4194,8 +4215,9 @@ define <8 x i32> @bitcast_v16bf16_to_v8i32(<16 x bfloat> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-FAKE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-FAKE16-NEXT: v_cmpx_ne_u32_e32 0, v8
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-FAKE16-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-FAKE16-NEXT: s_cbranch_execz .LBB22_2
; GFX11-FAKE16-NEXT: ; %bb.1: ; %cmp.true
@@ -4839,6 +4861,7 @@ define inreg <8 x i32> @bitcast_v16bf16_to_v8i32_scalar(<16 x bfloat> inreg %a,
; GFX11-TRUE16-NEXT: s_and_b32 s8, s8, exec_lo
; GFX11-TRUE16-NEXT: s_cselect_b32 s8, 1, 0
; GFX11-TRUE16-NEXT: s_cmp_lg_u32 s8, 1
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_cbranch_scc1 .LBB23_4
; GFX11-TRUE16-NEXT: ; %bb.3: ; %cmp.true
; GFX11-TRUE16-NEXT: s_and_b32 s8, s7, 0xffff0000
@@ -5023,6 +5046,7 @@ define inreg <8 x i32> @bitcast_v16bf16_to_v8i32_scalar(<16 x bfloat> inreg %a,
; GFX11-FAKE16-NEXT: s_and_b32 s8, s8, exec_lo
; GFX11-FAKE16-NEXT: s_cselect_b32 s8, 1, 0
; GFX11-FAKE16-NEXT: s_cmp_lg_u32 s8, 1
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_cbranch_scc1 .LBB23_4
; GFX11-FAKE16-NEXT: ; %bb.3: ; %cmp.true
; GFX11-FAKE16-NEXT: s_lshl_b32 s8, s7, 16
@@ -5647,6 +5671,7 @@ define <32 x i8> @bitcast_v8i32_to_v32i8(<8 x i32> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v1, 8, v3
; GFX11-TRUE16-NEXT: .LBB24_4: ; %end
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v0.l, v3.l
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v3.l, v35.l
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v8.l, v11.l
@@ -5756,6 +5781,7 @@ define <32 x i8> @bitcast_v8i32_to_v32i8(<8 x i32> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v1, 8, v38
; GFX11-FAKE16-NEXT: .LBB24_4: ; %end
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v0, v38
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v4, v39
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v8, v36
@@ -6239,6 +6265,7 @@ define inreg <32 x i8> @bitcast_v8i32_to_v32i8_scalar(<8 x i32> inreg %a, i32 in
; GFX11-NEXT: s_and_b32 s5, s46, exec_lo
; GFX11-NEXT: s_cselect_b32 s5, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s5, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB25_5
; GFX11-NEXT: ; %bb.4: ; %cmp.true
; GFX11-NEXT: s_add_i32 s1, s1, 3
@@ -6889,6 +6916,7 @@ define <8 x i32> @bitcast_v32i8_to_v8i32(<32 x i8> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: ; implicit-def: $vgpr0_vgpr1_vgpr2_vgpr3_vgpr4_vgpr5_vgpr6_vgpr7
; GFX11-TRUE16-NEXT: s_waitcnt vmcnt(1)
; GFX11-TRUE16-NEXT: v_cmpx_ne_u32_e32 0, v48
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB26_3
; GFX11-TRUE16-NEXT: ; %bb.1: ; %Flow
@@ -7864,6 +7892,7 @@ define inreg <8 x i32> @bitcast_v32i8_to_v8i32_scalar(<32 x i8> inreg %a, i32 in
; GFX11-NEXT: s_and_b32 s42, s42, exec_lo
; GFX11-NEXT: s_cselect_b32 s42, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s42, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB27_5
; GFX11-NEXT: ; %bb.4: ; %cmp.true
; GFX11-NEXT: v_mov_b32_e32 v3, 0xc0c0004
@@ -8024,8 +8053,9 @@ define <4 x i64> @bitcast_v8f32_to_v4i64(<8 x float> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v8
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_dual_add_f32 v7, 1.0, v7 :: v_dual_add_f32 v6, 1.0, v6
@@ -8178,6 +8208,7 @@ define inreg <4 x i64> @bitcast_v8f32_to_v4i64_scalar(<8 x float> inreg %a, i32
; GFX11-NEXT: s_and_b32 s8, s8, exec_lo
; GFX11-NEXT: s_cselect_b32 s8, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s8, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB29_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_add_f32_e64 v7, s7, 1.0
@@ -8278,18 +8309,17 @@ define <8 x float> @bitcast_v4i64_to_v8f32(<4 x i64> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v8
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB30_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_add_co_u32 v6, vcc_lo, v6, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v7, null, 0, v7, vcc_lo
; GFX11-NEXT: v_add_co_u32 v4, vcc_lo, v4, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v5, null, 0, v5, vcc_lo
; GFX11-NEXT: v_add_co_u32 v2, vcc_lo, v2, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX11-NEXT: v_add_co_u32 v0, vcc_lo, v0, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
@@ -8432,6 +8462,7 @@ define inreg <8 x float> @bitcast_v4i64_to_v8f32_scalar(<4 x i64> inreg %a, i32
; GFX11-NEXT: s_and_b32 s4, s4, exec_lo
; GFX11-NEXT: s_cselect_b32 s4, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s4, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB31_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: s_add_u32 s18, s18, 3
@@ -8532,8 +8563,9 @@ define <4 x double> @bitcast_v8f32_to_v4f64(<8 x float> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v8
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_dual_add_f32 v7, 1.0, v7 :: v_dual_add_f32 v6, 1.0, v6
@@ -8686,6 +8718,7 @@ define inreg <4 x double> @bitcast_v8f32_to_v4f64_scalar(<8 x float> inreg %a, i
; GFX11-NEXT: s_and_b32 s8, s8, exec_lo
; GFX11-NEXT: s_cselect_b32 s8, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s8, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB33_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_add_f32_e64 v7, s7, 1.0
@@ -8776,8 +8809,9 @@ define <8 x float> @bitcast_v4f64_to_v8f32(<4 x double> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v8
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB34_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -8919,6 +8953,7 @@ define inreg <8 x float> @bitcast_v4f64_to_v8f32_scalar(<4 x double> inreg %a, i
; GFX11-NEXT: s_and_b32 s8, s8, exec_lo
; GFX11-NEXT: s_cselect_b32 s8, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s8, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB35_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_add_f64 v[6:7], s[6:7], 1.0
@@ -9066,8 +9101,9 @@ define <16 x i16> @bitcast_v8f32_to_v16i16(<8 x float> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v8
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_dual_add_f32 v7, 1.0, v7 :: v_dual_add_f32 v6, 1.0, v6
@@ -9277,6 +9313,7 @@ define inreg <16 x i16> @bitcast_v8f32_to_v16i16_scalar(<8 x float> inreg %a, i3
; GFX11-NEXT: s_and_b32 s8, s8, exec_lo
; GFX11-NEXT: s_cselect_b32 s8, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s8, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB37_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_add_f32_e64 v7, s7, 1.0
@@ -9485,8 +9522,9 @@ define <8 x float> @bitcast_v16i16_to_v8f32(<16 x i16> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v8
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB38_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -9739,6 +9777,7 @@ define inreg <8 x float> @bitcast_v16i16_to_v8f32_scalar(<16 x i16> inreg %a, i3
; GFX11-NEXT: s_and_b32 s8, s8, exec_lo
; GFX11-NEXT: s_cselect_b32 s8, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s8, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB39_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_pk_add_u16 v7, s7, 3 op_sel_hi:[1,0]
@@ -9890,8 +9929,9 @@ define <16 x half> @bitcast_v8f32_to_v16f16(<8 x float> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v8
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_dual_add_f32 v7, 1.0, v7 :: v_dual_add_f32 v6, 1.0, v6
@@ -10101,6 +10141,7 @@ define inreg <16 x half> @bitcast_v8f32_to_v16f16_scalar(<8 x float> inreg %a, i
; GFX11-NEXT: s_and_b32 s8, s8, exec_lo
; GFX11-NEXT: s_cselect_b32 s8, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s8, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB41_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_add_f32_e64 v7, s7, 1.0
@@ -10341,8 +10382,9 @@ define <8 x float> @bitcast_v16f16_to_v8f32(<16 x half> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v8
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB42_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -10623,6 +10665,7 @@ define inreg <8 x float> @bitcast_v16f16_to_v8f32_scalar(<16 x half> inreg %a, i
; GFX11-NEXT: s_and_b32 s8, s8, exec_lo
; GFX11-NEXT: s_cselect_b32 s8, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s8, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB43_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_pk_add_f16 v7, 0x200, s7 op_sel_hi:[0,1]
@@ -10815,8 +10858,9 @@ define <16 x bfloat> @bitcast_v8f32_to_v16bf16(<8 x float> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v8
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_dual_add_f32 v7, 1.0, v7 :: v_dual_add_f32 v6, 1.0, v6
@@ -11058,6 +11102,7 @@ define inreg <16 x bfloat> @bitcast_v8f32_to_v16bf16_scalar(<8 x float> inreg %a
; GFX11-NEXT: s_and_b32 s8, s8, exec_lo
; GFX11-NEXT: s_cselect_b32 s8, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s8, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB45_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_add_f32_e64 v7, s7, 1.0
@@ -11524,8 +11569,9 @@ define <8 x float> @bitcast_v16bf16_to_v8f32(<16 x bfloat> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-TRUE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-TRUE16-NEXT: v_cmpx_ne_u32_e32 0, v8
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-TRUE16-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-TRUE16-NEXT: s_cbranch_execz .LBB46_2
; GFX11-TRUE16-NEXT: ; %bb.1: ; %cmp.true
@@ -11685,8 +11731,9 @@ define <8 x float> @bitcast_v16bf16_to_v8f32(<16 x bfloat> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-FAKE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-FAKE16-NEXT: v_cmpx_ne_u32_e32 0, v8
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-FAKE16-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-FAKE16-NEXT: s_cbranch_execz .LBB46_2
; GFX11-FAKE16-NEXT: ; %bb.1: ; %cmp.true
@@ -12330,6 +12377,7 @@ define inreg <8 x float> @bitcast_v16bf16_to_v8f32_scalar(<16 x bfloat> inreg %a
; GFX11-TRUE16-NEXT: s_and_b32 s8, s8, exec_lo
; GFX11-TRUE16-NEXT: s_cselect_b32 s8, 1, 0
; GFX11-TRUE16-NEXT: s_cmp_lg_u32 s8, 1
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_cbranch_scc1 .LBB47_4
; GFX11-TRUE16-NEXT: ; %bb.3: ; %cmp.true
; GFX11-TRUE16-NEXT: s_and_b32 s8, s7, 0xffff0000
@@ -12514,6 +12562,7 @@ define inreg <8 x float> @bitcast_v16bf16_to_v8f32_scalar(<16 x bfloat> inreg %a
; GFX11-FAKE16-NEXT: s_and_b32 s8, s8, exec_lo
; GFX11-FAKE16-NEXT: s_cselect_b32 s8, 1, 0
; GFX11-FAKE16-NEXT: s_cmp_lg_u32 s8, 1
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_cbranch_scc1 .LBB47_4
; GFX11-FAKE16-NEXT: ; %bb.3: ; %cmp.true
; GFX11-FAKE16-NEXT: s_lshl_b32 s8, s7, 16
@@ -13135,6 +13184,7 @@ define <32 x i8> @bitcast_v8f32_to_v32i8(<8 x float> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v1, 8, v3
; GFX11-TRUE16-NEXT: .LBB48_4: ; %end
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v0.l, v3.l
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v3.l, v35.l
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v8.l, v11.l
@@ -13242,6 +13292,7 @@ define <32 x i8> @bitcast_v8f32_to_v32i8(<8 x float> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v1, 8, v38
; GFX11-FAKE16-NEXT: .LBB48_4: ; %end
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v0, v38
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v4, v39
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v8, v36
@@ -13759,6 +13810,7 @@ define inreg <32 x i8> @bitcast_v8f32_to_v32i8_scalar(<8 x float> inreg %a, i32
; GFX11-NEXT: s_and_b32 s5, s12, exec_lo
; GFX11-NEXT: s_cselect_b32 s5, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s5, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB49_5
; GFX11-NEXT: ; %bb.4: ; %cmp.true
; GFX11-NEXT: v_add_f32_e64 v39, s1, 1.0
@@ -14422,6 +14474,7 @@ define <8 x float> @bitcast_v32i8_to_v8f32(<32 x i8> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: ; implicit-def: $vgpr0_vgpr1_vgpr2_vgpr3_vgpr4_vgpr5_vgpr6_vgpr7
; GFX11-TRUE16-NEXT: s_waitcnt vmcnt(1)
; GFX11-TRUE16-NEXT: v_cmpx_ne_u32_e32 0, v48
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB50_3
; GFX11-TRUE16-NEXT: ; %bb.1: ; %Flow
@@ -15397,6 +15450,7 @@ define inreg <8 x float> @bitcast_v32i8_to_v8f32_scalar(<32 x i8> inreg %a, i32
; GFX11-NEXT: s_and_b32 s42, s42, exec_lo
; GFX11-NEXT: s_cselect_b32 s42, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s42, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB51_5
; GFX11-NEXT: ; %bb.4: ; %cmp.true
; GFX11-NEXT: v_mov_b32_e32 v3, 0xc0c0004
@@ -15557,18 +15611,17 @@ define <4 x double> @bitcast_v4i64_to_v4f64(<4 x i64> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v8
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB52_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_add_co_u32 v0, vcc_lo, v0, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-NEXT: v_add_co_u32 v2, vcc_lo, v2, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX11-NEXT: v_add_co_u32 v4, vcc_lo, v4, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v5, null, 0, v5, vcc_lo
; GFX11-NEXT: v_add_co_u32 v6, vcc_lo, v6, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v7, null, 0, v7, vcc_lo
@@ -15711,6 +15764,7 @@ define inreg <4 x double> @bitcast_v4i64_to_v4f64_scalar(<4 x i64> inreg %a, i32
; GFX11-NEXT: s_and_b32 s4, s4, exec_lo
; GFX11-NEXT: s_cselect_b32 s4, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s4, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB53_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: s_add_u32 s0, s0, 3
@@ -15800,8 +15854,9 @@ define <4 x i64> @bitcast_v4f64_to_v4i64(<4 x double> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v8
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB54_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -15943,6 +15998,7 @@ define inreg <4 x i64> @bitcast_v4f64_to_v4i64_scalar(<4 x double> inreg %a, i32
; GFX11-NEXT: s_and_b32 s8, s8, exec_lo
; GFX11-NEXT: s_cselect_b32 s8, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s8, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB55_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_add_f64 v[0:1], s[0:1], 1.0
@@ -16090,18 +16146,17 @@ define <16 x i16> @bitcast_v4i64_to_v16i16(<4 x i64> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v8
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB56_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_add_co_u32 v6, vcc_lo, v6, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v7, null, 0, v7, vcc_lo
; GFX11-NEXT: v_add_co_u32 v4, vcc_lo, v4, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v5, null, 0, v5, vcc_lo
; GFX11-NEXT: v_add_co_u32 v2, vcc_lo, v2, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX11-NEXT: v_add_co_u32 v0, vcc_lo, v0, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
@@ -16292,6 +16347,7 @@ define inreg <16 x i16> @bitcast_v4i64_to_v16i16_scalar(<4 x i64> inreg %a, i32
; GFX11-NEXT: s_and_b32 s4, s4, exec_lo
; GFX11-NEXT: s_cselect_b32 s4, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s4, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB57_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: s_add_u32 s18, s18, 3
@@ -16500,8 +16556,9 @@ define <4 x i64> @bitcast_v16i16_to_v4i64(<16 x i16> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v8
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB58_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -16754,6 +16811,7 @@ define inreg <4 x i64> @bitcast_v16i16_to_v4i64_scalar(<16 x i16> inreg %a, i32
; GFX11-NEXT: s_and_b32 s8, s8, exec_lo
; GFX11-NEXT: s_cselect_b32 s8, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s8, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB59_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_pk_add_u16 v7, s7, 3 op_sel_hi:[1,0]
@@ -16905,18 +16963,17 @@ define <16 x half> @bitcast_v4i64_to_v16f16(<4 x i64> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v8
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB60_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_add_co_u32 v6, vcc_lo, v6, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v7, null, 0, v7, vcc_lo
; GFX11-NEXT: v_add_co_u32 v4, vcc_lo, v4, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v5, null, 0, v5, vcc_lo
; GFX11-NEXT: v_add_co_u32 v2, vcc_lo, v2, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX11-NEXT: v_add_co_u32 v0, vcc_lo, v0, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
@@ -17107,6 +17164,7 @@ define inreg <16 x half> @bitcast_v4i64_to_v16f16_scalar(<4 x i64> inreg %a, i32
; GFX11-NEXT: s_and_b32 s4, s4, exec_lo
; GFX11-NEXT: s_cselect_b32 s4, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s4, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB61_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: s_add_u32 s18, s18, 3
@@ -17347,8 +17405,9 @@ define <4 x i64> @bitcast_v16f16_to_v4i64(<16 x half> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v8
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB62_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -17629,6 +17688,7 @@ define inreg <4 x i64> @bitcast_v16f16_to_v4i64_scalar(<16 x half> inreg %a, i32
; GFX11-NEXT: s_and_b32 s8, s8, exec_lo
; GFX11-NEXT: s_cselect_b32 s8, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s8, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB63_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_pk_add_f16 v7, 0x200, s7 op_sel_hi:[0,1]
@@ -17821,18 +17881,17 @@ define <16 x bfloat> @bitcast_v4i64_to_v16bf16(<4 x i64> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v8
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB64_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_add_co_u32 v6, vcc_lo, v6, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v7, null, 0, v7, vcc_lo
; GFX11-NEXT: v_add_co_u32 v4, vcc_lo, v4, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v5, null, 0, v5, vcc_lo
; GFX11-NEXT: v_add_co_u32 v2, vcc_lo, v2, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX11-NEXT: v_add_co_u32 v0, vcc_lo, v0, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
@@ -18047,6 +18106,7 @@ define inreg <16 x bfloat> @bitcast_v4i64_to_v16bf16_scalar(<4 x i64> inreg %a,
; GFX11-NEXT: s_and_b32 s4, s4, exec_lo
; GFX11-NEXT: s_cselect_b32 s4, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s4, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB65_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: s_add_u32 s18, s18, 3
@@ -18513,8 +18573,9 @@ define <4 x i64> @bitcast_v16bf16_to_v4i64(<16 x bfloat> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-TRUE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-TRUE16-NEXT: v_cmpx_ne_u32_e32 0, v8
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-TRUE16-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-TRUE16-NEXT: s_cbranch_execz .LBB66_2
; GFX11-TRUE16-NEXT: ; %bb.1: ; %cmp.true
@@ -18674,8 +18735,9 @@ define <4 x i64> @bitcast_v16bf16_to_v4i64(<16 x bfloat> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-FAKE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-FAKE16-NEXT: v_cmpx_ne_u32_e32 0, v8
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-FAKE16-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-FAKE16-NEXT: s_cbranch_execz .LBB66_2
; GFX11-FAKE16-NEXT: ; %bb.1: ; %cmp.true
@@ -19319,6 +19381,7 @@ define inreg <4 x i64> @bitcast_v16bf16_to_v4i64_scalar(<16 x bfloat> inreg %a,
; GFX11-TRUE16-NEXT: s_and_b32 s8, s8, exec_lo
; GFX11-TRUE16-NEXT: s_cselect_b32 s8, 1, 0
; GFX11-TRUE16-NEXT: s_cmp_lg_u32 s8, 1
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_cbranch_scc1 .LBB67_4
; GFX11-TRUE16-NEXT: ; %bb.3: ; %cmp.true
; GFX11-TRUE16-NEXT: s_and_b32 s8, s7, 0xffff0000
@@ -19503,6 +19566,7 @@ define inreg <4 x i64> @bitcast_v16bf16_to_v4i64_scalar(<16 x bfloat> inreg %a,
; GFX11-FAKE16-NEXT: s_and_b32 s8, s8, exec_lo
; GFX11-FAKE16-NEXT: s_cselect_b32 s8, 1, 0
; GFX11-FAKE16-NEXT: s_cmp_lg_u32 s8, 1
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_cbranch_scc1 .LBB67_4
; GFX11-FAKE16-NEXT: ; %bb.3: ; %cmp.true
; GFX11-FAKE16-NEXT: s_lshl_b32 s8, s7, 16
@@ -20094,12 +20158,10 @@ define <32 x i8> @bitcast_v4i64_to_v32i8(<4 x i64> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: s_cbranch_execz .LBB68_4
; GFX11-TRUE16-NEXT: ; %bb.3: ; %cmp.true
; GFX11-TRUE16-NEXT: v_add_co_u32 v11, vcc_lo, v11, 3
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v12, null, 0, v12, vcc_lo
; GFX11-TRUE16-NEXT: v_add_co_u32 v19, vcc_lo, v19, 3
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v20, null, 0, v20, vcc_lo
; GFX11-TRUE16-NEXT: v_add_co_u32 v27, vcc_lo, v27, 3
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v28, null, 0, v28, vcc_lo
; GFX11-TRUE16-NEXT: v_add_co_u32 v3, vcc_lo, v3, 3
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v4, null, 0, v4, vcc_lo
@@ -20130,6 +20192,7 @@ define <32 x i8> @bitcast_v4i64_to_v32i8(<4 x i64> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v1, 8, v3
; GFX11-TRUE16-NEXT: .LBB68_4: ; %end
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v0.l, v3.l
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v3.l, v35.l
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v8.l, v11.l
@@ -20206,12 +20269,10 @@ define <32 x i8> @bitcast_v4i64_to_v32i8(<4 x i64> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: s_cbranch_execz .LBB68_4
; GFX11-FAKE16-NEXT: ; %bb.3: ; %cmp.true
; GFX11-FAKE16-NEXT: v_add_co_u32 v36, vcc_lo, v36, 3
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v37, null, 0, v37, vcc_lo
; GFX11-FAKE16-NEXT: v_add_co_u32 v34, vcc_lo, v34, 3
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v35, null, 0, v35, vcc_lo
; GFX11-FAKE16-NEXT: v_add_co_u32 v32, vcc_lo, v32, 3
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v33, null, 0, v33, vcc_lo
; GFX11-FAKE16-NEXT: v_add_co_u32 v38, vcc_lo, v38, 3
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v39, null, 0, v39, vcc_lo
@@ -20242,6 +20303,7 @@ define <32 x i8> @bitcast_v4i64_to_v32i8(<4 x i64> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v1, 8, v38
; GFX11-FAKE16-NEXT: .LBB68_4: ; %end
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v0, v38
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v4, v39
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v8, v36
@@ -20725,6 +20787,7 @@ define inreg <32 x i8> @bitcast_v4i64_to_v32i8_scalar(<4 x i64> inreg %a, i32 in
; GFX11-NEXT: s_and_b32 s5, s46, exec_lo
; GFX11-NEXT: s_cselect_b32 s5, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s5, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB69_5
; GFX11-NEXT: ; %bb.4: ; %cmp.true
; GFX11-NEXT: s_add_u32 s0, s0, 3
@@ -21375,6 +21438,7 @@ define <4 x i64> @bitcast_v32i8_to_v4i64(<32 x i8> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: ; implicit-def: $vgpr0_vgpr1_vgpr2_vgpr3_vgpr4_vgpr5_vgpr6_vgpr7
; GFX11-TRUE16-NEXT: s_waitcnt vmcnt(1)
; GFX11-TRUE16-NEXT: v_cmpx_ne_u32_e32 0, v48
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB70_3
; GFX11-TRUE16-NEXT: ; %bb.1: ; %Flow
@@ -22350,6 +22414,7 @@ define inreg <4 x i64> @bitcast_v32i8_to_v4i64_scalar(<32 x i8> inreg %a, i32 in
; GFX11-NEXT: s_and_b32 s42, s42, exec_lo
; GFX11-NEXT: s_cselect_b32 s42, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s42, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB71_5
; GFX11-NEXT: ; %bb.4: ; %cmp.true
; GFX11-NEXT: v_mov_b32_e32 v3, 0xc0c0004
@@ -22551,8 +22616,9 @@ define <16 x i16> @bitcast_v4f64_to_v16i16(<4 x double> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v8
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB72_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -22751,6 +22817,7 @@ define inreg <16 x i16> @bitcast_v4f64_to_v16i16_scalar(<4 x double> inreg %a, i
; GFX11-NEXT: s_and_b32 s8, s8, exec_lo
; GFX11-NEXT: s_cselect_b32 s8, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s8, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB73_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_add_f64 v[6:7], s[6:7], 1.0
@@ -22955,8 +23022,9 @@ define <4 x double> @bitcast_v16i16_to_v4f64(<16 x i16> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v8
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB74_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -23209,6 +23277,7 @@ define inreg <4 x double> @bitcast_v16i16_to_v4f64_scalar(<16 x i16> inreg %a, i
; GFX11-NEXT: s_and_b32 s8, s8, exec_lo
; GFX11-NEXT: s_cselect_b32 s8, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s8, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB75_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_pk_add_u16 v7, s7, 3 op_sel_hi:[1,0]
@@ -23350,8 +23419,9 @@ define <16 x half> @bitcast_v4f64_to_v16f16(<4 x double> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v8
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB76_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -23550,6 +23620,7 @@ define inreg <16 x half> @bitcast_v4f64_to_v16f16_scalar(<4 x double> inreg %a,
; GFX11-NEXT: s_and_b32 s8, s8, exec_lo
; GFX11-NEXT: s_cselect_b32 s8, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s8, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB77_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_add_f64 v[6:7], s[6:7], 1.0
@@ -23786,8 +23857,9 @@ define <4 x double> @bitcast_v16f16_to_v4f64(<16 x half> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v8
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB78_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -24068,6 +24140,7 @@ define inreg <4 x double> @bitcast_v16f16_to_v4f64_scalar(<16 x half> inreg %a,
; GFX11-NEXT: s_and_b32 s8, s8, exec_lo
; GFX11-NEXT: s_cselect_b32 s8, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s8, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB79_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_pk_add_f16 v7, 0x200, s7 op_sel_hi:[0,1]
@@ -24246,8 +24319,9 @@ define <16 x bfloat> @bitcast_v4f64_to_v16bf16(<4 x double> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v8
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB80_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -24478,6 +24552,7 @@ define inreg <16 x bfloat> @bitcast_v4f64_to_v16bf16_scalar(<4 x double> inreg %
; GFX11-NEXT: s_and_b32 s8, s8, exec_lo
; GFX11-NEXT: s_cselect_b32 s8, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s8, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB81_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_add_f64 v[6:7], s[6:7], 1.0
@@ -24940,8 +25015,9 @@ define <4 x double> @bitcast_v16bf16_to_v4f64(<16 x bfloat> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-TRUE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-TRUE16-NEXT: v_cmpx_ne_u32_e32 0, v8
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-TRUE16-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-TRUE16-NEXT: s_cbranch_execz .LBB82_2
; GFX11-TRUE16-NEXT: ; %bb.1: ; %cmp.true
@@ -25101,8 +25177,9 @@ define <4 x double> @bitcast_v16bf16_to_v4f64(<16 x bfloat> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-FAKE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-FAKE16-NEXT: v_cmpx_ne_u32_e32 0, v8
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-FAKE16-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-FAKE16-NEXT: s_cbranch_execz .LBB82_2
; GFX11-FAKE16-NEXT: ; %bb.1: ; %cmp.true
@@ -25746,6 +25823,7 @@ define inreg <4 x double> @bitcast_v16bf16_to_v4f64_scalar(<16 x bfloat> inreg %
; GFX11-TRUE16-NEXT: s_and_b32 s8, s8, exec_lo
; GFX11-TRUE16-NEXT: s_cselect_b32 s8, 1, 0
; GFX11-TRUE16-NEXT: s_cmp_lg_u32 s8, 1
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_cbranch_scc1 .LBB83_4
; GFX11-TRUE16-NEXT: ; %bb.3: ; %cmp.true
; GFX11-TRUE16-NEXT: s_and_b32 s8, s7, 0xffff0000
@@ -25930,6 +26008,7 @@ define inreg <4 x double> @bitcast_v16bf16_to_v4f64_scalar(<16 x bfloat> inreg %
; GFX11-FAKE16-NEXT: s_and_b32 s8, s8, exec_lo
; GFX11-FAKE16-NEXT: s_cselect_b32 s8, 1, 0
; GFX11-FAKE16-NEXT: s_cmp_lg_u32 s8, 1
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_cbranch_scc1 .LBB83_4
; GFX11-FAKE16-NEXT: ; %bb.3: ; %cmp.true
; GFX11-FAKE16-NEXT: s_lshl_b32 s8, s7, 16
@@ -26546,6 +26625,7 @@ define <32 x i8> @bitcast_v4f64_to_v32i8(<4 x double> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v1, 8, v3
; GFX11-TRUE16-NEXT: .LBB84_4: ; %end
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v0.l, v3.l
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v3.l, v35.l
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v8.l, v11.l
@@ -26653,6 +26733,7 @@ define <32 x i8> @bitcast_v4f64_to_v32i8(<4 x double> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v1, 8, v38
; GFX11-FAKE16-NEXT: .LBB84_4: ; %end
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v0, v38
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v4, v39
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v8, v36
@@ -27158,6 +27239,7 @@ define inreg <32 x i8> @bitcast_v4f64_to_v32i8_scalar(<4 x double> inreg %a, i32
; GFX11-NEXT: s_and_b32 s5, s12, exec_lo
; GFX11-NEXT: s_cselect_b32 s5, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s5, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB85_5
; GFX11-NEXT: ; %bb.4: ; %cmp.true
; GFX11-NEXT: v_add_f64 v[32:33], s[18:19], 1.0
@@ -27819,6 +27901,7 @@ define <4 x double> @bitcast_v32i8_to_v4f64(<32 x i8> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: ; implicit-def: $vgpr0_vgpr1_vgpr2_vgpr3_vgpr4_vgpr5_vgpr6_vgpr7
; GFX11-TRUE16-NEXT: s_waitcnt vmcnt(1)
; GFX11-TRUE16-NEXT: v_cmpx_ne_u32_e32 0, v48
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB86_3
; GFX11-TRUE16-NEXT: ; %bb.1: ; %Flow
@@ -28794,6 +28877,7 @@ define inreg <4 x double> @bitcast_v32i8_to_v4f64_scalar(<32 x i8> inreg %a, i32
; GFX11-NEXT: s_and_b32 s42, s42, exec_lo
; GFX11-NEXT: s_cselect_b32 s42, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s42, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB87_5
; GFX11-NEXT: ; %bb.4: ; %cmp.true
; GFX11-NEXT: v_mov_b32_e32 v3, 0xc0c0004
@@ -29097,8 +29181,9 @@ define <16 x half> @bitcast_v16i16_to_v16f16(<16 x i16> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v8
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB88_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -29398,6 +29483,7 @@ define inreg <16 x half> @bitcast_v16i16_to_v16f16_scalar(<16 x i16> inreg %a, i
; GFX11-NEXT: s_and_b32 s8, s8, exec_lo
; GFX11-NEXT: s_cselect_b32 s8, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s8, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB89_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_pk_add_u16 v7, s7, 3 op_sel_hi:[1,0]
@@ -29610,8 +29696,9 @@ define <16 x i16> @bitcast_v16f16_to_v16i16(<16 x half> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v8
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB90_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -29904,6 +29991,7 @@ define inreg <16 x i16> @bitcast_v16f16_to_v16i16_scalar(<16 x half> inreg %a, i
; GFX11-NEXT: s_and_b32 s8, s8, exec_lo
; GFX11-NEXT: s_cselect_b32 s8, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s8, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB91_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_pk_add_f16 v7, 0x200, s7 op_sel_hi:[0,1]
@@ -30138,8 +30226,9 @@ define <16 x bfloat> @bitcast_v16i16_to_v16bf16(<16 x i16> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v8
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB92_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -30439,6 +30528,7 @@ define inreg <16 x bfloat> @bitcast_v16i16_to_v16bf16_scalar(<16 x i16> inreg %a
; GFX11-NEXT: s_and_b32 s8, s8, exec_lo
; GFX11-NEXT: s_cselect_b32 s8, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s8, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB93_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_pk_add_u16 v7, s7, 3 op_sel_hi:[1,0]
@@ -30948,8 +31038,9 @@ define <16 x i16> @bitcast_v16bf16_to_v16i16(<16 x bfloat> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-TRUE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-TRUE16-NEXT: v_cmpx_ne_u32_e32 0, v8
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-TRUE16-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-TRUE16-NEXT: s_cbranch_execz .LBB94_2
; GFX11-TRUE16-NEXT: ; %bb.1: ; %cmp.true
@@ -31108,8 +31199,9 @@ define <16 x i16> @bitcast_v16bf16_to_v16i16(<16 x bfloat> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-FAKE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-FAKE16-NEXT: v_cmpx_ne_u32_e32 0, v8
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-FAKE16-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-FAKE16-NEXT: s_cbranch_execz .LBB94_2
; GFX11-FAKE16-NEXT: ; %bb.1: ; %cmp.true
@@ -31798,6 +31890,7 @@ define inreg <16 x i16> @bitcast_v16bf16_to_v16i16_scalar(<16 x bfloat> inreg %a
; GFX11-TRUE16-NEXT: s_and_b32 s8, s8, exec_lo
; GFX11-TRUE16-NEXT: s_cselect_b32 s8, 1, 0
; GFX11-TRUE16-NEXT: s_cmp_lg_u32 s8, 1
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_cbranch_scc1 .LBB95_4
; GFX11-TRUE16-NEXT: ; %bb.3: ; %cmp.true
; GFX11-TRUE16-NEXT: s_and_b32 s8, s0, 0xffff0000
@@ -31960,6 +32053,7 @@ define inreg <16 x i16> @bitcast_v16bf16_to_v16i16_scalar(<16 x bfloat> inreg %a
; GFX11-FAKE16-NEXT: s_and_b32 s8, s8, exec_lo
; GFX11-FAKE16-NEXT: s_cselect_b32 s8, 1, 0
; GFX11-FAKE16-NEXT: s_cmp_lg_u32 s8, 1
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_cbranch_scc1 .LBB95_4
; GFX11-FAKE16-NEXT: ; %bb.3: ; %cmp.true
; GFX11-FAKE16-NEXT: s_and_b32 s8, s0, 0xffff0000
@@ -32659,6 +32753,7 @@ define <32 x i8> @bitcast_v16i16_to_v32i8(<16 x i16> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v1, 8, v3
; GFX11-TRUE16-NEXT: .LBB96_4: ; %end
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v0.l, v3.l
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v3.l, v35.l
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v8.l, v11.l
@@ -32768,6 +32863,7 @@ define <32 x i8> @bitcast_v16i16_to_v32i8(<16 x i16> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v1, 8, v38
; GFX11-FAKE16-NEXT: .LBB96_4: ; %end
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v0, v38
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v4, v39
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v8, v36
@@ -33353,6 +33449,7 @@ define inreg <32 x i8> @bitcast_v16i16_to_v32i8_scalar(<16 x i16> inreg %a, i32
; GFX11-NEXT: s_and_b32 s5, s12, exec_lo
; GFX11-NEXT: s_cselect_b32 s5, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s5, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB97_5
; GFX11-NEXT: ; %bb.4: ; %cmp.true
; GFX11-NEXT: v_pk_add_u16 v39, s1, 3 op_sel_hi:[1,0]
@@ -34085,6 +34182,7 @@ define <16 x i16> @bitcast_v32i8_to_v16i16(<32 x i8> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: ; implicit-def: $vgpr0_vgpr1_vgpr2_vgpr3_vgpr4_vgpr5_vgpr6_vgpr7
; GFX11-TRUE16-NEXT: v_lshlrev_b16 v10.h, 8, v31.h
; GFX11-TRUE16-NEXT: v_cmpx_ne_u32_e32 0, v37
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB98_3
; GFX11-TRUE16-NEXT: ; %bb.1: ; %Flow
@@ -35100,6 +35198,7 @@ define inreg <16 x i16> @bitcast_v32i8_to_v16i16_scalar(<32 x i8> inreg %a, i32
; GFX11-TRUE16-NEXT: s_and_b32 s42, s42, exec_lo
; GFX11-TRUE16-NEXT: s_cselect_b32 s42, 1, 0
; GFX11-TRUE16-NEXT: s_cmp_lg_u32 s42, 1
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_cbranch_scc1 .LBB99_5
; GFX11-TRUE16-NEXT: ; %bb.4: ; %cmp.true
; GFX11-TRUE16-NEXT: v_mov_b32_e32 v0, 0xc0c0004
@@ -35228,6 +35327,7 @@ define inreg <16 x i16> @bitcast_v32i8_to_v16i16_scalar(<32 x i8> inreg %a, i32
; GFX11-FAKE16-NEXT: s_and_b32 s42, s42, exec_lo
; GFX11-FAKE16-NEXT: s_cselect_b32 s42, 1, 0
; GFX11-FAKE16-NEXT: s_cmp_lg_u32 s42, 1
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_cbranch_scc1 .LBB99_5
; GFX11-FAKE16-NEXT: ; %bb.4: ; %cmp.true
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v0, 0xc0c0004
@@ -35547,8 +35647,9 @@ define <16 x bfloat> @bitcast_v16f16_to_v16bf16(<16 x half> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v8
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB100_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -35877,6 +35978,7 @@ define inreg <16 x bfloat> @bitcast_v16f16_to_v16bf16_scalar(<16 x half> inreg %
; GFX11-NEXT: s_and_b32 s8, s8, exec_lo
; GFX11-NEXT: s_cselect_b32 s8, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s8, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB101_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_pk_add_f16 v7, 0x200, s7 op_sel_hi:[0,1]
@@ -36394,8 +36496,9 @@ define <16 x half> @bitcast_v16bf16_to_v16f16(<16 x bfloat> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-TRUE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-TRUE16-NEXT: v_cmpx_ne_u32_e32 0, v8
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-TRUE16-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-TRUE16-NEXT: s_cbranch_execz .LBB102_2
; GFX11-TRUE16-NEXT: ; %bb.1: ; %cmp.true
@@ -36556,8 +36659,9 @@ define <16 x half> @bitcast_v16bf16_to_v16f16(<16 x bfloat> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-FAKE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-FAKE16-NEXT: v_cmpx_ne_u32_e32 0, v8
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-FAKE16-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-FAKE16-NEXT: s_cbranch_execz .LBB102_2
; GFX11-FAKE16-NEXT: ; %bb.1: ; %cmp.true
@@ -37257,6 +37361,7 @@ define inreg <16 x half> @bitcast_v16bf16_to_v16f16_scalar(<16 x bfloat> inreg %
; GFX11-TRUE16-NEXT: s_and_b32 s8, s8, exec_lo
; GFX11-TRUE16-NEXT: s_cselect_b32 s8, 1, 0
; GFX11-TRUE16-NEXT: s_cmp_lg_u32 s8, 1
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_cbranch_scc1 .LBB103_4
; GFX11-TRUE16-NEXT: ; %bb.3: ; %cmp.true
; GFX11-TRUE16-NEXT: s_and_b32 s8, s0, 0xffff0000
@@ -37437,6 +37542,7 @@ define inreg <16 x half> @bitcast_v16bf16_to_v16f16_scalar(<16 x bfloat> inreg %
; GFX11-FAKE16-NEXT: s_and_b32 s8, s8, exec_lo
; GFX11-FAKE16-NEXT: s_cselect_b32 s8, 1, 0
; GFX11-FAKE16-NEXT: s_cmp_lg_u32 s8, 1
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_cbranch_scc1 .LBB103_4
; GFX11-FAKE16-NEXT: ; %bb.3: ; %cmp.true
; GFX11-FAKE16-NEXT: s_and_b32 s8, s0, 0xffff0000
@@ -38160,6 +38266,7 @@ define <32 x i8> @bitcast_v16f16_to_v32i8(<16 x half> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v1, 8, v3
; GFX11-TRUE16-NEXT: .LBB104_4: ; %end
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v0.l, v3.l
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v3.l, v35.l
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v8.l, v11.l
@@ -38269,6 +38376,7 @@ define <32 x i8> @bitcast_v16f16_to_v32i8(<16 x half> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v1, 8, v38
; GFX11-FAKE16-NEXT: .LBB104_4: ; %end
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v0, v38
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v4, v39
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v8, v36
@@ -38888,6 +38996,7 @@ define inreg <32 x i8> @bitcast_v16f16_to_v32i8_scalar(<16 x half> inreg %a, i32
; GFX11-NEXT: s_and_b32 s5, s12, exec_lo
; GFX11-NEXT: s_cselect_b32 s5, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s5, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB105_5
; GFX11-NEXT: ; %bb.4: ; %cmp.true
; GFX11-NEXT: v_pk_add_f16 v39, 0x200, s1 op_sel_hi:[0,1]
@@ -39620,6 +39729,7 @@ define <16 x half> @bitcast_v32i8_to_v16f16(<32 x i8> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: ; implicit-def: $vgpr0_vgpr1_vgpr2_vgpr3_vgpr4_vgpr5_vgpr6_vgpr7
; GFX11-TRUE16-NEXT: v_lshlrev_b16 v10.h, 8, v31.h
; GFX11-TRUE16-NEXT: v_cmpx_ne_u32_e32 0, v37
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB106_3
; GFX11-TRUE16-NEXT: ; %bb.1: ; %Flow
@@ -40635,6 +40745,7 @@ define inreg <16 x half> @bitcast_v32i8_to_v16f16_scalar(<32 x i8> inreg %a, i32
; GFX11-TRUE16-NEXT: s_and_b32 s42, s42, exec_lo
; GFX11-TRUE16-NEXT: s_cselect_b32 s42, 1, 0
; GFX11-TRUE16-NEXT: s_cmp_lg_u32 s42, 1
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_cbranch_scc1 .LBB107_5
; GFX11-TRUE16-NEXT: ; %bb.4: ; %cmp.true
; GFX11-TRUE16-NEXT: v_mov_b32_e32 v0, 0xc0c0004
@@ -40763,6 +40874,7 @@ define inreg <16 x half> @bitcast_v32i8_to_v16f16_scalar(<32 x i8> inreg %a, i32
; GFX11-FAKE16-NEXT: s_and_b32 s42, s42, exec_lo
; GFX11-FAKE16-NEXT: s_cselect_b32 s42, 1, 0
; GFX11-FAKE16-NEXT: s_cmp_lg_u32 s42, 1
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_cbranch_scc1 .LBB107_5
; GFX11-FAKE16-NEXT: ; %bb.4: ; %cmp.true
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v0, 0xc0c0004
@@ -41765,8 +41877,8 @@ define <32 x i8> @bitcast_v16bf16_to_v32i8(<16 x bfloat> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v25, 8, v27
; GFX11-TRUE16-NEXT: .LBB108_4: ; %end
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(VALU_DEP_3)
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v0.l, v3.l
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v3.l, v35.l
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v8.l, v11.l
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v11.l, v34.l
@@ -42001,6 +42113,7 @@ define <32 x i8> @bitcast_v16bf16_to_v32i8(<16 x bfloat> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v1, 8, v0
; GFX11-FAKE16-NEXT: .LBB108_4: ; %end
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v0, v38
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v4, v39
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v8, v36
@@ -42875,6 +42988,7 @@ define inreg <32 x i8> @bitcast_v16bf16_to_v32i8_scalar(<16 x bfloat> inreg %a,
; GFX11-TRUE16-NEXT: s_and_b32 s5, s12, exec_lo
; GFX11-TRUE16-NEXT: s_cselect_b32 s5, 1, 0
; GFX11-TRUE16-NEXT: s_cmp_lg_u32 s5, 1
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_cbranch_scc1 .LBB109_5
; GFX11-TRUE16-NEXT: ; %bb.4: ; %cmp.true
; GFX11-TRUE16-NEXT: s_lshl_b32 s4, s1, 16
@@ -43145,6 +43259,7 @@ define inreg <32 x i8> @bitcast_v16bf16_to_v32i8_scalar(<16 x bfloat> inreg %a,
; GFX11-FAKE16-NEXT: s_and_b32 s5, s12, exec_lo
; GFX11-FAKE16-NEXT: s_cselect_b32 s5, 1, 0
; GFX11-FAKE16-NEXT: s_cmp_lg_u32 s5, 1
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_cbranch_scc1 .LBB109_5
; GFX11-FAKE16-NEXT: ; %bb.4: ; %cmp.true
; GFX11-FAKE16-NEXT: s_lshl_b32 s4, s1, 16
@@ -44025,6 +44140,7 @@ define <16 x bfloat> @bitcast_v32i8_to_v16bf16(<32 x i8> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: ; implicit-def: $vgpr0_vgpr1_vgpr2_vgpr3_vgpr4_vgpr5_vgpr6_vgpr7
; GFX11-TRUE16-NEXT: v_lshlrev_b16 v10.h, 8, v31.h
; GFX11-TRUE16-NEXT: v_cmpx_ne_u32_e32 0, v37
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB110_3
; GFX11-TRUE16-NEXT: ; %bb.1: ; %Flow
@@ -45032,6 +45148,7 @@ define inreg <16 x bfloat> @bitcast_v32i8_to_v16bf16_scalar(<32 x i8> inreg %a,
; GFX11-TRUE16-NEXT: s_and_b32 s42, s42, exec_lo
; GFX11-TRUE16-NEXT: s_cselect_b32 s42, 1, 0
; GFX11-TRUE16-NEXT: s_cmp_lg_u32 s42, 1
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_cbranch_scc1 .LBB111_5
; GFX11-TRUE16-NEXT: ; %bb.4: ; %cmp.true
; GFX11-TRUE16-NEXT: v_mov_b32_e32 v0, 0xc0c0004
@@ -45160,6 +45277,7 @@ define inreg <16 x bfloat> @bitcast_v32i8_to_v16bf16_scalar(<32 x i8> inreg %a,
; GFX11-FAKE16-NEXT: s_and_b32 s42, s42, exec_lo
; GFX11-FAKE16-NEXT: s_cselect_b32 s42, 1, 0
; GFX11-FAKE16-NEXT: s_cmp_lg_u32 s42, 1
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_cbranch_scc1 .LBB111_5
; GFX11-FAKE16-NEXT: ; %bb.4: ; %cmp.true
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v0, 0xc0c0004
diff --git a/llvm/test/CodeGen/AMDGPU/amdgcn.bitcast.288bit.ll b/llvm/test/CodeGen/AMDGPU/amdgcn.bitcast.288bit.ll
index f27721779bc862..202760f43c1141 100644
--- a/llvm/test/CodeGen/AMDGPU/amdgcn.bitcast.288bit.ll
+++ b/llvm/test/CodeGen/AMDGPU/amdgcn.bitcast.288bit.ll
@@ -75,8 +75,9 @@ define <9 x float> @bitcast_v9i32_to_v9f32(<9 x i32> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v9
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB0_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -234,6 +235,7 @@ define inreg <9 x float> @bitcast_v9i32_to_v9f32_scalar(<9 x i32> inreg %a, i32
; GFX11-NEXT: s_and_b32 s4, s4, exec_lo
; GFX11-NEXT: s_cselect_b32 s4, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s4, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB1_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: s_add_i32 s20, s20, 3
@@ -339,8 +341,9 @@ define <9 x i32> @bitcast_v9f32_to_v9i32(<9 x float> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v9
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_dual_add_f32 v8, 1.0, v8 :: v_dual_add_f32 v7, 1.0, v7
@@ -500,6 +503,7 @@ define inreg <9 x i32> @bitcast_v9f32_to_v9i32_scalar(<9 x float> inreg %a, i32
; GFX11-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB3_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_add_f32_e64 v8, s20, 1.0
@@ -662,8 +666,9 @@ define <18 x i16> @bitcast_v9i32_to_v18i16(<9 x i32> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v9
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB4_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -875,6 +880,7 @@ define inreg <18 x i16> @bitcast_v9i32_to_v18i16_scalar(<9 x i32> inreg %a, i32
; GFX11-NEXT: s_and_b32 s4, s4, exec_lo
; GFX11-NEXT: s_cselect_b32 s4, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s4, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB5_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: s_add_i32 s20, s20, 3
@@ -1100,8 +1106,9 @@ define <9 x i32> @bitcast_v18i16_to_v9i32(<18 x i16> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v9
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB6_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -1373,6 +1380,7 @@ define inreg <9 x i32> @bitcast_v18i16_to_v9i32_scalar(<18 x i16> inreg %a, i32
; GFX11-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB7_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_pk_add_u16 v8, s20, 3 op_sel_hi:[1,0]
@@ -1535,8 +1543,9 @@ define <18 x half> @bitcast_v9i32_to_v18f16(<9 x i32> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v9
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB8_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -1748,6 +1757,7 @@ define inreg <18 x half> @bitcast_v9i32_to_v18f16_scalar(<9 x i32> inreg %a, i32
; GFX11-NEXT: s_and_b32 s4, s4, exec_lo
; GFX11-NEXT: s_cselect_b32 s4, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s4, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB9_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: s_add_i32 s20, s20, 3
@@ -2010,8 +2020,9 @@ define <9 x i32> @bitcast_v18f16_to_v9i32(<18 x half> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v9
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB10_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -2314,6 +2325,7 @@ define inreg <9 x i32> @bitcast_v18f16_to_v9i32_scalar(<18 x half> inreg %a, i32
; GFX11-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB11_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_pk_add_f16 v8, 0x200, s20 op_sel_hi:[0,1]
@@ -2476,8 +2488,9 @@ define <18 x i16> @bitcast_v9f32_to_v18i16(<9 x float> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v9
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_dual_add_f32 v8, 1.0, v8 :: v_dual_add_f32 v7, 1.0, v7
@@ -2715,6 +2728,7 @@ define inreg <18 x i16> @bitcast_v9f32_to_v18i16_scalar(<9 x float> inreg %a, i3
; GFX11-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB13_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_add_f32_e64 v8, s20, 1.0
@@ -2943,8 +2957,9 @@ define <9 x float> @bitcast_v18i16_to_v9f32(<18 x i16> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v9
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB14_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -3216,6 +3231,7 @@ define inreg <9 x float> @bitcast_v18i16_to_v9f32_scalar(<18 x i16> inreg %a, i3
; GFX11-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB15_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_pk_add_u16 v8, s20, 3 op_sel_hi:[1,0]
@@ -3378,8 +3394,9 @@ define <18 x half> @bitcast_v9f32_to_v18f16(<9 x float> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v9
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_dual_add_f32 v8, 1.0, v8 :: v_dual_add_f32 v7, 1.0, v7
@@ -3617,6 +3634,7 @@ define inreg <18 x half> @bitcast_v9f32_to_v18f16_scalar(<9 x float> inreg %a, i
; GFX11-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB17_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_add_f32_e64 v8, s20, 1.0
@@ -3882,8 +3900,9 @@ define <9 x float> @bitcast_v18f16_to_v9f32(<18 x half> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v9
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB18_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -4186,6 +4205,7 @@ define inreg <9 x float> @bitcast_v18f16_to_v9f32_scalar(<18 x half> inreg %a, i
; GFX11-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB19_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_pk_add_f16 v8, 0x200, s20 op_sel_hi:[0,1]
@@ -4450,8 +4470,9 @@ define <18 x half> @bitcast_v18i16_to_v18f16(<18 x i16> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v9
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB20_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -4782,6 +4803,7 @@ define inreg <18 x half> @bitcast_v18i16_to_v18f16_scalar(<18 x i16> inreg %a, i
; GFX11-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB21_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_pk_add_u16 v8, s20, 3 op_sel_hi:[1,0]
@@ -5016,8 +5038,9 @@ define <18 x i16> @bitcast_v18f16_to_v18i16(<18 x half> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v9
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB22_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -5347,6 +5370,7 @@ define inreg <18 x i16> @bitcast_v18f16_to_v18i16_scalar(<18 x half> inreg %a, i
; GFX11-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB23_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_pk_add_f16 v8, 0x200, s20 op_sel_hi:[0,1]
diff --git a/llvm/test/CodeGen/AMDGPU/amdgcn.bitcast.320bit.ll b/llvm/test/CodeGen/AMDGPU/amdgcn.bitcast.320bit.ll
index c108fb664ca936..e6fe67cba3a3e9 100644
--- a/llvm/test/CodeGen/AMDGPU/amdgcn.bitcast.320bit.ll
+++ b/llvm/test/CodeGen/AMDGPU/amdgcn.bitcast.320bit.ll
@@ -81,8 +81,9 @@ define <10 x float> @bitcast_v10i32_to_v10f32(<10 x i32> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v10
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB0_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -247,6 +248,7 @@ define inreg <10 x float> @bitcast_v10i32_to_v10f32_scalar(<10 x i32> inreg %a,
; GFX11-NEXT: s_and_b32 s4, s4, exec_lo
; GFX11-NEXT: s_cselect_b32 s4, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s4, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB1_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: s_add_i32 s21, s21, 3
@@ -359,8 +361,9 @@ define <10 x i32> @bitcast_v10f32_to_v10i32(<10 x float> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v10
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_dual_add_f32 v9, 1.0, v9 :: v_dual_add_f32 v8, 1.0, v8
@@ -526,6 +529,7 @@ define inreg <10 x i32> @bitcast_v10f32_to_v10i32_scalar(<10 x float> inreg %a,
; GFX11-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB3_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_add_f32_e64 v9, s21, 1.0
@@ -701,8 +705,9 @@ define <20 x i16> @bitcast_v10i32_to_v20i16(<10 x i32> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v10
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB4_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -927,6 +932,7 @@ define inreg <20 x i16> @bitcast_v10i32_to_v20i16_scalar(<10 x i32> inreg %a, i3
; GFX11-NEXT: s_and_b32 s4, s4, exec_lo
; GFX11-NEXT: s_cselect_b32 s4, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s4, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB5_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: s_add_i32 s21, s21, 3
@@ -1169,8 +1175,9 @@ define <10 x i32> @bitcast_v20i16_to_v10i32(<20 x i16> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v10
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB6_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -1461,6 +1468,7 @@ define inreg <10 x i32> @bitcast_v20i16_to_v10i32_scalar(<20 x i16> inreg %a, i3
; GFX11-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB7_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_pk_add_u16 v9, s21, 3 op_sel_hi:[1,0]
@@ -1636,8 +1644,9 @@ define <20 x half> @bitcast_v10i32_to_v20f16(<10 x i32> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v10
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB8_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -1862,6 +1871,7 @@ define inreg <20 x half> @bitcast_v10i32_to_v20f16_scalar(<10 x i32> inreg %a, i
; GFX11-NEXT: s_and_b32 s4, s4, exec_lo
; GFX11-NEXT: s_cselect_b32 s4, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s4, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB9_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: s_add_i32 s21, s21, 3
@@ -2144,8 +2154,9 @@ define <10 x i32> @bitcast_v20f16_to_v10i32(<20 x half> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v10
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB10_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -2470,6 +2481,7 @@ define inreg <10 x i32> @bitcast_v20f16_to_v10i32_scalar(<20 x half> inreg %a, i
; GFX11-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB11_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_pk_add_f16 v9, 0x200, s21 op_sel_hi:[0,1]
@@ -3175,6 +3187,7 @@ define <40 x i8> @bitcast_v10i32_to_v40i8(<10 x i32> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v16, 8, v1
; GFX11-TRUE16-NEXT: .LBB12_4: ; %end
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_perm_b32 v3, v3, v36, 0xc0c0004
; GFX11-TRUE16-NEXT: v_perm_b32 v14, v35, v14, 0xc0c0004
; GFX11-TRUE16-NEXT: v_perm_b32 v5, v5, v31, 0xc0c0004
@@ -3326,6 +3339,7 @@ define <40 x i8> @bitcast_v10i32_to_v40i8(<10 x i32> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v16, 8, v1
; GFX11-FAKE16-NEXT: .LBB12_4: ; %end
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_perm_b32 v3, v3, v36, 0xc0c0004
; GFX11-FAKE16-NEXT: v_perm_b32 v14, v35, v14, 0xc0c0004
; GFX11-FAKE16-NEXT: v_perm_b32 v5, v5, v31, 0xc0c0004
@@ -4074,6 +4088,7 @@ define inreg <40 x i8> @bitcast_v10i32_to_v40i8_scalar(<10 x i32> inreg %a, i32
; GFX11-NEXT: s_and_b32 s5, s63, exec_lo
; GFX11-NEXT: s_cselect_b32 s5, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s5, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB13_5
; GFX11-NEXT: ; %bb.4: ; %cmp.true
; GFX11-NEXT: s_add_i32 s1, s1, 3
@@ -4928,6 +4943,7 @@ define <10 x i32> @bitcast_v40i8_to_v10i32(<40 x i8> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: ; implicit-def: $vgpr0_vgpr1_vgpr2_vgpr3_vgpr4_vgpr5_vgpr6_vgpr7_vgpr8_vgpr9
; GFX11-TRUE16-NEXT: s_waitcnt vmcnt(5)
; GFX11-TRUE16-NEXT: v_cmpx_ne_u32_e32 0, v66
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB14_3
; GFX11-TRUE16-NEXT: ; %bb.1: ; %Flow
@@ -6136,6 +6152,7 @@ define inreg <10 x i32> @bitcast_v40i8_to_v10i32_scalar(<40 x i8> inreg %a, i32
; GFX11-NEXT: s_and_b32 s58, s58, exec_lo
; GFX11-NEXT: s_cselect_b32 s58, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s58, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB15_5
; GFX11-NEXT: ; %bb.4: ; %cmp.true
; GFX11-NEXT: v_mov_b32_e32 v5, 0xc0c0004
@@ -6323,8 +6340,9 @@ define <5 x double> @bitcast_v10i32_to_v5f64(<10 x i32> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v10
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB16_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -6489,6 +6507,7 @@ define inreg <5 x double> @bitcast_v10i32_to_v5f64_scalar(<10 x i32> inreg %a, i
; GFX11-NEXT: s_and_b32 s4, s4, exec_lo
; GFX11-NEXT: s_cselect_b32 s4, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s4, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB17_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: s_add_i32 s21, s21, 3
@@ -6586,8 +6605,9 @@ define <10 x i32> @bitcast_v5f64_to_v10i32(<5 x double> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v10
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB18_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -6739,6 +6759,7 @@ define inreg <10 x i32> @bitcast_v5f64_to_v10i32_scalar(<5 x double> inreg %a, i
; GFX11-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB19_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_add_f64 v[8:9], s[20:21], 1.0
@@ -6846,8 +6867,9 @@ define <5 x i64> @bitcast_v10i32_to_v5i64(<10 x i32> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v10
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB20_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -7012,6 +7034,7 @@ define inreg <5 x i64> @bitcast_v10i32_to_v5i64_scalar(<10 x i32> inreg %a, i32
; GFX11-NEXT: s_and_b32 s4, s4, exec_lo
; GFX11-NEXT: s_cselect_b32 s4, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s4, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB21_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: s_add_i32 s21, s21, 3
@@ -7124,23 +7147,21 @@ define <10 x i32> @bitcast_v5i64_to_v10i32(<5 x i64> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v10
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB22_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_add_co_u32 v8, vcc_lo, v8, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v9, null, 0, v9, vcc_lo
; GFX11-NEXT: v_add_co_u32 v6, vcc_lo, v6, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v7, null, 0, v7, vcc_lo
; GFX11-NEXT: v_add_co_u32 v4, vcc_lo, v4, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v5, null, 0, v5, vcc_lo
; GFX11-NEXT: v_add_co_u32 v2, vcc_lo, v2, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX11-NEXT: v_add_co_u32 v0, vcc_lo, v0, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-NEXT: .LBB22_2: ; %end
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -7293,6 +7314,7 @@ define inreg <10 x i32> @bitcast_v5i64_to_v10i32_scalar(<5 x i64> inreg %a, i32
; GFX11-NEXT: s_and_b32 s4, s4, exec_lo
; GFX11-NEXT: s_cselect_b32 s4, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s4, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB23_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: s_add_u32 s20, s20, 3
@@ -7468,8 +7490,9 @@ define <20 x i16> @bitcast_v10f32_to_v20i16(<10 x float> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v10
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_dual_add_f32 v9, 1.0, v9 :: v_dual_add_f32 v8, 1.0, v8
@@ -7718,6 +7741,7 @@ define inreg <20 x i16> @bitcast_v10f32_to_v20i16_scalar(<10 x float> inreg %a,
; GFX11-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB25_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_add_f32_e64 v9, s21, 1.0
@@ -7963,8 +7987,9 @@ define <10 x float> @bitcast_v20i16_to_v10f32(<20 x i16> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v10
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB26_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -8255,6 +8280,7 @@ define inreg <10 x float> @bitcast_v20i16_to_v10f32_scalar(<20 x i16> inreg %a,
; GFX11-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB27_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_pk_add_u16 v9, s21, 3 op_sel_hi:[1,0]
@@ -8430,8 +8456,9 @@ define <20 x half> @bitcast_v10f32_to_v20f16(<10 x float> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v10
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_dual_add_f32 v9, 1.0, v9 :: v_dual_add_f32 v8, 1.0, v8
@@ -8680,6 +8707,7 @@ define inreg <20 x half> @bitcast_v10f32_to_v20f16_scalar(<10 x float> inreg %a,
; GFX11-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB29_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_add_f32_e64 v9, s21, 1.0
@@ -8965,8 +8993,9 @@ define <10 x float> @bitcast_v20f16_to_v10f32(<20 x half> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v10
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB30_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -9291,6 +9320,7 @@ define inreg <10 x float> @bitcast_v20f16_to_v10f32_scalar(<20 x half> inreg %a,
; GFX11-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB31_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_pk_add_f16 v9, 0x200, s21 op_sel_hi:[0,1]
@@ -9992,6 +10022,7 @@ define <40 x i8> @bitcast_v10f32_to_v40i8(<10 x float> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v16, 8, v1
; GFX11-TRUE16-NEXT: .LBB32_4: ; %end
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_perm_b32 v3, v3, v36, 0xc0c0004
; GFX11-TRUE16-NEXT: v_perm_b32 v14, v35, v14, 0xc0c0004
; GFX11-TRUE16-NEXT: v_perm_b32 v5, v5, v31, 0xc0c0004
@@ -10139,6 +10170,7 @@ define <40 x i8> @bitcast_v10f32_to_v40i8(<10 x float> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v16, 8, v1
; GFX11-FAKE16-NEXT: .LBB32_4: ; %end
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_perm_b32 v3, v3, v36, 0xc0c0004
; GFX11-FAKE16-NEXT: v_perm_b32 v14, v35, v14, 0xc0c0004
; GFX11-FAKE16-NEXT: v_perm_b32 v5, v5, v31, 0xc0c0004
@@ -10959,6 +10991,7 @@ define inreg <40 x i8> @bitcast_v10f32_to_v40i8_scalar(<10 x float> inreg %a, i3
; GFX11-NEXT: s_and_b32 s5, s14, exec_lo
; GFX11-NEXT: s_cselect_b32 s5, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s5, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB33_5
; GFX11-NEXT: ; %bb.4: ; %cmp.true
; GFX11-NEXT: v_add_f32_e64 v2, s21, 1.0
@@ -11834,6 +11867,7 @@ define <10 x float> @bitcast_v40i8_to_v10f32(<40 x i8> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: ; implicit-def: $vgpr0_vgpr1_vgpr2_vgpr3_vgpr4_vgpr5_vgpr6_vgpr7_vgpr8_vgpr9
; GFX11-TRUE16-NEXT: s_waitcnt vmcnt(5)
; GFX11-TRUE16-NEXT: v_cmpx_ne_u32_e32 0, v66
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB34_3
; GFX11-TRUE16-NEXT: ; %bb.1: ; %Flow
@@ -13042,6 +13076,7 @@ define inreg <10 x float> @bitcast_v40i8_to_v10f32_scalar(<40 x i8> inreg %a, i3
; GFX11-NEXT: s_and_b32 s58, s58, exec_lo
; GFX11-NEXT: s_cselect_b32 s58, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s58, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB35_5
; GFX11-NEXT: ; %bb.4: ; %cmp.true
; GFX11-NEXT: v_mov_b32_e32 v5, 0xc0c0004
@@ -13229,8 +13264,9 @@ define <5 x double> @bitcast_v10f32_to_v5f64(<10 x float> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v10
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_dual_add_f32 v9, 1.0, v9 :: v_dual_add_f32 v8, 1.0, v8
@@ -13414,6 +13450,7 @@ define inreg <5 x double> @bitcast_v10f32_to_v5f64_scalar(<10 x float> inreg %a,
; GFX11-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB37_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_add_f32_e64 v9, s21, 1.0
@@ -13514,8 +13551,9 @@ define <10 x float> @bitcast_v5f64_to_v10f32(<5 x double> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v10
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB38_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -13667,6 +13705,7 @@ define inreg <10 x float> @bitcast_v5f64_to_v10f32_scalar(<5 x double> inreg %a,
; GFX11-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB39_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_add_f64 v[8:9], s[20:21], 1.0
@@ -13774,8 +13813,9 @@ define <5 x i64> @bitcast_v10f32_to_v5i64(<10 x float> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v10
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_dual_add_f32 v9, 1.0, v9 :: v_dual_add_f32 v8, 1.0, v8
@@ -13959,6 +13999,7 @@ define inreg <5 x i64> @bitcast_v10f32_to_v5i64_scalar(<10 x float> inreg %a, i3
; GFX11-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB41_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_add_f32_e64 v9, s21, 1.0
@@ -14074,23 +14115,21 @@ define <10 x float> @bitcast_v5i64_to_v10f32(<5 x i64> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v10
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB42_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_add_co_u32 v8, vcc_lo, v8, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v9, null, 0, v9, vcc_lo
; GFX11-NEXT: v_add_co_u32 v6, vcc_lo, v6, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v7, null, 0, v7, vcc_lo
; GFX11-NEXT: v_add_co_u32 v4, vcc_lo, v4, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v5, null, 0, v5, vcc_lo
; GFX11-NEXT: v_add_co_u32 v2, vcc_lo, v2, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX11-NEXT: v_add_co_u32 v0, vcc_lo, v0, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-NEXT: .LBB42_2: ; %end
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -14243,6 +14282,7 @@ define inreg <10 x float> @bitcast_v5i64_to_v10f32_scalar(<5 x i64> inreg %a, i3
; GFX11-NEXT: s_and_b32 s4, s4, exec_lo
; GFX11-NEXT: s_cselect_b32 s4, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s4, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB43_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: s_add_u32 s20, s20, 3
@@ -14530,8 +14570,9 @@ define <20 x half> @bitcast_v20i16_to_v20f16(<20 x i16> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v10
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB44_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -14887,6 +14928,7 @@ define inreg <20 x half> @bitcast_v20i16_to_v20f16_scalar(<20 x i16> inreg %a, i
; GFX11-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB45_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_pk_add_u16 v9, s21, 3 op_sel_hi:[1,0]
@@ -15139,8 +15181,9 @@ define <20 x i16> @bitcast_v20f16_to_v20i16(<20 x half> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v10
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB46_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -15492,6 +15535,7 @@ define inreg <20 x i16> @bitcast_v20f16_to_v20i16_scalar(<20 x half> inreg %a, i
; GFX11-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB47_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_pk_add_f16 v9, 0x200, s21 op_sel_hi:[0,1]
@@ -16365,6 +16409,7 @@ define <40 x i8> @bitcast_v20i16_to_v40i8(<20 x i16> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v16, 8, v1
; GFX11-TRUE16-NEXT: .LBB48_4: ; %end
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_perm_b32 v3, v3, v36, 0xc0c0004
; GFX11-TRUE16-NEXT: v_perm_b32 v14, v35, v14, 0xc0c0004
; GFX11-TRUE16-NEXT: v_perm_b32 v5, v5, v31, 0xc0c0004
@@ -16516,6 +16561,7 @@ define <40 x i8> @bitcast_v20i16_to_v40i8(<20 x i16> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v16, 8, v1
; GFX11-FAKE16-NEXT: .LBB48_4: ; %end
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_perm_b32 v3, v3, v36, 0xc0c0004
; GFX11-FAKE16-NEXT: v_perm_b32 v14, v35, v14, 0xc0c0004
; GFX11-FAKE16-NEXT: v_perm_b32 v5, v5, v31, 0xc0c0004
@@ -17413,6 +17459,7 @@ define inreg <40 x i8> @bitcast_v20i16_to_v40i8_scalar(<20 x i16> inreg %a, i32
; GFX11-NEXT: s_and_b32 s5, s14, exec_lo
; GFX11-NEXT: s_cselect_b32 s5, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s5, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB49_5
; GFX11-NEXT: ; %bb.4: ; %cmp.true
; GFX11-NEXT: v_pk_add_u16 v2, s21, 3 op_sel_hi:[1,0]
@@ -18437,6 +18484,7 @@ define <20 x i16> @bitcast_v40i8_to_v20i16(<40 x i8> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: v_lshlrev_b16 v16.l, 8, v33.h
; GFX11-TRUE16-NEXT: v_lshlrev_b16 v16.h, 8, v34.h
; GFX11-TRUE16-NEXT: v_cmpx_ne_u32_e32 0, v70
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB50_3
; GFX11-TRUE16-NEXT: ; %bb.1: ; %Flow
@@ -19730,6 +19778,7 @@ define inreg <20 x i16> @bitcast_v40i8_to_v20i16_scalar(<40 x i8> inreg %a, i32
; GFX11-TRUE16-NEXT: s_and_b32 s58, s58, exec_lo
; GFX11-TRUE16-NEXT: s_cselect_b32 s58, 1, 0
; GFX11-TRUE16-NEXT: s_cmp_lg_u32 s58, 1
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_cbranch_scc1 .LBB51_5
; GFX11-TRUE16-NEXT: ; %bb.4: ; %cmp.true
; GFX11-TRUE16-NEXT: v_mov_b32_e32 v0, 0xc0c0004
@@ -19888,6 +19937,7 @@ define inreg <20 x i16> @bitcast_v40i8_to_v20i16_scalar(<40 x i8> inreg %a, i32
; GFX11-FAKE16-NEXT: s_and_b32 s58, s58, exec_lo
; GFX11-FAKE16-NEXT: s_cselect_b32 s58, 1, 0
; GFX11-FAKE16-NEXT: s_cmp_lg_u32 s58, 1
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_cbranch_scc1 .LBB51_5
; GFX11-FAKE16-NEXT: ; %bb.4: ; %cmp.true
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v0, 0xc0c0004
@@ -20195,8 +20245,9 @@ define <5 x double> @bitcast_v20i16_to_v5f64(<20 x i16> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v10
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB52_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -20517,6 +20568,7 @@ define inreg <5 x double> @bitcast_v20i16_to_v5f64_scalar(<20 x i16> inreg %a, i
; GFX11-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB53_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_pk_add_u16 v9, s21, 3 op_sel_hi:[1,0]
@@ -20680,8 +20732,9 @@ define <20 x i16> @bitcast_v5f64_to_v20i16(<5 x double> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v10
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB54_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -20916,6 +20969,7 @@ define inreg <20 x i16> @bitcast_v5f64_to_v20i16_scalar(<5 x double> inreg %a, i
; GFX11-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB55_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_add_f64 v[8:9], s[20:21], 1.0
@@ -21156,8 +21210,9 @@ define <5 x i64> @bitcast_v20i16_to_v5i64(<20 x i16> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v10
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB56_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -21478,6 +21533,7 @@ define inreg <5 x i64> @bitcast_v20i16_to_v5i64_scalar(<20 x i16> inreg %a, i32
; GFX11-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB57_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_pk_add_u16 v9, s21, 3 op_sel_hi:[1,0]
@@ -21656,23 +21712,21 @@ define <20 x i16> @bitcast_v5i64_to_v20i16(<5 x i64> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v10
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB58_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_add_co_u32 v8, vcc_lo, v8, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v9, null, 0, v9, vcc_lo
; GFX11-NEXT: v_add_co_u32 v6, vcc_lo, v6, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v7, null, 0, v7, vcc_lo
; GFX11-NEXT: v_add_co_u32 v4, vcc_lo, v4, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v5, null, 0, v5, vcc_lo
; GFX11-NEXT: v_add_co_u32 v2, vcc_lo, v2, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX11-NEXT: v_add_co_u32 v0, vcc_lo, v0, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-NEXT: .LBB58_2: ; %end
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -21885,6 +21939,7 @@ define inreg <20 x i16> @bitcast_v5i64_to_v20i16_scalar(<5 x i64> inreg %a, i32
; GFX11-NEXT: s_and_b32 s4, s4, exec_lo
; GFX11-NEXT: s_cselect_b32 s4, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s4, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB59_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: s_add_u32 s20, s20, 3
@@ -22739,6 +22794,7 @@ define <40 x i8> @bitcast_v20f16_to_v40i8(<20 x half> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v16, 8, v1
; GFX11-TRUE16-NEXT: .LBB60_4: ; %end
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_perm_b32 v3, v3, v36, 0xc0c0004
; GFX11-TRUE16-NEXT: v_perm_b32 v14, v35, v14, 0xc0c0004
; GFX11-TRUE16-NEXT: v_perm_b32 v5, v5, v31, 0xc0c0004
@@ -22890,6 +22946,7 @@ define <40 x i8> @bitcast_v20f16_to_v40i8(<20 x half> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v16, 8, v1
; GFX11-FAKE16-NEXT: .LBB60_4: ; %end
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_perm_b32 v3, v3, v36, 0xc0c0004
; GFX11-FAKE16-NEXT: v_perm_b32 v14, v35, v14, 0xc0c0004
; GFX11-FAKE16-NEXT: v_perm_b32 v5, v5, v31, 0xc0c0004
@@ -23852,6 +23909,7 @@ define inreg <40 x i8> @bitcast_v20f16_to_v40i8_scalar(<20 x half> inreg %a, i32
; GFX11-NEXT: s_and_b32 s5, s14, exec_lo
; GFX11-NEXT: s_cselect_b32 s5, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s5, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB61_5
; GFX11-NEXT: ; %bb.4: ; %cmp.true
; GFX11-NEXT: v_pk_add_f16 v2, 0x200, s21 op_sel_hi:[0,1]
@@ -24876,6 +24934,7 @@ define <20 x half> @bitcast_v40i8_to_v20f16(<40 x i8> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: v_lshlrev_b16 v16.l, 8, v33.h
; GFX11-TRUE16-NEXT: v_lshlrev_b16 v16.h, 8, v34.h
; GFX11-TRUE16-NEXT: v_cmpx_ne_u32_e32 0, v70
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB62_3
; GFX11-TRUE16-NEXT: ; %bb.1: ; %Flow
@@ -26169,6 +26228,7 @@ define inreg <20 x half> @bitcast_v40i8_to_v20f16_scalar(<40 x i8> inreg %a, i32
; GFX11-TRUE16-NEXT: s_and_b32 s58, s58, exec_lo
; GFX11-TRUE16-NEXT: s_cselect_b32 s58, 1, 0
; GFX11-TRUE16-NEXT: s_cmp_lg_u32 s58, 1
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_cbranch_scc1 .LBB63_5
; GFX11-TRUE16-NEXT: ; %bb.4: ; %cmp.true
; GFX11-TRUE16-NEXT: v_mov_b32_e32 v0, 0xc0c0004
@@ -26327,6 +26387,7 @@ define inreg <20 x half> @bitcast_v40i8_to_v20f16_scalar(<40 x i8> inreg %a, i32
; GFX11-FAKE16-NEXT: s_and_b32 s58, s58, exec_lo
; GFX11-FAKE16-NEXT: s_cselect_b32 s58, 1, 0
; GFX11-FAKE16-NEXT: s_cmp_lg_u32 s58, 1
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_cbranch_scc1 .LBB63_5
; GFX11-FAKE16-NEXT: ; %bb.4: ; %cmp.true
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v0, 0xc0c0004
@@ -26674,8 +26735,9 @@ define <5 x double> @bitcast_v20f16_to_v5f64(<20 x half> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v10
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB64_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -27043,6 +27105,7 @@ define inreg <5 x double> @bitcast_v20f16_to_v5f64_scalar(<20 x half> inreg %a,
; GFX11-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB65_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_pk_add_f16 v9, 0x200, s21 op_sel_hi:[0,1]
@@ -27206,8 +27269,9 @@ define <20 x half> @bitcast_v5f64_to_v20f16(<5 x double> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v10
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB66_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -27442,6 +27506,7 @@ define inreg <20 x half> @bitcast_v5f64_to_v20f16_scalar(<5 x double> inreg %a,
; GFX11-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB67_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_add_f64 v[8:9], s[20:21], 1.0
@@ -27722,8 +27787,9 @@ define <5 x i64> @bitcast_v20f16_to_v5i64(<20 x half> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v10
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB68_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -28091,6 +28157,7 @@ define inreg <5 x i64> @bitcast_v20f16_to_v5i64_scalar(<20 x half> inreg %a, i32
; GFX11-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB69_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_pk_add_f16 v9, 0x200, s21 op_sel_hi:[0,1]
@@ -28269,23 +28336,21 @@ define <20 x half> @bitcast_v5i64_to_v20f16(<5 x i64> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v10
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB70_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_add_co_u32 v8, vcc_lo, v8, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v9, null, 0, v9, vcc_lo
; GFX11-NEXT: v_add_co_u32 v6, vcc_lo, v6, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v7, null, 0, v7, vcc_lo
; GFX11-NEXT: v_add_co_u32 v4, vcc_lo, v4, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v5, null, 0, v5, vcc_lo
; GFX11-NEXT: v_add_co_u32 v2, vcc_lo, v2, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX11-NEXT: v_add_co_u32 v0, vcc_lo, v0, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-NEXT: .LBB70_2: ; %end
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -28498,6 +28563,7 @@ define inreg <20 x half> @bitcast_v5i64_to_v20f16_scalar(<5 x i64> inreg %a, i32
; GFX11-NEXT: s_and_b32 s4, s4, exec_lo
; GFX11-NEXT: s_cselect_b32 s4, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s4, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB71_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: s_add_u32 s20, s20, 3
@@ -29336,6 +29402,7 @@ define <5 x double> @bitcast_v40i8_to_v5f64(<40 x i8> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: ; implicit-def: $vgpr0_vgpr1_vgpr2_vgpr3_vgpr4_vgpr5_vgpr6_vgpr7_vgpr8_vgpr9_vgpr10_vgpr11_vgpr12_vgpr13_vgpr14_vgpr15
; GFX11-TRUE16-NEXT: s_waitcnt vmcnt(5)
; GFX11-TRUE16-NEXT: v_cmpx_ne_u32_e32 0, v80
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB72_3
; GFX11-TRUE16-NEXT: ; %bb.1: ; %Flow
@@ -30571,6 +30638,7 @@ define inreg <5 x double> @bitcast_v40i8_to_v5f64_scalar(<40 x i8> inreg %a, i32
; GFX11-NEXT: s_and_b32 s58, s58, exec_lo
; GFX11-NEXT: s_cselect_b32 s58, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s58, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB73_5
; GFX11-NEXT: ; %bb.4: ; %cmp.true
; GFX11-NEXT: v_mov_b32_e32 v5, 0xc0c0004
@@ -31331,6 +31399,7 @@ define <40 x i8> @bitcast_v5f64_to_v40i8(<5 x double> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v16, 8, v1
; GFX11-TRUE16-NEXT: .LBB74_4: ; %end
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_perm_b32 v3, v3, v36, 0xc0c0004
; GFX11-TRUE16-NEXT: v_perm_b32 v14, v35, v14, 0xc0c0004
; GFX11-TRUE16-NEXT: v_perm_b32 v5, v5, v31, 0xc0c0004
@@ -31477,6 +31546,7 @@ define <40 x i8> @bitcast_v5f64_to_v40i8(<5 x double> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v16, 8, v1
; GFX11-FAKE16-NEXT: .LBB74_4: ; %end
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_perm_b32 v3, v3, v36, 0xc0c0004
; GFX11-FAKE16-NEXT: v_perm_b32 v14, v35, v14, 0xc0c0004
; GFX11-FAKE16-NEXT: v_perm_b32 v5, v5, v31, 0xc0c0004
@@ -32277,6 +32347,7 @@ define inreg <40 x i8> @bitcast_v5f64_to_v40i8_scalar(<5 x double> inreg %a, i32
; GFX11-NEXT: s_and_b32 s5, s14, exec_lo
; GFX11-NEXT: s_cselect_b32 s5, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s5, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB75_5
; GFX11-NEXT: ; %bb.4: ; %cmp.true
; GFX11-NEXT: v_add_f64 v[5:6], s[16:17], 1.0
@@ -33201,6 +33272,7 @@ define <5 x i64> @bitcast_v40i8_to_v5i64(<40 x i8> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: ; implicit-def: $vgpr0_vgpr1_vgpr2_vgpr3_vgpr4_vgpr5_vgpr6_vgpr7_vgpr8_vgpr9_vgpr10_vgpr11_vgpr12_vgpr13_vgpr14_vgpr15
; GFX11-TRUE16-NEXT: s_waitcnt vmcnt(5)
; GFX11-TRUE16-NEXT: v_cmpx_ne_u32_e32 0, v80
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB76_3
; GFX11-TRUE16-NEXT: ; %bb.1: ; %Flow
@@ -34436,6 +34508,7 @@ define inreg <5 x i64> @bitcast_v40i8_to_v5i64_scalar(<40 x i8> inreg %a, i32 in
; GFX11-NEXT: s_and_b32 s58, s58, exec_lo
; GFX11-NEXT: s_cselect_b32 s58, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s58, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB77_5
; GFX11-NEXT: ; %bb.4: ; %cmp.true
; GFX11-NEXT: v_mov_b32_e32 v5, 0xc0c0004
@@ -35175,18 +35248,16 @@ define <40 x i8> @bitcast_v5i64_to_v40i8(<5 x i64> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: s_cbranch_execz .LBB78_4
; GFX11-TRUE16-NEXT: ; %bb.3: ; %cmp.true
; GFX11-TRUE16-NEXT: v_add_co_u32 v3, vcc_lo, v3, 3
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v4, null, 0, v4, vcc_lo
; GFX11-TRUE16-NEXT: v_add_co_u32 v5, vcc_lo, v5, 3
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v6, null, 0, v6, vcc_lo
; GFX11-TRUE16-NEXT: v_add_co_u32 v7, vcc_lo, v7, 3
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v8, null, 0, v8, vcc_lo
; GFX11-TRUE16-NEXT: v_add_co_u32 v9, vcc_lo, v9, 3
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v10, null, 0, v10, vcc_lo
; GFX11-TRUE16-NEXT: v_add_co_u32 v1, vcc_lo, v1, 3
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_3)
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v2, null, 0, v2, vcc_lo
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX11-TRUE16-NEXT: v_lshrrev_b64 v[11:12], 24, v[9:10]
; GFX11-TRUE16-NEXT: v_lshrrev_b64 v[12:13], 24, v[7:8]
; GFX11-TRUE16-NEXT: v_lshrrev_b64 v[13:14], 24, v[5:6]
@@ -35219,6 +35290,7 @@ define <40 x i8> @bitcast_v5i64_to_v40i8(<5 x i64> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v16, 8, v1
; GFX11-TRUE16-NEXT: .LBB78_4: ; %end
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_perm_b32 v3, v3, v36, 0xc0c0004
; GFX11-TRUE16-NEXT: v_perm_b32 v14, v35, v14, 0xc0c0004
; GFX11-TRUE16-NEXT: v_perm_b32 v5, v5, v31, 0xc0c0004
@@ -35329,18 +35401,16 @@ define <40 x i8> @bitcast_v5i64_to_v40i8(<5 x i64> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: s_cbranch_execz .LBB78_4
; GFX11-FAKE16-NEXT: ; %bb.3: ; %cmp.true
; GFX11-FAKE16-NEXT: v_add_co_u32 v3, vcc_lo, v3, 3
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v4, null, 0, v4, vcc_lo
; GFX11-FAKE16-NEXT: v_add_co_u32 v5, vcc_lo, v5, 3
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v6, null, 0, v6, vcc_lo
; GFX11-FAKE16-NEXT: v_add_co_u32 v7, vcc_lo, v7, 3
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v8, null, 0, v8, vcc_lo
; GFX11-FAKE16-NEXT: v_add_co_u32 v9, vcc_lo, v9, 3
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v10, null, 0, v10, vcc_lo
; GFX11-FAKE16-NEXT: v_add_co_u32 v1, vcc_lo, v1, 3
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v2, null, 0, v2, vcc_lo
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_lshrrev_b64 v[11:12], 24, v[9:10]
; GFX11-FAKE16-NEXT: v_lshrrev_b64 v[12:13], 24, v[7:8]
; GFX11-FAKE16-NEXT: v_lshrrev_b64 v[13:14], 24, v[5:6]
@@ -35373,6 +35443,7 @@ define <40 x i8> @bitcast_v5i64_to_v40i8(<5 x i64> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v16, 8, v1
; GFX11-FAKE16-NEXT: .LBB78_4: ; %end
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_perm_b32 v3, v3, v36, 0xc0c0004
; GFX11-FAKE16-NEXT: v_perm_b32 v14, v35, v14, 0xc0c0004
; GFX11-FAKE16-NEXT: v_perm_b32 v5, v5, v31, 0xc0c0004
@@ -36121,6 +36192,7 @@ define inreg <40 x i8> @bitcast_v5i64_to_v40i8_scalar(<5 x i64> inreg %a, i32 in
; GFX11-NEXT: s_and_b32 s5, s63, exec_lo
; GFX11-NEXT: s_cselect_b32 s5, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s5, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB79_5
; GFX11-NEXT: ; %bb.4: ; %cmp.true
; GFX11-NEXT: s_add_u32 s0, s0, 3
@@ -36288,8 +36360,9 @@ define <5 x i64> @bitcast_v5f64_to_v5i64(<5 x double> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v10
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB80_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -36459,6 +36532,7 @@ define inreg <5 x i64> @bitcast_v5f64_to_v5i64_scalar(<5 x double> inreg %a, i32
; GFX11-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB81_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_add_f64 v[0:1], s[12:13], 1.0
@@ -36569,23 +36643,21 @@ define <5 x double> @bitcast_v5i64_to_v5f64(<5 x i64> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v10
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB82_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_add_co_u32 v0, vcc_lo, v0, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-NEXT: v_add_co_u32 v2, vcc_lo, v2, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX11-NEXT: v_add_co_u32 v4, vcc_lo, v4, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v5, null, 0, v5, vcc_lo
; GFX11-NEXT: v_add_co_u32 v6, vcc_lo, v6, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v7, null, 0, v7, vcc_lo
; GFX11-NEXT: v_add_co_u32 v8, vcc_lo, v8, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v9, null, 0, v9, vcc_lo
; GFX11-NEXT: .LBB82_2: ; %end
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -36738,6 +36810,7 @@ define inreg <5 x double> @bitcast_v5i64_to_v5f64_scalar(<5 x i64> inreg %a, i32
; GFX11-NEXT: s_and_b32 s4, s4, exec_lo
; GFX11-NEXT: s_cselect_b32 s4, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s4, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB83_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: s_add_u32 s0, s0, 3
diff --git a/llvm/test/CodeGen/AMDGPU/amdgcn.bitcast.32bit.ll b/llvm/test/CodeGen/AMDGPU/amdgcn.bitcast.32bit.ll
index 38a3d781676eff..fd88c16a180e89 100644
--- a/llvm/test/CodeGen/AMDGPU/amdgcn.bitcast.32bit.ll
+++ b/llvm/test/CodeGen/AMDGPU/amdgcn.bitcast.32bit.ll
@@ -51,8 +51,9 @@ define float @bitcast_i32_to_f32(i32 %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v1
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_add_nc_u32_e32 v0, 3, v0
@@ -153,6 +154,7 @@ define inreg float @bitcast_i32_to_f32_scalar(i32 inreg %a, i32 inreg %b) #0 {
; GFX11-NEXT: s_and_b32 s1, s1, exec_lo
; GFX11-NEXT: s_cselect_b32 s1, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s1, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB1_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: s_add_i32 s0, s0, 3
@@ -222,8 +224,9 @@ define i32 @bitcast_f32_to_i32(float %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v1
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_add_f32_e32 v0, 1.0, v0
@@ -327,6 +330,7 @@ define inreg i32 @bitcast_f32_to_i32_scalar(float inreg %a, i32 inreg %b) #0 {
; GFX11-NEXT: s_and_b32 s1, s1, exec_lo
; GFX11-NEXT: s_cselect_b32 s1, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s1, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB3_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_add_f32_e64 v0, s0, 1.0
@@ -404,8 +408,9 @@ define <2 x i16> @bitcast_i32_to_v2i16(i32 %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v1
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_add_nc_u32_e32 v0, 3, v0
@@ -512,6 +517,7 @@ define inreg <2 x i16> @bitcast_i32_to_v2i16_scalar(i32 inreg %a, i32 inreg %b)
; GFX11-NEXT: s_and_b32 s1, s1, exec_lo
; GFX11-NEXT: s_cselect_b32 s1, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s1, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB5_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: s_add_i32 s0, s0, 3
@@ -603,8 +609,9 @@ define i32 @bitcast_v2i16_to_i32(<2 x i16> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v1
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_pk_add_u16 v0, v0, 3 op_sel_hi:[1,0]
@@ -719,6 +726,7 @@ define inreg i32 @bitcast_v2i16_to_i32_scalar(<2 x i16> inreg %a, i32 inreg %b)
; GFX11-NEXT: s_and_b32 s1, s1, exec_lo
; GFX11-NEXT: s_cselect_b32 s1, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s1, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB7_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_pk_add_u16 v0, s0, 3 op_sel_hi:[1,0]
@@ -796,8 +804,9 @@ define <2 x half> @bitcast_i32_to_v2f16(i32 %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v1
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_add_nc_u32_e32 v0, 3, v0
@@ -904,6 +913,7 @@ define inreg <2 x half> @bitcast_i32_to_v2f16_scalar(i32 inreg %a, i32 inreg %b)
; GFX11-NEXT: s_and_b32 s1, s1, exec_lo
; GFX11-NEXT: s_cselect_b32 s1, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s1, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB9_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: s_add_i32 s0, s0, 3
@@ -995,8 +1005,9 @@ define i32 @bitcast_v2f16_to_i32(<2 x half> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v1
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_pk_add_f16 v0, 0x200, v0 op_sel_hi:[0,1]
@@ -1118,6 +1129,7 @@ define inreg i32 @bitcast_v2f16_to_i32_scalar(<2 x half> inreg %a, i32 inreg %b)
; GFX11-NEXT: s_and_b32 s1, s1, exec_lo
; GFX11-NEXT: s_cselect_b32 s1, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s1, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB11_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_pk_add_f16 v0, 0x200, s0 op_sel_hi:[0,1]
@@ -1200,8 +1212,9 @@ define <2 x bfloat> @bitcast_i32_to_v2bf16(i32 %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v1
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_add_nc_u32_e32 v0, 3, v0
@@ -1311,6 +1324,7 @@ define inreg <2 x bfloat> @bitcast_i32_to_v2bf16_scalar(i32 inreg %a, i32 inreg
; GFX11-NEXT: s_and_b32 s1, s1, exec_lo
; GFX11-NEXT: s_cselect_b32 s1, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s1, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB13_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: s_add_i32 s0, s0, 3
@@ -1437,8 +1451,9 @@ define i32 @bitcast_v2bf16_to_i32(<2 x bfloat> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-TRUE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-TRUE16-NEXT: v_cmpx_ne_u32_e32 0, v1
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-TRUE16-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-TRUE16-NEXT: s_cbranch_execz .LBB14_2
; GFX11-TRUE16-NEXT: ; %bb.1: ; %cmp.true
@@ -1472,8 +1487,9 @@ define i32 @bitcast_v2bf16_to_i32(<2 x bfloat> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-FAKE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-FAKE16-NEXT: v_cmpx_ne_u32_e32 0, v1
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-FAKE16-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-FAKE16-NEXT: s_cbranch_execz .LBB14_2
; GFX11-FAKE16-NEXT: ; %bb.1: ; %cmp.true
@@ -1642,6 +1658,7 @@ define inreg i32 @bitcast_v2bf16_to_i32_scalar(<2 x bfloat> inreg %a, i32 inreg
; GFX11-TRUE16-NEXT: s_and_b32 s1, s1, exec_lo
; GFX11-TRUE16-NEXT: s_cselect_b32 s1, 1, 0
; GFX11-TRUE16-NEXT: s_cmp_lg_u32 s1, 1
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_cbranch_scc1 .LBB15_4
; GFX11-TRUE16-NEXT: ; %bb.3: ; %cmp.true
; GFX11-TRUE16-NEXT: s_and_b32 s1, s0, 0xffff0000
@@ -1686,6 +1703,7 @@ define inreg i32 @bitcast_v2bf16_to_i32_scalar(<2 x bfloat> inreg %a, i32 inreg
; GFX11-FAKE16-NEXT: s_and_b32 s1, s1, exec_lo
; GFX11-FAKE16-NEXT: s_cselect_b32 s1, 1, 0
; GFX11-FAKE16-NEXT: s_cmp_lg_u32 s1, 1
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_cbranch_scc1 .LBB15_4
; GFX11-FAKE16-NEXT: ; %bb.3: ; %cmp.true
; GFX11-FAKE16-NEXT: s_lshl_b32 s1, s0, 16
@@ -1778,8 +1796,9 @@ define <1 x i32> @bitcast_i32_to_v1i32(i32 %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v1
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_add_nc_u32_e32 v0, 3, v0
@@ -1880,6 +1899,7 @@ define inreg <1 x i32> @bitcast_i32_to_v1i32_scalar(i32 inreg %a, i32 inreg %b)
; GFX11-NEXT: s_and_b32 s1, s1, exec_lo
; GFX11-NEXT: s_cselect_b32 s1, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s1, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB17_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: s_add_i32 s0, s0, 3
@@ -1949,8 +1969,9 @@ define i32 @bitcast_v1i32_to_i32(<1 x i32> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v1
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_add_nc_u32_e32 v0, 3, v0
@@ -2051,6 +2072,7 @@ define inreg i32 @bitcast_v1i32_to_i32_scalar(<1 x i32> inreg %a, i32 inreg %b)
; GFX11-NEXT: s_and_b32 s1, s1, exec_lo
; GFX11-NEXT: s_cselect_b32 s1, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s1, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB19_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: s_add_i32 s0, s0, 3
@@ -2364,6 +2386,7 @@ define inreg <4 x i8> @bitcast_i32_to_v4i8_scalar(i32 inreg %a, i32 inreg %b) #0
; GFX11-NEXT: s_and_b32 s4, s4, exec_lo
; GFX11-NEXT: s_cselect_b32 s4, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s4, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB21_5
; GFX11-NEXT: ; %bb.4: ; %cmp.true
; GFX11-NEXT: s_add_i32 s0, s0, 3
@@ -2527,6 +2550,7 @@ define i32 @bitcast_v4i8_to_i32(<4 x i8> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-TRUE16-NEXT: ; implicit-def: $vgpr0
; GFX11-TRUE16-NEXT: v_cmpx_ne_u32_e32 0, v4
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB22_3
; GFX11-TRUE16-NEXT: ; %bb.1: ; %Flow
@@ -2570,6 +2594,7 @@ define i32 @bitcast_v4i8_to_i32(<4 x i8> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-FAKE16-NEXT: ; implicit-def: $vgpr0
; GFX11-FAKE16-NEXT: v_cmpx_ne_u32_e32 0, v4
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB22_3
; GFX11-FAKE16-NEXT: ; %bb.1: ; %Flow
@@ -2768,6 +2793,7 @@ define inreg i32 @bitcast_v4i8_to_i32_scalar(<4 x i8> inreg %a, i32 inreg %b) #0
; GFX11-NEXT: s_and_b32 s4, s4, exec_lo
; GFX11-NEXT: s_cselect_b32 s4, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s4, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB23_5
; GFX11-NEXT: ; %bb.4: ; %cmp.true
; GFX11-NEXT: v_mov_b32_e32 v0, 0xc0c0004
@@ -2855,8 +2881,9 @@ define <2 x i16> @bitcast_f32_to_v2i16(float %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v1
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_add_f32_e32 v0, 1.0, v0
@@ -2968,6 +2995,7 @@ define inreg <2 x i16> @bitcast_f32_to_v2i16_scalar(float inreg %a, i32 inreg %b
; GFX11-NEXT: s_and_b32 s1, s1, exec_lo
; GFX11-NEXT: s_cselect_b32 s1, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s1, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB25_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_add_f32_e64 v0, s0, 1.0
@@ -3059,8 +3087,9 @@ define float @bitcast_v2i16_to_f32(<2 x i16> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v1
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_pk_add_u16 v0, v0, 3 op_sel_hi:[1,0]
@@ -3175,6 +3204,7 @@ define inreg float @bitcast_v2i16_to_f32_scalar(<2 x i16> inreg %a, i32 inreg %b
; GFX11-NEXT: s_and_b32 s1, s1, exec_lo
; GFX11-NEXT: s_cselect_b32 s1, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s1, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB27_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_pk_add_u16 v0, s0, 3 op_sel_hi:[1,0]
@@ -3252,8 +3282,9 @@ define <2 x half> @bitcast_f32_to_v2f16(float %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v1
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_add_f32_e32 v0, 1.0, v0
@@ -3365,6 +3396,7 @@ define inreg <2 x half> @bitcast_f32_to_v2f16_scalar(float inreg %a, i32 inreg %
; GFX11-NEXT: s_and_b32 s1, s1, exec_lo
; GFX11-NEXT: s_cselect_b32 s1, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s1, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB29_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_add_f32_e64 v0, s0, 1.0
@@ -3456,8 +3488,9 @@ define float @bitcast_v2f16_to_f32(<2 x half> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v1
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_pk_add_f16 v0, 0x200, v0 op_sel_hi:[0,1]
@@ -3579,6 +3612,7 @@ define inreg float @bitcast_v2f16_to_f32_scalar(<2 x half> inreg %a, i32 inreg %
; GFX11-NEXT: s_and_b32 s1, s1, exec_lo
; GFX11-NEXT: s_cselect_b32 s1, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s1, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB31_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_pk_add_f16 v0, 0x200, s0 op_sel_hi:[0,1]
@@ -3661,8 +3695,9 @@ define <2 x bfloat> @bitcast_f32_to_v2bf16(float %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v1
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_add_f32_e32 v0, 1.0, v0
@@ -3778,6 +3813,7 @@ define inreg <2 x bfloat> @bitcast_f32_to_v2bf16_scalar(float inreg %a, i32 inre
; GFX11-NEXT: s_and_b32 s1, s1, exec_lo
; GFX11-NEXT: s_cselect_b32 s1, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s1, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB33_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_add_f32_e64 v0, s0, 1.0
@@ -3904,8 +3940,9 @@ define float @bitcast_v2bf16_to_f32(<2 x bfloat> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-TRUE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-TRUE16-NEXT: v_cmpx_ne_u32_e32 0, v1
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-TRUE16-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-TRUE16-NEXT: s_cbranch_execz .LBB34_2
; GFX11-TRUE16-NEXT: ; %bb.1: ; %cmp.true
@@ -3939,8 +3976,9 @@ define float @bitcast_v2bf16_to_f32(<2 x bfloat> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-FAKE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-FAKE16-NEXT: v_cmpx_ne_u32_e32 0, v1
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-FAKE16-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-FAKE16-NEXT: s_cbranch_execz .LBB34_2
; GFX11-FAKE16-NEXT: ; %bb.1: ; %cmp.true
@@ -4109,6 +4147,7 @@ define inreg float @bitcast_v2bf16_to_f32_scalar(<2 x bfloat> inreg %a, i32 inre
; GFX11-TRUE16-NEXT: s_and_b32 s1, s1, exec_lo
; GFX11-TRUE16-NEXT: s_cselect_b32 s1, 1, 0
; GFX11-TRUE16-NEXT: s_cmp_lg_u32 s1, 1
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_cbranch_scc1 .LBB35_4
; GFX11-TRUE16-NEXT: ; %bb.3: ; %cmp.true
; GFX11-TRUE16-NEXT: s_and_b32 s1, s0, 0xffff0000
@@ -4153,6 +4192,7 @@ define inreg float @bitcast_v2bf16_to_f32_scalar(<2 x bfloat> inreg %a, i32 inre
; GFX11-FAKE16-NEXT: s_and_b32 s1, s1, exec_lo
; GFX11-FAKE16-NEXT: s_cselect_b32 s1, 1, 0
; GFX11-FAKE16-NEXT: s_cmp_lg_u32 s1, 1
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_cbranch_scc1 .LBB35_4
; GFX11-FAKE16-NEXT: ; %bb.3: ; %cmp.true
; GFX11-FAKE16-NEXT: s_lshl_b32 s1, s0, 16
@@ -4245,8 +4285,9 @@ define <1 x i32> @bitcast_f32_to_v1i32(float %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v1
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_add_f32_e32 v0, 1.0, v0
@@ -4350,6 +4391,7 @@ define inreg <1 x i32> @bitcast_f32_to_v1i32_scalar(float inreg %a, i32 inreg %b
; GFX11-NEXT: s_and_b32 s1, s1, exec_lo
; GFX11-NEXT: s_cselect_b32 s1, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s1, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB37_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_add_f32_e64 v0, s0, 1.0
@@ -4419,8 +4461,9 @@ define float @bitcast_v1i32_to_f32(<1 x i32> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v1
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_add_nc_u32_e32 v0, 3, v0
@@ -4521,6 +4564,7 @@ define inreg float @bitcast_v1i32_to_f32_scalar(<1 x i32> inreg %a, i32 inreg %b
; GFX11-NEXT: s_and_b32 s1, s1, exec_lo
; GFX11-NEXT: s_cselect_b32 s1, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s1, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB39_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: s_add_i32 s0, s0, 3
@@ -4837,6 +4881,7 @@ define inreg <4 x i8> @bitcast_f32_to_v4i8_scalar(float inreg %a, i32 inreg %b)
; GFX11-NEXT: s_and_b32 s3, s3, exec_lo
; GFX11-NEXT: s_cselect_b32 s3, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s3, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB41_5
; GFX11-NEXT: ; %bb.4: ; %cmp.true
; GFX11-NEXT: v_add_f32_e64 v0, s0, 1.0
@@ -5000,6 +5045,7 @@ define float @bitcast_v4i8_to_f32(<4 x i8> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-TRUE16-NEXT: ; implicit-def: $vgpr0
; GFX11-TRUE16-NEXT: v_cmpx_ne_u32_e32 0, v4
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB42_3
; GFX11-TRUE16-NEXT: ; %bb.1: ; %Flow
@@ -5043,6 +5089,7 @@ define float @bitcast_v4i8_to_f32(<4 x i8> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-FAKE16-NEXT: ; implicit-def: $vgpr0
; GFX11-FAKE16-NEXT: v_cmpx_ne_u32_e32 0, v4
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB42_3
; GFX11-FAKE16-NEXT: ; %bb.1: ; %Flow
@@ -5241,6 +5288,7 @@ define inreg float @bitcast_v4i8_to_f32_scalar(<4 x i8> inreg %a, i32 inreg %b)
; GFX11-NEXT: s_and_b32 s4, s4, exec_lo
; GFX11-NEXT: s_cselect_b32 s4, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s4, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB43_5
; GFX11-NEXT: ; %bb.4: ; %cmp.true
; GFX11-NEXT: v_mov_b32_e32 v0, 0xc0c0004
@@ -5339,8 +5387,9 @@ define <2 x half> @bitcast_v2i16_to_v2f16(<2 x i16> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v1
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_pk_add_u16 v0, v0, 3 op_sel_hi:[1,0]
@@ -5459,6 +5508,7 @@ define inreg <2 x half> @bitcast_v2i16_to_v2f16_scalar(<2 x i16> inreg %a, i32 i
; GFX11-NEXT: s_and_b32 s1, s1, exec_lo
; GFX11-NEXT: s_cselect_b32 s1, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s1, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB45_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_pk_add_u16 v0, s0, 3 op_sel_hi:[1,0]
@@ -5544,8 +5594,9 @@ define <2 x i16> @bitcast_v2f16_to_v2i16(<2 x half> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v1
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_pk_add_f16 v0, 0x200, v0 op_sel_hi:[0,1]
@@ -5668,6 +5719,7 @@ define inreg <2 x i16> @bitcast_v2f16_to_v2i16_scalar(<2 x half> inreg %a, i32 i
; GFX11-NEXT: s_and_b32 s1, s1, exec_lo
; GFX11-NEXT: s_cselect_b32 s1, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s1, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB47_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_pk_add_f16 v0, 0x200, s0 op_sel_hi:[0,1]
@@ -5748,8 +5800,9 @@ define <2 x bfloat> @bitcast_v2i16_to_v2bf16(<2 x i16> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v1
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_pk_add_u16 v0, v0, 3 op_sel_hi:[1,0]
@@ -5866,6 +5919,7 @@ define inreg <2 x bfloat> @bitcast_v2i16_to_v2bf16_scalar(<2 x i16> inreg %a, i3
; GFX11-NEXT: s_and_b32 s1, s1, exec_lo
; GFX11-NEXT: s_cselect_b32 s1, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s1, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB49_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_pk_add_u16 v0, s0, 3 op_sel_hi:[1,0]
@@ -5990,8 +6044,9 @@ define <2 x i16> @bitcast_v2bf16_to_v2i16(<2 x bfloat> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-TRUE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-TRUE16-NEXT: v_cmpx_ne_u32_e32 0, v1
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-TRUE16-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-TRUE16-NEXT: s_cbranch_execz .LBB50_2
; GFX11-TRUE16-NEXT: ; %bb.1: ; %cmp.true
@@ -6025,8 +6080,9 @@ define <2 x i16> @bitcast_v2bf16_to_v2i16(<2 x bfloat> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-FAKE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-FAKE16-NEXT: v_cmpx_ne_u32_e32 0, v1
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-FAKE16-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-FAKE16-NEXT: s_cbranch_execz .LBB50_2
; GFX11-FAKE16-NEXT: ; %bb.1: ; %cmp.true
@@ -6199,6 +6255,7 @@ define inreg <2 x i16> @bitcast_v2bf16_to_v2i16_scalar(<2 x bfloat> inreg %a, i3
; GFX11-TRUE16-NEXT: s_and_b32 s1, s1, exec_lo
; GFX11-TRUE16-NEXT: s_cselect_b32 s1, 1, 0
; GFX11-TRUE16-NEXT: s_cmp_lg_u32 s1, 1
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_cbranch_scc1 .LBB51_4
; GFX11-TRUE16-NEXT: ; %bb.3: ; %cmp.true
; GFX11-TRUE16-NEXT: s_lshl_b32 s1, s0, 16
@@ -6239,6 +6296,7 @@ define inreg <2 x i16> @bitcast_v2bf16_to_v2i16_scalar(<2 x bfloat> inreg %a, i3
; GFX11-FAKE16-NEXT: s_and_b32 s1, s1, exec_lo
; GFX11-FAKE16-NEXT: s_cselect_b32 s1, 1, 0
; GFX11-FAKE16-NEXT: s_cmp_lg_u32 s1, 1
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_cbranch_scc1 .LBB51_4
; GFX11-FAKE16-NEXT: ; %bb.3: ; %cmp.true
; GFX11-FAKE16-NEXT: s_lshl_b32 s1, s0, 16
@@ -6351,8 +6409,9 @@ define <1 x i32> @bitcast_v2i16_to_v1i32(<2 x i16> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v1
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_pk_add_u16 v0, v0, 3 op_sel_hi:[1,0]
@@ -6467,6 +6526,7 @@ define inreg <1 x i32> @bitcast_v2i16_to_v1i32_scalar(<2 x i16> inreg %a, i32 in
; GFX11-NEXT: s_and_b32 s1, s1, exec_lo
; GFX11-NEXT: s_cselect_b32 s1, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s1, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB53_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_pk_add_u16 v0, s0, 3 op_sel_hi:[1,0]
@@ -6544,8 +6604,9 @@ define <2 x i16> @bitcast_v1i32_to_v2i16(<1 x i32> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v1
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_add_nc_u32_e32 v0, 3, v0
@@ -6652,6 +6713,7 @@ define inreg <2 x i16> @bitcast_v1i32_to_v2i16_scalar(<1 x i32> inreg %a, i32 in
; GFX11-NEXT: s_and_b32 s1, s1, exec_lo
; GFX11-NEXT: s_cselect_b32 s1, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s1, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB55_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: s_add_i32 s0, s0, 3
@@ -6994,6 +7056,7 @@ define inreg <4 x i8> @bitcast_v2i16_to_v4i8_scalar(<2 x i16> inreg %a, i32 inre
; GFX11-NEXT: s_and_b32 s3, s3, exec_lo
; GFX11-NEXT: s_cselect_b32 s3, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s3, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB57_5
; GFX11-NEXT: ; %bb.4: ; %cmp.true
; GFX11-NEXT: v_pk_add_u16 v0, s0, 3 op_sel_hi:[1,0]
@@ -7157,6 +7220,7 @@ define <2 x i16> @bitcast_v4i8_to_v2i16(<4 x i8> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-TRUE16-NEXT: ; implicit-def: $vgpr0
; GFX11-TRUE16-NEXT: v_cmpx_ne_u32_e32 0, v4
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB58_3
; GFX11-TRUE16-NEXT: ; %bb.1: ; %Flow
@@ -7200,6 +7264,7 @@ define <2 x i16> @bitcast_v4i8_to_v2i16(<4 x i8> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-FAKE16-NEXT: ; implicit-def: $vgpr0
; GFX11-FAKE16-NEXT: v_cmpx_ne_u32_e32 0, v4
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB58_3
; GFX11-FAKE16-NEXT: ; %bb.1: ; %Flow
@@ -7404,6 +7469,7 @@ define inreg <2 x i16> @bitcast_v4i8_to_v2i16_scalar(<4 x i8> inreg %a, i32 inre
; GFX11-NEXT: s_and_b32 s4, s4, exec_lo
; GFX11-NEXT: s_cselect_b32 s4, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s4, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB59_5
; GFX11-NEXT: ; %bb.4: ; %cmp.true
; GFX11-NEXT: v_mov_b32_e32 v0, 0xc0c0004
@@ -7508,8 +7574,9 @@ define <2 x bfloat> @bitcast_v2f16_to_v2bf16(<2 x half> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v1
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_pk_add_f16 v0, 0x200, v0 op_sel_hi:[0,1]
@@ -7637,6 +7704,7 @@ define inreg <2 x bfloat> @bitcast_v2f16_to_v2bf16_scalar(<2 x half> inreg %a, i
; GFX11-NEXT: s_and_b32 s1, s1, exec_lo
; GFX11-NEXT: s_cselect_b32 s1, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s1, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB61_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_pk_add_f16 v0, 0x200, s0 op_sel_hi:[0,1]
@@ -7761,8 +7829,9 @@ define <2 x half> @bitcast_v2bf16_to_v2f16(<2 x bfloat> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-TRUE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-TRUE16-NEXT: v_cmpx_ne_u32_e32 0, v1
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-TRUE16-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-TRUE16-NEXT: s_cbranch_execz .LBB62_2
; GFX11-TRUE16-NEXT: ; %bb.1: ; %cmp.true
@@ -7796,8 +7865,9 @@ define <2 x half> @bitcast_v2bf16_to_v2f16(<2 x bfloat> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-FAKE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-FAKE16-NEXT: v_cmpx_ne_u32_e32 0, v1
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-FAKE16-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-FAKE16-NEXT: s_cbranch_execz .LBB62_2
; GFX11-FAKE16-NEXT: ; %bb.1: ; %cmp.true
@@ -7969,6 +8039,7 @@ define inreg <2 x half> @bitcast_v2bf16_to_v2f16_scalar(<2 x bfloat> inreg %a, i
; GFX11-TRUE16-NEXT: s_and_b32 s1, s1, exec_lo
; GFX11-TRUE16-NEXT: s_cselect_b32 s1, 1, 0
; GFX11-TRUE16-NEXT: s_cmp_lg_u32 s1, 1
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_cbranch_scc1 .LBB63_4
; GFX11-TRUE16-NEXT: ; %bb.3: ; %cmp.true
; GFX11-TRUE16-NEXT: s_and_b32 s1, s0, 0xffff0000
@@ -8013,6 +8084,7 @@ define inreg <2 x half> @bitcast_v2bf16_to_v2f16_scalar(<2 x bfloat> inreg %a, i
; GFX11-FAKE16-NEXT: s_and_b32 s1, s1, exec_lo
; GFX11-FAKE16-NEXT: s_cselect_b32 s1, 1, 0
; GFX11-FAKE16-NEXT: s_cmp_lg_u32 s1, 1
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_cbranch_scc1 .LBB63_4
; GFX11-FAKE16-NEXT: ; %bb.3: ; %cmp.true
; GFX11-FAKE16-NEXT: s_lshl_b32 s1, s0, 16
@@ -8127,8 +8199,9 @@ define <1 x i32> @bitcast_v2f16_to_v1i32(<2 x half> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v1
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_pk_add_f16 v0, 0x200, v0 op_sel_hi:[0,1]
@@ -8250,6 +8323,7 @@ define inreg <1 x i32> @bitcast_v2f16_to_v1i32_scalar(<2 x half> inreg %a, i32 i
; GFX11-NEXT: s_and_b32 s1, s1, exec_lo
; GFX11-NEXT: s_cselect_b32 s1, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s1, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB65_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_pk_add_f16 v0, 0x200, s0 op_sel_hi:[0,1]
@@ -8327,8 +8401,9 @@ define <2 x half> @bitcast_v1i32_to_v2f16(<1 x i32> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v1
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_add_nc_u32_e32 v0, 3, v0
@@ -8435,6 +8510,7 @@ define inreg <2 x half> @bitcast_v1i32_to_v2f16_scalar(<1 x i32> inreg %a, i32 i
; GFX11-NEXT: s_and_b32 s1, s1, exec_lo
; GFX11-NEXT: s_cselect_b32 s1, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s1, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB67_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: s_add_i32 s0, s0, 3
@@ -8779,6 +8855,7 @@ define inreg <4 x i8> @bitcast_v2f16_to_v4i8_scalar(<2 x half> inreg %a, i32 inr
; GFX11-NEXT: s_and_b32 s3, s3, exec_lo
; GFX11-NEXT: s_cselect_b32 s3, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s3, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB69_5
; GFX11-NEXT: ; %bb.4: ; %cmp.true
; GFX11-NEXT: v_pk_add_f16 v0, 0x200, s0 op_sel_hi:[0,1]
@@ -8942,6 +9019,7 @@ define <2 x half> @bitcast_v4i8_to_v2f16(<4 x i8> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-TRUE16-NEXT: ; implicit-def: $vgpr0
; GFX11-TRUE16-NEXT: v_cmpx_ne_u32_e32 0, v4
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB70_3
; GFX11-TRUE16-NEXT: ; %bb.1: ; %Flow
@@ -8985,6 +9063,7 @@ define <2 x half> @bitcast_v4i8_to_v2f16(<4 x i8> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-FAKE16-NEXT: ; implicit-def: $vgpr0
; GFX11-FAKE16-NEXT: v_cmpx_ne_u32_e32 0, v4
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB70_3
; GFX11-FAKE16-NEXT: ; %bb.1: ; %Flow
@@ -9189,6 +9268,7 @@ define inreg <2 x half> @bitcast_v4i8_to_v2f16_scalar(<4 x i8> inreg %a, i32 inr
; GFX11-NEXT: s_and_b32 s4, s4, exec_lo
; GFX11-NEXT: s_cselect_b32 s4, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s4, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB71_5
; GFX11-NEXT: ; %bb.4: ; %cmp.true
; GFX11-NEXT: v_mov_b32_e32 v0, 0xc0c0004
@@ -9325,8 +9405,9 @@ define <1 x i32> @bitcast_v2bf16_to_v1i32(<2 x bfloat> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-TRUE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-TRUE16-NEXT: v_cmpx_ne_u32_e32 0, v1
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-TRUE16-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-TRUE16-NEXT: s_cbranch_execz .LBB72_2
; GFX11-TRUE16-NEXT: ; %bb.1: ; %cmp.true
@@ -9360,8 +9441,9 @@ define <1 x i32> @bitcast_v2bf16_to_v1i32(<2 x bfloat> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-FAKE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-FAKE16-NEXT: v_cmpx_ne_u32_e32 0, v1
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-FAKE16-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-FAKE16-NEXT: s_cbranch_execz .LBB72_2
; GFX11-FAKE16-NEXT: ; %bb.1: ; %cmp.true
@@ -9530,6 +9612,7 @@ define inreg <1 x i32> @bitcast_v2bf16_to_v1i32_scalar(<2 x bfloat> inreg %a, i3
; GFX11-TRUE16-NEXT: s_and_b32 s1, s1, exec_lo
; GFX11-TRUE16-NEXT: s_cselect_b32 s1, 1, 0
; GFX11-TRUE16-NEXT: s_cmp_lg_u32 s1, 1
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_cbranch_scc1 .LBB73_4
; GFX11-TRUE16-NEXT: ; %bb.3: ; %cmp.true
; GFX11-TRUE16-NEXT: s_and_b32 s1, s0, 0xffff0000
@@ -9574,6 +9657,7 @@ define inreg <1 x i32> @bitcast_v2bf16_to_v1i32_scalar(<2 x bfloat> inreg %a, i3
; GFX11-FAKE16-NEXT: s_and_b32 s1, s1, exec_lo
; GFX11-FAKE16-NEXT: s_cselect_b32 s1, 1, 0
; GFX11-FAKE16-NEXT: s_cmp_lg_u32 s1, 1
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_cbranch_scc1 .LBB73_4
; GFX11-FAKE16-NEXT: ; %bb.3: ; %cmp.true
; GFX11-FAKE16-NEXT: s_lshl_b32 s1, s0, 16
@@ -9679,8 +9763,9 @@ define <2 x bfloat> @bitcast_v1i32_to_v2bf16(<1 x i32> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v1
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_add_nc_u32_e32 v0, 3, v0
@@ -9790,6 +9875,7 @@ define inreg <2 x bfloat> @bitcast_v1i32_to_v2bf16_scalar(<1 x i32> inreg %a, i3
; GFX11-NEXT: s_and_b32 s1, s1, exec_lo
; GFX11-NEXT: s_cselect_b32 s1, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s1, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB75_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: s_add_i32 s0, s0, 3
@@ -10232,6 +10318,7 @@ define inreg <4 x i8> @bitcast_v2bf16_to_v4i8_scalar(<2 x bfloat> inreg %a, i32
; GFX11-TRUE16-NEXT: s_and_b32 s2, s2, exec_lo
; GFX11-TRUE16-NEXT: s_cselect_b32 s2, 1, 0
; GFX11-TRUE16-NEXT: s_cmp_lg_u32 s2, 1
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_cbranch_scc1 .LBB77_5
; GFX11-TRUE16-NEXT: ; %bb.4: ; %cmp.true
; GFX11-TRUE16-NEXT: s_lshl_b32 s1, s0, 16
@@ -10289,6 +10376,7 @@ define inreg <4 x i8> @bitcast_v2bf16_to_v4i8_scalar(<2 x bfloat> inreg %a, i32
; GFX11-FAKE16-NEXT: s_and_b32 s2, s2, exec_lo
; GFX11-FAKE16-NEXT: s_cselect_b32 s2, 1, 0
; GFX11-FAKE16-NEXT: s_cmp_lg_u32 s2, 1
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_cbranch_scc1 .LBB77_5
; GFX11-FAKE16-NEXT: ; %bb.4: ; %cmp.true
; GFX11-FAKE16-NEXT: s_lshl_b32 s1, s0, 16
@@ -10472,6 +10560,7 @@ define <2 x bfloat> @bitcast_v4i8_to_v2bf16(<4 x i8> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-TRUE16-NEXT: ; implicit-def: $vgpr0
; GFX11-TRUE16-NEXT: v_cmpx_ne_u32_e32 0, v4
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB78_3
; GFX11-TRUE16-NEXT: ; %bb.1: ; %Flow
@@ -10515,6 +10604,7 @@ define <2 x bfloat> @bitcast_v4i8_to_v2bf16(<4 x i8> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-FAKE16-NEXT: ; implicit-def: $vgpr0
; GFX11-FAKE16-NEXT: v_cmpx_ne_u32_e32 0, v4
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB78_3
; GFX11-FAKE16-NEXT: ; %bb.1: ; %Flow
@@ -10715,6 +10805,7 @@ define inreg <2 x bfloat> @bitcast_v4i8_to_v2bf16_scalar(<4 x i8> inreg %a, i32
; GFX11-NEXT: s_and_b32 s4, s4, exec_lo
; GFX11-NEXT: s_cselect_b32 s4, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s4, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB79_5
; GFX11-NEXT: ; %bb.4: ; %cmp.true
; GFX11-NEXT: v_mov_b32_e32 v0, 0xc0c0004
@@ -11038,6 +11129,7 @@ define inreg <4 x i8> @bitcast_v1i32_to_v4i8_scalar(<1 x i32> inreg %a, i32 inre
; GFX11-NEXT: s_and_b32 s4, s4, exec_lo
; GFX11-NEXT: s_cselect_b32 s4, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s4, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB81_5
; GFX11-NEXT: ; %bb.4: ; %cmp.true
; GFX11-NEXT: s_add_i32 s0, s0, 3
@@ -11201,6 +11293,7 @@ define <1 x i32> @bitcast_v4i8_to_v1i32(<4 x i8> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-TRUE16-NEXT: ; implicit-def: $vgpr0
; GFX11-TRUE16-NEXT: v_cmpx_ne_u32_e32 0, v4
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB82_3
; GFX11-TRUE16-NEXT: ; %bb.1: ; %Flow
@@ -11244,6 +11337,7 @@ define <1 x i32> @bitcast_v4i8_to_v1i32(<4 x i8> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-FAKE16-NEXT: ; implicit-def: $vgpr0
; GFX11-FAKE16-NEXT: v_cmpx_ne_u32_e32 0, v4
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB82_3
; GFX11-FAKE16-NEXT: ; %bb.1: ; %Flow
@@ -11442,6 +11536,7 @@ define inreg <1 x i32> @bitcast_v4i8_to_v1i32_scalar(<4 x i8> inreg %a, i32 inre
; GFX11-NEXT: s_and_b32 s4, s4, exec_lo
; GFX11-NEXT: s_cselect_b32 s4, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s4, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB83_5
; GFX11-NEXT: ; %bb.4: ; %cmp.true
; GFX11-NEXT: v_mov_b32_e32 v0, 0xc0c0004
diff --git a/llvm/test/CodeGen/AMDGPU/amdgcn.bitcast.352bit.ll b/llvm/test/CodeGen/AMDGPU/amdgcn.bitcast.352bit.ll
index ec56217a96b6dd..cd6797db4a666a 100644
--- a/llvm/test/CodeGen/AMDGPU/amdgcn.bitcast.352bit.ll
+++ b/llvm/test/CodeGen/AMDGPU/amdgcn.bitcast.352bit.ll
@@ -84,8 +84,9 @@ define <11 x float> @bitcast_v11i32_to_v11f32(<11 x i32> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v11
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB0_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -257,6 +258,7 @@ define inreg <11 x float> @bitcast_v11i32_to_v11f32_scalar(<11 x i32> inreg %a,
; GFX11-NEXT: s_and_b32 s4, s4, exec_lo
; GFX11-NEXT: s_cselect_b32 s4, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s4, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB1_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: s_add_i32 s22, s22, 3
@@ -374,8 +376,9 @@ define <11 x i32> @bitcast_v11f32_to_v11i32(<11 x float> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v11
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_dual_add_f32 v10, 1.0, v10 :: v_dual_add_f32 v9, 1.0, v9
@@ -548,6 +551,7 @@ define inreg <11 x i32> @bitcast_v11f32_to_v11i32_scalar(<11 x float> inreg %a,
; GFX11-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB3_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_add_f32_e64 v10, s22, 1.0
@@ -734,8 +738,9 @@ define <22 x i16> @bitcast_v11i32_to_v22i16(<11 x i32> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v11
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB4_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -973,6 +978,7 @@ define inreg <22 x i16> @bitcast_v11i32_to_v22i16_scalar(<11 x i32> inreg %a, i3
; GFX11-NEXT: s_and_b32 s4, s4, exec_lo
; GFX11-NEXT: s_cselect_b32 s4, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s4, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB5_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: s_add_i32 s22, s22, 3
@@ -1232,8 +1238,9 @@ define <11 x i32> @bitcast_v22i16_to_v11i32(<22 x i16> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v11
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB6_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -1543,6 +1550,7 @@ define inreg <11 x i32> @bitcast_v22i16_to_v11i32_scalar(<22 x i16> inreg %a, i3
; GFX11-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB7_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_pk_add_u16 v10, s22, 3 op_sel_hi:[1,0]
@@ -1729,8 +1737,9 @@ define <22 x half> @bitcast_v11i32_to_v22f16(<11 x i32> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v11
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB8_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -1968,6 +1977,7 @@ define inreg <22 x half> @bitcast_v11i32_to_v22f16_scalar(<11 x i32> inreg %a, i
; GFX11-NEXT: s_and_b32 s4, s4, exec_lo
; GFX11-NEXT: s_cselect_b32 s4, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s4, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB9_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: s_add_i32 s22, s22, 3
@@ -2271,8 +2281,9 @@ define <11 x i32> @bitcast_v22f16_to_v11i32(<22 x half> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v11
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB10_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -2619,6 +2630,7 @@ define inreg <11 x i32> @bitcast_v22f16_to_v11i32_scalar(<22 x half> inreg %a, i
; GFX11-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB11_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_pk_add_f16 v10, 0x200, s22 op_sel_hi:[0,1]
@@ -2805,8 +2817,9 @@ define <22 x i16> @bitcast_v11f32_to_v22i16(<11 x float> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v11
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_dual_add_f32 v10, 1.0, v10 :: v_dual_add_f32 v9, 1.0, v9
@@ -3067,6 +3080,7 @@ define inreg <22 x i16> @bitcast_v11f32_to_v22i16_scalar(<11 x float> inreg %a,
; GFX11-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB13_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_add_f32_e64 v10, s22, 1.0
@@ -3328,8 +3342,9 @@ define <11 x float> @bitcast_v22i16_to_v11f32(<22 x i16> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v11
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB14_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -3639,6 +3654,7 @@ define inreg <11 x float> @bitcast_v22i16_to_v11f32_scalar(<22 x i16> inreg %a,
; GFX11-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB15_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_pk_add_u16 v10, s22, 3 op_sel_hi:[1,0]
@@ -3825,8 +3841,9 @@ define <22 x half> @bitcast_v11f32_to_v22f16(<11 x float> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v11
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_dual_add_f32 v10, 1.0, v10 :: v_dual_add_f32 v9, 1.0, v9
@@ -4087,6 +4104,7 @@ define inreg <22 x half> @bitcast_v11f32_to_v22f16_scalar(<11 x float> inreg %a,
; GFX11-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB17_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_add_f32_e64 v10, s22, 1.0
@@ -4392,8 +4410,9 @@ define <11 x float> @bitcast_v22f16_to_v11f32(<22 x half> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v11
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB18_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -4740,6 +4759,7 @@ define inreg <11 x float> @bitcast_v22f16_to_v11f32_scalar(<22 x half> inreg %a,
; GFX11-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB19_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_pk_add_f16 v10, 0x200, s22 op_sel_hi:[0,1]
@@ -5048,8 +5068,9 @@ define <22 x half> @bitcast_v22i16_to_v22f16(<22 x i16> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v11
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB20_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -5428,6 +5449,7 @@ define inreg <22 x half> @bitcast_v22i16_to_v22f16_scalar(<22 x i16> inreg %a, i
; GFX11-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB21_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_pk_add_u16 v10, s22, 3 op_sel_hi:[1,0]
@@ -5697,8 +5719,9 @@ define <22 x i16> @bitcast_v22f16_to_v22i16(<22 x half> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v11
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB22_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -6071,6 +6094,7 @@ define inreg <22 x i16> @bitcast_v22f16_to_v22i16_scalar(<22 x half> inreg %a, i
; GFX11-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB23_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_pk_add_f16 v10, 0x200, s22 op_sel_hi:[0,1]
diff --git a/llvm/test/CodeGen/AMDGPU/amdgcn.bitcast.384bit.ll b/llvm/test/CodeGen/AMDGPU/amdgcn.bitcast.384bit.ll
index ab00b98ec0560b..c9701a93d3a945 100644
--- a/llvm/test/CodeGen/AMDGPU/amdgcn.bitcast.384bit.ll
+++ b/llvm/test/CodeGen/AMDGPU/amdgcn.bitcast.384bit.ll
@@ -87,8 +87,9 @@ define <12 x float> @bitcast_v12i32_to_v12f32(<12 x i32> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v12
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB0_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -267,6 +268,7 @@ define inreg <12 x float> @bitcast_v12i32_to_v12f32_scalar(<12 x i32> inreg %a,
; GFX11-NEXT: s_and_b32 s4, s4, exec_lo
; GFX11-NEXT: s_cselect_b32 s4, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s4, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB1_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: s_add_i32 s23, s23, 3
@@ -388,8 +390,9 @@ define <12 x i32> @bitcast_v12f32_to_v12i32(<12 x float> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v12
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_dual_add_f32 v11, 1.0, v11 :: v_dual_add_f32 v10, 1.0, v10
@@ -568,6 +571,7 @@ define inreg <12 x i32> @bitcast_v12f32_to_v12i32_scalar(<12 x float> inreg %a,
; GFX11-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB3_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_add_f32_e64 v11, s23, 1.0
@@ -689,8 +693,9 @@ define <6 x double> @bitcast_v12i32_to_v6f64(<12 x i32> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v12
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB4_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -869,6 +874,7 @@ define inreg <6 x double> @bitcast_v12i32_to_v6f64_scalar(<12 x i32> inreg %a, i
; GFX11-NEXT: s_and_b32 s4, s4, exec_lo
; GFX11-NEXT: s_cselect_b32 s4, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s4, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB5_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: s_add_i32 s23, s23, 3
@@ -972,8 +978,9 @@ define <12 x i32> @bitcast_v6f64_to_v12i32(<6 x double> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v12
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB6_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -1135,6 +1142,7 @@ define inreg <12 x i32> @bitcast_v6f64_to_v12i32_scalar(<6 x double> inreg %a, i
; GFX11-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB7_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_add_f64 v[10:11], s[22:23], 1.0
@@ -1250,8 +1258,9 @@ define <6 x i64> @bitcast_v12i32_to_v6i64(<12 x i32> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v12
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB8_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -1430,6 +1439,7 @@ define inreg <6 x i64> @bitcast_v12i32_to_v6i64_scalar(<12 x i32> inreg %a, i32
; GFX11-NEXT: s_and_b32 s4, s4, exec_lo
; GFX11-NEXT: s_cselect_b32 s4, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s4, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB9_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: s_add_i32 s23, s23, 3
@@ -1551,23 +1561,21 @@ define <12 x i32> @bitcast_v6i64_to_v12i32(<6 x i64> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v12
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB10_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_add_co_u32 v10, vcc_lo, v10, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v11, null, 0, v11, vcc_lo
; GFX11-NEXT: v_add_co_u32 v8, vcc_lo, v8, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v9, null, 0, v9, vcc_lo
; GFX11-NEXT: v_add_co_u32 v6, vcc_lo, v6, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v7, null, 0, v7, vcc_lo
; GFX11-NEXT: v_add_co_u32 v4, vcc_lo, v4, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v5, null, 0, v5, vcc_lo
; GFX11-NEXT: v_add_co_u32 v2, vcc_lo, v2, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX11-NEXT: v_add_co_u32 v0, vcc_lo, v0, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
@@ -1734,6 +1742,7 @@ define inreg <12 x i32> @bitcast_v6i64_to_v12i32_scalar(<6 x i64> inreg %a, i32
; GFX11-NEXT: s_and_b32 s4, s4, exec_lo
; GFX11-NEXT: s_cselect_b32 s4, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s4, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB11_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: s_add_u32 s22, s22, 3
@@ -1930,8 +1939,9 @@ define <24 x i16> @bitcast_v12i32_to_v24i16(<12 x i32> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v12
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB12_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -2182,6 +2192,7 @@ define inreg <24 x i16> @bitcast_v12i32_to_v24i16_scalar(<12 x i32> inreg %a, i3
; GFX11-NEXT: s_and_b32 s4, s4, exec_lo
; GFX11-NEXT: s_cselect_b32 s4, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s4, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB13_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: s_add_i32 s23, s23, 3
@@ -2457,8 +2468,9 @@ define <12 x i32> @bitcast_v24i16_to_v12i32(<24 x i16> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v12
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB14_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -2787,6 +2799,7 @@ define inreg <12 x i32> @bitcast_v24i16_to_v12i32_scalar(<24 x i16> inreg %a, i3
; GFX11-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB15_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_pk_add_u16 v11, s23, 3 op_sel_hi:[1,0]
@@ -2983,8 +2996,9 @@ define <24 x half> @bitcast_v12i32_to_v24f16(<12 x i32> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v12
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB16_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -3235,6 +3249,7 @@ define inreg <24 x half> @bitcast_v12i32_to_v24f16_scalar(<12 x i32> inreg %a, i
; GFX11-NEXT: s_and_b32 s4, s4, exec_lo
; GFX11-NEXT: s_cselect_b32 s4, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s4, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB17_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: s_add_i32 s23, s23, 3
@@ -3558,8 +3573,9 @@ define <12 x i32> @bitcast_v24f16_to_v12i32(<24 x half> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v12
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB18_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -3928,6 +3944,7 @@ define inreg <12 x i32> @bitcast_v24f16_to_v12i32_scalar(<24 x half> inreg %a, i
; GFX11-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB19_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_pk_add_f16 v11, 0x200, s23 op_sel_hi:[0,1]
@@ -4049,8 +4066,9 @@ define <6 x double> @bitcast_v12f32_to_v6f64(<12 x float> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v12
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_dual_add_f32 v11, 1.0, v11 :: v_dual_add_f32 v10, 1.0, v10
@@ -4241,6 +4259,7 @@ define inreg <6 x double> @bitcast_v12f32_to_v6f64_scalar(<12 x float> inreg %a,
; GFX11-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB21_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_add_f32_e64 v11, s23, 1.0
@@ -4346,8 +4365,9 @@ define <12 x float> @bitcast_v6f64_to_v12f32(<6 x double> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v12
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB22_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -4509,6 +4529,7 @@ define inreg <12 x float> @bitcast_v6f64_to_v12f32_scalar(<6 x double> inreg %a,
; GFX11-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB23_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_add_f64 v[10:11], s[22:23], 1.0
@@ -4624,8 +4645,9 @@ define <6 x i64> @bitcast_v12f32_to_v6i64(<12 x float> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v12
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_dual_add_f32 v11, 1.0, v11 :: v_dual_add_f32 v10, 1.0, v10
@@ -4816,6 +4838,7 @@ define inreg <6 x i64> @bitcast_v12f32_to_v6i64_scalar(<12 x float> inreg %a, i3
; GFX11-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB25_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_add_f32_e64 v11, s23, 1.0
@@ -4939,23 +4962,21 @@ define <12 x float> @bitcast_v6i64_to_v12f32(<6 x i64> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v12
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB26_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_add_co_u32 v10, vcc_lo, v10, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v11, null, 0, v11, vcc_lo
; GFX11-NEXT: v_add_co_u32 v8, vcc_lo, v8, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v9, null, 0, v9, vcc_lo
; GFX11-NEXT: v_add_co_u32 v6, vcc_lo, v6, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v7, null, 0, v7, vcc_lo
; GFX11-NEXT: v_add_co_u32 v4, vcc_lo, v4, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v5, null, 0, v5, vcc_lo
; GFX11-NEXT: v_add_co_u32 v2, vcc_lo, v2, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX11-NEXT: v_add_co_u32 v0, vcc_lo, v0, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
@@ -5122,6 +5143,7 @@ define inreg <12 x float> @bitcast_v6i64_to_v12f32_scalar(<6 x i64> inreg %a, i3
; GFX11-NEXT: s_and_b32 s4, s4, exec_lo
; GFX11-NEXT: s_cselect_b32 s4, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s4, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB27_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: s_add_u32 s22, s22, 3
@@ -5318,8 +5340,9 @@ define <24 x i16> @bitcast_v12f32_to_v24i16(<12 x float> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v12
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_dual_add_f32 v11, 1.0, v11 :: v_dual_add_f32 v10, 1.0, v10
@@ -5591,6 +5614,7 @@ define inreg <24 x i16> @bitcast_v12f32_to_v24i16_scalar(<12 x float> inreg %a,
; GFX11-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB29_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_add_f32_e64 v11, s23, 1.0
@@ -5868,8 +5892,9 @@ define <12 x float> @bitcast_v24i16_to_v12f32(<24 x i16> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v12
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB30_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -6198,6 +6223,7 @@ define inreg <12 x float> @bitcast_v24i16_to_v12f32_scalar(<24 x i16> inreg %a,
; GFX11-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB31_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_pk_add_u16 v11, s23, 3 op_sel_hi:[1,0]
@@ -6394,8 +6420,9 @@ define <24 x half> @bitcast_v12f32_to_v24f16(<12 x float> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v12
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_dual_add_f32 v11, 1.0, v11 :: v_dual_add_f32 v10, 1.0, v10
@@ -6667,6 +6694,7 @@ define inreg <24 x half> @bitcast_v12f32_to_v24f16_scalar(<12 x float> inreg %a,
; GFX11-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB33_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_add_f32_e64 v11, s23, 1.0
@@ -6992,8 +7020,9 @@ define <12 x float> @bitcast_v24f16_to_v12f32(<24 x half> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v12
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB34_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -7362,6 +7391,7 @@ define inreg <12 x float> @bitcast_v24f16_to_v12f32_scalar(<24 x half> inreg %a,
; GFX11-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB35_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_pk_add_f16 v11, 0x200, s23 op_sel_hi:[0,1]
@@ -7465,8 +7495,9 @@ define <6 x i64> @bitcast_v6f64_to_v6i64(<6 x double> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v12
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB36_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -7640,6 +7671,7 @@ define inreg <6 x i64> @bitcast_v6f64_to_v6i64_scalar(<6 x double> inreg %a, i32
; GFX11-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB37_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_add_f64 v[0:1], s[12:13], 1.0
@@ -7757,23 +7789,21 @@ define <6 x double> @bitcast_v6i64_to_v6f64(<6 x i64> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v12
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB38_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_add_co_u32 v0, vcc_lo, v0, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-NEXT: v_add_co_u32 v2, vcc_lo, v2, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX11-NEXT: v_add_co_u32 v4, vcc_lo, v4, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v5, null, 0, v5, vcc_lo
; GFX11-NEXT: v_add_co_u32 v6, vcc_lo, v6, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v7, null, 0, v7, vcc_lo
; GFX11-NEXT: v_add_co_u32 v8, vcc_lo, v8, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v9, null, 0, v9, vcc_lo
; GFX11-NEXT: v_add_co_u32 v10, vcc_lo, v10, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v11, null, 0, v11, vcc_lo
@@ -7940,6 +7970,7 @@ define inreg <6 x double> @bitcast_v6i64_to_v6f64_scalar(<6 x i64> inreg %a, i32
; GFX11-NEXT: s_and_b32 s4, s4, exec_lo
; GFX11-NEXT: s_cselect_b32 s4, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s4, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB39_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: s_add_u32 s0, s0, 3
@@ -8117,8 +8148,9 @@ define <24 x i16> @bitcast_v6f64_to_v24i16(<6 x double> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v12
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB40_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -8373,6 +8405,7 @@ define inreg <24 x i16> @bitcast_v6f64_to_v24i16_scalar(<6 x double> inreg %a, i
; GFX11-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB41_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_add_f64 v[10:11], s[22:23], 1.0
@@ -8644,8 +8677,9 @@ define <6 x double> @bitcast_v24i16_to_v6f64(<24 x i16> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v12
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB42_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -9002,6 +9036,7 @@ define inreg <6 x double> @bitcast_v24i16_to_v6f64_scalar(<24 x i16> inreg %a, i
; GFX11-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB43_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_pk_add_u16 v11, s23, 3 op_sel_hi:[1,0]
@@ -9182,8 +9217,9 @@ define <24 x half> @bitcast_v6f64_to_v24f16(<6 x double> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v12
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB44_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -9438,6 +9474,7 @@ define inreg <24 x half> @bitcast_v6f64_to_v24f16_scalar(<6 x double> inreg %a,
; GFX11-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB45_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_add_f64 v[10:11], s[22:23], 1.0
@@ -9757,8 +9794,9 @@ define <6 x double> @bitcast_v24f16_to_v6f64(<24 x half> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v12
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB46_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -10164,6 +10202,7 @@ define inreg <6 x double> @bitcast_v24f16_to_v6f64_scalar(<24 x half> inreg %a,
; GFX11-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB47_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_pk_add_f16 v11, 0x200, s23 op_sel_hi:[0,1]
@@ -10362,23 +10401,21 @@ define <24 x i16> @bitcast_v6i64_to_v24i16(<6 x i64> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v12
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB48_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_add_co_u32 v10, vcc_lo, v10, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v11, null, 0, v11, vcc_lo
; GFX11-NEXT: v_add_co_u32 v8, vcc_lo, v8, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v9, null, 0, v9, vcc_lo
; GFX11-NEXT: v_add_co_u32 v6, vcc_lo, v6, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v7, null, 0, v7, vcc_lo
; GFX11-NEXT: v_add_co_u32 v4, vcc_lo, v4, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v5, null, 0, v5, vcc_lo
; GFX11-NEXT: v_add_co_u32 v2, vcc_lo, v2, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX11-NEXT: v_add_co_u32 v0, vcc_lo, v0, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
@@ -10617,6 +10654,7 @@ define inreg <24 x i16> @bitcast_v6i64_to_v24i16_scalar(<6 x i64> inreg %a, i32
; GFX11-NEXT: s_and_b32 s4, s4, exec_lo
; GFX11-NEXT: s_cselect_b32 s4, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s4, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB49_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: s_add_u32 s22, s22, 3
@@ -10892,8 +10930,9 @@ define <6 x i64> @bitcast_v24i16_to_v6i64(<24 x i16> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v12
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB50_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -11250,6 +11289,7 @@ define inreg <6 x i64> @bitcast_v24i16_to_v6i64_scalar(<24 x i16> inreg %a, i32
; GFX11-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB51_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_pk_add_u16 v11, s23, 3 op_sel_hi:[1,0]
@@ -11448,23 +11488,21 @@ define <24 x half> @bitcast_v6i64_to_v24f16(<6 x i64> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v12
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB52_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_add_co_u32 v10, vcc_lo, v10, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v11, null, 0, v11, vcc_lo
; GFX11-NEXT: v_add_co_u32 v8, vcc_lo, v8, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v9, null, 0, v9, vcc_lo
; GFX11-NEXT: v_add_co_u32 v6, vcc_lo, v6, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v7, null, 0, v7, vcc_lo
; GFX11-NEXT: v_add_co_u32 v4, vcc_lo, v4, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v5, null, 0, v5, vcc_lo
; GFX11-NEXT: v_add_co_u32 v2, vcc_lo, v2, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX11-NEXT: v_add_co_u32 v0, vcc_lo, v0, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
@@ -11703,6 +11741,7 @@ define inreg <24 x half> @bitcast_v6i64_to_v24f16_scalar(<6 x i64> inreg %a, i32
; GFX11-NEXT: s_and_b32 s4, s4, exec_lo
; GFX11-NEXT: s_cselect_b32 s4, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s4, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB53_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: s_add_u32 s22, s22, 3
@@ -12026,8 +12065,9 @@ define <6 x i64> @bitcast_v24f16_to_v6i64(<24 x half> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v12
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB54_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -12433,6 +12473,7 @@ define inreg <6 x i64> @bitcast_v24f16_to_v6i64_scalar(<24 x half> inreg %a, i32
; GFX11-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB55_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_pk_add_f16 v11, 0x200, s23 op_sel_hi:[0,1]
@@ -12765,8 +12806,9 @@ define <24 x half> @bitcast_v24i16_to_v24f16(<24 x i16> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v12
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB56_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -13170,6 +13212,7 @@ define inreg <24 x half> @bitcast_v24i16_to_v24f16_scalar(<24 x i16> inreg %a, i
; GFX11-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB57_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_pk_add_u16 v11, s23, 3 op_sel_hi:[1,0]
@@ -13457,8 +13500,9 @@ define <24 x i16> @bitcast_v24f16_to_v24i16(<24 x half> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v12
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB58_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -13853,6 +13897,7 @@ define inreg <24 x i16> @bitcast_v24f16_to_v24i16_scalar(<24 x half> inreg %a, i
; GFX11-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB59_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_pk_add_f16 v11, 0x200, s23 op_sel_hi:[0,1]
diff --git a/llvm/test/CodeGen/AMDGPU/amdgcn.bitcast.448bit.ll b/llvm/test/CodeGen/AMDGPU/amdgcn.bitcast.448bit.ll
index cb85f607aca61b..aa44935e005bb0 100644
--- a/llvm/test/CodeGen/AMDGPU/amdgcn.bitcast.448bit.ll
+++ b/llvm/test/CodeGen/AMDGPU/amdgcn.bitcast.448bit.ll
@@ -93,8 +93,9 @@ define <14 x float> @bitcast_v14i32_to_v14f32(<14 x i32> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v14
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB0_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -290,6 +291,7 @@ define inreg <14 x float> @bitcast_v14i32_to_v14f32_scalar(<14 x i32> inreg %a,
; GFX11-NEXT: s_and_b32 s4, s4, exec_lo
; GFX11-NEXT: s_cselect_b32 s4, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s4, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB1_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: s_add_i32 s25, s25, 3
@@ -420,8 +422,9 @@ define <14 x i32> @bitcast_v14f32_to_v14i32(<14 x float> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v14
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB2_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -623,6 +626,7 @@ define inreg <14 x i32> @bitcast_v14f32_to_v14i32_scalar(<14 x float> inreg %a,
; GFX11-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB3_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_add_f32_e64 v13, s25, 1.0
@@ -754,8 +758,9 @@ define <7 x i64> @bitcast_v14i32_to_v7i64(<14 x i32> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v14
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB4_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -951,6 +956,7 @@ define inreg <7 x i64> @bitcast_v14i32_to_v7i64_scalar(<14 x i32> inreg %a, i32
; GFX11-NEXT: s_and_b32 s4, s4, exec_lo
; GFX11-NEXT: s_cselect_b32 s4, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s4, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB5_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: s_add_i32 s25, s25, 3
@@ -1081,28 +1087,25 @@ define <14 x i32> @bitcast_v7i64_to_v14i32(<7 x i64> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v14
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB6_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_add_co_u32 v12, vcc_lo, v12, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v13, null, 0, v13, vcc_lo
; GFX11-NEXT: v_add_co_u32 v10, vcc_lo, v10, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v11, null, 0, v11, vcc_lo
; GFX11-NEXT: v_add_co_u32 v8, vcc_lo, v8, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v9, null, 0, v9, vcc_lo
; GFX11-NEXT: v_add_co_u32 v6, vcc_lo, v6, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v7, null, 0, v7, vcc_lo
; GFX11-NEXT: v_add_co_u32 v4, vcc_lo, v4, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v5, null, 0, v5, vcc_lo
; GFX11-NEXT: v_add_co_u32 v2, vcc_lo, v2, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX11-NEXT: v_add_co_u32 v0, vcc_lo, v0, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-NEXT: .LBB6_2: ; %end
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -1282,6 +1285,7 @@ define inreg <14 x i32> @bitcast_v7i64_to_v14i32_scalar(<7 x i64> inreg %a, i32
; GFX11-NEXT: s_and_b32 s4, s4, exec_lo
; GFX11-NEXT: s_cselect_b32 s4, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s4, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB7_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: s_add_u32 s24, s24, 3
@@ -1412,8 +1416,9 @@ define <7 x double> @bitcast_v14i32_to_v7f64(<14 x i32> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v14
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB8_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -1609,6 +1614,7 @@ define inreg <7 x double> @bitcast_v14i32_to_v7f64_scalar(<14 x i32> inreg %a, i
; GFX11-NEXT: s_and_b32 s4, s4, exec_lo
; GFX11-NEXT: s_cselect_b32 s4, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s4, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB9_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: s_add_i32 s25, s25, 3
@@ -1718,8 +1724,9 @@ define <14 x i32> @bitcast_v7f64_to_v14i32(<7 x double> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v14
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB10_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -1900,6 +1907,7 @@ define inreg <14 x i32> @bitcast_v7f64_to_v14i32_scalar(<7 x double> inreg %a, i
; GFX11-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB11_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_add_f64 v[12:13], s[24:25], 1.0
@@ -2111,8 +2119,9 @@ define <28 x i16> @bitcast_v14i32_to_v28i16(<14 x i32> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v14
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB12_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -2392,6 +2401,7 @@ define inreg <28 x i16> @bitcast_v14i32_to_v28i16_scalar(<14 x i32> inreg %a, i3
; GFX11-NEXT: s_and_b32 s4, s4, exec_lo
; GFX11-NEXT: s_cselect_b32 s4, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s4, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB13_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: s_add_i32 s25, s25, 3
@@ -2700,8 +2710,9 @@ define <14 x i32> @bitcast_v28i16_to_v14i32(<28 x i16> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v14
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB14_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -3097,6 +3108,7 @@ define inreg <14 x i32> @bitcast_v28i16_to_v14i32_scalar(<28 x i16> inreg %a, i3
; GFX11-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB15_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_pk_add_u16 v13, s25, 3 op_sel_hi:[1,0]
@@ -3315,8 +3327,9 @@ define <28 x half> @bitcast_v14i32_to_v28f16(<14 x i32> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v14
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB16_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -3596,6 +3609,7 @@ define inreg <28 x half> @bitcast_v14i32_to_v28f16_scalar(<14 x i32> inreg %a, i
; GFX11-NEXT: s_and_b32 s4, s4, exec_lo
; GFX11-NEXT: s_cselect_b32 s4, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s4, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB17_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: s_add_i32 s25, s25, 3
@@ -3960,8 +3974,9 @@ define <14 x i32> @bitcast_v28f16_to_v14i32(<28 x half> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v14
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB18_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -4408,6 +4423,7 @@ define inreg <14 x i32> @bitcast_v28f16_to_v14i32_scalar(<28 x half> inreg %a, i
; GFX11-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB19_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_pk_add_f16 v13, 0x200, s25 op_sel_hi:[0,1]
@@ -4539,8 +4555,9 @@ define <7 x i64> @bitcast_v14f32_to_v7i64(<14 x float> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v14
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB20_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -4742,6 +4759,7 @@ define inreg <7 x i64> @bitcast_v14f32_to_v7i64_scalar(<14 x float> inreg %a, i3
; GFX11-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB21_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_add_f32_e64 v13, s25, 1.0
@@ -4873,28 +4891,25 @@ define <14 x float> @bitcast_v7i64_to_v14f32(<7 x i64> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v14
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB22_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_add_co_u32 v12, vcc_lo, v12, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v13, null, 0, v13, vcc_lo
; GFX11-NEXT: v_add_co_u32 v10, vcc_lo, v10, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v11, null, 0, v11, vcc_lo
; GFX11-NEXT: v_add_co_u32 v8, vcc_lo, v8, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v9, null, 0, v9, vcc_lo
; GFX11-NEXT: v_add_co_u32 v6, vcc_lo, v6, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v7, null, 0, v7, vcc_lo
; GFX11-NEXT: v_add_co_u32 v4, vcc_lo, v4, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v5, null, 0, v5, vcc_lo
; GFX11-NEXT: v_add_co_u32 v2, vcc_lo, v2, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX11-NEXT: v_add_co_u32 v0, vcc_lo, v0, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-NEXT: .LBB22_2: ; %end
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -5074,6 +5089,7 @@ define inreg <14 x float> @bitcast_v7i64_to_v14f32_scalar(<7 x i64> inreg %a, i3
; GFX11-NEXT: s_and_b32 s4, s4, exec_lo
; GFX11-NEXT: s_cselect_b32 s4, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s4, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB23_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: s_add_u32 s24, s24, 3
@@ -5204,8 +5220,9 @@ define <7 x double> @bitcast_v14f32_to_v7f64(<14 x float> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v14
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB24_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -5407,6 +5424,7 @@ define inreg <7 x double> @bitcast_v14f32_to_v7f64_scalar(<14 x float> inreg %a,
; GFX11-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB25_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_add_f32_e64 v13, s25, 1.0
@@ -5517,8 +5535,9 @@ define <14 x float> @bitcast_v7f64_to_v14f32(<7 x double> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v14
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB26_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -5699,6 +5718,7 @@ define inreg <14 x float> @bitcast_v7f64_to_v14f32_scalar(<7 x double> inreg %a,
; GFX11-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB27_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_add_f64 v[12:13], s[24:25], 1.0
@@ -5910,8 +5930,9 @@ define <28 x i16> @bitcast_v14f32_to_v28i16(<14 x float> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v14
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB28_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -6210,6 +6231,7 @@ define inreg <28 x i16> @bitcast_v14f32_to_v28i16_scalar(<14 x float> inreg %a,
; GFX11-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB29_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_add_f32_e64 v13, s25, 1.0
@@ -6519,8 +6541,9 @@ define <14 x float> @bitcast_v28i16_to_v14f32(<28 x i16> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v14
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB30_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -6916,6 +6939,7 @@ define inreg <14 x float> @bitcast_v28i16_to_v14f32_scalar(<28 x i16> inreg %a,
; GFX11-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB31_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_pk_add_u16 v13, s25, 3 op_sel_hi:[1,0]
@@ -7134,8 +7158,9 @@ define <28 x half> @bitcast_v14f32_to_v28f16(<14 x float> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v14
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB32_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -7434,6 +7459,7 @@ define inreg <28 x half> @bitcast_v14f32_to_v28f16_scalar(<14 x float> inreg %a,
; GFX11-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB33_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_add_f32_e64 v13, s25, 1.0
@@ -7799,8 +7825,9 @@ define <14 x float> @bitcast_v28f16_to_v14f32(<28 x half> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v14
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB34_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -8247,6 +8274,7 @@ define inreg <14 x float> @bitcast_v28f16_to_v14f32_scalar(<28 x half> inreg %a,
; GFX11-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB35_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_pk_add_f16 v13, 0x200, s25 op_sel_hi:[0,1]
@@ -8378,28 +8406,25 @@ define <7 x double> @bitcast_v7i64_to_v7f64(<7 x i64> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v14
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB36_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_add_co_u32 v0, vcc_lo, v0, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-NEXT: v_add_co_u32 v2, vcc_lo, v2, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX11-NEXT: v_add_co_u32 v4, vcc_lo, v4, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v5, null, 0, v5, vcc_lo
; GFX11-NEXT: v_add_co_u32 v6, vcc_lo, v6, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v7, null, 0, v7, vcc_lo
; GFX11-NEXT: v_add_co_u32 v8, vcc_lo, v8, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v9, null, 0, v9, vcc_lo
; GFX11-NEXT: v_add_co_u32 v10, vcc_lo, v10, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v11, null, 0, v11, vcc_lo
; GFX11-NEXT: v_add_co_u32 v12, vcc_lo, v12, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v13, null, 0, v13, vcc_lo
; GFX11-NEXT: .LBB36_2: ; %end
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -8579,6 +8604,7 @@ define inreg <7 x double> @bitcast_v7i64_to_v7f64_scalar(<7 x i64> inreg %a, i32
; GFX11-NEXT: s_and_b32 s4, s4, exec_lo
; GFX11-NEXT: s_cselect_b32 s4, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s4, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB37_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: s_add_u32 s0, s0, 3
@@ -8687,8 +8713,9 @@ define <7 x i64> @bitcast_v7f64_to_v7i64(<7 x double> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v14
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB38_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -8869,6 +8896,7 @@ define inreg <7 x i64> @bitcast_v7f64_to_v7i64_scalar(<7 x double> inreg %a, i32
; GFX11-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB39_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_add_f64 v[0:1], s[12:13], 1.0
@@ -9080,28 +9108,25 @@ define <28 x i16> @bitcast_v7i64_to_v28i16(<7 x i64> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v14
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB40_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_add_co_u32 v12, vcc_lo, v12, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v13, null, 0, v13, vcc_lo
; GFX11-NEXT: v_add_co_u32 v10, vcc_lo, v10, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v11, null, 0, v11, vcc_lo
; GFX11-NEXT: v_add_co_u32 v8, vcc_lo, v8, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v9, null, 0, v9, vcc_lo
; GFX11-NEXT: v_add_co_u32 v6, vcc_lo, v6, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v7, null, 0, v7, vcc_lo
; GFX11-NEXT: v_add_co_u32 v4, vcc_lo, v4, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v5, null, 0, v5, vcc_lo
; GFX11-NEXT: v_add_co_u32 v2, vcc_lo, v2, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX11-NEXT: v_add_co_u32 v0, vcc_lo, v0, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-NEXT: .LBB40_2: ; %end
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -9365,6 +9390,7 @@ define inreg <28 x i16> @bitcast_v7i64_to_v28i16_scalar(<7 x i64> inreg %a, i32
; GFX11-NEXT: s_and_b32 s4, s4, exec_lo
; GFX11-NEXT: s_cselect_b32 s4, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s4, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB41_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: s_add_u32 s24, s24, 3
@@ -9673,8 +9699,9 @@ define <7 x i64> @bitcast_v28i16_to_v7i64(<28 x i16> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v14
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB42_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -10070,6 +10097,7 @@ define inreg <7 x i64> @bitcast_v28i16_to_v7i64_scalar(<28 x i16> inreg %a, i32
; GFX11-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB43_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_pk_add_u16 v13, s25, 3 op_sel_hi:[1,0]
@@ -10288,28 +10316,25 @@ define <28 x half> @bitcast_v7i64_to_v28f16(<7 x i64> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v14
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB44_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_add_co_u32 v12, vcc_lo, v12, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v13, null, 0, v13, vcc_lo
; GFX11-NEXT: v_add_co_u32 v10, vcc_lo, v10, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v11, null, 0, v11, vcc_lo
; GFX11-NEXT: v_add_co_u32 v8, vcc_lo, v8, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v9, null, 0, v9, vcc_lo
; GFX11-NEXT: v_add_co_u32 v6, vcc_lo, v6, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v7, null, 0, v7, vcc_lo
; GFX11-NEXT: v_add_co_u32 v4, vcc_lo, v4, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v5, null, 0, v5, vcc_lo
; GFX11-NEXT: v_add_co_u32 v2, vcc_lo, v2, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX11-NEXT: v_add_co_u32 v0, vcc_lo, v0, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-NEXT: .LBB44_2: ; %end
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -10573,6 +10598,7 @@ define inreg <28 x half> @bitcast_v7i64_to_v28f16_scalar(<7 x i64> inreg %a, i32
; GFX11-NEXT: s_and_b32 s4, s4, exec_lo
; GFX11-NEXT: s_cselect_b32 s4, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s4, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB45_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: s_add_u32 s24, s24, 3
@@ -10937,8 +10963,9 @@ define <7 x i64> @bitcast_v28f16_to_v7i64(<28 x half> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v14
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB46_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -11385,6 +11412,7 @@ define inreg <7 x i64> @bitcast_v28f16_to_v7i64_scalar(<28 x half> inreg %a, i32
; GFX11-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB47_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_pk_add_f16 v13, 0x200, s25 op_sel_hi:[0,1]
@@ -11582,8 +11610,9 @@ define <28 x i16> @bitcast_v7f64_to_v28i16(<7 x double> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v14
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB48_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -11861,6 +11890,7 @@ define inreg <28 x i16> @bitcast_v7f64_to_v28i16_scalar(<7 x double> inreg %a, i
; GFX11-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB49_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_add_f64 v[12:13], s[24:25], 1.0
@@ -12163,8 +12193,9 @@ define <7 x double> @bitcast_v28i16_to_v7f64(<28 x i16> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v14
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB50_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -12560,6 +12591,7 @@ define inreg <7 x double> @bitcast_v28i16_to_v7f64_scalar(<28 x i16> inreg %a, i
; GFX11-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB51_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_pk_add_u16 v13, s25, 3 op_sel_hi:[1,0]
@@ -12757,8 +12789,9 @@ define <28 x half> @bitcast_v7f64_to_v28f16(<7 x double> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v14
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB52_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -13036,6 +13069,7 @@ define inreg <28 x half> @bitcast_v7f64_to_v28f16_scalar(<7 x double> inreg %a,
; GFX11-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB53_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_add_f64 v[12:13], s[24:25], 1.0
@@ -13394,8 +13428,9 @@ define <7 x double> @bitcast_v28f16_to_v7f64(<28 x half> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v14
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB54_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -13842,6 +13877,7 @@ define inreg <7 x double> @bitcast_v28f16_to_v7f64_scalar(<28 x half> inreg %a,
; GFX11-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB55_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_pk_add_f16 v13, 0x200, s25 op_sel_hi:[0,1]
@@ -14237,8 +14273,9 @@ define <28 x half> @bitcast_v28i16_to_v28f16(<28 x i16> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v14
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB56_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -14693,6 +14730,7 @@ define inreg <28 x half> @bitcast_v28i16_to_v28f16_scalar(<28 x i16> inreg %a, i
; GFX11-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB57_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_pk_add_u16 v13, s25, 3 op_sel_hi:[1,0]
@@ -15015,8 +15053,9 @@ define <28 x i16> @bitcast_v28f16_to_v28i16(<28 x half> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v14
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB58_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -15457,6 +15496,7 @@ define inreg <28 x i16> @bitcast_v28f16_to_v28i16_scalar(<28 x half> inreg %a, i
; GFX11-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB59_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_pk_add_f16 v13, 0x200, s25 op_sel_hi:[0,1]
diff --git a/llvm/test/CodeGen/AMDGPU/amdgcn.bitcast.48bit.ll b/llvm/test/CodeGen/AMDGPU/amdgcn.bitcast.48bit.ll
index ffff9da12fe9fe..972d8348b9ef85 100644
--- a/llvm/test/CodeGen/AMDGPU/amdgcn.bitcast.48bit.ll
+++ b/llvm/test/CodeGen/AMDGPU/amdgcn.bitcast.48bit.ll
@@ -139,8 +139,9 @@ define <3 x half> @bitcast_v3bf16_to_v3f16(<3 x bfloat> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-TRUE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-TRUE16-NEXT: v_cmpx_ne_u32_e32 0, v2
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-TRUE16-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-TRUE16-NEXT: s_cbranch_execz .LBB0_2
; GFX11-TRUE16-NEXT: ; %bb.1: ; %cmp.true
@@ -186,8 +187,9 @@ define <3 x half> @bitcast_v3bf16_to_v3f16(<3 x bfloat> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-FAKE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-FAKE16-NEXT: v_cmpx_ne_u32_e32 0, v2
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-FAKE16-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-FAKE16-NEXT: s_cbranch_execz .LBB0_2
; GFX11-FAKE16-NEXT: ; %bb.1: ; %cmp.true
@@ -403,6 +405,7 @@ define inreg <3 x half> @bitcast_v3bf16_to_v3f16_scalar(<3 x bfloat> inreg %a, i
; GFX11-TRUE16-NEXT: s_and_b32 s2, s2, exec_lo
; GFX11-TRUE16-NEXT: s_cselect_b32 s2, 1, 0
; GFX11-TRUE16-NEXT: s_cmp_lg_u32 s2, 1
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_cbranch_scc1 .LBB1_4
; GFX11-TRUE16-NEXT: ; %bb.3: ; %cmp.true
; GFX11-TRUE16-NEXT: s_pack_lh_b32_b16 s2, 0, s0
@@ -459,6 +462,7 @@ define inreg <3 x half> @bitcast_v3bf16_to_v3f16_scalar(<3 x bfloat> inreg %a, i
; GFX11-FAKE16-NEXT: s_and_b32 s2, s2, exec_lo
; GFX11-FAKE16-NEXT: s_cselect_b32 s2, 1, 0
; GFX11-FAKE16-NEXT: s_cmp_lg_u32 s2, 1
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_cbranch_scc1 .LBB1_4
; GFX11-FAKE16-NEXT: ; %bb.3: ; %cmp.true
; GFX11-FAKE16-NEXT: s_lshl_b32 s2, s0, 16
@@ -601,8 +605,9 @@ define <3 x bfloat> @bitcast_v3f16_to_v3bf16(<3 x half> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v2
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_pk_add_f16 v1, 0x200, v1
@@ -745,6 +750,7 @@ define inreg <3 x bfloat> @bitcast_v3f16_to_v3bf16_scalar(<3 x half> inreg %a, i
; GFX11-NEXT: s_and_b32 s2, s2, exec_lo
; GFX11-NEXT: s_cselect_b32 s2, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s2, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB3_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_pk_add_f16 v1, 0x200, s1
@@ -901,8 +907,9 @@ define <3 x i16> @bitcast_v3bf16_to_v3i16(<3 x bfloat> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-TRUE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-TRUE16-NEXT: v_cmpx_ne_u32_e32 0, v2
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-TRUE16-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-TRUE16-NEXT: s_cbranch_execz .LBB4_2
; GFX11-TRUE16-NEXT: ; %bb.1: ; %cmp.true
@@ -947,8 +954,9 @@ define <3 x i16> @bitcast_v3bf16_to_v3i16(<3 x bfloat> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-FAKE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-FAKE16-NEXT: v_cmpx_ne_u32_e32 0, v2
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-FAKE16-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-FAKE16-NEXT: s_cbranch_execz .LBB4_2
; GFX11-FAKE16-NEXT: ; %bb.1: ; %cmp.true
@@ -1162,6 +1170,7 @@ define inreg <3 x i16> @bitcast_v3bf16_to_v3i16_scalar(<3 x bfloat> inreg %a, i3
; GFX11-TRUE16-NEXT: s_and_b32 s2, s2, exec_lo
; GFX11-TRUE16-NEXT: s_cselect_b32 s2, 1, 0
; GFX11-TRUE16-NEXT: s_cmp_lg_u32 s2, 1
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_cbranch_scc1 .LBB5_4
; GFX11-TRUE16-NEXT: ; %bb.3: ; %cmp.true
; GFX11-TRUE16-NEXT: s_lshl_b32 s2, s0, 16
@@ -1214,6 +1223,7 @@ define inreg <3 x i16> @bitcast_v3bf16_to_v3i16_scalar(<3 x bfloat> inreg %a, i3
; GFX11-FAKE16-NEXT: s_and_b32 s2, s2, exec_lo
; GFX11-FAKE16-NEXT: s_cselect_b32 s2, 1, 0
; GFX11-FAKE16-NEXT: s_cmp_lg_u32 s2, 1
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_cbranch_scc1 .LBB5_4
; GFX11-FAKE16-NEXT: ; %bb.3: ; %cmp.true
; GFX11-FAKE16-NEXT: s_lshl_b32 s2, s0, 16
@@ -1330,7 +1340,7 @@ define <3 x bfloat> @bitcast_v3i16_to_v3bf16(<3 x i16> %a, i32 %b) #0 {
; GFX9-NEXT: s_xor_b64 s[4:5], exec, s[4:5]
; GFX9-NEXT: s_andn2_saveexec_b64 s[4:5], s[4:5]
; GFX9-NEXT: ; %bb.1: ; %cmp.true
-; GFX9-NEXT: v_pk_add_u16 v1, v1, 3
+; GFX9-NEXT: v_pk_add_u16 v1, v1, 3 op_sel_hi:[1,0]
; GFX9-NEXT: v_pk_add_u16 v0, v0, 3 op_sel_hi:[1,0]
; GFX9-NEXT: ; %bb.2: ; %end
; GFX9-NEXT: s_or_b64 exec, exec, s[4:5]
@@ -1341,11 +1351,12 @@ define <3 x bfloat> @bitcast_v3i16_to_v3bf16(<3 x i16> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v2
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: ; %bb.1: ; %cmp.true
-; GFX11-NEXT: v_pk_add_u16 v1, v1, 3
+; GFX11-NEXT: v_pk_add_u16 v1, v1, 3 op_sel_hi:[1,0]
; GFX11-NEXT: v_pk_add_u16 v0, v0, 3 op_sel_hi:[1,0]
; GFX11-NEXT: ; %bb.2: ; %end
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -1452,7 +1463,7 @@ define inreg <3 x bfloat> @bitcast_v3i16_to_v3bf16_scalar(<3 x i16> inreg %a, i3
; GFX9-NEXT: s_cmp_lg_u32 s4, 1
; GFX9-NEXT: s_cbranch_scc1 .LBB7_5
; GFX9-NEXT: ; %bb.4: ; %cmp.true
-; GFX9-NEXT: v_pk_add_u16 v1, s17, 3
+; GFX9-NEXT: v_pk_add_u16 v1, s17, 3 op_sel_hi:[1,0]
; GFX9-NEXT: v_pk_add_u16 v0, s16, 3 op_sel_hi:[1,0]
; GFX9-NEXT: s_setpc_b64 s[30:31]
; GFX9-NEXT: .LBB7_5:
@@ -1473,9 +1484,10 @@ define inreg <3 x bfloat> @bitcast_v3i16_to_v3bf16_scalar(<3 x i16> inreg %a, i3
; GFX11-NEXT: s_and_b32 s2, s2, exec_lo
; GFX11-NEXT: s_cselect_b32 s2, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s2, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB7_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
-; GFX11-NEXT: v_pk_add_u16 v1, s1, 3
+; GFX11-NEXT: v_pk_add_u16 v1, s1, 3 op_sel_hi:[1,0]
; GFX11-NEXT: v_pk_add_u16 v0, s0, 3 op_sel_hi:[1,0]
; GFX11-NEXT: s_setpc_b64 s[30:31]
; GFX11-NEXT: .LBB7_4:
@@ -1566,8 +1578,9 @@ define <3 x i16> @bitcast_v3f16_to_v3i16(<3 x half> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v2
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_pk_add_f16 v1, 0x200, v1
@@ -1701,6 +1714,7 @@ define inreg <3 x i16> @bitcast_v3f16_to_v3i16_scalar(<3 x half> inreg %a, i32 i
; GFX11-NEXT: s_and_b32 s2, s2, exec_lo
; GFX11-NEXT: s_cselect_b32 s2, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s2, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB9_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_pk_add_f16 v1, 0x200, s1
@@ -1789,7 +1803,7 @@ define <3 x half> @bitcast_v3i16_to_v3f16(<3 x i16> %a, i32 %b) #0 {
; GFX9-NEXT: s_xor_b64 s[4:5], exec, s[4:5]
; GFX9-NEXT: s_andn2_saveexec_b64 s[4:5], s[4:5]
; GFX9-NEXT: ; %bb.1: ; %cmp.true
-; GFX9-NEXT: v_pk_add_u16 v1, v1, 3
+; GFX9-NEXT: v_pk_add_u16 v1, v1, 3 op_sel_hi:[1,0]
; GFX9-NEXT: v_pk_add_u16 v0, v0, 3 op_sel_hi:[1,0]
; GFX9-NEXT: ; %bb.2: ; %end
; GFX9-NEXT: s_or_b64 exec, exec, s[4:5]
@@ -1800,11 +1814,12 @@ define <3 x half> @bitcast_v3i16_to_v3f16(<3 x i16> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v2
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: ; %bb.1: ; %cmp.true
-; GFX11-NEXT: v_pk_add_u16 v1, v1, 3
+; GFX11-NEXT: v_pk_add_u16 v1, v1, 3 op_sel_hi:[1,0]
; GFX11-NEXT: v_pk_add_u16 v0, v0, 3 op_sel_hi:[1,0]
; GFX11-NEXT: ; %bb.2: ; %end
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -1912,7 +1927,7 @@ define inreg <3 x half> @bitcast_v3i16_to_v3f16_scalar(<3 x i16> inreg %a, i32 i
; GFX9-NEXT: s_cmp_lg_u32 s4, 1
; GFX9-NEXT: s_cbranch_scc1 .LBB11_5
; GFX9-NEXT: ; %bb.4: ; %cmp.true
-; GFX9-NEXT: v_pk_add_u16 v1, s17, 3
+; GFX9-NEXT: v_pk_add_u16 v1, s17, 3 op_sel_hi:[1,0]
; GFX9-NEXT: v_pk_add_u16 v0, s16, 3 op_sel_hi:[1,0]
; GFX9-NEXT: s_setpc_b64 s[30:31]
; GFX9-NEXT: .LBB11_5:
@@ -1933,9 +1948,10 @@ define inreg <3 x half> @bitcast_v3i16_to_v3f16_scalar(<3 x i16> inreg %a, i32 i
; GFX11-NEXT: s_and_b32 s2, s2, exec_lo
; GFX11-NEXT: s_cselect_b32 s2, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s2, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB11_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
-; GFX11-NEXT: v_pk_add_u16 v1, s1, 3
+; GFX11-NEXT: v_pk_add_u16 v1, s1, 3 op_sel_hi:[1,0]
; GFX11-NEXT: v_pk_add_u16 v0, s0, 3 op_sel_hi:[1,0]
; GFX11-NEXT: s_setpc_b64 s[30:31]
; GFX11-NEXT: .LBB11_4:
diff --git a/llvm/test/CodeGen/AMDGPU/amdgcn.bitcast.512bit.ll b/llvm/test/CodeGen/AMDGPU/amdgcn.bitcast.512bit.ll
index 280a0f17a4bc8a..12176bcb140ea8 100644
--- a/llvm/test/CodeGen/AMDGPU/amdgcn.bitcast.512bit.ll
+++ b/llvm/test/CodeGen/AMDGPU/amdgcn.bitcast.512bit.ll
@@ -99,8 +99,9 @@ define <16 x float> @bitcast_v16i32_to_v16f32(<16 x i32> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v16
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB0_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -316,6 +317,7 @@ define inreg <16 x float> @bitcast_v16i32_to_v16f32_scalar(<16 x i32> inreg %a,
; GFX11-NEXT: s_and_b32 s4, s4, exec_lo
; GFX11-NEXT: s_cselect_b32 s4, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s4, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB1_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: s_add_i32 s27, s27, 3
@@ -455,8 +457,9 @@ define <16 x i32> @bitcast_v16f32_to_v16i32(<16 x float> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v16
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB2_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -708,6 +711,7 @@ define inreg <16 x i32> @bitcast_v16f32_to_v16i32_scalar(<16 x float> inreg %a,
; GFX11-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB3_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_add_f32_e64 v15, s27, 1.0
@@ -847,8 +851,9 @@ define <8 x i64> @bitcast_v16i32_to_v8i64(<16 x i32> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v16
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB4_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -1064,6 +1069,7 @@ define inreg <8 x i64> @bitcast_v16i32_to_v8i64_scalar(<16 x i32> inreg %a, i32
; GFX11-NEXT: s_and_b32 s4, s4, exec_lo
; GFX11-NEXT: s_cselect_b32 s4, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s4, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB5_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: s_add_i32 s27, s27, 3
@@ -1203,28 +1209,25 @@ define <16 x i32> @bitcast_v8i64_to_v16i32(<8 x i64> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v16
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB6_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_add_co_u32 v14, vcc_lo, v14, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v15, null, 0, v15, vcc_lo
; GFX11-NEXT: v_add_co_u32 v12, vcc_lo, v12, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v13, null, 0, v13, vcc_lo
; GFX11-NEXT: v_add_co_u32 v10, vcc_lo, v10, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v11, null, 0, v11, vcc_lo
; GFX11-NEXT: v_add_co_u32 v8, vcc_lo, v8, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v9, null, 0, v9, vcc_lo
; GFX11-NEXT: v_add_co_u32 v6, vcc_lo, v6, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v7, null, 0, v7, vcc_lo
; GFX11-NEXT: v_add_co_u32 v4, vcc_lo, v4, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v5, null, 0, v5, vcc_lo
; GFX11-NEXT: v_add_co_u32 v2, vcc_lo, v2, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX11-NEXT: v_add_co_u32 v0, vcc_lo, v0, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
@@ -1424,6 +1427,7 @@ define inreg <16 x i32> @bitcast_v8i64_to_v16i32_scalar(<8 x i64> inreg %a, i32
; GFX11-NEXT: s_and_b32 s4, s4, exec_lo
; GFX11-NEXT: s_cselect_b32 s4, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s4, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB7_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: s_add_u32 s26, s26, 3
@@ -1563,8 +1567,9 @@ define <8 x double> @bitcast_v16i32_to_v8f64(<16 x i32> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v16
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB8_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -1780,6 +1785,7 @@ define inreg <8 x double> @bitcast_v16i32_to_v8f64_scalar(<16 x i32> inreg %a, i
; GFX11-NEXT: s_and_b32 s4, s4, exec_lo
; GFX11-NEXT: s_cselect_b32 s4, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s4, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB9_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: s_add_i32 s27, s27, 3
@@ -1895,8 +1901,9 @@ define <16 x i32> @bitcast_v8f64_to_v16i32(<8 x double> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v16
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB10_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -2124,6 +2131,7 @@ define inreg <16 x i32> @bitcast_v8f64_to_v16i32_scalar(<8 x double> inreg %a, i
; GFX11-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB11_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_add_f64 v[14:15], s[26:27], 1.0
@@ -2354,8 +2362,9 @@ define <32 x i16> @bitcast_v16i32_to_v32i16(<16 x i32> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v16
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB12_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -2667,6 +2676,7 @@ define inreg <32 x i16> @bitcast_v16i32_to_v32i16_scalar(<16 x i32> inreg %a, i3
; GFX11-NEXT: s_and_b32 s4, s4, exec_lo
; GFX11-NEXT: s_cselect_b32 s4, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s4, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB13_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: s_add_i32 s27, s27, 3
@@ -3008,8 +3018,9 @@ define <16 x i32> @bitcast_v32i16_to_v16i32(<32 x i16> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v16
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB14_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -3459,6 +3470,7 @@ define inreg <16 x i32> @bitcast_v32i16_to_v16i32_scalar(<32 x i16> inreg %a, i3
; GFX11-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB15_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_pk_add_u16 v15, s27, 3 op_sel_hi:[1,0]
@@ -3697,8 +3709,9 @@ define <32 x half> @bitcast_v16i32_to_v32f16(<16 x i32> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v16
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB16_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -4010,6 +4023,7 @@ define inreg <32 x half> @bitcast_v16i32_to_v32f16_scalar(<16 x i32> inreg %a, i
; GFX11-NEXT: s_and_b32 s4, s4, exec_lo
; GFX11-NEXT: s_cselect_b32 s4, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s4, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB17_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: s_add_i32 s27, s27, 3
@@ -4415,8 +4429,9 @@ define <16 x i32> @bitcast_v32f16_to_v16i32(<32 x half> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v16
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB18_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -4931,6 +4946,7 @@ define inreg <16 x i32> @bitcast_v32f16_to_v16i32_scalar(<32 x half> inreg %a, i
; GFX11-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB19_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_pk_add_f16 v15, 0x200, s27 op_sel_hi:[0,1]
@@ -5249,8 +5265,9 @@ define <32 x bfloat> @bitcast_v16i32_to_v32bf16(<16 x i32> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v16
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB20_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -5610,6 +5627,7 @@ define inreg <32 x bfloat> @bitcast_v16i32_to_v32bf16_scalar(<16 x i32> inreg %a
; GFX11-NEXT: s_and_b32 s4, s4, exec_lo
; GFX11-NEXT: s_cselect_b32 s4, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s4, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB21_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: s_add_i32 s27, s27, 3
@@ -6464,8 +6482,9 @@ define <16 x i32> @bitcast_v32bf16_to_v16i32(<32 x bfloat> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-TRUE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-TRUE16-NEXT: v_cmpx_ne_u32_e32 0, v16
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-TRUE16-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-TRUE16-NEXT: s_cbranch_execz .LBB22_2
; GFX11-TRUE16-NEXT: ; %bb.1: ; %cmp.true
@@ -6759,8 +6778,9 @@ define <16 x i32> @bitcast_v32bf16_to_v16i32(<32 x bfloat> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-FAKE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-FAKE16-NEXT: v_cmpx_ne_u32_e32 0, v16
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-FAKE16-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-FAKE16-NEXT: s_cbranch_execz .LBB22_2
; GFX11-FAKE16-NEXT: ; %bb.1: ; %cmp.true
@@ -8033,6 +8053,7 @@ define inreg <16 x i32> @bitcast_v32bf16_to_v16i32_scalar(<32 x bfloat> inreg %a
; GFX11-TRUE16-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-TRUE16-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-TRUE16-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_cbranch_scc1 .LBB23_4
; GFX11-TRUE16-NEXT: ; %bb.3: ; %cmp.true
; GFX11-TRUE16-NEXT: s_and_b32 s0, s27, 0xffff0000
@@ -8372,6 +8393,7 @@ define inreg <16 x i32> @bitcast_v32bf16_to_v16i32_scalar(<32 x bfloat> inreg %a
; GFX11-FAKE16-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-FAKE16-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-FAKE16-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_cbranch_scc1 .LBB23_4
; GFX11-FAKE16-NEXT: ; %bb.3: ; %cmp.true
; GFX11-FAKE16-NEXT: s_and_b32 s1, s27, 0xffff0000
@@ -9890,8 +9912,9 @@ define <64 x i8> @bitcast_v16i32_to_v64i8(<16 x i32> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v25, 8, v1
; GFX11-TRUE16-NEXT: .LBB24_4: ; %end
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_perm_b32 v1, v1, v25, 0xc0c0004
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX11-TRUE16-NEXT: v_perm_b32 v24, v96, v24, 0xc0c0004
; GFX11-TRUE16-NEXT: v_perm_b32 v5, v5, v71, 0xc0c0004
; GFX11-TRUE16-NEXT: v_perm_b32 v22, v70, v22, 0xc0c0004
@@ -10121,8 +10144,9 @@ define <64 x i8> @bitcast_v16i32_to_v64i8(<16 x i32> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v25, 8, v1
; GFX11-FAKE16-NEXT: .LBB24_4: ; %end
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_perm_b32 v1, v1, v25, 0xc0c0004
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_perm_b32 v24, v96, v24, 0xc0c0004
; GFX11-FAKE16-NEXT: v_perm_b32 v5, v5, v71, 0xc0c0004
; GFX11-FAKE16-NEXT: v_perm_b32 v22, v70, v22, 0xc0c0004
@@ -11424,6 +11448,7 @@ define inreg <64 x i8> @bitcast_v16i32_to_v64i8_scalar(<16 x i32> inreg %a, i32
; GFX11-NEXT: s_and_b32 s5, s48, exec_lo
; GFX11-NEXT: s_cselect_b32 s5, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s5, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB25_5
; GFX11-NEXT: ; %bb.4: ; %cmp.true
; GFX11-NEXT: s_add_i32 s1, s1, 3
@@ -13171,6 +13196,7 @@ define <16 x i32> @bitcast_v64i8_to_v16i32(<64 x i8> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: ; implicit-def: $vgpr0_vgpr1_vgpr2_vgpr3_vgpr4_vgpr5_vgpr6_vgpr7_vgpr8_vgpr9_vgpr10_vgpr11_vgpr12_vgpr13_vgpr14_vgpr15
; GFX11-TRUE16-NEXT: s_waitcnt vmcnt(5)
; GFX11-TRUE16-NEXT: v_cmpx_ne_u32_e32 0, v128
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB26_3
; GFX11-TRUE16-NEXT: ; %bb.1: ; %Flow
@@ -15060,6 +15086,7 @@ define inreg <16 x i32> @bitcast_v64i8_to_v16i32_scalar(<64 x i8> inreg %a, i32
; GFX11-TRUE16-NEXT: s_and_b32 s75, s75, exec_lo
; GFX11-TRUE16-NEXT: s_cselect_b32 s75, 1, 0
; GFX11-TRUE16-NEXT: s_cmp_lg_u32 s75, 1
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_cbranch_scc1 .LBB27_5
; GFX11-TRUE16-NEXT: ; %bb.4: ; %cmp.true
; GFX11-TRUE16-NEXT: v_mov_b32_e32 v11, 0xc0c0004
@@ -15349,6 +15376,7 @@ define inreg <16 x i32> @bitcast_v64i8_to_v16i32_scalar(<64 x i8> inreg %a, i32
; GFX11-FAKE16-NEXT: s_and_b32 s75, s75, exec_lo
; GFX11-FAKE16-NEXT: s_cselect_b32 s75, 1, 0
; GFX11-FAKE16-NEXT: s_cmp_lg_u32 s75, 1
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_cbranch_scc1 .LBB27_5
; GFX11-FAKE16-NEXT: ; %bb.4: ; %cmp.true
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v11, 0xc0c0004
@@ -15611,8 +15639,9 @@ define <8 x i64> @bitcast_v16f32_to_v8i64(<16 x float> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v16
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB28_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -15864,6 +15893,7 @@ define inreg <8 x i64> @bitcast_v16f32_to_v8i64_scalar(<16 x float> inreg %a, i3
; GFX11-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB29_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_add_f32_e64 v15, s27, 1.0
@@ -16003,28 +16033,25 @@ define <16 x float> @bitcast_v8i64_to_v16f32(<8 x i64> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v16
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB30_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_add_co_u32 v14, vcc_lo, v14, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v15, null, 0, v15, vcc_lo
; GFX11-NEXT: v_add_co_u32 v12, vcc_lo, v12, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v13, null, 0, v13, vcc_lo
; GFX11-NEXT: v_add_co_u32 v10, vcc_lo, v10, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v11, null, 0, v11, vcc_lo
; GFX11-NEXT: v_add_co_u32 v8, vcc_lo, v8, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v9, null, 0, v9, vcc_lo
; GFX11-NEXT: v_add_co_u32 v6, vcc_lo, v6, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v7, null, 0, v7, vcc_lo
; GFX11-NEXT: v_add_co_u32 v4, vcc_lo, v4, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v5, null, 0, v5, vcc_lo
; GFX11-NEXT: v_add_co_u32 v2, vcc_lo, v2, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX11-NEXT: v_add_co_u32 v0, vcc_lo, v0, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
@@ -16224,6 +16251,7 @@ define inreg <16 x float> @bitcast_v8i64_to_v16f32_scalar(<8 x i64> inreg %a, i3
; GFX11-NEXT: s_and_b32 s4, s4, exec_lo
; GFX11-NEXT: s_cselect_b32 s4, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s4, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB31_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: s_add_u32 s26, s26, 3
@@ -16363,8 +16391,9 @@ define <8 x double> @bitcast_v16f32_to_v8f64(<16 x float> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v16
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB32_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -16616,6 +16645,7 @@ define inreg <8 x double> @bitcast_v16f32_to_v8f64_scalar(<16 x float> inreg %a,
; GFX11-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB33_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_add_f32_e64 v15, s27, 1.0
@@ -16731,8 +16761,9 @@ define <16 x float> @bitcast_v8f64_to_v16f32(<8 x double> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v16
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB34_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -16960,6 +16991,7 @@ define inreg <16 x float> @bitcast_v8f64_to_v16f32_scalar(<8 x double> inreg %a,
; GFX11-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB35_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_add_f64 v[14:15], s[26:27], 1.0
@@ -17190,8 +17222,9 @@ define <32 x i16> @bitcast_v16f32_to_v32i16(<16 x float> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v16
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB36_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -17543,6 +17576,7 @@ define inreg <32 x i16> @bitcast_v16f32_to_v32i16_scalar(<16 x float> inreg %a,
; GFX11-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB37_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_add_f32_e64 v15, s27, 1.0
@@ -17884,8 +17918,9 @@ define <16 x float> @bitcast_v32i16_to_v16f32(<32 x i16> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v16
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB38_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -18335,6 +18370,7 @@ define inreg <16 x float> @bitcast_v32i16_to_v16f32_scalar(<32 x i16> inreg %a,
; GFX11-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB39_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_pk_add_u16 v15, s27, 3 op_sel_hi:[1,0]
@@ -18573,8 +18609,9 @@ define <32 x half> @bitcast_v16f32_to_v32f16(<16 x float> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v16
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB40_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -18926,6 +18963,7 @@ define inreg <32 x half> @bitcast_v16f32_to_v32f16_scalar(<16 x float> inreg %a,
; GFX11-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB41_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_add_f32_e64 v15, s27, 1.0
@@ -19331,8 +19369,9 @@ define <16 x float> @bitcast_v32f16_to_v16f32(<32 x half> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v16
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB42_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -19847,6 +19886,7 @@ define inreg <16 x float> @bitcast_v32f16_to_v16f32_scalar(<32 x half> inreg %a,
; GFX11-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB43_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_pk_add_f16 v15, 0x200, s27 op_sel_hi:[0,1]
@@ -20165,8 +20205,9 @@ define <32 x bfloat> @bitcast_v16f32_to_v32bf16(<16 x float> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v16
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB44_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -20582,6 +20623,7 @@ define inreg <32 x bfloat> @bitcast_v16f32_to_v32bf16_scalar(<16 x float> inreg
; GFX11-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB45_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_add_f32_e64 v15, s27, 1.0
@@ -21436,8 +21478,9 @@ define <16 x float> @bitcast_v32bf16_to_v16f32(<32 x bfloat> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-TRUE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-TRUE16-NEXT: v_cmpx_ne_u32_e32 0, v16
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-TRUE16-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-TRUE16-NEXT: s_cbranch_execz .LBB46_2
; GFX11-TRUE16-NEXT: ; %bb.1: ; %cmp.true
@@ -21731,8 +21774,9 @@ define <16 x float> @bitcast_v32bf16_to_v16f32(<32 x bfloat> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-FAKE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-FAKE16-NEXT: v_cmpx_ne_u32_e32 0, v16
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-FAKE16-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-FAKE16-NEXT: s_cbranch_execz .LBB46_2
; GFX11-FAKE16-NEXT: ; %bb.1: ; %cmp.true
@@ -23005,6 +23049,7 @@ define inreg <16 x float> @bitcast_v32bf16_to_v16f32_scalar(<32 x bfloat> inreg
; GFX11-TRUE16-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-TRUE16-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-TRUE16-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_cbranch_scc1 .LBB47_4
; GFX11-TRUE16-NEXT: ; %bb.3: ; %cmp.true
; GFX11-TRUE16-NEXT: s_and_b32 s0, s27, 0xffff0000
@@ -23344,6 +23389,7 @@ define inreg <16 x float> @bitcast_v32bf16_to_v16f32_scalar(<32 x bfloat> inreg
; GFX11-FAKE16-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-FAKE16-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-FAKE16-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_cbranch_scc1 .LBB47_4
; GFX11-FAKE16-NEXT: ; %bb.3: ; %cmp.true
; GFX11-FAKE16-NEXT: s_and_b32 s1, s27, 0xffff0000
@@ -24854,8 +24900,9 @@ define <64 x i8> @bitcast_v16f32_to_v64i8(<16 x float> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v25, 8, v1
; GFX11-TRUE16-NEXT: .LBB48_4: ; %end
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_perm_b32 v1, v1, v25, 0xc0c0004
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX11-TRUE16-NEXT: v_perm_b32 v24, v96, v24, 0xc0c0004
; GFX11-TRUE16-NEXT: v_perm_b32 v5, v5, v71, 0xc0c0004
; GFX11-TRUE16-NEXT: v_perm_b32 v22, v70, v22, 0xc0c0004
@@ -25077,8 +25124,9 @@ define <64 x i8> @bitcast_v16f32_to_v64i8(<16 x float> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v25, 8, v1
; GFX11-FAKE16-NEXT: .LBB48_4: ; %end
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_perm_b32 v1, v1, v25, 0xc0c0004
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_perm_b32 v24, v96, v24, 0xc0c0004
; GFX11-FAKE16-NEXT: v_perm_b32 v5, v5, v71, 0xc0c0004
; GFX11-FAKE16-NEXT: v_perm_b32 v22, v70, v22, 0xc0c0004
@@ -26586,6 +26634,7 @@ define inreg <64 x i8> @bitcast_v16f32_to_v64i8_scalar(<16 x float> inreg %a, i3
; GFX11-NEXT: s_and_b32 s5, s42, exec_lo
; GFX11-NEXT: s_cselect_b32 s5, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s5, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB49_5
; GFX11-NEXT: ; %bb.4: ; %cmp.true
; GFX11-NEXT: v_add_f32_e64 v4, s25, 1.0
@@ -28366,6 +28415,7 @@ define <16 x float> @bitcast_v64i8_to_v16f32(<64 x i8> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: ; implicit-def: $vgpr0_vgpr1_vgpr2_vgpr3_vgpr4_vgpr5_vgpr6_vgpr7_vgpr8_vgpr9_vgpr10_vgpr11_vgpr12_vgpr13_vgpr14_vgpr15
; GFX11-TRUE16-NEXT: s_waitcnt vmcnt(5)
; GFX11-TRUE16-NEXT: v_cmpx_ne_u32_e32 0, v128
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB50_3
; GFX11-TRUE16-NEXT: ; %bb.1: ; %Flow
@@ -30255,6 +30305,7 @@ define inreg <16 x float> @bitcast_v64i8_to_v16f32_scalar(<64 x i8> inreg %a, i3
; GFX11-TRUE16-NEXT: s_and_b32 s75, s75, exec_lo
; GFX11-TRUE16-NEXT: s_cselect_b32 s75, 1, 0
; GFX11-TRUE16-NEXT: s_cmp_lg_u32 s75, 1
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_cbranch_scc1 .LBB51_5
; GFX11-TRUE16-NEXT: ; %bb.4: ; %cmp.true
; GFX11-TRUE16-NEXT: v_mov_b32_e32 v11, 0xc0c0004
@@ -30544,6 +30595,7 @@ define inreg <16 x float> @bitcast_v64i8_to_v16f32_scalar(<64 x i8> inreg %a, i3
; GFX11-FAKE16-NEXT: s_and_b32 s75, s75, exec_lo
; GFX11-FAKE16-NEXT: s_cselect_b32 s75, 1, 0
; GFX11-FAKE16-NEXT: s_cmp_lg_u32 s75, 1
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_cbranch_scc1 .LBB51_5
; GFX11-FAKE16-NEXT: ; %bb.4: ; %cmp.true
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v11, 0xc0c0004
@@ -30806,28 +30858,25 @@ define <8 x double> @bitcast_v8i64_to_v8f64(<8 x i64> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v16
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB52_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_add_co_u32 v0, vcc_lo, v0, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-NEXT: v_add_co_u32 v2, vcc_lo, v2, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX11-NEXT: v_add_co_u32 v4, vcc_lo, v4, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v5, null, 0, v5, vcc_lo
; GFX11-NEXT: v_add_co_u32 v6, vcc_lo, v6, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v7, null, 0, v7, vcc_lo
; GFX11-NEXT: v_add_co_u32 v8, vcc_lo, v8, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v9, null, 0, v9, vcc_lo
; GFX11-NEXT: v_add_co_u32 v10, vcc_lo, v10, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v11, null, 0, v11, vcc_lo
; GFX11-NEXT: v_add_co_u32 v12, vcc_lo, v12, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v13, null, 0, v13, vcc_lo
; GFX11-NEXT: v_add_co_u32 v14, vcc_lo, v14, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v15, null, 0, v15, vcc_lo
@@ -31027,6 +31076,7 @@ define inreg <8 x double> @bitcast_v8i64_to_v8f64_scalar(<8 x i64> inreg %a, i32
; GFX11-NEXT: s_and_b32 s4, s4, exec_lo
; GFX11-NEXT: s_cselect_b32 s4, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s4, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB53_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: s_add_u32 s0, s0, 3
@@ -31141,8 +31191,9 @@ define <8 x i64> @bitcast_v8f64_to_v8i64(<8 x double> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v16
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB54_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -31370,6 +31421,7 @@ define inreg <8 x i64> @bitcast_v8f64_to_v8i64_scalar(<8 x double> inreg %a, i32
; GFX11-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB55_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_add_f64 v[0:1], s[12:13], 1.0
@@ -31600,28 +31652,25 @@ define <32 x i16> @bitcast_v8i64_to_v32i16(<8 x i64> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v16
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB56_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_add_co_u32 v14, vcc_lo, v14, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v15, null, 0, v15, vcc_lo
; GFX11-NEXT: v_add_co_u32 v12, vcc_lo, v12, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v13, null, 0, v13, vcc_lo
; GFX11-NEXT: v_add_co_u32 v10, vcc_lo, v10, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v11, null, 0, v11, vcc_lo
; GFX11-NEXT: v_add_co_u32 v8, vcc_lo, v8, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v9, null, 0, v9, vcc_lo
; GFX11-NEXT: v_add_co_u32 v6, vcc_lo, v6, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v7, null, 0, v7, vcc_lo
; GFX11-NEXT: v_add_co_u32 v4, vcc_lo, v4, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v5, null, 0, v5, vcc_lo
; GFX11-NEXT: v_add_co_u32 v2, vcc_lo, v2, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX11-NEXT: v_add_co_u32 v0, vcc_lo, v0, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
@@ -31917,6 +31966,7 @@ define inreg <32 x i16> @bitcast_v8i64_to_v32i16_scalar(<8 x i64> inreg %a, i32
; GFX11-NEXT: s_and_b32 s4, s4, exec_lo
; GFX11-NEXT: s_cselect_b32 s4, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s4, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB57_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: s_add_u32 s26, s26, 3
@@ -32258,8 +32308,9 @@ define <8 x i64> @bitcast_v32i16_to_v8i64(<32 x i16> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v16
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB58_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -32709,6 +32760,7 @@ define inreg <8 x i64> @bitcast_v32i16_to_v8i64_scalar(<32 x i16> inreg %a, i32
; GFX11-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB59_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_pk_add_u16 v15, s27, 3 op_sel_hi:[1,0]
@@ -32947,28 +32999,25 @@ define <32 x half> @bitcast_v8i64_to_v32f16(<8 x i64> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v16
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB60_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_add_co_u32 v14, vcc_lo, v14, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v15, null, 0, v15, vcc_lo
; GFX11-NEXT: v_add_co_u32 v12, vcc_lo, v12, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v13, null, 0, v13, vcc_lo
; GFX11-NEXT: v_add_co_u32 v10, vcc_lo, v10, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v11, null, 0, v11, vcc_lo
; GFX11-NEXT: v_add_co_u32 v8, vcc_lo, v8, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v9, null, 0, v9, vcc_lo
; GFX11-NEXT: v_add_co_u32 v6, vcc_lo, v6, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v7, null, 0, v7, vcc_lo
; GFX11-NEXT: v_add_co_u32 v4, vcc_lo, v4, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v5, null, 0, v5, vcc_lo
; GFX11-NEXT: v_add_co_u32 v2, vcc_lo, v2, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX11-NEXT: v_add_co_u32 v0, vcc_lo, v0, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
@@ -33264,6 +33313,7 @@ define inreg <32 x half> @bitcast_v8i64_to_v32f16_scalar(<8 x i64> inreg %a, i32
; GFX11-NEXT: s_and_b32 s4, s4, exec_lo
; GFX11-NEXT: s_cselect_b32 s4, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s4, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB61_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: s_add_u32 s26, s26, 3
@@ -33669,8 +33719,9 @@ define <8 x i64> @bitcast_v32f16_to_v8i64(<32 x half> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v16
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB62_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -34185,6 +34236,7 @@ define inreg <8 x i64> @bitcast_v32f16_to_v8i64_scalar(<32 x half> inreg %a, i32
; GFX11-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB63_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_pk_add_f16 v15, 0x200, s27 op_sel_hi:[0,1]
@@ -34503,28 +34555,25 @@ define <32 x bfloat> @bitcast_v8i64_to_v32bf16(<8 x i64> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v16
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB64_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_add_co_u32 v14, vcc_lo, v14, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v15, null, 0, v15, vcc_lo
; GFX11-NEXT: v_add_co_u32 v12, vcc_lo, v12, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v13, null, 0, v13, vcc_lo
; GFX11-NEXT: v_add_co_u32 v10, vcc_lo, v10, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v11, null, 0, v11, vcc_lo
; GFX11-NEXT: v_add_co_u32 v8, vcc_lo, v8, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v9, null, 0, v9, vcc_lo
; GFX11-NEXT: v_add_co_u32 v6, vcc_lo, v6, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v7, null, 0, v7, vcc_lo
; GFX11-NEXT: v_add_co_u32 v4, vcc_lo, v4, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v5, null, 0, v5, vcc_lo
; GFX11-NEXT: v_add_co_u32 v2, vcc_lo, v2, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX11-NEXT: v_add_co_u32 v0, vcc_lo, v0, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
@@ -34868,6 +34917,7 @@ define inreg <32 x bfloat> @bitcast_v8i64_to_v32bf16_scalar(<8 x i64> inreg %a,
; GFX11-NEXT: s_and_b32 s4, s4, exec_lo
; GFX11-NEXT: s_cselect_b32 s4, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s4, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB65_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: s_add_u32 s26, s26, 3
@@ -35722,8 +35772,9 @@ define <8 x i64> @bitcast_v32bf16_to_v8i64(<32 x bfloat> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-TRUE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-TRUE16-NEXT: v_cmpx_ne_u32_e32 0, v16
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-TRUE16-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-TRUE16-NEXT: s_cbranch_execz .LBB66_2
; GFX11-TRUE16-NEXT: ; %bb.1: ; %cmp.true
@@ -36017,8 +36068,9 @@ define <8 x i64> @bitcast_v32bf16_to_v8i64(<32 x bfloat> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-FAKE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-FAKE16-NEXT: v_cmpx_ne_u32_e32 0, v16
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-FAKE16-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-FAKE16-NEXT: s_cbranch_execz .LBB66_2
; GFX11-FAKE16-NEXT: ; %bb.1: ; %cmp.true
@@ -37291,6 +37343,7 @@ define inreg <8 x i64> @bitcast_v32bf16_to_v8i64_scalar(<32 x bfloat> inreg %a,
; GFX11-TRUE16-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-TRUE16-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-TRUE16-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_cbranch_scc1 .LBB67_4
; GFX11-TRUE16-NEXT: ; %bb.3: ; %cmp.true
; GFX11-TRUE16-NEXT: s_and_b32 s0, s27, 0xffff0000
@@ -37630,6 +37683,7 @@ define inreg <8 x i64> @bitcast_v32bf16_to_v8i64_scalar(<32 x bfloat> inreg %a,
; GFX11-FAKE16-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-FAKE16-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-FAKE16-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_cbranch_scc1 .LBB67_4
; GFX11-FAKE16-NEXT: ; %bb.3: ; %cmp.true
; GFX11-FAKE16-NEXT: s_and_b32 s1, s27, 0xffff0000
@@ -39083,22 +39137,18 @@ define <64 x i8> @bitcast_v8i64_to_v64i8(<8 x i64> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: s_cbranch_execz .LBB68_4
; GFX11-TRUE16-NEXT: ; %bb.3: ; %cmp.true
; GFX11-TRUE16-NEXT: v_add_co_u32 v1, vcc_lo, v1, 3
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v2, null, 0, v2, vcc_lo
; GFX11-TRUE16-NEXT: v_add_co_u32 v3, vcc_lo, v3, 3
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v4, null, 0, v4, vcc_lo
; GFX11-TRUE16-NEXT: v_add_co_u32 v5, vcc_lo, v5, 3
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v6, null, 0, v6, vcc_lo
; GFX11-TRUE16-NEXT: v_add_co_u32 v9, vcc_lo, v9, 3
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v10, null, 0, v10, vcc_lo
; GFX11-TRUE16-NEXT: v_add_co_u32 v11, vcc_lo, v11, 3
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v12, null, 0, v12, vcc_lo
; GFX11-TRUE16-NEXT: v_add_co_u32 v13, vcc_lo, v13, 3
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v14, null, 0, v14, vcc_lo
; GFX11-TRUE16-NEXT: v_add_co_u32 v15, vcc_lo, v15, 3
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v16, null, 0, v16, vcc_lo
; GFX11-TRUE16-NEXT: v_add_co_u32 v7, vcc_lo, v7, 3
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v8, null, 0, v8, vcc_lo
@@ -39153,8 +39203,9 @@ define <64 x i8> @bitcast_v8i64_to_v64i8(<8 x i64> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v25, 8, v1
; GFX11-TRUE16-NEXT: .LBB68_4: ; %end
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_perm_b32 v1, v1, v25, 0xc0c0004
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX11-TRUE16-NEXT: v_perm_b32 v24, v96, v24, 0xc0c0004
; GFX11-TRUE16-NEXT: v_perm_b32 v5, v5, v71, 0xc0c0004
; GFX11-TRUE16-NEXT: v_perm_b32 v22, v70, v22, 0xc0c0004
@@ -39319,22 +39370,18 @@ define <64 x i8> @bitcast_v8i64_to_v64i8(<8 x i64> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: s_cbranch_execz .LBB68_4
; GFX11-FAKE16-NEXT: ; %bb.3: ; %cmp.true
; GFX11-FAKE16-NEXT: v_add_co_u32 v1, vcc_lo, v1, 3
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v2, null, 0, v2, vcc_lo
; GFX11-FAKE16-NEXT: v_add_co_u32 v3, vcc_lo, v3, 3
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v4, null, 0, v4, vcc_lo
; GFX11-FAKE16-NEXT: v_add_co_u32 v5, vcc_lo, v5, 3
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v6, null, 0, v6, vcc_lo
; GFX11-FAKE16-NEXT: v_add_co_u32 v9, vcc_lo, v9, 3
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v10, null, 0, v10, vcc_lo
; GFX11-FAKE16-NEXT: v_add_co_u32 v11, vcc_lo, v11, 3
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v12, null, 0, v12, vcc_lo
; GFX11-FAKE16-NEXT: v_add_co_u32 v13, vcc_lo, v13, 3
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v14, null, 0, v14, vcc_lo
; GFX11-FAKE16-NEXT: v_add_co_u32 v15, vcc_lo, v15, 3
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v16, null, 0, v16, vcc_lo
; GFX11-FAKE16-NEXT: v_add_co_u32 v7, vcc_lo, v7, 3
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v8, null, 0, v8, vcc_lo
@@ -39389,8 +39436,9 @@ define <64 x i8> @bitcast_v8i64_to_v64i8(<8 x i64> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v25, 8, v1
; GFX11-FAKE16-NEXT: .LBB68_4: ; %end
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_perm_b32 v1, v1, v25, 0xc0c0004
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_perm_b32 v24, v96, v24, 0xc0c0004
; GFX11-FAKE16-NEXT: v_perm_b32 v5, v5, v71, 0xc0c0004
; GFX11-FAKE16-NEXT: v_perm_b32 v22, v70, v22, 0xc0c0004
@@ -40692,6 +40740,7 @@ define inreg <64 x i8> @bitcast_v8i64_to_v64i8_scalar(<8 x i64> inreg %a, i32 in
; GFX11-NEXT: s_and_b32 s5, s48, exec_lo
; GFX11-NEXT: s_cselect_b32 s5, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s5, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB69_5
; GFX11-NEXT: ; %bb.4: ; %cmp.true
; GFX11-NEXT: s_add_u32 s0, s0, 3
@@ -42439,6 +42488,7 @@ define <8 x i64> @bitcast_v64i8_to_v8i64(<64 x i8> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: ; implicit-def: $vgpr0_vgpr1_vgpr2_vgpr3_vgpr4_vgpr5_vgpr6_vgpr7_vgpr8_vgpr9_vgpr10_vgpr11_vgpr12_vgpr13_vgpr14_vgpr15
; GFX11-TRUE16-NEXT: s_waitcnt vmcnt(5)
; GFX11-TRUE16-NEXT: v_cmpx_ne_u32_e32 0, v128
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB70_3
; GFX11-TRUE16-NEXT: ; %bb.1: ; %Flow
@@ -44328,6 +44378,7 @@ define inreg <8 x i64> @bitcast_v64i8_to_v8i64_scalar(<64 x i8> inreg %a, i32 in
; GFX11-TRUE16-NEXT: s_and_b32 s75, s75, exec_lo
; GFX11-TRUE16-NEXT: s_cselect_b32 s75, 1, 0
; GFX11-TRUE16-NEXT: s_cmp_lg_u32 s75, 1
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_cbranch_scc1 .LBB71_5
; GFX11-TRUE16-NEXT: ; %bb.4: ; %cmp.true
; GFX11-TRUE16-NEXT: v_mov_b32_e32 v11, 0xc0c0004
@@ -44617,6 +44668,7 @@ define inreg <8 x i64> @bitcast_v64i8_to_v8i64_scalar(<64 x i8> inreg %a, i32 in
; GFX11-FAKE16-NEXT: s_and_b32 s75, s75, exec_lo
; GFX11-FAKE16-NEXT: s_cselect_b32 s75, 1, 0
; GFX11-FAKE16-NEXT: s_cmp_lg_u32 s75, 1
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_cbranch_scc1 .LBB71_5
; GFX11-FAKE16-NEXT: ; %bb.4: ; %cmp.true
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v11, 0xc0c0004
@@ -44954,8 +45006,9 @@ define <32 x i16> @bitcast_v8f64_to_v32i16(<8 x double> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v16
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB72_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -45283,6 +45336,7 @@ define inreg <32 x i16> @bitcast_v8f64_to_v32i16_scalar(<8 x double> inreg %a, i
; GFX11-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB73_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_add_f64 v[14:15], s[26:27], 1.0
@@ -45616,8 +45670,9 @@ define <8 x double> @bitcast_v32i16_to_v8f64(<32 x i16> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v16
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB74_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -46067,6 +46122,7 @@ define inreg <8 x double> @bitcast_v32i16_to_v8f64_scalar(<32 x i16> inreg %a, i
; GFX11-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB75_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_pk_add_u16 v15, s27, 3 op_sel_hi:[1,0]
@@ -46281,8 +46337,9 @@ define <32 x half> @bitcast_v8f64_to_v32f16(<8 x double> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v16
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB76_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -46610,6 +46667,7 @@ define inreg <32 x half> @bitcast_v8f64_to_v32f16_scalar(<8 x double> inreg %a,
; GFX11-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB77_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_add_f64 v[14:15], s[26:27], 1.0
@@ -47007,8 +47065,9 @@ define <8 x double> @bitcast_v32f16_to_v8f64(<32 x half> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v16
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB78_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -47523,6 +47582,7 @@ define inreg <8 x double> @bitcast_v32f16_to_v8f64_scalar(<32 x half> inreg %a,
; GFX11-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB79_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_pk_add_f16 v15, 0x200, s27 op_sel_hi:[0,1]
@@ -47809,8 +47869,9 @@ define <32 x bfloat> @bitcast_v8f64_to_v32bf16(<8 x double> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v16
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB80_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -48202,6 +48263,7 @@ define inreg <32 x bfloat> @bitcast_v8f64_to_v32bf16_scalar(<8 x double> inreg %
; GFX11-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB81_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_add_f64 v[14:15], s[26:27], 1.0
@@ -49048,8 +49110,9 @@ define <8 x double> @bitcast_v32bf16_to_v8f64(<32 x bfloat> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-TRUE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-TRUE16-NEXT: v_cmpx_ne_u32_e32 0, v16
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-TRUE16-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-TRUE16-NEXT: s_cbranch_execz .LBB82_2
; GFX11-TRUE16-NEXT: ; %bb.1: ; %cmp.true
@@ -49343,8 +49406,9 @@ define <8 x double> @bitcast_v32bf16_to_v8f64(<32 x bfloat> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-FAKE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-FAKE16-NEXT: v_cmpx_ne_u32_e32 0, v16
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-FAKE16-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-FAKE16-NEXT: s_cbranch_execz .LBB82_2
; GFX11-FAKE16-NEXT: ; %bb.1: ; %cmp.true
@@ -50617,6 +50681,7 @@ define inreg <8 x double> @bitcast_v32bf16_to_v8f64_scalar(<32 x bfloat> inreg %
; GFX11-TRUE16-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-TRUE16-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-TRUE16-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_cbranch_scc1 .LBB83_4
; GFX11-TRUE16-NEXT: ; %bb.3: ; %cmp.true
; GFX11-TRUE16-NEXT: s_and_b32 s0, s27, 0xffff0000
@@ -50956,6 +51021,7 @@ define inreg <8 x double> @bitcast_v32bf16_to_v8f64_scalar(<32 x bfloat> inreg %
; GFX11-FAKE16-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-FAKE16-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-FAKE16-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_cbranch_scc1 .LBB83_4
; GFX11-FAKE16-NEXT: ; %bb.3: ; %cmp.true
; GFX11-FAKE16-NEXT: s_and_b32 s1, s27, 0xffff0000
@@ -52442,8 +52508,9 @@ define <64 x i8> @bitcast_v8f64_to_v64i8(<8 x double> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v25, 8, v1
; GFX11-TRUE16-NEXT: .LBB84_4: ; %end
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_perm_b32 v1, v1, v25, 0xc0c0004
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX11-TRUE16-NEXT: v_perm_b32 v24, v96, v24, 0xc0c0004
; GFX11-TRUE16-NEXT: v_perm_b32 v5, v5, v71, 0xc0c0004
; GFX11-TRUE16-NEXT: v_perm_b32 v22, v70, v22, 0xc0c0004
@@ -52665,8 +52732,9 @@ define <64 x i8> @bitcast_v8f64_to_v64i8(<8 x double> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v25, 8, v1
; GFX11-FAKE16-NEXT: .LBB84_4: ; %end
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_perm_b32 v1, v1, v25, 0xc0c0004
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_perm_b32 v24, v96, v24, 0xc0c0004
; GFX11-FAKE16-NEXT: v_perm_b32 v5, v5, v71, 0xc0c0004
; GFX11-FAKE16-NEXT: v_perm_b32 v22, v70, v22, 0xc0c0004
@@ -54158,6 +54226,7 @@ define inreg <64 x i8> @bitcast_v8f64_to_v64i8_scalar(<8 x double> inreg %a, i32
; GFX11-NEXT: s_and_b32 s5, s42, exec_lo
; GFX11-NEXT: s_cselect_b32 s5, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s5, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB85_5
; GFX11-NEXT: ; %bb.4: ; %cmp.true
; GFX11-NEXT: v_add_f64 v[7:8], s[20:21], 1.0
@@ -55930,6 +55999,7 @@ define <8 x double> @bitcast_v64i8_to_v8f64(<64 x i8> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: ; implicit-def: $vgpr0_vgpr1_vgpr2_vgpr3_vgpr4_vgpr5_vgpr6_vgpr7_vgpr8_vgpr9_vgpr10_vgpr11_vgpr12_vgpr13_vgpr14_vgpr15
; GFX11-TRUE16-NEXT: s_waitcnt vmcnt(5)
; GFX11-TRUE16-NEXT: v_cmpx_ne_u32_e32 0, v128
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB86_3
; GFX11-TRUE16-NEXT: ; %bb.1: ; %Flow
@@ -57819,6 +57889,7 @@ define inreg <8 x double> @bitcast_v64i8_to_v8f64_scalar(<64 x i8> inreg %a, i32
; GFX11-TRUE16-NEXT: s_and_b32 s75, s75, exec_lo
; GFX11-TRUE16-NEXT: s_cselect_b32 s75, 1, 0
; GFX11-TRUE16-NEXT: s_cmp_lg_u32 s75, 1
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_cbranch_scc1 .LBB87_5
; GFX11-TRUE16-NEXT: ; %bb.4: ; %cmp.true
; GFX11-TRUE16-NEXT: v_mov_b32_e32 v11, 0xc0c0004
@@ -58108,6 +58179,7 @@ define inreg <8 x double> @bitcast_v64i8_to_v8f64_scalar(<64 x i8> inreg %a, i32
; GFX11-FAKE16-NEXT: s_and_b32 s75, s75, exec_lo
; GFX11-FAKE16-NEXT: s_cselect_b32 s75, 1, 0
; GFX11-FAKE16-NEXT: s_cmp_lg_u32 s75, 1
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_cbranch_scc1 .LBB87_5
; GFX11-FAKE16-NEXT: ; %bb.4: ; %cmp.true
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v11, 0xc0c0004
@@ -58684,8 +58756,9 @@ define <32 x half> @bitcast_v32i16_to_v32f16(<32 x i16> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v16
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB88_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -59226,6 +59299,7 @@ define inreg <32 x half> @bitcast_v32i16_to_v32f16_scalar(<32 x i16> inreg %a, i
; GFX11-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB89_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_pk_add_u16 v15, s27, 3 op_sel_hi:[1,0]
@@ -59583,8 +59657,9 @@ define <32 x i16> @bitcast_v32f16_to_v32i16(<32 x half> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v16
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB90_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -60098,6 +60173,7 @@ define inreg <32 x i16> @bitcast_v32f16_to_v32i16_scalar(<32 x half> inreg %a, i
; GFX11-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB91_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_pk_add_f16 v15, 0x200, s27 op_sel_hi:[0,1]
@@ -60498,8 +60574,9 @@ define <32 x bfloat> @bitcast_v32i16_to_v32bf16(<32 x i16> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v16
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB92_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -61040,6 +61117,7 @@ define inreg <32 x bfloat> @bitcast_v32i16_to_v32bf16_scalar(<32 x i16> inreg %a
; GFX11-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB93_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_pk_add_u16 v15, s27, 3 op_sel_hi:[1,0]
@@ -62019,8 +62097,9 @@ define <32 x i16> @bitcast_v32bf16_to_v32i16(<32 x bfloat> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-TRUE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-TRUE16-NEXT: v_cmpx_ne_u32_e32 0, v16
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-TRUE16-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-TRUE16-NEXT: s_cbranch_execz .LBB94_2
; GFX11-TRUE16-NEXT: ; %bb.1: ; %cmp.true
@@ -62330,8 +62409,9 @@ define <32 x i16> @bitcast_v32bf16_to_v32i16(<32 x bfloat> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-FAKE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-FAKE16-NEXT: v_cmpx_ne_u32_e32 0, v16
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-FAKE16-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-FAKE16-NEXT: s_cbranch_execz .LBB94_2
; GFX11-FAKE16-NEXT: ; %bb.1: ; %cmp.true
@@ -63653,6 +63733,7 @@ define inreg <32 x i16> @bitcast_v32bf16_to_v32i16_scalar(<32 x bfloat> inreg %a
; GFX11-TRUE16-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-TRUE16-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-TRUE16-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_cbranch_scc1 .LBB95_4
; GFX11-TRUE16-NEXT: ; %bb.3: ; %cmp.true
; GFX11-TRUE16-NEXT: s_and_b32 s0, s12, 0xffff0000
@@ -63951,6 +64032,7 @@ define inreg <32 x i16> @bitcast_v32bf16_to_v32i16_scalar(<32 x bfloat> inreg %a
; GFX11-FAKE16-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-FAKE16-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-FAKE16-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_cbranch_scc1 .LBB95_4
; GFX11-FAKE16-NEXT: ; %bb.3: ; %cmp.true
; GFX11-FAKE16-NEXT: s_and_b32 s0, s12, 0xffff0000
@@ -65936,8 +66018,9 @@ define <64 x i8> @bitcast_v32i16_to_v64i8(<32 x i16> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v25, 8, v1
; GFX11-TRUE16-NEXT: .LBB96_4: ; %end
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_perm_b32 v1, v1, v25, 0xc0c0004
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX11-TRUE16-NEXT: v_perm_b32 v24, v96, v24, 0xc0c0004
; GFX11-TRUE16-NEXT: v_perm_b32 v5, v5, v71, 0xc0c0004
; GFX11-TRUE16-NEXT: v_perm_b32 v22, v70, v22, 0xc0c0004
@@ -66167,8 +66250,9 @@ define <64 x i8> @bitcast_v32i16_to_v64i8(<32 x i16> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v25, 8, v1
; GFX11-FAKE16-NEXT: .LBB96_4: ; %end
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_perm_b32 v1, v1, v25, 0xc0c0004
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_perm_b32 v24, v96, v24, 0xc0c0004
; GFX11-FAKE16-NEXT: v_perm_b32 v5, v5, v71, 0xc0c0004
; GFX11-FAKE16-NEXT: v_perm_b32 v22, v70, v22, 0xc0c0004
@@ -67787,6 +67871,7 @@ define inreg <64 x i8> @bitcast_v32i16_to_v64i8_scalar(<32 x i16> inreg %a, i32
; GFX11-NEXT: s_and_b32 s5, s42, exec_lo
; GFX11-NEXT: s_cselect_b32 s5, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s5, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB97_5
; GFX11-NEXT: ; %bb.4: ; %cmp.true
; GFX11-NEXT: v_pk_add_u16 v4, s25, 3 op_sel_hi:[1,0]
@@ -69775,6 +69860,7 @@ define <32 x i16> @bitcast_v64i8_to_v32i16(<64 x i8> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: v_lshlrev_b16 v17.h, 8, v55.h
; GFX11-TRUE16-NEXT: v_lshlrev_b16 v19.l, 8, v64.h
; GFX11-TRUE16-NEXT: v_cmpx_ne_u32_e32 0, v100
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB98_3
; GFX11-TRUE16-NEXT: ; %bb.1: ; %Flow
@@ -71998,6 +72084,7 @@ define inreg <32 x i16> @bitcast_v64i8_to_v32i16_scalar(<64 x i8> inreg %a, i32
; GFX11-TRUE16-NEXT: s_and_b32 s75, s75, exec_lo
; GFX11-TRUE16-NEXT: s_cselect_b32 s75, 1, 0
; GFX11-TRUE16-NEXT: s_cmp_lg_u32 s75, 1
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_cbranch_scc1 .LBB99_5
; GFX11-TRUE16-NEXT: ; %bb.4: ; %cmp.true
; GFX11-TRUE16-NEXT: v_add_nc_u32_e32 v1, 3, v36
@@ -72254,6 +72341,7 @@ define inreg <32 x i16> @bitcast_v64i8_to_v32i16_scalar(<64 x i8> inreg %a, i32
; GFX11-FAKE16-NEXT: s_and_b32 s75, s75, exec_lo
; GFX11-FAKE16-NEXT: s_cselect_b32 s75, 1, 0
; GFX11-FAKE16-NEXT: s_cmp_lg_u32 s75, 1
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_cbranch_scc1 .LBB99_5
; GFX11-FAKE16-NEXT: ; %bb.4: ; %cmp.true
; GFX11-FAKE16-NEXT: v_add_nc_u32_e32 v0, 3, v52
@@ -72860,8 +72948,9 @@ define <32 x bfloat> @bitcast_v32f16_to_v32bf16(<32 x half> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v16
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB100_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -73467,6 +73556,7 @@ define inreg <32 x bfloat> @bitcast_v32f16_to_v32bf16_scalar(<32 x half> inreg %
; GFX11-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB101_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_pk_add_f16 v15, 0x200, s27 op_sel_hi:[0,1]
@@ -74462,8 +74552,9 @@ define <32 x half> @bitcast_v32bf16_to_v32f16(<32 x bfloat> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-TRUE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-TRUE16-NEXT: v_cmpx_ne_u32_e32 0, v16
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-TRUE16-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-TRUE16-NEXT: s_cbranch_execz .LBB102_2
; GFX11-TRUE16-NEXT: ; %bb.1: ; %cmp.true
@@ -74756,8 +74847,9 @@ define <32 x half> @bitcast_v32bf16_to_v32f16(<32 x bfloat> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-FAKE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-FAKE16-NEXT: v_cmpx_ne_u32_e32 0, v16
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-FAKE16-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-FAKE16-NEXT: s_cbranch_execz .LBB102_2
; GFX11-FAKE16-NEXT: ; %bb.1: ; %cmp.true
@@ -76319,6 +76411,7 @@ define inreg <32 x half> @bitcast_v32bf16_to_v32f16_scalar(<32 x bfloat> inreg %
; GFX11-TRUE16-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-TRUE16-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-TRUE16-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_cbranch_scc1 .LBB103_4
; GFX11-TRUE16-NEXT: ; %bb.3: ; %cmp.true
; GFX11-TRUE16-NEXT: s_and_b32 s0, s12, 0xffff0000
@@ -76651,6 +76744,7 @@ define inreg <32 x half> @bitcast_v32bf16_to_v32f16_scalar(<32 x bfloat> inreg %
; GFX11-FAKE16-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-FAKE16-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-FAKE16-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_cbranch_scc1 .LBB103_4
; GFX11-FAKE16-NEXT: ; %bb.3: ; %cmp.true
; GFX11-FAKE16-NEXT: s_and_b32 s0, s12, 0xffff0000
@@ -78537,8 +78631,9 @@ define <64 x i8> @bitcast_v32f16_to_v64i8(<32 x half> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v25, 8, v1
; GFX11-TRUE16-NEXT: .LBB104_4: ; %end
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_perm_b32 v1, v1, v25, 0xc0c0004
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX11-TRUE16-NEXT: v_perm_b32 v24, v96, v24, 0xc0c0004
; GFX11-TRUE16-NEXT: v_perm_b32 v5, v5, v71, 0xc0c0004
; GFX11-TRUE16-NEXT: v_perm_b32 v22, v70, v22, 0xc0c0004
@@ -78768,8 +78863,9 @@ define <64 x i8> @bitcast_v32f16_to_v64i8(<32 x half> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v25, 8, v1
; GFX11-FAKE16-NEXT: .LBB104_4: ; %end
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_perm_b32 v1, v1, v25, 0xc0c0004
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_perm_b32 v24, v96, v24, 0xc0c0004
; GFX11-FAKE16-NEXT: v_perm_b32 v5, v5, v71, 0xc0c0004
; GFX11-FAKE16-NEXT: v_perm_b32 v22, v70, v22, 0xc0c0004
@@ -80524,6 +80620,7 @@ define inreg <64 x i8> @bitcast_v32f16_to_v64i8_scalar(<32 x half> inreg %a, i32
; GFX11-NEXT: s_and_b32 s5, s42, exec_lo
; GFX11-NEXT: s_cselect_b32 s5, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s5, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB105_5
; GFX11-NEXT: ; %bb.4: ; %cmp.true
; GFX11-NEXT: v_pk_add_f16 v4, 0x200, s25 op_sel_hi:[0,1]
@@ -82512,6 +82609,7 @@ define <32 x half> @bitcast_v64i8_to_v32f16(<64 x i8> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: v_lshlrev_b16 v17.h, 8, v55.h
; GFX11-TRUE16-NEXT: v_lshlrev_b16 v19.l, 8, v64.h
; GFX11-TRUE16-NEXT: v_cmpx_ne_u32_e32 0, v100
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB106_3
; GFX11-TRUE16-NEXT: ; %bb.1: ; %Flow
@@ -84735,6 +84833,7 @@ define inreg <32 x half> @bitcast_v64i8_to_v32f16_scalar(<64 x i8> inreg %a, i32
; GFX11-TRUE16-NEXT: s_and_b32 s75, s75, exec_lo
; GFX11-TRUE16-NEXT: s_cselect_b32 s75, 1, 0
; GFX11-TRUE16-NEXT: s_cmp_lg_u32 s75, 1
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_cbranch_scc1 .LBB107_5
; GFX11-TRUE16-NEXT: ; %bb.4: ; %cmp.true
; GFX11-TRUE16-NEXT: v_add_nc_u32_e32 v1, 3, v36
@@ -84991,6 +85090,7 @@ define inreg <32 x half> @bitcast_v64i8_to_v32f16_scalar(<64 x i8> inreg %a, i32
; GFX11-FAKE16-NEXT: s_and_b32 s75, s75, exec_lo
; GFX11-FAKE16-NEXT: s_cselect_b32 s75, 1, 0
; GFX11-FAKE16-NEXT: s_cmp_lg_u32 s75, 1
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_cbranch_scc1 .LBB107_5
; GFX11-FAKE16-NEXT: ; %bb.4: ; %cmp.true
; GFX11-FAKE16-NEXT: v_add_nc_u32_e32 v0, 3, v52
@@ -87514,8 +87614,9 @@ define <64 x i8> @bitcast_v32bf16_to_v64i8(<32 x bfloat> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v25, 8, v1
; GFX11-TRUE16-NEXT: .LBB108_4: ; %end
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_4)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_perm_b32 v1, v1, v25, 0xc0c0004
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_4)
; GFX11-TRUE16-NEXT: v_perm_b32 v24, v28, v24, 0xc0c0004
; GFX11-TRUE16-NEXT: v_perm_b32 v5, v5, v83, 0xc0c0004
; GFX11-TRUE16-NEXT: v_perm_b32 v22, v32, v22, 0xc0c0004
@@ -88007,7 +88108,7 @@ define <64 x i8> @bitcast_v32bf16_to_v64i8(<32 x bfloat> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v25, 8, v26
; GFX11-FAKE16-NEXT: .LBB108_4: ; %end
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_perm_b32 v1, v1, v25, 0xc0c0004
; GFX11-FAKE16-NEXT: v_perm_b32 v24, v81, v24, 0xc0c0004
; GFX11-FAKE16-NEXT: v_perm_b32 v5, v5, v87, 0xc0c0004
@@ -90292,6 +90393,7 @@ define inreg <64 x i8> @bitcast_v32bf16_to_v64i8_scalar(<32 x bfloat> inreg %a,
; GFX11-TRUE16-NEXT: s_and_b32 s5, s42, exec_lo
; GFX11-TRUE16-NEXT: s_cselect_b32 s5, 1, 0
; GFX11-TRUE16-NEXT: s_cmp_lg_u32 s5, 1
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_cbranch_scc1 .LBB109_5
; GFX11-TRUE16-NEXT: ; %bb.4: ; %cmp.true
; GFX11-TRUE16-NEXT: s_lshl_b32 s4, s1, 16
@@ -90889,6 +90991,7 @@ define inreg <64 x i8> @bitcast_v32bf16_to_v64i8_scalar(<32 x bfloat> inreg %a,
; GFX11-FAKE16-NEXT: s_and_b32 s5, s42, exec_lo
; GFX11-FAKE16-NEXT: s_cselect_b32 s5, 1, 0
; GFX11-FAKE16-NEXT: s_cmp_lg_u32 s5, 1
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_cbranch_scc1 .LBB109_5
; GFX11-FAKE16-NEXT: ; %bb.4: ; %cmp.true
; GFX11-FAKE16-NEXT: s_lshl_b32 s4, s1, 16
@@ -93182,6 +93285,7 @@ define <32 x bfloat> @bitcast_v64i8_to_v32bf16(<64 x i8> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: v_lshlrev_b16 v17.h, 8, v55.h
; GFX11-TRUE16-NEXT: v_lshlrev_b16 v19.l, 8, v64.h
; GFX11-TRUE16-NEXT: v_cmpx_ne_u32_e32 0, v100
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB110_3
; GFX11-TRUE16-NEXT: ; %bb.1: ; %Flow
@@ -95374,6 +95478,7 @@ define inreg <32 x bfloat> @bitcast_v64i8_to_v32bf16_scalar(<64 x i8> inreg %a,
; GFX11-TRUE16-NEXT: s_and_b32 s75, s75, exec_lo
; GFX11-TRUE16-NEXT: s_cselect_b32 s75, 1, 0
; GFX11-TRUE16-NEXT: s_cmp_lg_u32 s75, 1
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_cbranch_scc1 .LBB111_5
; GFX11-TRUE16-NEXT: ; %bb.4: ; %cmp.true
; GFX11-TRUE16-NEXT: v_add_nc_u32_e32 v1, 3, v36
@@ -95630,6 +95735,7 @@ define inreg <32 x bfloat> @bitcast_v64i8_to_v32bf16_scalar(<64 x i8> inreg %a,
; GFX11-FAKE16-NEXT: s_and_b32 s75, s75, exec_lo
; GFX11-FAKE16-NEXT: s_cselect_b32 s75, 1, 0
; GFX11-FAKE16-NEXT: s_cmp_lg_u32 s75, 1
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_cbranch_scc1 .LBB111_5
; GFX11-FAKE16-NEXT: ; %bb.4: ; %cmp.true
; GFX11-FAKE16-NEXT: v_add_nc_u32_e32 v0, 3, v52
diff --git a/llvm/test/CodeGen/AMDGPU/amdgcn.bitcast.576bit.ll b/llvm/test/CodeGen/AMDGPU/amdgcn.bitcast.576bit.ll
index 9b079dccf1023a..7ae172197d52b5 100644
--- a/llvm/test/CodeGen/AMDGPU/amdgcn.bitcast.576bit.ll
+++ b/llvm/test/CodeGen/AMDGPU/amdgcn.bitcast.576bit.ll
@@ -105,8 +105,9 @@ define <18 x float> @bitcast_v18i32_to_v18f32(<18 x i32> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v18
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB0_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -343,6 +344,7 @@ define inreg <18 x float> @bitcast_v18i32_to_v18f32_scalar(<18 x i32> inreg %a,
; GFX11-NEXT: s_and_b32 s4, s4, exec_lo
; GFX11-NEXT: s_cselect_b32 s4, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s4, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB1_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: s_add_i32 s29, s29, 3
@@ -491,8 +493,9 @@ define <18 x i32> @bitcast_v18f32_to_v18i32(<18 x float> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v18
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB2_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -923,6 +926,7 @@ define inreg <18 x i32> @bitcast_v18f32_to_v18i32_scalar(<18 x float> inreg %a,
; GFX11-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB3_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_add_f32_e64 v17, s53, 1.0
@@ -1093,8 +1097,9 @@ define <9 x i64> @bitcast_v18i32_to_v9i64(<18 x i32> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v18
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB4_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -1331,6 +1336,7 @@ define inreg <9 x i64> @bitcast_v18i32_to_v9i64_scalar(<18 x i32> inreg %a, i32
; GFX11-NEXT: s_and_b32 s4, s4, exec_lo
; GFX11-NEXT: s_cselect_b32 s4, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s4, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB5_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: s_add_i32 s29, s29, 3
@@ -1479,33 +1485,29 @@ define <18 x i32> @bitcast_v9i64_to_v18i32(<9 x i64> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v18
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB6_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_add_co_u32 v16, vcc_lo, v16, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v17, null, 0, v17, vcc_lo
; GFX11-NEXT: v_add_co_u32 v14, vcc_lo, v14, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v15, null, 0, v15, vcc_lo
; GFX11-NEXT: v_add_co_u32 v12, vcc_lo, v12, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v13, null, 0, v13, vcc_lo
; GFX11-NEXT: v_add_co_u32 v10, vcc_lo, v10, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v11, null, 0, v11, vcc_lo
; GFX11-NEXT: v_add_co_u32 v8, vcc_lo, v8, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v9, null, 0, v9, vcc_lo
; GFX11-NEXT: v_add_co_u32 v6, vcc_lo, v6, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v7, null, 0, v7, vcc_lo
; GFX11-NEXT: v_add_co_u32 v4, vcc_lo, v4, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v5, null, 0, v5, vcc_lo
; GFX11-NEXT: v_add_co_u32 v2, vcc_lo, v2, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX11-NEXT: v_add_co_u32 v0, vcc_lo, v0, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-NEXT: .LBB6_2: ; %end
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -1722,6 +1724,7 @@ define inreg <18 x i32> @bitcast_v9i64_to_v18i32_scalar(<9 x i64> inreg %a, i32
; GFX11-NEXT: s_and_b32 s4, s4, exec_lo
; GFX11-NEXT: s_cselect_b32 s4, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s4, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB7_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: s_add_u32 s28, s28, 3
@@ -1870,8 +1873,9 @@ define <9 x double> @bitcast_v18i32_to_v9f64(<18 x i32> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v18
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB8_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -2108,6 +2112,7 @@ define inreg <9 x double> @bitcast_v18i32_to_v9f64_scalar(<18 x i32> inreg %a, i
; GFX11-NEXT: s_and_b32 s4, s4, exec_lo
; GFX11-NEXT: s_cselect_b32 s4, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s4, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB9_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: s_add_i32 s29, s29, 3
@@ -2229,8 +2234,9 @@ define <18 x i32> @bitcast_v9f64_to_v18i32(<9 x double> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v18
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB10_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -2634,6 +2640,7 @@ define inreg <18 x i32> @bitcast_v9f64_to_v18i32_scalar(<9 x double> inreg %a, i
; GFX11-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB11_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_add_f64 v[16:17], s[52:53], 1.0
@@ -3157,8 +3164,9 @@ define <36 x i16> @bitcast_v18i32_to_v36i16(<18 x i32> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v35, 16, v0
; GFX11-TRUE16-NEXT: .LBB12_4: ; %end
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v0.h, v35.l
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v1.h, v34.l
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v2.h, v33.l
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v3.h, v32.l
@@ -3265,8 +3273,9 @@ define <36 x i16> @bitcast_v18i32_to_v36i16(<18 x i32> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v35, 16, v0
; GFX11-FAKE16-NEXT: .LBB12_4: ; %end
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_perm_b32 v0, v35, v0, 0x5040100
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_perm_b32 v1, v34, v1, 0x5040100
; GFX11-FAKE16-NEXT: v_perm_b32 v2, v33, v2, 0x5040100
; GFX11-FAKE16-NEXT: v_perm_b32 v3, v32, v3, 0x5040100
@@ -3777,6 +3786,7 @@ define inreg <36 x i16> @bitcast_v18i32_to_v36i16_scalar(<18 x i32> inreg %a, i3
; GFX11-NEXT: v_readfirstlane_b32 s4, v0
; GFX11-NEXT: s_mov_b32 s46, 0
; GFX11-NEXT: s_cmp_lg_u32 s4, 0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc0 .LBB13_2
; GFX11-NEXT: ; %bb.1: ; %cmp.false
; GFX11-NEXT: s_lshr_b32 s4, s29, 16
@@ -3823,6 +3833,7 @@ define inreg <36 x i16> @bitcast_v18i32_to_v36i16_scalar(<18 x i32> inreg %a, i3
; GFX11-NEXT: s_and_b32 s46, s46, exec_lo
; GFX11-NEXT: s_cselect_b32 s46, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s46, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB13_5
; GFX11-NEXT: ; %bb.4: ; %cmp.true
; GFX11-NEXT: s_add_i32 s29, s29, 3
@@ -4556,6 +4567,7 @@ define <18 x i32> @bitcast_v36i16_to_v18i32(<36 x i16> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v36, 16, v0
; GFX11-TRUE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-TRUE16-NEXT: v_cmpx_ne_u32_e32 0, v18
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB14_3
; GFX11-TRUE16-NEXT: ; %bb.1: ; %Flow
@@ -4684,8 +4696,9 @@ define <18 x i32> @bitcast_v36i16_to_v18i32(<36 x i16> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: v_perm_b32 v17, v19, v17, 0x5040100
; GFX11-FAKE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-FAKE16-NEXT: v_cmpx_ne_u32_e32 0, v18
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-FAKE16-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-FAKE16-NEXT: s_cbranch_execz .LBB14_2
; GFX11-FAKE16-NEXT: ; %bb.1: ; %cmp.true
@@ -5406,6 +5419,7 @@ define inreg <18 x i32> @bitcast_v36i16_to_v18i32_scalar(<36 x i16> inreg %a, i3
; GFX11-NEXT: s_and_b32 s40, s40, exec_lo
; GFX11-NEXT: s_cselect_b32 s40, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s40, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB15_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_pk_add_u16 v0, s0, 3 op_sel_hi:[1,0]
@@ -5923,8 +5937,9 @@ define <36 x half> @bitcast_v18i32_to_v36f16(<18 x i32> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v35, 16, v0
; GFX11-TRUE16-NEXT: .LBB16_4: ; %end
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v0.h, v35.l
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v1.h, v34.l
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v2.h, v33.l
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v3.h, v32.l
@@ -6031,8 +6046,9 @@ define <36 x half> @bitcast_v18i32_to_v36f16(<18 x i32> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v35, 16, v0
; GFX11-FAKE16-NEXT: .LBB16_4: ; %end
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_perm_b32 v0, v35, v0, 0x5040100
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_perm_b32 v1, v34, v1, 0x5040100
; GFX11-FAKE16-NEXT: v_perm_b32 v2, v33, v2, 0x5040100
; GFX11-FAKE16-NEXT: v_perm_b32 v3, v32, v3, 0x5040100
@@ -6543,6 +6559,7 @@ define inreg <36 x half> @bitcast_v18i32_to_v36f16_scalar(<18 x i32> inreg %a, i
; GFX11-NEXT: v_readfirstlane_b32 s4, v0
; GFX11-NEXT: s_mov_b32 s46, 0
; GFX11-NEXT: s_cmp_lg_u32 s4, 0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc0 .LBB17_2
; GFX11-NEXT: ; %bb.1: ; %cmp.false
; GFX11-NEXT: s_lshr_b32 s4, s29, 16
@@ -6589,6 +6606,7 @@ define inreg <36 x half> @bitcast_v18i32_to_v36f16_scalar(<18 x i32> inreg %a, i
; GFX11-NEXT: s_and_b32 s46, s46, exec_lo
; GFX11-NEXT: s_cselect_b32 s46, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s46, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB17_5
; GFX11-NEXT: ; %bb.4: ; %cmp.true
; GFX11-NEXT: s_add_i32 s29, s29, 3
@@ -7396,6 +7414,7 @@ define <18 x i32> @bitcast_v36f16_to_v18i32(<36 x half> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v36, 16, v0
; GFX11-TRUE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-TRUE16-NEXT: v_cmpx_ne_u32_e32 0, v18
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB18_3
; GFX11-TRUE16-NEXT: ; %bb.1: ; %Flow
@@ -7524,8 +7543,9 @@ define <18 x i32> @bitcast_v36f16_to_v18i32(<36 x half> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: v_perm_b32 v17, v19, v17, 0x5040100
; GFX11-FAKE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-FAKE16-NEXT: v_cmpx_ne_u32_e32 0, v18
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-FAKE16-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-FAKE16-NEXT: s_cbranch_execz .LBB18_2
; GFX11-FAKE16-NEXT: ; %bb.1: ; %cmp.true
@@ -8316,6 +8336,7 @@ define inreg <18 x i32> @bitcast_v36f16_to_v18i32_scalar(<36 x half> inreg %a, i
; GFX11-NEXT: s_and_b32 s40, s40, exec_lo
; GFX11-NEXT: s_cselect_b32 s40, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s40, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB19_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_pk_add_f16 v0, 0x200, s0 op_sel_hi:[0,1]
@@ -8471,8 +8492,9 @@ define <9 x i64> @bitcast_v18f32_to_v9i64(<18 x float> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v18
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB20_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -8903,6 +8925,7 @@ define inreg <9 x i64> @bitcast_v18f32_to_v9i64_scalar(<18 x float> inreg %a, i3
; GFX11-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB21_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_add_f32_e64 v17, s53, 1.0
@@ -9073,33 +9096,29 @@ define <18 x float> @bitcast_v9i64_to_v18f32(<9 x i64> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v18
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB22_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_add_co_u32 v16, vcc_lo, v16, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v17, null, 0, v17, vcc_lo
; GFX11-NEXT: v_add_co_u32 v14, vcc_lo, v14, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v15, null, 0, v15, vcc_lo
; GFX11-NEXT: v_add_co_u32 v12, vcc_lo, v12, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v13, null, 0, v13, vcc_lo
; GFX11-NEXT: v_add_co_u32 v10, vcc_lo, v10, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v11, null, 0, v11, vcc_lo
; GFX11-NEXT: v_add_co_u32 v8, vcc_lo, v8, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v9, null, 0, v9, vcc_lo
; GFX11-NEXT: v_add_co_u32 v6, vcc_lo, v6, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v7, null, 0, v7, vcc_lo
; GFX11-NEXT: v_add_co_u32 v4, vcc_lo, v4, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v5, null, 0, v5, vcc_lo
; GFX11-NEXT: v_add_co_u32 v2, vcc_lo, v2, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX11-NEXT: v_add_co_u32 v0, vcc_lo, v0, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-NEXT: .LBB22_2: ; %end
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -9316,6 +9335,7 @@ define inreg <18 x float> @bitcast_v9i64_to_v18f32_scalar(<9 x i64> inreg %a, i3
; GFX11-NEXT: s_and_b32 s4, s4, exec_lo
; GFX11-NEXT: s_cselect_b32 s4, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s4, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB23_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: s_add_u32 s28, s28, 3
@@ -9464,8 +9484,9 @@ define <9 x double> @bitcast_v18f32_to_v9f64(<18 x float> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v18
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB24_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -9896,6 +9917,7 @@ define inreg <9 x double> @bitcast_v18f32_to_v9f64_scalar(<18 x float> inreg %a,
; GFX11-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB25_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_add_f32_e64 v17, s53, 1.0
@@ -10039,8 +10061,9 @@ define <18 x float> @bitcast_v9f64_to_v18f32(<9 x double> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v18
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB26_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -10444,6 +10467,7 @@ define inreg <18 x float> @bitcast_v9f64_to_v18f32_scalar(<9 x double> inreg %a,
; GFX11-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB27_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_add_f64 v[16:17], s[52:53], 1.0
@@ -10958,8 +10982,9 @@ define <36 x i16> @bitcast_v18f32_to_v36i16(<18 x float> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v35, 16, v0
; GFX11-TRUE16-NEXT: .LBB28_4: ; %end
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v0.h, v35.l
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v1.h, v34.l
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v2.h, v33.l
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v3.h, v32.l
@@ -11057,8 +11082,9 @@ define <36 x i16> @bitcast_v18f32_to_v36i16(<18 x float> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v35, 16, v0
; GFX11-FAKE16-NEXT: .LBB28_4: ; %end
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_perm_b32 v0, v35, v0, 0x5040100
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_perm_b32 v1, v34, v1, 0x5040100
; GFX11-FAKE16-NEXT: v_perm_b32 v2, v33, v2, 0x5040100
; GFX11-FAKE16-NEXT: v_perm_b32 v3, v32, v3, 0x5040100
@@ -11629,6 +11655,7 @@ define inreg <36 x i16> @bitcast_v18f32_to_v36i16_scalar(<18 x float> inreg %a,
; GFX11-TRUE16-NEXT: v_readfirstlane_b32 s4, v0
; GFX11-TRUE16-NEXT: s_mov_b32 s6, 0
; GFX11-TRUE16-NEXT: s_cmp_lg_u32 s4, 0
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_cbranch_scc0 .LBB29_2
; GFX11-TRUE16-NEXT: ; %bb.1: ; %cmp.false
; GFX11-TRUE16-NEXT: s_lshr_b32 s4, s29, 16
@@ -11675,6 +11702,7 @@ define inreg <36 x i16> @bitcast_v18f32_to_v36i16_scalar(<18 x float> inreg %a,
; GFX11-TRUE16-NEXT: s_and_b32 s6, s6, exec_lo
; GFX11-TRUE16-NEXT: s_cselect_b32 s6, 1, 0
; GFX11-TRUE16-NEXT: s_cmp_lg_u32 s6, 1
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_cbranch_scc1 .LBB29_5
; GFX11-TRUE16-NEXT: ; %bb.4: ; %cmp.true
; GFX11-TRUE16-NEXT: v_add_f32_e64 v17, s29, 1.0
@@ -11761,6 +11789,7 @@ define inreg <36 x i16> @bitcast_v18f32_to_v36i16_scalar(<18 x float> inreg %a,
; GFX11-FAKE16-NEXT: v_readfirstlane_b32 s4, v0
; GFX11-FAKE16-NEXT: s_mov_b32 s6, 0
; GFX11-FAKE16-NEXT: s_cmp_lg_u32 s4, 0
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_cbranch_scc0 .LBB29_2
; GFX11-FAKE16-NEXT: ; %bb.1: ; %cmp.false
; GFX11-FAKE16-NEXT: s_lshr_b32 s4, s29, 16
@@ -11807,6 +11836,7 @@ define inreg <36 x i16> @bitcast_v18f32_to_v36i16_scalar(<18 x float> inreg %a,
; GFX11-FAKE16-NEXT: s_and_b32 s6, s6, exec_lo
; GFX11-FAKE16-NEXT: s_cselect_b32 s6, 1, 0
; GFX11-FAKE16-NEXT: s_cmp_lg_u32 s6, 1
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_cbranch_scc1 .LBB29_5
; GFX11-FAKE16-NEXT: ; %bb.4: ; %cmp.true
; GFX11-FAKE16-NEXT: v_add_f32_e64 v13, s29, 1.0
@@ -12568,6 +12598,7 @@ define <18 x float> @bitcast_v36i16_to_v18f32(<36 x i16> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v36, 16, v0
; GFX11-TRUE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-TRUE16-NEXT: v_cmpx_ne_u32_e32 0, v18
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB30_3
; GFX11-TRUE16-NEXT: ; %bb.1: ; %Flow
@@ -12696,8 +12727,9 @@ define <18 x float> @bitcast_v36i16_to_v18f32(<36 x i16> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: v_perm_b32 v17, v19, v17, 0x5040100
; GFX11-FAKE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-FAKE16-NEXT: v_cmpx_ne_u32_e32 0, v18
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-FAKE16-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-FAKE16-NEXT: s_cbranch_execz .LBB30_2
; GFX11-FAKE16-NEXT: ; %bb.1: ; %cmp.true
@@ -13418,6 +13450,7 @@ define inreg <18 x float> @bitcast_v36i16_to_v18f32_scalar(<36 x i16> inreg %a,
; GFX11-NEXT: s_and_b32 s40, s40, exec_lo
; GFX11-NEXT: s_cselect_b32 s40, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s40, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB31_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_pk_add_u16 v0, s0, 3 op_sel_hi:[1,0]
@@ -13926,8 +13959,9 @@ define <36 x half> @bitcast_v18f32_to_v36f16(<18 x float> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v35, 16, v0
; GFX11-TRUE16-NEXT: .LBB32_4: ; %end
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v0.h, v35.l
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v1.h, v34.l
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v2.h, v33.l
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v3.h, v32.l
@@ -14025,8 +14059,9 @@ define <36 x half> @bitcast_v18f32_to_v36f16(<18 x float> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v35, 16, v0
; GFX11-FAKE16-NEXT: .LBB32_4: ; %end
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_perm_b32 v0, v35, v0, 0x5040100
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_perm_b32 v1, v34, v1, 0x5040100
; GFX11-FAKE16-NEXT: v_perm_b32 v2, v33, v2, 0x5040100
; GFX11-FAKE16-NEXT: v_perm_b32 v3, v32, v3, 0x5040100
@@ -14597,6 +14632,7 @@ define inreg <36 x half> @bitcast_v18f32_to_v36f16_scalar(<18 x float> inreg %a,
; GFX11-TRUE16-NEXT: v_readfirstlane_b32 s4, v0
; GFX11-TRUE16-NEXT: s_mov_b32 s6, 0
; GFX11-TRUE16-NEXT: s_cmp_lg_u32 s4, 0
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_cbranch_scc0 .LBB33_2
; GFX11-TRUE16-NEXT: ; %bb.1: ; %cmp.false
; GFX11-TRUE16-NEXT: s_lshr_b32 s4, s29, 16
@@ -14643,6 +14679,7 @@ define inreg <36 x half> @bitcast_v18f32_to_v36f16_scalar(<18 x float> inreg %a,
; GFX11-TRUE16-NEXT: s_and_b32 s6, s6, exec_lo
; GFX11-TRUE16-NEXT: s_cselect_b32 s6, 1, 0
; GFX11-TRUE16-NEXT: s_cmp_lg_u32 s6, 1
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_cbranch_scc1 .LBB33_5
; GFX11-TRUE16-NEXT: ; %bb.4: ; %cmp.true
; GFX11-TRUE16-NEXT: v_add_f32_e64 v17, s29, 1.0
@@ -14729,6 +14766,7 @@ define inreg <36 x half> @bitcast_v18f32_to_v36f16_scalar(<18 x float> inreg %a,
; GFX11-FAKE16-NEXT: v_readfirstlane_b32 s4, v0
; GFX11-FAKE16-NEXT: s_mov_b32 s6, 0
; GFX11-FAKE16-NEXT: s_cmp_lg_u32 s4, 0
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_cbranch_scc0 .LBB33_2
; GFX11-FAKE16-NEXT: ; %bb.1: ; %cmp.false
; GFX11-FAKE16-NEXT: s_lshr_b32 s4, s29, 16
@@ -14775,6 +14813,7 @@ define inreg <36 x half> @bitcast_v18f32_to_v36f16_scalar(<18 x float> inreg %a,
; GFX11-FAKE16-NEXT: s_and_b32 s6, s6, exec_lo
; GFX11-FAKE16-NEXT: s_cselect_b32 s6, 1, 0
; GFX11-FAKE16-NEXT: s_cmp_lg_u32 s6, 1
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_cbranch_scc1 .LBB33_5
; GFX11-FAKE16-NEXT: ; %bb.4: ; %cmp.true
; GFX11-FAKE16-NEXT: v_add_f32_e64 v13, s29, 1.0
@@ -15610,6 +15649,7 @@ define <18 x float> @bitcast_v36f16_to_v18f32(<36 x half> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v36, 16, v0
; GFX11-TRUE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-TRUE16-NEXT: v_cmpx_ne_u32_e32 0, v18
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB34_3
; GFX11-TRUE16-NEXT: ; %bb.1: ; %Flow
@@ -15738,8 +15778,9 @@ define <18 x float> @bitcast_v36f16_to_v18f32(<36 x half> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: v_perm_b32 v17, v19, v17, 0x5040100
; GFX11-FAKE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-FAKE16-NEXT: v_cmpx_ne_u32_e32 0, v18
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-FAKE16-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-FAKE16-NEXT: s_cbranch_execz .LBB34_2
; GFX11-FAKE16-NEXT: ; %bb.1: ; %cmp.true
@@ -16530,6 +16571,7 @@ define inreg <18 x float> @bitcast_v36f16_to_v18f32_scalar(<36 x half> inreg %a,
; GFX11-NEXT: s_and_b32 s40, s40, exec_lo
; GFX11-NEXT: s_cselect_b32 s40, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s40, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB35_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_pk_add_f16 v0, 0x200, s0 op_sel_hi:[0,1]
@@ -16685,33 +16727,29 @@ define <9 x double> @bitcast_v9i64_to_v9f64(<9 x i64> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v18
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB36_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_add_co_u32 v0, vcc_lo, v0, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-NEXT: v_add_co_u32 v2, vcc_lo, v2, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX11-NEXT: v_add_co_u32 v4, vcc_lo, v4, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v5, null, 0, v5, vcc_lo
; GFX11-NEXT: v_add_co_u32 v6, vcc_lo, v6, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v7, null, 0, v7, vcc_lo
; GFX11-NEXT: v_add_co_u32 v8, vcc_lo, v8, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v9, null, 0, v9, vcc_lo
; GFX11-NEXT: v_add_co_u32 v10, vcc_lo, v10, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v11, null, 0, v11, vcc_lo
; GFX11-NEXT: v_add_co_u32 v12, vcc_lo, v12, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v13, null, 0, v13, vcc_lo
; GFX11-NEXT: v_add_co_u32 v14, vcc_lo, v14, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v15, null, 0, v15, vcc_lo
; GFX11-NEXT: v_add_co_u32 v16, vcc_lo, v16, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v17, null, 0, v17, vcc_lo
; GFX11-NEXT: .LBB36_2: ; %end
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -16928,6 +16966,7 @@ define inreg <9 x double> @bitcast_v9i64_to_v9f64_scalar(<9 x i64> inreg %a, i32
; GFX11-NEXT: s_and_b32 s4, s4, exec_lo
; GFX11-NEXT: s_cselect_b32 s4, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s4, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB37_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: s_add_u32 s0, s0, 3
@@ -17048,8 +17087,9 @@ define <9 x i64> @bitcast_v9f64_to_v9i64(<9 x double> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v18
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB38_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -17453,6 +17493,7 @@ define inreg <9 x i64> @bitcast_v9f64_to_v9i64_scalar(<9 x double> inreg %a, i32
; GFX11-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB39_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_add_f64 v[0:1], s[36:37], 1.0
@@ -17939,27 +17980,22 @@ define <36 x i16> @bitcast_v9i64_to_v36i16(<9 x i64> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: s_cbranch_execz .LBB40_4
; GFX11-TRUE16-NEXT: ; %bb.3: ; %cmp.true
; GFX11-TRUE16-NEXT: v_add_co_u32 v16, vcc_lo, v16, 3
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v17, null, 0, v17, vcc_lo
; GFX11-TRUE16-NEXT: v_add_co_u32 v14, vcc_lo, v14, 3
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v15, null, 0, v15, vcc_lo
; GFX11-TRUE16-NEXT: v_add_co_u32 v12, vcc_lo, v12, 3
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v13, null, 0, v13, vcc_lo
; GFX11-TRUE16-NEXT: v_add_co_u32 v10, vcc_lo, v10, 3
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v11, null, 0, v11, vcc_lo
; GFX11-TRUE16-NEXT: v_add_co_u32 v8, vcc_lo, v8, 3
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v9, null, 0, v9, vcc_lo
; GFX11-TRUE16-NEXT: v_add_co_u32 v6, vcc_lo, v6, 3
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v7, null, 0, v7, vcc_lo
; GFX11-TRUE16-NEXT: v_add_co_u32 v4, vcc_lo, v4, 3
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v5, null, 0, v5, vcc_lo
; GFX11-TRUE16-NEXT: v_add_co_u32 v2, vcc_lo, v2, 3
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX11-TRUE16-NEXT: v_add_co_u32 v0, vcc_lo, v0, 3
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v18, 16, v17
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v19, 16, v16
@@ -17981,8 +18017,9 @@ define <36 x i16> @bitcast_v9i64_to_v36i16(<9 x i64> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v35, 16, v0
; GFX11-TRUE16-NEXT: .LBB40_4: ; %end
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v0.h, v35.l
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v1.h, v34.l
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v2.h, v33.l
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v3.h, v32.l
@@ -18052,27 +18089,22 @@ define <36 x i16> @bitcast_v9i64_to_v36i16(<9 x i64> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: s_cbranch_execz .LBB40_4
; GFX11-FAKE16-NEXT: ; %bb.3: ; %cmp.true
; GFX11-FAKE16-NEXT: v_add_co_u32 v16, vcc_lo, v16, 3
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v17, null, 0, v17, vcc_lo
; GFX11-FAKE16-NEXT: v_add_co_u32 v14, vcc_lo, v14, 3
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v15, null, 0, v15, vcc_lo
; GFX11-FAKE16-NEXT: v_add_co_u32 v12, vcc_lo, v12, 3
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v13, null, 0, v13, vcc_lo
; GFX11-FAKE16-NEXT: v_add_co_u32 v10, vcc_lo, v10, 3
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v11, null, 0, v11, vcc_lo
; GFX11-FAKE16-NEXT: v_add_co_u32 v8, vcc_lo, v8, 3
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v9, null, 0, v9, vcc_lo
; GFX11-FAKE16-NEXT: v_add_co_u32 v6, vcc_lo, v6, 3
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v7, null, 0, v7, vcc_lo
; GFX11-FAKE16-NEXT: v_add_co_u32 v4, vcc_lo, v4, 3
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v5, null, 0, v5, vcc_lo
; GFX11-FAKE16-NEXT: v_add_co_u32 v2, vcc_lo, v2, 3
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX11-FAKE16-NEXT: v_add_co_u32 v0, vcc_lo, v0, 3
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v18, 16, v17
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v19, 16, v16
@@ -18094,8 +18126,9 @@ define <36 x i16> @bitcast_v9i64_to_v36i16(<9 x i64> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v35, 16, v0
; GFX11-FAKE16-NEXT: .LBB40_4: ; %end
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_perm_b32 v0, v35, v0, 0x5040100
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_perm_b32 v1, v34, v1, 0x5040100
; GFX11-FAKE16-NEXT: v_perm_b32 v2, v33, v2, 0x5040100
; GFX11-FAKE16-NEXT: v_perm_b32 v3, v32, v3, 0x5040100
@@ -18606,6 +18639,7 @@ define inreg <36 x i16> @bitcast_v9i64_to_v36i16_scalar(<9 x i64> inreg %a, i32
; GFX11-NEXT: v_readfirstlane_b32 s4, v0
; GFX11-NEXT: s_mov_b32 s46, 0
; GFX11-NEXT: s_cmp_lg_u32 s4, 0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc0 .LBB41_2
; GFX11-NEXT: ; %bb.1: ; %cmp.false
; GFX11-NEXT: s_lshr_b32 s4, s29, 16
@@ -18652,6 +18686,7 @@ define inreg <36 x i16> @bitcast_v9i64_to_v36i16_scalar(<9 x i64> inreg %a, i32
; GFX11-NEXT: s_and_b32 s46, s46, exec_lo
; GFX11-NEXT: s_cselect_b32 s46, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s46, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB41_5
; GFX11-NEXT: ; %bb.4: ; %cmp.true
; GFX11-NEXT: s_add_u32 s28, s28, 3
@@ -19385,6 +19420,7 @@ define <9 x i64> @bitcast_v36i16_to_v9i64(<36 x i16> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v36, 16, v0
; GFX11-TRUE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-TRUE16-NEXT: v_cmpx_ne_u32_e32 0, v18
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB42_3
; GFX11-TRUE16-NEXT: ; %bb.1: ; %Flow
@@ -19513,8 +19549,9 @@ define <9 x i64> @bitcast_v36i16_to_v9i64(<36 x i16> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: v_perm_b32 v17, v19, v17, 0x5040100
; GFX11-FAKE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-FAKE16-NEXT: v_cmpx_ne_u32_e32 0, v18
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-FAKE16-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-FAKE16-NEXT: s_cbranch_execz .LBB42_2
; GFX11-FAKE16-NEXT: ; %bb.1: ; %cmp.true
@@ -20235,6 +20272,7 @@ define inreg <9 x i64> @bitcast_v36i16_to_v9i64_scalar(<36 x i16> inreg %a, i32
; GFX11-NEXT: s_and_b32 s40, s40, exec_lo
; GFX11-NEXT: s_cselect_b32 s40, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s40, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB43_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_pk_add_u16 v0, s0, 3 op_sel_hi:[1,0]
@@ -20715,27 +20753,22 @@ define <36 x half> @bitcast_v9i64_to_v36f16(<9 x i64> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: s_cbranch_execz .LBB44_4
; GFX11-TRUE16-NEXT: ; %bb.3: ; %cmp.true
; GFX11-TRUE16-NEXT: v_add_co_u32 v16, vcc_lo, v16, 3
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v17, null, 0, v17, vcc_lo
; GFX11-TRUE16-NEXT: v_add_co_u32 v14, vcc_lo, v14, 3
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v15, null, 0, v15, vcc_lo
; GFX11-TRUE16-NEXT: v_add_co_u32 v12, vcc_lo, v12, 3
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v13, null, 0, v13, vcc_lo
; GFX11-TRUE16-NEXT: v_add_co_u32 v10, vcc_lo, v10, 3
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v11, null, 0, v11, vcc_lo
; GFX11-TRUE16-NEXT: v_add_co_u32 v8, vcc_lo, v8, 3
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v9, null, 0, v9, vcc_lo
; GFX11-TRUE16-NEXT: v_add_co_u32 v6, vcc_lo, v6, 3
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v7, null, 0, v7, vcc_lo
; GFX11-TRUE16-NEXT: v_add_co_u32 v4, vcc_lo, v4, 3
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v5, null, 0, v5, vcc_lo
; GFX11-TRUE16-NEXT: v_add_co_u32 v2, vcc_lo, v2, 3
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX11-TRUE16-NEXT: v_add_co_u32 v0, vcc_lo, v0, 3
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v18, 16, v17
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v19, 16, v16
@@ -20757,8 +20790,9 @@ define <36 x half> @bitcast_v9i64_to_v36f16(<9 x i64> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v35, 16, v0
; GFX11-TRUE16-NEXT: .LBB44_4: ; %end
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v0.h, v35.l
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v1.h, v34.l
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v2.h, v33.l
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v3.h, v32.l
@@ -20828,27 +20862,22 @@ define <36 x half> @bitcast_v9i64_to_v36f16(<9 x i64> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: s_cbranch_execz .LBB44_4
; GFX11-FAKE16-NEXT: ; %bb.3: ; %cmp.true
; GFX11-FAKE16-NEXT: v_add_co_u32 v16, vcc_lo, v16, 3
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v17, null, 0, v17, vcc_lo
; GFX11-FAKE16-NEXT: v_add_co_u32 v14, vcc_lo, v14, 3
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v15, null, 0, v15, vcc_lo
; GFX11-FAKE16-NEXT: v_add_co_u32 v12, vcc_lo, v12, 3
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v13, null, 0, v13, vcc_lo
; GFX11-FAKE16-NEXT: v_add_co_u32 v10, vcc_lo, v10, 3
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v11, null, 0, v11, vcc_lo
; GFX11-FAKE16-NEXT: v_add_co_u32 v8, vcc_lo, v8, 3
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v9, null, 0, v9, vcc_lo
; GFX11-FAKE16-NEXT: v_add_co_u32 v6, vcc_lo, v6, 3
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v7, null, 0, v7, vcc_lo
; GFX11-FAKE16-NEXT: v_add_co_u32 v4, vcc_lo, v4, 3
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v5, null, 0, v5, vcc_lo
; GFX11-FAKE16-NEXT: v_add_co_u32 v2, vcc_lo, v2, 3
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX11-FAKE16-NEXT: v_add_co_u32 v0, vcc_lo, v0, 3
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v18, 16, v17
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v19, 16, v16
@@ -20870,8 +20899,9 @@ define <36 x half> @bitcast_v9i64_to_v36f16(<9 x i64> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v35, 16, v0
; GFX11-FAKE16-NEXT: .LBB44_4: ; %end
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_perm_b32 v0, v35, v0, 0x5040100
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_perm_b32 v1, v34, v1, 0x5040100
; GFX11-FAKE16-NEXT: v_perm_b32 v2, v33, v2, 0x5040100
; GFX11-FAKE16-NEXT: v_perm_b32 v3, v32, v3, 0x5040100
@@ -21382,6 +21412,7 @@ define inreg <36 x half> @bitcast_v9i64_to_v36f16_scalar(<9 x i64> inreg %a, i32
; GFX11-NEXT: v_readfirstlane_b32 s4, v0
; GFX11-NEXT: s_mov_b32 s46, 0
; GFX11-NEXT: s_cmp_lg_u32 s4, 0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc0 .LBB45_2
; GFX11-NEXT: ; %bb.1: ; %cmp.false
; GFX11-NEXT: s_lshr_b32 s4, s29, 16
@@ -21428,6 +21459,7 @@ define inreg <36 x half> @bitcast_v9i64_to_v36f16_scalar(<9 x i64> inreg %a, i32
; GFX11-NEXT: s_and_b32 s46, s46, exec_lo
; GFX11-NEXT: s_cselect_b32 s46, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s46, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB45_5
; GFX11-NEXT: ; %bb.4: ; %cmp.true
; GFX11-NEXT: s_add_u32 s28, s28, 3
@@ -22235,6 +22267,7 @@ define <9 x i64> @bitcast_v36f16_to_v9i64(<36 x half> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v36, 16, v0
; GFX11-TRUE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-TRUE16-NEXT: v_cmpx_ne_u32_e32 0, v18
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB46_3
; GFX11-TRUE16-NEXT: ; %bb.1: ; %Flow
@@ -22363,8 +22396,9 @@ define <9 x i64> @bitcast_v36f16_to_v9i64(<36 x half> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: v_perm_b32 v17, v19, v17, 0x5040100
; GFX11-FAKE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-FAKE16-NEXT: v_cmpx_ne_u32_e32 0, v18
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-FAKE16-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-FAKE16-NEXT: s_cbranch_execz .LBB46_2
; GFX11-FAKE16-NEXT: ; %bb.1: ; %cmp.true
@@ -23155,6 +23189,7 @@ define inreg <9 x i64> @bitcast_v36f16_to_v9i64_scalar(<36 x half> inreg %a, i32
; GFX11-NEXT: s_and_b32 s40, s40, exec_lo
; GFX11-NEXT: s_cselect_b32 s40, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s40, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB47_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_pk_add_f16 v0, 0x200, s0 op_sel_hi:[0,1]
@@ -23636,8 +23671,9 @@ define <36 x i16> @bitcast_v9f64_to_v36i16(<9 x double> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v35, 16, v0
; GFX11-TRUE16-NEXT: .LBB48_4: ; %end
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v0.h, v35.l
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v1.h, v34.l
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v2.h, v33.l
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v3.h, v32.l
@@ -23735,8 +23771,9 @@ define <36 x i16> @bitcast_v9f64_to_v36i16(<9 x double> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v35, 16, v0
; GFX11-FAKE16-NEXT: .LBB48_4: ; %end
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_perm_b32 v0, v35, v0, 0x5040100
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_perm_b32 v1, v34, v1, 0x5040100
; GFX11-FAKE16-NEXT: v_perm_b32 v2, v33, v2, 0x5040100
; GFX11-FAKE16-NEXT: v_perm_b32 v3, v32, v3, 0x5040100
@@ -24280,6 +24317,7 @@ define inreg <36 x i16> @bitcast_v9f64_to_v36i16_scalar(<9 x double> inreg %a, i
; GFX11-TRUE16-NEXT: v_readfirstlane_b32 s4, v0
; GFX11-TRUE16-NEXT: s_mov_b32 s6, 0
; GFX11-TRUE16-NEXT: s_cmp_lg_u32 s4, 0
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_cbranch_scc0 .LBB49_2
; GFX11-TRUE16-NEXT: ; %bb.1: ; %cmp.false
; GFX11-TRUE16-NEXT: s_lshr_b32 s4, s29, 16
@@ -24326,6 +24364,7 @@ define inreg <36 x i16> @bitcast_v9f64_to_v36i16_scalar(<9 x double> inreg %a, i
; GFX11-TRUE16-NEXT: s_and_b32 s6, s6, exec_lo
; GFX11-TRUE16-NEXT: s_cselect_b32 s6, 1, 0
; GFX11-TRUE16-NEXT: s_cmp_lg_u32 s6, 1
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_cbranch_scc1 .LBB49_5
; GFX11-TRUE16-NEXT: ; %bb.4: ; %cmp.true
; GFX11-TRUE16-NEXT: v_add_f64 v[16:17], s[28:29], 1.0
@@ -24403,6 +24442,7 @@ define inreg <36 x i16> @bitcast_v9f64_to_v36i16_scalar(<9 x double> inreg %a, i
; GFX11-FAKE16-NEXT: v_readfirstlane_b32 s4, v0
; GFX11-FAKE16-NEXT: s_mov_b32 s6, 0
; GFX11-FAKE16-NEXT: s_cmp_lg_u32 s4, 0
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_cbranch_scc0 .LBB49_2
; GFX11-FAKE16-NEXT: ; %bb.1: ; %cmp.false
; GFX11-FAKE16-NEXT: s_lshr_b32 s4, s29, 16
@@ -24449,6 +24489,7 @@ define inreg <36 x i16> @bitcast_v9f64_to_v36i16_scalar(<9 x double> inreg %a, i
; GFX11-FAKE16-NEXT: s_and_b32 s6, s6, exec_lo
; GFX11-FAKE16-NEXT: s_cselect_b32 s6, 1, 0
; GFX11-FAKE16-NEXT: s_cmp_lg_u32 s6, 1
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_cbranch_scc1 .LBB49_5
; GFX11-FAKE16-NEXT: ; %bb.4: ; %cmp.true
; GFX11-FAKE16-NEXT: v_add_f64 v[13:14], s[28:29], 1.0
@@ -25201,6 +25242,7 @@ define <9 x double> @bitcast_v36i16_to_v9f64(<36 x i16> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v36, 16, v0
; GFX11-TRUE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-TRUE16-NEXT: v_cmpx_ne_u32_e32 0, v18
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB50_3
; GFX11-TRUE16-NEXT: ; %bb.1: ; %Flow
@@ -25329,8 +25371,9 @@ define <9 x double> @bitcast_v36i16_to_v9f64(<36 x i16> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: v_perm_b32 v17, v19, v17, 0x5040100
; GFX11-FAKE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-FAKE16-NEXT: v_cmpx_ne_u32_e32 0, v18
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-FAKE16-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-FAKE16-NEXT: s_cbranch_execz .LBB50_2
; GFX11-FAKE16-NEXT: ; %bb.1: ; %cmp.true
@@ -26051,6 +26094,7 @@ define inreg <9 x double> @bitcast_v36i16_to_v9f64_scalar(<36 x i16> inreg %a, i
; GFX11-NEXT: s_and_b32 s40, s40, exec_lo
; GFX11-NEXT: s_cselect_b32 s40, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s40, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB51_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_pk_add_u16 v0, s0, 3 op_sel_hi:[1,0]
@@ -26532,8 +26576,9 @@ define <36 x half> @bitcast_v9f64_to_v36f16(<9 x double> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v35, 16, v0
; GFX11-TRUE16-NEXT: .LBB52_4: ; %end
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v0.h, v35.l
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v1.h, v34.l
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v2.h, v33.l
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v3.h, v32.l
@@ -26631,8 +26676,9 @@ define <36 x half> @bitcast_v9f64_to_v36f16(<9 x double> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v35, 16, v0
; GFX11-FAKE16-NEXT: .LBB52_4: ; %end
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_perm_b32 v0, v35, v0, 0x5040100
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_perm_b32 v1, v34, v1, 0x5040100
; GFX11-FAKE16-NEXT: v_perm_b32 v2, v33, v2, 0x5040100
; GFX11-FAKE16-NEXT: v_perm_b32 v3, v32, v3, 0x5040100
@@ -27176,6 +27222,7 @@ define inreg <36 x half> @bitcast_v9f64_to_v36f16_scalar(<9 x double> inreg %a,
; GFX11-TRUE16-NEXT: v_readfirstlane_b32 s4, v0
; GFX11-TRUE16-NEXT: s_mov_b32 s6, 0
; GFX11-TRUE16-NEXT: s_cmp_lg_u32 s4, 0
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_cbranch_scc0 .LBB53_2
; GFX11-TRUE16-NEXT: ; %bb.1: ; %cmp.false
; GFX11-TRUE16-NEXT: s_lshr_b32 s4, s29, 16
@@ -27222,6 +27269,7 @@ define inreg <36 x half> @bitcast_v9f64_to_v36f16_scalar(<9 x double> inreg %a,
; GFX11-TRUE16-NEXT: s_and_b32 s6, s6, exec_lo
; GFX11-TRUE16-NEXT: s_cselect_b32 s6, 1, 0
; GFX11-TRUE16-NEXT: s_cmp_lg_u32 s6, 1
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_cbranch_scc1 .LBB53_5
; GFX11-TRUE16-NEXT: ; %bb.4: ; %cmp.true
; GFX11-TRUE16-NEXT: v_add_f64 v[16:17], s[28:29], 1.0
@@ -27299,6 +27347,7 @@ define inreg <36 x half> @bitcast_v9f64_to_v36f16_scalar(<9 x double> inreg %a,
; GFX11-FAKE16-NEXT: v_readfirstlane_b32 s4, v0
; GFX11-FAKE16-NEXT: s_mov_b32 s6, 0
; GFX11-FAKE16-NEXT: s_cmp_lg_u32 s4, 0
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_cbranch_scc0 .LBB53_2
; GFX11-FAKE16-NEXT: ; %bb.1: ; %cmp.false
; GFX11-FAKE16-NEXT: s_lshr_b32 s4, s29, 16
@@ -27345,6 +27394,7 @@ define inreg <36 x half> @bitcast_v9f64_to_v36f16_scalar(<9 x double> inreg %a,
; GFX11-FAKE16-NEXT: s_and_b32 s6, s6, exec_lo
; GFX11-FAKE16-NEXT: s_cselect_b32 s6, 1, 0
; GFX11-FAKE16-NEXT: s_cmp_lg_u32 s6, 1
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_cbranch_scc1 .LBB53_5
; GFX11-FAKE16-NEXT: ; %bb.4: ; %cmp.true
; GFX11-FAKE16-NEXT: v_add_f64 v[13:14], s[28:29], 1.0
@@ -28171,6 +28221,7 @@ define <9 x double> @bitcast_v36f16_to_v9f64(<36 x half> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v36, 16, v0
; GFX11-TRUE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-TRUE16-NEXT: v_cmpx_ne_u32_e32 0, v18
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB54_3
; GFX11-TRUE16-NEXT: ; %bb.1: ; %Flow
@@ -28299,8 +28350,9 @@ define <9 x double> @bitcast_v36f16_to_v9f64(<36 x half> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: v_perm_b32 v17, v19, v17, 0x5040100
; GFX11-FAKE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-FAKE16-NEXT: v_cmpx_ne_u32_e32 0, v18
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-FAKE16-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-FAKE16-NEXT: s_cbranch_execz .LBB54_2
; GFX11-FAKE16-NEXT: ; %bb.1: ; %cmp.true
@@ -29091,6 +29143,7 @@ define inreg <9 x double> @bitcast_v36f16_to_v9f64_scalar(<36 x half> inreg %a,
; GFX11-NEXT: s_and_b32 s40, s40, exec_lo
; GFX11-NEXT: s_cselect_b32 s40, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s40, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB55_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_pk_add_f16 v0, 0x200, s0 op_sel_hi:[0,1]
@@ -29772,8 +29825,9 @@ define <36 x half> @bitcast_v36i16_to_v36f16(<36 x i16> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v23, 16, v0
; GFX11-TRUE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-TRUE16-NEXT: v_cmpx_ne_u32_e32 0, v18
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-TRUE16-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-TRUE16-NEXT: s_cbranch_execz .LBB56_2
; GFX11-TRUE16-NEXT: ; %bb.1: ; %cmp.true
@@ -29833,6 +29887,7 @@ define <36 x half> @bitcast_v36i16_to_v36f16(<36 x i16> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v36, 16, v17
; GFX11-TRUE16-NEXT: .LBB56_2: ; %end
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v0.h, v23.l
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v1.h, v22.l
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v2.h, v21.l
@@ -29876,8 +29931,9 @@ define <36 x half> @bitcast_v36i16_to_v36f16(<36 x i16> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v19, 16, v0
; GFX11-FAKE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-FAKE16-NEXT: v_cmpx_ne_u32_e32 0, v18
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-FAKE16-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-FAKE16-NEXT: s_cbranch_execz .LBB56_2
; GFX11-FAKE16-NEXT: ; %bb.1: ; %cmp.true
@@ -29937,6 +29993,7 @@ define <36 x half> @bitcast_v36i16_to_v36f16(<36 x i16> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v36, 16, v17
; GFX11-FAKE16-NEXT: .LBB56_2: ; %end
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_perm_b32 v0, v19, v0, 0x5040100
; GFX11-FAKE16-NEXT: v_perm_b32 v1, v20, v1, 0x5040100
; GFX11-FAKE16-NEXT: v_perm_b32 v2, v21, v2, 0x5040100
@@ -30674,6 +30731,7 @@ define inreg <36 x half> @bitcast_v36i16_to_v36f16_scalar(<36 x i16> inreg %a, i
; GFX11-TRUE16-NEXT: s_and_b32 s46, s46, exec_lo
; GFX11-TRUE16-NEXT: s_cselect_b32 s46, 1, 0
; GFX11-TRUE16-NEXT: s_cmp_lg_u32 s46, 1
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_cbranch_scc1 .LBB57_4
; GFX11-TRUE16-NEXT: ; %bb.3: ; %cmp.true
; GFX11-TRUE16-NEXT: s_pack_ll_b32_b16 s29, s29, s45
@@ -30805,6 +30863,7 @@ define inreg <36 x half> @bitcast_v36i16_to_v36f16_scalar(<36 x i16> inreg %a, i
; GFX11-FAKE16-NEXT: s_and_b32 s46, s46, exec_lo
; GFX11-FAKE16-NEXT: s_cselect_b32 s46, 1, 0
; GFX11-FAKE16-NEXT: s_cmp_lg_u32 s46, 1
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_cbranch_scc1 .LBB57_4
; GFX11-FAKE16-NEXT: ; %bb.3: ; %cmp.true
; GFX11-FAKE16-NEXT: s_pack_ll_b32_b16 s29, s29, s45
@@ -31407,8 +31466,9 @@ define <36 x i16> @bitcast_v36f16_to_v36i16(<36 x half> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v23, 16, v0
; GFX11-TRUE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-TRUE16-NEXT: v_cmpx_ne_u32_e32 0, v18
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-TRUE16-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-TRUE16-NEXT: s_cbranch_execz .LBB58_2
; GFX11-TRUE16-NEXT: ; %bb.1: ; %cmp.true
@@ -31468,6 +31528,7 @@ define <36 x i16> @bitcast_v36f16_to_v36i16(<36 x half> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v36, 16, v17
; GFX11-TRUE16-NEXT: .LBB58_2: ; %end
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v0.h, v23.l
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v1.h, v22.l
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v2.h, v21.l
@@ -31511,8 +31572,9 @@ define <36 x i16> @bitcast_v36f16_to_v36i16(<36 x half> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v19, 16, v0
; GFX11-FAKE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-FAKE16-NEXT: v_cmpx_ne_u32_e32 0, v18
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-FAKE16-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-FAKE16-NEXT: s_cbranch_execz .LBB58_2
; GFX11-FAKE16-NEXT: ; %bb.1: ; %cmp.true
@@ -31572,6 +31634,7 @@ define <36 x i16> @bitcast_v36f16_to_v36i16(<36 x half> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v36, 16, v17
; GFX11-FAKE16-NEXT: .LBB58_2: ; %end
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_perm_b32 v0, v19, v0, 0x5040100
; GFX11-FAKE16-NEXT: v_perm_b32 v1, v20, v1, 0x5040100
; GFX11-FAKE16-NEXT: v_perm_b32 v2, v21, v2, 0x5040100
@@ -32270,6 +32333,7 @@ define inreg <36 x i16> @bitcast_v36f16_to_v36i16_scalar(<36 x half> inreg %a, i
; GFX11-TRUE16-NEXT: s_and_b32 s46, s46, exec_lo
; GFX11-TRUE16-NEXT: s_cselect_b32 s46, 1, 0
; GFX11-TRUE16-NEXT: s_cmp_lg_u32 s46, 1
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_cbranch_scc1 .LBB59_4
; GFX11-TRUE16-NEXT: ; %bb.3: ; %cmp.true
; GFX11-TRUE16-NEXT: s_pack_ll_b32_b16 s29, s29, s45
@@ -32401,6 +32465,7 @@ define inreg <36 x i16> @bitcast_v36f16_to_v36i16_scalar(<36 x half> inreg %a, i
; GFX11-FAKE16-NEXT: s_and_b32 s46, s46, exec_lo
; GFX11-FAKE16-NEXT: s_cselect_b32 s46, 1, 0
; GFX11-FAKE16-NEXT: s_cmp_lg_u32 s46, 1
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_cbranch_scc1 .LBB59_4
; GFX11-FAKE16-NEXT: ; %bb.3: ; %cmp.true
; GFX11-FAKE16-NEXT: s_pack_ll_b32_b16 s29, s29, s45
diff --git a/llvm/test/CodeGen/AMDGPU/amdgcn.bitcast.640bit.ll b/llvm/test/CodeGen/AMDGPU/amdgcn.bitcast.640bit.ll
index 78b54755887fca..6128f1e2f8aa53 100644
--- a/llvm/test/CodeGen/AMDGPU/amdgcn.bitcast.640bit.ll
+++ b/llvm/test/CodeGen/AMDGPU/amdgcn.bitcast.640bit.ll
@@ -111,8 +111,9 @@ define <20 x float> @bitcast_v20i32_to_v20f32(<20 x i32> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v20
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB0_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -371,6 +372,7 @@ define inreg <20 x float> @bitcast_v20i32_to_v20f32_scalar(<20 x i32> inreg %a,
; GFX11-NEXT: s_and_b32 s6, s6, exec_lo
; GFX11-NEXT: s_cselect_b32 s6, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s6, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB1_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: s_add_i32 s4, s4, 3
@@ -528,8 +530,9 @@ define <20 x i32> @bitcast_v20f32_to_v20i32(<20 x float> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v20
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB2_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -989,6 +992,7 @@ define inreg <20 x i32> @bitcast_v20f32_to_v20i32_scalar(<20 x float> inreg %a,
; GFX11-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB3_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_add_f32_e64 v19, s55, 1.0
@@ -1169,8 +1173,9 @@ define <10 x i64> @bitcast_v20i32_to_v10i64(<20 x i32> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v20
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB4_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -1429,6 +1434,7 @@ define inreg <10 x i64> @bitcast_v20i32_to_v10i64_scalar(<20 x i32> inreg %a, i3
; GFX11-NEXT: s_and_b32 s6, s6, exec_lo
; GFX11-NEXT: s_cselect_b32 s6, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s6, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB5_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: s_add_i32 s4, s4, 3
@@ -1586,33 +1592,29 @@ define <20 x i32> @bitcast_v10i64_to_v20i32(<10 x i64> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v20
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB6_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_add_co_u32 v18, vcc_lo, v18, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v19, null, 0, v19, vcc_lo
; GFX11-NEXT: v_add_co_u32 v16, vcc_lo, v16, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v17, null, 0, v17, vcc_lo
; GFX11-NEXT: v_add_co_u32 v14, vcc_lo, v14, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v15, null, 0, v15, vcc_lo
; GFX11-NEXT: v_add_co_u32 v12, vcc_lo, v12, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v13, null, 0, v13, vcc_lo
; GFX11-NEXT: v_add_co_u32 v10, vcc_lo, v10, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v11, null, 0, v11, vcc_lo
; GFX11-NEXT: v_add_co_u32 v8, vcc_lo, v8, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v9, null, 0, v9, vcc_lo
; GFX11-NEXT: v_add_co_u32 v6, vcc_lo, v6, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v7, null, 0, v7, vcc_lo
; GFX11-NEXT: v_add_co_u32 v4, vcc_lo, v4, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v5, null, 0, v5, vcc_lo
; GFX11-NEXT: v_add_co_u32 v2, vcc_lo, v2, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX11-NEXT: v_add_co_u32 v0, vcc_lo, v0, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
@@ -1851,6 +1853,7 @@ define inreg <20 x i32> @bitcast_v10i64_to_v20i32_scalar(<10 x i64> inreg %a, i3
; GFX11-NEXT: s_and_b32 s6, s6, exec_lo
; GFX11-NEXT: s_cselect_b32 s6, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s6, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB7_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: s_add_u32 s5, s5, 3
@@ -2008,8 +2011,9 @@ define <10 x double> @bitcast_v20i32_to_v10f64(<20 x i32> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v20
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB8_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -2268,6 +2272,7 @@ define inreg <10 x double> @bitcast_v20i32_to_v10f64_scalar(<20 x i32> inreg %a,
; GFX11-NEXT: s_and_b32 s6, s6, exec_lo
; GFX11-NEXT: s_cselect_b32 s6, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s6, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB9_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: s_add_i32 s4, s4, 3
@@ -2395,8 +2400,9 @@ define <20 x i32> @bitcast_v10f64_to_v20i32(<10 x double> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v20
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB10_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -2826,6 +2832,7 @@ define inreg <20 x i32> @bitcast_v10f64_to_v20i32_scalar(<10 x double> inreg %a,
; GFX11-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB11_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_add_f64 v[18:19], s[54:55], 1.0
@@ -3396,8 +3403,9 @@ define <40 x i16> @bitcast_v20i32_to_v40i16(<20 x i32> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v39, 16, v0
; GFX11-TRUE16-NEXT: .LBB12_4: ; %end
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v0.h, v39.l
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v1.h, v38.l
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v2.h, v37.l
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v3.h, v36.l
@@ -3514,8 +3522,9 @@ define <40 x i16> @bitcast_v20i32_to_v40i16(<20 x i32> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v39, 16, v0
; GFX11-FAKE16-NEXT: .LBB12_4: ; %end
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_perm_b32 v0, v39, v0, 0x5040100
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_perm_b32 v1, v38, v1, 0x5040100
; GFX11-FAKE16-NEXT: v_perm_b32 v2, v37, v2, 0x5040100
; GFX11-FAKE16-NEXT: v_perm_b32 v3, v36, v3, 0x5040100
@@ -4080,6 +4089,7 @@ define inreg <40 x i16> @bitcast_v20i32_to_v40i16_scalar(<20 x i32> inreg %a, i3
; GFX11-NEXT: v_readfirstlane_b32 s5, v0
; GFX11-NEXT: s_mov_b32 s58, 0
; GFX11-NEXT: s_cmp_lg_u32 s6, 0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc0 .LBB13_2
; GFX11-NEXT: ; %bb.1: ; %cmp.false
; GFX11-NEXT: s_lshr_b32 s6, s4, 16
@@ -4130,6 +4140,7 @@ define inreg <40 x i16> @bitcast_v20i32_to_v40i16_scalar(<20 x i32> inreg %a, i3
; GFX11-NEXT: s_and_b32 s58, s58, exec_lo
; GFX11-NEXT: s_cselect_b32 s58, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s58, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB13_5
; GFX11-NEXT: ; %bb.4: ; %cmp.true
; GFX11-NEXT: s_add_i32 s4, s4, 3
@@ -4971,6 +4982,7 @@ define <20 x i32> @bitcast_v40i16_to_v20i32(<40 x i16> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v48, 16, v0
; GFX11-TRUE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-TRUE16-NEXT: v_cmpx_ne_u32_e32 0, v20
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB14_3
; GFX11-TRUE16-NEXT: ; %bb.1: ; %Flow
@@ -5901,6 +5913,7 @@ define inreg <20 x i32> @bitcast_v40i16_to_v20i32_scalar(<40 x i16> inreg %a, i3
; GFX11-NEXT: s_and_b32 s40, s40, exec_lo
; GFX11-NEXT: s_cselect_b32 s40, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s40, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB15_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_pk_add_u16 v0, s0, 3 op_sel_hi:[1,0]
@@ -6464,8 +6477,9 @@ define <40 x half> @bitcast_v20i32_to_v40f16(<20 x i32> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v39, 16, v0
; GFX11-TRUE16-NEXT: .LBB16_4: ; %end
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v0.h, v39.l
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v1.h, v38.l
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v2.h, v37.l
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v3.h, v36.l
@@ -6582,8 +6596,9 @@ define <40 x half> @bitcast_v20i32_to_v40f16(<20 x i32> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v39, 16, v0
; GFX11-FAKE16-NEXT: .LBB16_4: ; %end
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_perm_b32 v0, v39, v0, 0x5040100
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_perm_b32 v1, v38, v1, 0x5040100
; GFX11-FAKE16-NEXT: v_perm_b32 v2, v37, v2, 0x5040100
; GFX11-FAKE16-NEXT: v_perm_b32 v3, v36, v3, 0x5040100
@@ -7148,6 +7163,7 @@ define inreg <40 x half> @bitcast_v20i32_to_v40f16_scalar(<20 x i32> inreg %a, i
; GFX11-NEXT: v_readfirstlane_b32 s5, v0
; GFX11-NEXT: s_mov_b32 s58, 0
; GFX11-NEXT: s_cmp_lg_u32 s6, 0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc0 .LBB17_2
; GFX11-NEXT: ; %bb.1: ; %cmp.false
; GFX11-NEXT: s_lshr_b32 s6, s4, 16
@@ -7198,6 +7214,7 @@ define inreg <40 x half> @bitcast_v20i32_to_v40f16_scalar(<20 x i32> inreg %a, i
; GFX11-NEXT: s_and_b32 s58, s58, exec_lo
; GFX11-NEXT: s_cselect_b32 s58, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s58, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB17_5
; GFX11-NEXT: ; %bb.4: ; %cmp.true
; GFX11-NEXT: s_add_i32 s4, s4, 3
@@ -8124,6 +8141,7 @@ define <20 x i32> @bitcast_v40f16_to_v20i32(<40 x half> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v48, 16, v0
; GFX11-TRUE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-TRUE16-NEXT: v_cmpx_ne_u32_e32 0, v20
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB18_3
; GFX11-TRUE16-NEXT: ; %bb.1: ; %Flow
@@ -9124,6 +9142,7 @@ define inreg <20 x i32> @bitcast_v40f16_to_v20i32_scalar(<40 x half> inreg %a, i
; GFX11-NEXT: s_and_b32 s40, s40, exec_lo
; GFX11-NEXT: s_cselect_b32 s40, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s40, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB19_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_pk_add_f16 v0, 0x200, s0 op_sel_hi:[0,1]
@@ -9287,8 +9306,9 @@ define <10 x i64> @bitcast_v20f32_to_v10i64(<20 x float> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v20
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB20_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -9748,6 +9768,7 @@ define inreg <10 x i64> @bitcast_v20f32_to_v10i64_scalar(<20 x float> inreg %a,
; GFX11-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB21_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_add_f32_e64 v19, s55, 1.0
@@ -9928,33 +9949,29 @@ define <20 x float> @bitcast_v10i64_to_v20f32(<10 x i64> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v20
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB22_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_add_co_u32 v18, vcc_lo, v18, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v19, null, 0, v19, vcc_lo
; GFX11-NEXT: v_add_co_u32 v16, vcc_lo, v16, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v17, null, 0, v17, vcc_lo
; GFX11-NEXT: v_add_co_u32 v14, vcc_lo, v14, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v15, null, 0, v15, vcc_lo
; GFX11-NEXT: v_add_co_u32 v12, vcc_lo, v12, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v13, null, 0, v13, vcc_lo
; GFX11-NEXT: v_add_co_u32 v10, vcc_lo, v10, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v11, null, 0, v11, vcc_lo
; GFX11-NEXT: v_add_co_u32 v8, vcc_lo, v8, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v9, null, 0, v9, vcc_lo
; GFX11-NEXT: v_add_co_u32 v6, vcc_lo, v6, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v7, null, 0, v7, vcc_lo
; GFX11-NEXT: v_add_co_u32 v4, vcc_lo, v4, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v5, null, 0, v5, vcc_lo
; GFX11-NEXT: v_add_co_u32 v2, vcc_lo, v2, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX11-NEXT: v_add_co_u32 v0, vcc_lo, v0, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
@@ -10193,6 +10210,7 @@ define inreg <20 x float> @bitcast_v10i64_to_v20f32_scalar(<10 x i64> inreg %a,
; GFX11-NEXT: s_and_b32 s6, s6, exec_lo
; GFX11-NEXT: s_cselect_b32 s6, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s6, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB23_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: s_add_u32 s5, s5, 3
@@ -10350,8 +10368,9 @@ define <10 x double> @bitcast_v20f32_to_v10f64(<20 x float> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v20
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB24_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -10811,6 +10830,7 @@ define inreg <10 x double> @bitcast_v20f32_to_v10f64_scalar(<20 x float> inreg %
; GFX11-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB25_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_add_f32_e64 v19, s55, 1.0
@@ -10961,8 +10981,9 @@ define <20 x float> @bitcast_v10f64_to_v20f32(<10 x double> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v20
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB26_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -11392,6 +11413,7 @@ define inreg <20 x float> @bitcast_v10f64_to_v20f32_scalar(<10 x double> inreg %
; GFX11-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB27_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_add_f64 v[18:19], s[54:55], 1.0
@@ -11952,8 +11974,9 @@ define <40 x i16> @bitcast_v20f32_to_v40i16(<20 x float> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v39, 16, v0
; GFX11-TRUE16-NEXT: .LBB28_4: ; %end
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v0.h, v39.l
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v1.h, v38.l
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v2.h, v37.l
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v3.h, v36.l
@@ -12060,8 +12083,9 @@ define <40 x i16> @bitcast_v20f32_to_v40i16(<20 x float> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v39, 16, v0
; GFX11-FAKE16-NEXT: .LBB28_4: ; %end
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_perm_b32 v0, v39, v0, 0x5040100
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_perm_b32 v1, v38, v1, 0x5040100
; GFX11-FAKE16-NEXT: v_perm_b32 v2, v37, v2, 0x5040100
; GFX11-FAKE16-NEXT: v_perm_b32 v3, v36, v3, 0x5040100
@@ -12692,6 +12716,7 @@ define inreg <40 x i16> @bitcast_v20f32_to_v40i16_scalar(<20 x float> inreg %a,
; GFX11-TRUE16-NEXT: v_readfirstlane_b32 s5, v0
; GFX11-TRUE16-NEXT: s_mov_b32 s8, 0
; GFX11-TRUE16-NEXT: s_cmp_lg_u32 s6, 0
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_cbranch_scc0 .LBB29_2
; GFX11-TRUE16-NEXT: ; %bb.1: ; %cmp.false
; GFX11-TRUE16-NEXT: s_lshr_b32 s6, s4, 16
@@ -12742,6 +12767,7 @@ define inreg <40 x i16> @bitcast_v20f32_to_v40i16_scalar(<20 x float> inreg %a,
; GFX11-TRUE16-NEXT: s_and_b32 s8, s8, exec_lo
; GFX11-TRUE16-NEXT: s_cselect_b32 s8, 1, 0
; GFX11-TRUE16-NEXT: s_cmp_lg_u32 s8, 1
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_cbranch_scc1 .LBB29_5
; GFX11-TRUE16-NEXT: ; %bb.4: ; %cmp.true
; GFX11-TRUE16-NEXT: v_add_f32_e64 v19, s4, 1.0
@@ -12838,6 +12864,7 @@ define inreg <40 x i16> @bitcast_v20f32_to_v40i16_scalar(<20 x float> inreg %a,
; GFX11-FAKE16-NEXT: v_readfirstlane_b32 s5, v0
; GFX11-FAKE16-NEXT: s_mov_b32 s8, 0
; GFX11-FAKE16-NEXT: s_cmp_lg_u32 s6, 0
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_cbranch_scc0 .LBB29_2
; GFX11-FAKE16-NEXT: ; %bb.1: ; %cmp.false
; GFX11-FAKE16-NEXT: s_lshr_b32 s6, s4, 16
@@ -12888,6 +12915,7 @@ define inreg <40 x i16> @bitcast_v20f32_to_v40i16_scalar(<20 x float> inreg %a,
; GFX11-FAKE16-NEXT: s_and_b32 s8, s8, exec_lo
; GFX11-FAKE16-NEXT: s_cselect_b32 s8, 1, 0
; GFX11-FAKE16-NEXT: s_cmp_lg_u32 s8, 1
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_cbranch_scc1 .LBB29_5
; GFX11-FAKE16-NEXT: ; %bb.4: ; %cmp.true
; GFX11-FAKE16-NEXT: v_add_f32_e64 v15, s4, 1.0
@@ -13760,6 +13788,7 @@ define <20 x float> @bitcast_v40i16_to_v20f32(<40 x i16> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v48, 16, v0
; GFX11-TRUE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-TRUE16-NEXT: v_cmpx_ne_u32_e32 0, v20
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB30_3
; GFX11-TRUE16-NEXT: ; %bb.1: ; %Flow
@@ -14690,6 +14719,7 @@ define inreg <20 x float> @bitcast_v40i16_to_v20f32_scalar(<40 x i16> inreg %a,
; GFX11-NEXT: s_and_b32 s40, s40, exec_lo
; GFX11-NEXT: s_cselect_b32 s40, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s40, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB31_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_pk_add_u16 v0, s0, 3 op_sel_hi:[1,0]
@@ -15243,8 +15273,9 @@ define <40 x half> @bitcast_v20f32_to_v40f16(<20 x float> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v39, 16, v0
; GFX11-TRUE16-NEXT: .LBB32_4: ; %end
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v0.h, v39.l
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v1.h, v38.l
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v2.h, v37.l
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v3.h, v36.l
@@ -15351,8 +15382,9 @@ define <40 x half> @bitcast_v20f32_to_v40f16(<20 x float> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v39, 16, v0
; GFX11-FAKE16-NEXT: .LBB32_4: ; %end
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_perm_b32 v0, v39, v0, 0x5040100
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_perm_b32 v1, v38, v1, 0x5040100
; GFX11-FAKE16-NEXT: v_perm_b32 v2, v37, v2, 0x5040100
; GFX11-FAKE16-NEXT: v_perm_b32 v3, v36, v3, 0x5040100
@@ -15983,6 +16015,7 @@ define inreg <40 x half> @bitcast_v20f32_to_v40f16_scalar(<20 x float> inreg %a,
; GFX11-TRUE16-NEXT: v_readfirstlane_b32 s5, v0
; GFX11-TRUE16-NEXT: s_mov_b32 s8, 0
; GFX11-TRUE16-NEXT: s_cmp_lg_u32 s6, 0
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_cbranch_scc0 .LBB33_2
; GFX11-TRUE16-NEXT: ; %bb.1: ; %cmp.false
; GFX11-TRUE16-NEXT: s_lshr_b32 s6, s4, 16
@@ -16033,6 +16066,7 @@ define inreg <40 x half> @bitcast_v20f32_to_v40f16_scalar(<20 x float> inreg %a,
; GFX11-TRUE16-NEXT: s_and_b32 s8, s8, exec_lo
; GFX11-TRUE16-NEXT: s_cselect_b32 s8, 1, 0
; GFX11-TRUE16-NEXT: s_cmp_lg_u32 s8, 1
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_cbranch_scc1 .LBB33_5
; GFX11-TRUE16-NEXT: ; %bb.4: ; %cmp.true
; GFX11-TRUE16-NEXT: v_add_f32_e64 v19, s4, 1.0
@@ -16129,6 +16163,7 @@ define inreg <40 x half> @bitcast_v20f32_to_v40f16_scalar(<20 x float> inreg %a,
; GFX11-FAKE16-NEXT: v_readfirstlane_b32 s5, v0
; GFX11-FAKE16-NEXT: s_mov_b32 s8, 0
; GFX11-FAKE16-NEXT: s_cmp_lg_u32 s6, 0
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_cbranch_scc0 .LBB33_2
; GFX11-FAKE16-NEXT: ; %bb.1: ; %cmp.false
; GFX11-FAKE16-NEXT: s_lshr_b32 s6, s4, 16
@@ -16179,6 +16214,7 @@ define inreg <40 x half> @bitcast_v20f32_to_v40f16_scalar(<20 x float> inreg %a,
; GFX11-FAKE16-NEXT: s_and_b32 s8, s8, exec_lo
; GFX11-FAKE16-NEXT: s_cselect_b32 s8, 1, 0
; GFX11-FAKE16-NEXT: s_cmp_lg_u32 s8, 1
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_cbranch_scc1 .LBB33_5
; GFX11-FAKE16-NEXT: ; %bb.4: ; %cmp.true
; GFX11-FAKE16-NEXT: v_add_f32_e64 v15, s4, 1.0
@@ -17136,6 +17172,7 @@ define <20 x float> @bitcast_v40f16_to_v20f32(<40 x half> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v48, 16, v0
; GFX11-TRUE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-TRUE16-NEXT: v_cmpx_ne_u32_e32 0, v20
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB34_3
; GFX11-TRUE16-NEXT: ; %bb.1: ; %Flow
@@ -18136,6 +18173,7 @@ define inreg <20 x float> @bitcast_v40f16_to_v20f32_scalar(<40 x half> inreg %a,
; GFX11-NEXT: s_and_b32 s40, s40, exec_lo
; GFX11-NEXT: s_cselect_b32 s40, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s40, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB35_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_pk_add_f16 v0, 0x200, s0 op_sel_hi:[0,1]
@@ -18299,33 +18337,29 @@ define <10 x double> @bitcast_v10i64_to_v10f64(<10 x i64> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v20
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB36_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_add_co_u32 v0, vcc_lo, v0, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-NEXT: v_add_co_u32 v2, vcc_lo, v2, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX11-NEXT: v_add_co_u32 v4, vcc_lo, v4, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v5, null, 0, v5, vcc_lo
; GFX11-NEXT: v_add_co_u32 v6, vcc_lo, v6, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v7, null, 0, v7, vcc_lo
; GFX11-NEXT: v_add_co_u32 v8, vcc_lo, v8, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v9, null, 0, v9, vcc_lo
; GFX11-NEXT: v_add_co_u32 v10, vcc_lo, v10, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v11, null, 0, v11, vcc_lo
; GFX11-NEXT: v_add_co_u32 v12, vcc_lo, v12, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v13, null, 0, v13, vcc_lo
; GFX11-NEXT: v_add_co_u32 v14, vcc_lo, v14, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v15, null, 0, v15, vcc_lo
; GFX11-NEXT: v_add_co_u32 v16, vcc_lo, v16, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v17, null, 0, v17, vcc_lo
; GFX11-NEXT: v_add_co_u32 v18, vcc_lo, v18, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v19, null, 0, v19, vcc_lo
@@ -18564,6 +18598,7 @@ define inreg <10 x double> @bitcast_v10i64_to_v10f64_scalar(<10 x i64> inreg %a,
; GFX11-NEXT: s_and_b32 s6, s6, exec_lo
; GFX11-NEXT: s_cselect_b32 s6, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s6, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB37_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: s_add_u32 s0, s0, 3
@@ -18690,8 +18725,9 @@ define <10 x i64> @bitcast_v10f64_to_v10i64(<10 x double> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v20
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB38_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -19121,6 +19157,7 @@ define inreg <10 x i64> @bitcast_v10f64_to_v10i64_scalar(<10 x double> inreg %a,
; GFX11-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB39_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_add_f64 v[0:1], s[36:37], 1.0
@@ -19650,27 +19687,22 @@ define <40 x i16> @bitcast_v10i64_to_v40i16(<10 x i64> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: s_cbranch_execz .LBB40_4
; GFX11-TRUE16-NEXT: ; %bb.3: ; %cmp.true
; GFX11-TRUE16-NEXT: v_add_co_u32 v18, vcc_lo, v18, 3
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v19, null, 0, v19, vcc_lo
; GFX11-TRUE16-NEXT: v_add_co_u32 v16, vcc_lo, v16, 3
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v17, null, 0, v17, vcc_lo
; GFX11-TRUE16-NEXT: v_add_co_u32 v14, vcc_lo, v14, 3
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v15, null, 0, v15, vcc_lo
; GFX11-TRUE16-NEXT: v_add_co_u32 v12, vcc_lo, v12, 3
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v13, null, 0, v13, vcc_lo
; GFX11-TRUE16-NEXT: v_add_co_u32 v10, vcc_lo, v10, 3
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v11, null, 0, v11, vcc_lo
; GFX11-TRUE16-NEXT: v_add_co_u32 v8, vcc_lo, v8, 3
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v9, null, 0, v9, vcc_lo
; GFX11-TRUE16-NEXT: v_add_co_u32 v6, vcc_lo, v6, 3
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v7, null, 0, v7, vcc_lo
; GFX11-TRUE16-NEXT: v_add_co_u32 v4, vcc_lo, v4, 3
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v5, null, 0, v5, vcc_lo
; GFX11-TRUE16-NEXT: v_add_co_u32 v2, vcc_lo, v2, 3
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX11-TRUE16-NEXT: v_add_co_u32 v0, vcc_lo, v0, 3
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
@@ -19696,8 +19728,9 @@ define <40 x i16> @bitcast_v10i64_to_v40i16(<10 x i64> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v39, 16, v0
; GFX11-TRUE16-NEXT: .LBB40_4: ; %end
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v0.h, v39.l
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v1.h, v38.l
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v2.h, v37.l
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v3.h, v36.l
@@ -19773,27 +19806,22 @@ define <40 x i16> @bitcast_v10i64_to_v40i16(<10 x i64> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: s_cbranch_execz .LBB40_4
; GFX11-FAKE16-NEXT: ; %bb.3: ; %cmp.true
; GFX11-FAKE16-NEXT: v_add_co_u32 v18, vcc_lo, v18, 3
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v19, null, 0, v19, vcc_lo
; GFX11-FAKE16-NEXT: v_add_co_u32 v16, vcc_lo, v16, 3
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v17, null, 0, v17, vcc_lo
; GFX11-FAKE16-NEXT: v_add_co_u32 v14, vcc_lo, v14, 3
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v15, null, 0, v15, vcc_lo
; GFX11-FAKE16-NEXT: v_add_co_u32 v12, vcc_lo, v12, 3
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v13, null, 0, v13, vcc_lo
; GFX11-FAKE16-NEXT: v_add_co_u32 v10, vcc_lo, v10, 3
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v11, null, 0, v11, vcc_lo
; GFX11-FAKE16-NEXT: v_add_co_u32 v8, vcc_lo, v8, 3
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v9, null, 0, v9, vcc_lo
; GFX11-FAKE16-NEXT: v_add_co_u32 v6, vcc_lo, v6, 3
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v7, null, 0, v7, vcc_lo
; GFX11-FAKE16-NEXT: v_add_co_u32 v4, vcc_lo, v4, 3
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v5, null, 0, v5, vcc_lo
; GFX11-FAKE16-NEXT: v_add_co_u32 v2, vcc_lo, v2, 3
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX11-FAKE16-NEXT: v_add_co_u32 v0, vcc_lo, v0, 3
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
@@ -19819,8 +19847,9 @@ define <40 x i16> @bitcast_v10i64_to_v40i16(<10 x i64> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v39, 16, v0
; GFX11-FAKE16-NEXT: .LBB40_4: ; %end
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_perm_b32 v0, v39, v0, 0x5040100
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_perm_b32 v1, v38, v1, 0x5040100
; GFX11-FAKE16-NEXT: v_perm_b32 v2, v37, v2, 0x5040100
; GFX11-FAKE16-NEXT: v_perm_b32 v3, v36, v3, 0x5040100
@@ -20385,6 +20414,7 @@ define inreg <40 x i16> @bitcast_v10i64_to_v40i16_scalar(<10 x i64> inreg %a, i3
; GFX11-NEXT: v_readfirstlane_b32 s5, v0
; GFX11-NEXT: s_mov_b32 s58, 0
; GFX11-NEXT: s_cmp_lg_u32 s6, 0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc0 .LBB41_2
; GFX11-NEXT: ; %bb.1: ; %cmp.false
; GFX11-NEXT: s_lshr_b32 s6, s4, 16
@@ -20435,6 +20465,7 @@ define inreg <40 x i16> @bitcast_v10i64_to_v40i16_scalar(<10 x i64> inreg %a, i3
; GFX11-NEXT: s_and_b32 s58, s58, exec_lo
; GFX11-NEXT: s_cselect_b32 s58, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s58, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB41_5
; GFX11-NEXT: ; %bb.4: ; %cmp.true
; GFX11-NEXT: s_add_u32 s5, s5, 3
@@ -21276,6 +21307,7 @@ define <10 x i64> @bitcast_v40i16_to_v10i64(<40 x i16> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v48, 16, v0
; GFX11-TRUE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-TRUE16-NEXT: v_cmpx_ne_u32_e32 0, v20
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB42_3
; GFX11-TRUE16-NEXT: ; %bb.1: ; %Flow
@@ -22206,6 +22238,7 @@ define inreg <10 x i64> @bitcast_v40i16_to_v10i64_scalar(<40 x i16> inreg %a, i3
; GFX11-NEXT: s_and_b32 s40, s40, exec_lo
; GFX11-NEXT: s_cselect_b32 s40, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s40, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB43_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_pk_add_u16 v0, s0, 3 op_sel_hi:[1,0]
@@ -22728,27 +22761,22 @@ define <40 x half> @bitcast_v10i64_to_v40f16(<10 x i64> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: s_cbranch_execz .LBB44_4
; GFX11-TRUE16-NEXT: ; %bb.3: ; %cmp.true
; GFX11-TRUE16-NEXT: v_add_co_u32 v18, vcc_lo, v18, 3
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v19, null, 0, v19, vcc_lo
; GFX11-TRUE16-NEXT: v_add_co_u32 v16, vcc_lo, v16, 3
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v17, null, 0, v17, vcc_lo
; GFX11-TRUE16-NEXT: v_add_co_u32 v14, vcc_lo, v14, 3
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v15, null, 0, v15, vcc_lo
; GFX11-TRUE16-NEXT: v_add_co_u32 v12, vcc_lo, v12, 3
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v13, null, 0, v13, vcc_lo
; GFX11-TRUE16-NEXT: v_add_co_u32 v10, vcc_lo, v10, 3
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v11, null, 0, v11, vcc_lo
; GFX11-TRUE16-NEXT: v_add_co_u32 v8, vcc_lo, v8, 3
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v9, null, 0, v9, vcc_lo
; GFX11-TRUE16-NEXT: v_add_co_u32 v6, vcc_lo, v6, 3
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v7, null, 0, v7, vcc_lo
; GFX11-TRUE16-NEXT: v_add_co_u32 v4, vcc_lo, v4, 3
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v5, null, 0, v5, vcc_lo
; GFX11-TRUE16-NEXT: v_add_co_u32 v2, vcc_lo, v2, 3
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX11-TRUE16-NEXT: v_add_co_u32 v0, vcc_lo, v0, 3
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
@@ -22774,8 +22802,9 @@ define <40 x half> @bitcast_v10i64_to_v40f16(<10 x i64> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v39, 16, v0
; GFX11-TRUE16-NEXT: .LBB44_4: ; %end
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v0.h, v39.l
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v1.h, v38.l
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v2.h, v37.l
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v3.h, v36.l
@@ -22851,27 +22880,22 @@ define <40 x half> @bitcast_v10i64_to_v40f16(<10 x i64> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: s_cbranch_execz .LBB44_4
; GFX11-FAKE16-NEXT: ; %bb.3: ; %cmp.true
; GFX11-FAKE16-NEXT: v_add_co_u32 v18, vcc_lo, v18, 3
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v19, null, 0, v19, vcc_lo
; GFX11-FAKE16-NEXT: v_add_co_u32 v16, vcc_lo, v16, 3
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v17, null, 0, v17, vcc_lo
; GFX11-FAKE16-NEXT: v_add_co_u32 v14, vcc_lo, v14, 3
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v15, null, 0, v15, vcc_lo
; GFX11-FAKE16-NEXT: v_add_co_u32 v12, vcc_lo, v12, 3
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v13, null, 0, v13, vcc_lo
; GFX11-FAKE16-NEXT: v_add_co_u32 v10, vcc_lo, v10, 3
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v11, null, 0, v11, vcc_lo
; GFX11-FAKE16-NEXT: v_add_co_u32 v8, vcc_lo, v8, 3
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v9, null, 0, v9, vcc_lo
; GFX11-FAKE16-NEXT: v_add_co_u32 v6, vcc_lo, v6, 3
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v7, null, 0, v7, vcc_lo
; GFX11-FAKE16-NEXT: v_add_co_u32 v4, vcc_lo, v4, 3
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v5, null, 0, v5, vcc_lo
; GFX11-FAKE16-NEXT: v_add_co_u32 v2, vcc_lo, v2, 3
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX11-FAKE16-NEXT: v_add_co_u32 v0, vcc_lo, v0, 3
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
@@ -22897,8 +22921,9 @@ define <40 x half> @bitcast_v10i64_to_v40f16(<10 x i64> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v39, 16, v0
; GFX11-FAKE16-NEXT: .LBB44_4: ; %end
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_perm_b32 v0, v39, v0, 0x5040100
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_perm_b32 v1, v38, v1, 0x5040100
; GFX11-FAKE16-NEXT: v_perm_b32 v2, v37, v2, 0x5040100
; GFX11-FAKE16-NEXT: v_perm_b32 v3, v36, v3, 0x5040100
@@ -23463,6 +23488,7 @@ define inreg <40 x half> @bitcast_v10i64_to_v40f16_scalar(<10 x i64> inreg %a, i
; GFX11-NEXT: v_readfirstlane_b32 s5, v0
; GFX11-NEXT: s_mov_b32 s58, 0
; GFX11-NEXT: s_cmp_lg_u32 s6, 0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc0 .LBB45_2
; GFX11-NEXT: ; %bb.1: ; %cmp.false
; GFX11-NEXT: s_lshr_b32 s6, s4, 16
@@ -23513,6 +23539,7 @@ define inreg <40 x half> @bitcast_v10i64_to_v40f16_scalar(<10 x i64> inreg %a, i
; GFX11-NEXT: s_and_b32 s58, s58, exec_lo
; GFX11-NEXT: s_cselect_b32 s58, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s58, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB45_5
; GFX11-NEXT: ; %bb.4: ; %cmp.true
; GFX11-NEXT: s_add_u32 s5, s5, 3
@@ -24439,6 +24466,7 @@ define <10 x i64> @bitcast_v40f16_to_v10i64(<40 x half> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v48, 16, v0
; GFX11-TRUE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-TRUE16-NEXT: v_cmpx_ne_u32_e32 0, v20
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB46_3
; GFX11-TRUE16-NEXT: ; %bb.1: ; %Flow
@@ -25439,6 +25467,7 @@ define inreg <10 x i64> @bitcast_v40f16_to_v10i64_scalar(<40 x half> inreg %a, i
; GFX11-NEXT: s_and_b32 s40, s40, exec_lo
; GFX11-NEXT: s_cselect_b32 s40, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s40, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB47_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_pk_add_f16 v0, 0x200, s0 op_sel_hi:[0,1]
@@ -25962,8 +25991,9 @@ define <40 x i16> @bitcast_v10f64_to_v40i16(<10 x double> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v39, 16, v0
; GFX11-TRUE16-NEXT: .LBB48_4: ; %end
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v0.h, v39.l
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v1.h, v38.l
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v2.h, v37.l
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v3.h, v36.l
@@ -26070,8 +26100,9 @@ define <40 x i16> @bitcast_v10f64_to_v40i16(<10 x double> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v39, 16, v0
; GFX11-FAKE16-NEXT: .LBB48_4: ; %end
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_perm_b32 v0, v39, v0, 0x5040100
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_perm_b32 v1, v38, v1, 0x5040100
; GFX11-FAKE16-NEXT: v_perm_b32 v2, v37, v2, 0x5040100
; GFX11-FAKE16-NEXT: v_perm_b32 v3, v36, v3, 0x5040100
@@ -26672,6 +26703,7 @@ define inreg <40 x i16> @bitcast_v10f64_to_v40i16_scalar(<10 x double> inreg %a,
; GFX11-TRUE16-NEXT: v_readfirstlane_b32 s4, v0
; GFX11-TRUE16-NEXT: s_mov_b32 s8, 0
; GFX11-TRUE16-NEXT: s_cmp_lg_u32 s6, 0
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_cbranch_scc0 .LBB49_2
; GFX11-TRUE16-NEXT: ; %bb.1: ; %cmp.false
; GFX11-TRUE16-NEXT: s_lshr_b32 s6, s5, 16
@@ -26722,6 +26754,7 @@ define inreg <40 x i16> @bitcast_v10f64_to_v40i16_scalar(<10 x double> inreg %a,
; GFX11-TRUE16-NEXT: s_and_b32 s8, s8, exec_lo
; GFX11-TRUE16-NEXT: s_cselect_b32 s8, 1, 0
; GFX11-TRUE16-NEXT: s_cmp_lg_u32 s8, 1
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_cbranch_scc1 .LBB49_5
; GFX11-TRUE16-NEXT: ; %bb.4: ; %cmp.true
; GFX11-TRUE16-NEXT: v_add_f64 v[18:19], s[4:5], 1.0
@@ -26808,6 +26841,7 @@ define inreg <40 x i16> @bitcast_v10f64_to_v40i16_scalar(<10 x double> inreg %a,
; GFX11-FAKE16-NEXT: v_readfirstlane_b32 s4, v0
; GFX11-FAKE16-NEXT: s_mov_b32 s8, 0
; GFX11-FAKE16-NEXT: s_cmp_lg_u32 s6, 0
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_cbranch_scc0 .LBB49_2
; GFX11-FAKE16-NEXT: ; %bb.1: ; %cmp.false
; GFX11-FAKE16-NEXT: s_lshr_b32 s6, s5, 16
@@ -26858,6 +26892,7 @@ define inreg <40 x i16> @bitcast_v10f64_to_v40i16_scalar(<10 x double> inreg %a,
; GFX11-FAKE16-NEXT: s_and_b32 s8, s8, exec_lo
; GFX11-FAKE16-NEXT: s_cselect_b32 s8, 1, 0
; GFX11-FAKE16-NEXT: s_cmp_lg_u32 s8, 1
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_cbranch_scc1 .LBB49_5
; GFX11-FAKE16-NEXT: ; %bb.4: ; %cmp.true
; GFX11-FAKE16-NEXT: v_add_f64 v[15:16], s[4:5], 1.0
@@ -27720,6 +27755,7 @@ define <10 x double> @bitcast_v40i16_to_v10f64(<40 x i16> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v48, 16, v0
; GFX11-TRUE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-TRUE16-NEXT: v_cmpx_ne_u32_e32 0, v20
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB50_3
; GFX11-TRUE16-NEXT: ; %bb.1: ; %Flow
@@ -28650,6 +28686,7 @@ define inreg <10 x double> @bitcast_v40i16_to_v10f64_scalar(<40 x i16> inreg %a,
; GFX11-NEXT: s_and_b32 s40, s40, exec_lo
; GFX11-NEXT: s_cselect_b32 s40, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s40, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB51_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_pk_add_u16 v0, s0, 3 op_sel_hi:[1,0]
@@ -29173,8 +29210,9 @@ define <40 x half> @bitcast_v10f64_to_v40f16(<10 x double> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v39, 16, v0
; GFX11-TRUE16-NEXT: .LBB52_4: ; %end
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v0.h, v39.l
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v1.h, v38.l
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v2.h, v37.l
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v3.h, v36.l
@@ -29281,8 +29319,9 @@ define <40 x half> @bitcast_v10f64_to_v40f16(<10 x double> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v39, 16, v0
; GFX11-FAKE16-NEXT: .LBB52_4: ; %end
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_perm_b32 v0, v39, v0, 0x5040100
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_perm_b32 v1, v38, v1, 0x5040100
; GFX11-FAKE16-NEXT: v_perm_b32 v2, v37, v2, 0x5040100
; GFX11-FAKE16-NEXT: v_perm_b32 v3, v36, v3, 0x5040100
@@ -29883,6 +29922,7 @@ define inreg <40 x half> @bitcast_v10f64_to_v40f16_scalar(<10 x double> inreg %a
; GFX11-TRUE16-NEXT: v_readfirstlane_b32 s4, v0
; GFX11-TRUE16-NEXT: s_mov_b32 s8, 0
; GFX11-TRUE16-NEXT: s_cmp_lg_u32 s6, 0
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_cbranch_scc0 .LBB53_2
; GFX11-TRUE16-NEXT: ; %bb.1: ; %cmp.false
; GFX11-TRUE16-NEXT: s_lshr_b32 s6, s5, 16
@@ -29933,6 +29973,7 @@ define inreg <40 x half> @bitcast_v10f64_to_v40f16_scalar(<10 x double> inreg %a
; GFX11-TRUE16-NEXT: s_and_b32 s8, s8, exec_lo
; GFX11-TRUE16-NEXT: s_cselect_b32 s8, 1, 0
; GFX11-TRUE16-NEXT: s_cmp_lg_u32 s8, 1
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_cbranch_scc1 .LBB53_5
; GFX11-TRUE16-NEXT: ; %bb.4: ; %cmp.true
; GFX11-TRUE16-NEXT: v_add_f64 v[18:19], s[4:5], 1.0
@@ -30019,6 +30060,7 @@ define inreg <40 x half> @bitcast_v10f64_to_v40f16_scalar(<10 x double> inreg %a
; GFX11-FAKE16-NEXT: v_readfirstlane_b32 s4, v0
; GFX11-FAKE16-NEXT: s_mov_b32 s8, 0
; GFX11-FAKE16-NEXT: s_cmp_lg_u32 s6, 0
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_cbranch_scc0 .LBB53_2
; GFX11-FAKE16-NEXT: ; %bb.1: ; %cmp.false
; GFX11-FAKE16-NEXT: s_lshr_b32 s6, s5, 16
@@ -30069,6 +30111,7 @@ define inreg <40 x half> @bitcast_v10f64_to_v40f16_scalar(<10 x double> inreg %a
; GFX11-FAKE16-NEXT: s_and_b32 s8, s8, exec_lo
; GFX11-FAKE16-NEXT: s_cselect_b32 s8, 1, 0
; GFX11-FAKE16-NEXT: s_cmp_lg_u32 s8, 1
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_cbranch_scc1 .LBB53_5
; GFX11-FAKE16-NEXT: ; %bb.4: ; %cmp.true
; GFX11-FAKE16-NEXT: v_add_f64 v[15:16], s[4:5], 1.0
@@ -31016,6 +31059,7 @@ define <10 x double> @bitcast_v40f16_to_v10f64(<40 x half> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v48, 16, v0
; GFX11-TRUE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-TRUE16-NEXT: v_cmpx_ne_u32_e32 0, v20
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB54_3
; GFX11-TRUE16-NEXT: ; %bb.1: ; %Flow
@@ -32016,6 +32060,7 @@ define inreg <10 x double> @bitcast_v40f16_to_v10f64_scalar(<40 x half> inreg %a
; GFX11-NEXT: s_and_b32 s40, s40, exec_lo
; GFX11-NEXT: s_cselect_b32 s40, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s40, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB55_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_pk_add_f16 v0, 0x200, s0 op_sel_hi:[0,1]
@@ -32803,8 +32848,9 @@ define <40 x half> @bitcast_v40i16_to_v40f16(<40 x i16> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v25, 16, v0
; GFX11-TRUE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-TRUE16-NEXT: v_cmpx_ne_u32_e32 0, v20
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-TRUE16-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-TRUE16-NEXT: s_cbranch_execz .LBB56_2
; GFX11-TRUE16-NEXT: ; %bb.1: ; %cmp.true
@@ -32870,6 +32916,7 @@ define <40 x half> @bitcast_v40i16_to_v40f16(<40 x i16> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v48, 16, v19
; GFX11-TRUE16-NEXT: .LBB56_2: ; %end
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v0.h, v25.l
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v1.h, v24.l
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v2.h, v23.l
@@ -32917,8 +32964,9 @@ define <40 x half> @bitcast_v40i16_to_v40f16(<40 x i16> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v21, 16, v0
; GFX11-FAKE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-FAKE16-NEXT: v_cmpx_ne_u32_e32 0, v20
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-FAKE16-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-FAKE16-NEXT: s_cbranch_execz .LBB56_2
; GFX11-FAKE16-NEXT: ; %bb.1: ; %cmp.true
@@ -32984,6 +33032,7 @@ define <40 x half> @bitcast_v40i16_to_v40f16(<40 x i16> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v48, 16, v19
; GFX11-FAKE16-NEXT: .LBB56_2: ; %end
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_perm_b32 v0, v21, v0, 0x5040100
; GFX11-FAKE16-NEXT: v_perm_b32 v1, v22, v1, 0x5040100
; GFX11-FAKE16-NEXT: v_perm_b32 v2, v23, v2, 0x5040100
@@ -33811,6 +33860,7 @@ define inreg <40 x half> @bitcast_v40i16_to_v40f16_scalar(<40 x i16> inreg %a, i
; GFX11-TRUE16-NEXT: s_and_b32 s58, s58, exec_lo
; GFX11-TRUE16-NEXT: s_cselect_b32 s58, 1, 0
; GFX11-TRUE16-NEXT: s_cmp_lg_u32 s58, 1
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_cbranch_scc1 .LBB57_4
; GFX11-TRUE16-NEXT: ; %bb.3: ; %cmp.true
; GFX11-TRUE16-NEXT: s_pack_ll_b32_b16 s47, s57, s47
@@ -33956,6 +34006,7 @@ define inreg <40 x half> @bitcast_v40i16_to_v40f16_scalar(<40 x i16> inreg %a, i
; GFX11-FAKE16-NEXT: s_and_b32 s58, s58, exec_lo
; GFX11-FAKE16-NEXT: s_cselect_b32 s58, 1, 0
; GFX11-FAKE16-NEXT: s_cmp_lg_u32 s58, 1
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_cbranch_scc1 .LBB57_4
; GFX11-FAKE16-NEXT: ; %bb.3: ; %cmp.true
; GFX11-FAKE16-NEXT: s_pack_ll_b32_b16 s47, s57, s47
@@ -34617,8 +34668,9 @@ define <40 x i16> @bitcast_v40f16_to_v40i16(<40 x half> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v25, 16, v0
; GFX11-TRUE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-TRUE16-NEXT: v_cmpx_ne_u32_e32 0, v20
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-TRUE16-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-TRUE16-NEXT: s_cbranch_execz .LBB58_2
; GFX11-TRUE16-NEXT: ; %bb.1: ; %cmp.true
@@ -34684,6 +34736,7 @@ define <40 x i16> @bitcast_v40f16_to_v40i16(<40 x half> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v48, 16, v19
; GFX11-TRUE16-NEXT: .LBB58_2: ; %end
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v0.h, v25.l
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v1.h, v24.l
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v2.h, v23.l
@@ -34731,8 +34784,9 @@ define <40 x i16> @bitcast_v40f16_to_v40i16(<40 x half> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v21, 16, v0
; GFX11-FAKE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-FAKE16-NEXT: v_cmpx_ne_u32_e32 0, v20
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-FAKE16-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-FAKE16-NEXT: s_cbranch_execz .LBB58_2
; GFX11-FAKE16-NEXT: ; %bb.1: ; %cmp.true
@@ -34798,6 +34852,7 @@ define <40 x i16> @bitcast_v40f16_to_v40i16(<40 x half> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v48, 16, v19
; GFX11-FAKE16-NEXT: .LBB58_2: ; %end
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_perm_b32 v0, v21, v0, 0x5040100
; GFX11-FAKE16-NEXT: v_perm_b32 v1, v22, v1, 0x5040100
; GFX11-FAKE16-NEXT: v_perm_b32 v2, v23, v2, 0x5040100
@@ -35579,6 +35634,7 @@ define inreg <40 x i16> @bitcast_v40f16_to_v40i16_scalar(<40 x half> inreg %a, i
; GFX11-TRUE16-NEXT: s_and_b32 s58, s58, exec_lo
; GFX11-TRUE16-NEXT: s_cselect_b32 s58, 1, 0
; GFX11-TRUE16-NEXT: s_cmp_lg_u32 s58, 1
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_cbranch_scc1 .LBB59_4
; GFX11-TRUE16-NEXT: ; %bb.3: ; %cmp.true
; GFX11-TRUE16-NEXT: s_pack_ll_b32_b16 s47, s57, s47
@@ -35724,6 +35780,7 @@ define inreg <40 x i16> @bitcast_v40f16_to_v40i16_scalar(<40 x half> inreg %a, i
; GFX11-FAKE16-NEXT: s_and_b32 s58, s58, exec_lo
; GFX11-FAKE16-NEXT: s_cselect_b32 s58, 1, 0
; GFX11-FAKE16-NEXT: s_cmp_lg_u32 s58, 1
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_cbranch_scc1 .LBB59_4
; GFX11-FAKE16-NEXT: ; %bb.3: ; %cmp.true
; GFX11-FAKE16-NEXT: s_pack_ll_b32_b16 s47, s57, s47
diff --git a/llvm/test/CodeGen/AMDGPU/amdgcn.bitcast.64bit.ll b/llvm/test/CodeGen/AMDGPU/amdgcn.bitcast.64bit.ll
index 26821b059cb6c0..085b41ad4f8bef 100644
--- a/llvm/test/CodeGen/AMDGPU/amdgcn.bitcast.64bit.ll
+++ b/llvm/test/CodeGen/AMDGPU/amdgcn.bitcast.64bit.ll
@@ -54,12 +54,12 @@ define double @bitcast_i64_to_f64(i64 %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v2
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_add_co_u32 v0, vcc_lo, v0, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-NEXT: ; %bb.2: ; %end
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -164,6 +164,7 @@ define inreg double @bitcast_i64_to_f64_scalar(i64 inreg %a, i32 inreg %b) #0 {
; GFX11-NEXT: s_and_b32 s2, s2, exec_lo
; GFX11-NEXT: s_cselect_b32 s2, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s2, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB1_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: s_add_u32 s0, s0, 3
@@ -234,8 +235,9 @@ define i64 @bitcast_f64_to_i64(double %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v2
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB2_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -343,6 +345,7 @@ define inreg i64 @bitcast_f64_to_i64_scalar(double inreg %a, i32 inreg %b) #0 {
; GFX11-NEXT: s_and_b32 s2, s2, exec_lo
; GFX11-NEXT: s_cselect_b32 s2, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s2, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB3_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_add_f64 v[0:1], s[0:1], 1.0
@@ -415,12 +418,12 @@ define <2 x i32> @bitcast_i64_to_v2i32(i64 %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v2
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_add_co_u32 v0, vcc_lo, v0, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-NEXT: ; %bb.2: ; %end
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -525,6 +528,7 @@ define inreg <2 x i32> @bitcast_i64_to_v2i32_scalar(i64 inreg %a, i32 inreg %b)
; GFX11-NEXT: s_and_b32 s2, s2, exec_lo
; GFX11-NEXT: s_cselect_b32 s2, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s2, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB5_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: s_add_u32 s0, s0, 3
@@ -598,8 +602,9 @@ define i64 @bitcast_v2i32_to_i64(<2 x i32> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v2
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_add_nc_u32_e32 v1, 3, v1
@@ -707,6 +712,7 @@ define inreg i64 @bitcast_v2i32_to_i64_scalar(<2 x i32> inreg %a, i32 inreg %b)
; GFX11-NEXT: s_and_b32 s2, s2, exec_lo
; GFX11-NEXT: s_cselect_b32 s2, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s2, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB7_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: s_add_i32 s1, s1, 3
@@ -780,12 +786,12 @@ define <2 x float> @bitcast_i64_to_v2f32(i64 %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v2
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_add_co_u32 v0, vcc_lo, v0, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-NEXT: ; %bb.2: ; %end
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -890,6 +896,7 @@ define inreg <2 x float> @bitcast_i64_to_v2f32_scalar(i64 inreg %a, i32 inreg %b
; GFX11-NEXT: s_and_b32 s2, s2, exec_lo
; GFX11-NEXT: s_cselect_b32 s2, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s2, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB9_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: s_add_u32 s0, s0, 3
@@ -963,8 +970,9 @@ define i64 @bitcast_v2f32_to_i64(<2 x float> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v2
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_dual_add_f32 v1, 1.0, v1 :: v_dual_add_f32 v0, 1.0, v0
@@ -1074,6 +1082,7 @@ define inreg i64 @bitcast_v2f32_to_i64_scalar(<2 x float> inreg %a, i32 inreg %b
; GFX11-NEXT: s_and_b32 s2, s2, exec_lo
; GFX11-NEXT: s_cselect_b32 s2, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s2, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB11_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_add_f32_e64 v1, s1, 1.0
@@ -1161,12 +1170,12 @@ define <4 x i16> @bitcast_i64_to_v4i16(i64 %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v2
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_add_co_u32 v0, vcc_lo, v0, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-NEXT: ; %bb.2: ; %end
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -1283,6 +1292,7 @@ define inreg <4 x i16> @bitcast_i64_to_v4i16_scalar(i64 inreg %a, i32 inreg %b)
; GFX11-NEXT: s_and_b32 s2, s2, exec_lo
; GFX11-NEXT: s_cselect_b32 s2, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s2, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB13_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: s_add_u32 s0, s0, 3
@@ -1390,8 +1400,9 @@ define i64 @bitcast_v4i16_to_i64(<4 x i16> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v2
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_pk_add_u16 v1, v1, 3 op_sel_hi:[1,0]
@@ -1525,6 +1536,7 @@ define inreg i64 @bitcast_v4i16_to_i64_scalar(<4 x i16> inreg %a, i32 inreg %b)
; GFX11-NEXT: s_and_b32 s2, s2, exec_lo
; GFX11-NEXT: s_cselect_b32 s2, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s2, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB15_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_pk_add_u16 v1, s1, 3 op_sel_hi:[1,0]
@@ -1612,12 +1624,12 @@ define <4 x half> @bitcast_i64_to_v4f16(i64 %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v2
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_add_co_u32 v0, vcc_lo, v0, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-NEXT: ; %bb.2: ; %end
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -1734,6 +1746,7 @@ define inreg <4 x half> @bitcast_i64_to_v4f16_scalar(i64 inreg %a, i32 inreg %b)
; GFX11-NEXT: s_and_b32 s2, s2, exec_lo
; GFX11-NEXT: s_cselect_b32 s2, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s2, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB17_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: s_add_u32 s0, s0, 3
@@ -1850,8 +1863,9 @@ define i64 @bitcast_v4f16_to_i64(<4 x half> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v2
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_pk_add_f16 v1, 0x200, v1 op_sel_hi:[0,1]
@@ -1995,6 +2009,7 @@ define inreg i64 @bitcast_v4f16_to_i64_scalar(<4 x half> inreg %a, i32 inreg %b)
; GFX11-NEXT: s_and_b32 s2, s2, exec_lo
; GFX11-NEXT: s_cselect_b32 s2, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s2, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB19_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_pk_add_f16 v1, 0x200, s1 op_sel_hi:[0,1]
@@ -2092,12 +2107,12 @@ define <4 x bfloat> @bitcast_i64_to_v4bf16(i64 %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v2
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_add_co_u32 v0, vcc_lo, v0, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-NEXT: ; %bb.2: ; %end
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -2220,6 +2235,7 @@ define inreg <4 x bfloat> @bitcast_i64_to_v4bf16_scalar(i64 inreg %a, i32 inreg
; GFX11-NEXT: s_and_b32 s2, s2, exec_lo
; GFX11-NEXT: s_cselect_b32 s2, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s2, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB21_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: s_add_u32 s0, s0, 3
@@ -2395,8 +2411,9 @@ define i64 @bitcast_v4bf16_to_i64(<4 x bfloat> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-TRUE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-TRUE16-NEXT: v_cmpx_ne_u32_e32 0, v2
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-TRUE16-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-TRUE16-NEXT: s_cbranch_execz .LBB22_2
; GFX11-TRUE16-NEXT: ; %bb.1: ; %cmp.true
@@ -2450,8 +2467,9 @@ define i64 @bitcast_v4bf16_to_i64(<4 x bfloat> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-FAKE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-FAKE16-NEXT: v_cmpx_ne_u32_e32 0, v2
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-FAKE16-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-FAKE16-NEXT: s_cbranch_execz .LBB22_2
; GFX11-FAKE16-NEXT: ; %bb.1: ; %cmp.true
@@ -2687,6 +2705,7 @@ define inreg i64 @bitcast_v4bf16_to_i64_scalar(<4 x bfloat> inreg %a, i32 inreg
; GFX11-TRUE16-NEXT: s_and_b32 s2, s2, exec_lo
; GFX11-TRUE16-NEXT: s_cselect_b32 s2, 1, 0
; GFX11-TRUE16-NEXT: s_cmp_lg_u32 s2, 1
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_cbranch_scc1 .LBB23_4
; GFX11-TRUE16-NEXT: ; %bb.3: ; %cmp.true
; GFX11-TRUE16-NEXT: s_pack_lh_b32_b16 s2, 0, s0
@@ -2747,6 +2766,7 @@ define inreg i64 @bitcast_v4bf16_to_i64_scalar(<4 x bfloat> inreg %a, i32 inreg
; GFX11-FAKE16-NEXT: s_and_b32 s2, s2, exec_lo
; GFX11-FAKE16-NEXT: s_cselect_b32 s2, 1, 0
; GFX11-FAKE16-NEXT: s_cmp_lg_u32 s2, 1
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_cbranch_scc1 .LBB23_4
; GFX11-FAKE16-NEXT: ; %bb.3: ; %cmp.true
; GFX11-FAKE16-NEXT: s_pack_lh_b32_b16 s2, 0, s0
@@ -2959,17 +2979,17 @@ define <8 x i8> @bitcast_i64_to_v8i8(i64 %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: s_cbranch_execz .LBB24_4
; GFX11-TRUE16-NEXT: ; %bb.3: ; %cmp.true
; GFX11-TRUE16-NEXT: v_add_co_u32 v3, vcc_lo, v3, 3
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v4, null, 0, v4, vcc_lo
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_3)
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v2, 16, v3
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v1, 8, v3
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX11-TRUE16-NEXT: v_lshrrev_b64 v[8:9], 24, v[3:4]
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v7, 24, v4
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v6, 16, v4
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v5, 8, v4
; GFX11-TRUE16-NEXT: .LBB24_4: ; %end
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v0.l, v3.l
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v3.l, v8.l
; GFX11-TRUE16-NEXT: s_setpc_b64 s[30:31]
@@ -3001,17 +3021,17 @@ define <8 x i8> @bitcast_i64_to_v8i8(i64 %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: s_cbranch_execz .LBB24_4
; GFX11-FAKE16-NEXT: ; %bb.3: ; %cmp.true
; GFX11-FAKE16-NEXT: v_add_co_u32 v8, vcc_lo, v8, 3
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v9, null, 0, v9, vcc_lo
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v2, 16, v8
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v1, 8, v8
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_lshrrev_b64 v[3:4], 24, v[8:9]
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v7, 24, v9
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v6, 16, v9
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v5, 8, v9
; GFX11-FAKE16-NEXT: .LBB24_4: ; %end
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v0, v8
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v4, v9
; GFX11-FAKE16-NEXT: s_setpc_b64 s[30:31]
@@ -3201,6 +3221,7 @@ define inreg <8 x i8> @bitcast_i64_to_v8i8_scalar(i64 inreg %a, i32 inreg %b) #0
; GFX11-NEXT: s_and_b32 s8, s8, exec_lo
; GFX11-NEXT: s_cselect_b32 s8, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s8, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB25_5
; GFX11-NEXT: ; %bb.4: ; %cmp.true
; GFX11-NEXT: s_add_u32 s0, s0, 3
@@ -3429,6 +3450,7 @@ define i64 @bitcast_v8i8_to_i64(<8 x i8> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-TRUE16-NEXT: ; implicit-def: $vgpr0_vgpr1
; GFX11-TRUE16-NEXT: v_cmpx_ne_u32_e32 0, v8
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB26_3
; GFX11-TRUE16-NEXT: ; %bb.1: ; %Flow
@@ -3489,6 +3511,7 @@ define i64 @bitcast_v8i8_to_i64(<8 x i8> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-FAKE16-NEXT: ; implicit-def: $vgpr0_vgpr1
; GFX11-FAKE16-NEXT: v_cmpx_ne_u32_e32 0, v8
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB26_3
; GFX11-FAKE16-NEXT: ; %bb.1: ; %Flow
@@ -3766,6 +3789,7 @@ define inreg i64 @bitcast_v8i8_to_i64_scalar(<8 x i8> inreg %a, i32 inreg %b) #0
; GFX11-NEXT: s_and_b32 s4, s4, exec_lo
; GFX11-NEXT: s_cselect_b32 s4, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s4, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB27_5
; GFX11-NEXT: ; %bb.4: ; %cmp.true
; GFX11-NEXT: v_mov_b32_e32 v0, 0xc0c0004
@@ -3856,8 +3880,9 @@ define <2 x i32> @bitcast_f64_to_v2i32(double %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v2
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB28_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -3965,6 +3990,7 @@ define inreg <2 x i32> @bitcast_f64_to_v2i32_scalar(double inreg %a, i32 inreg %
; GFX11-NEXT: s_and_b32 s2, s2, exec_lo
; GFX11-NEXT: s_cselect_b32 s2, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s2, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB29_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_add_f64 v[0:1], s[0:1], 1.0
@@ -4037,8 +4063,9 @@ define double @bitcast_v2i32_to_f64(<2 x i32> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v2
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_add_nc_u32_e32 v1, 3, v1
@@ -4146,6 +4173,7 @@ define inreg double @bitcast_v2i32_to_f64_scalar(<2 x i32> inreg %a, i32 inreg %
; GFX11-NEXT: s_and_b32 s2, s2, exec_lo
; GFX11-NEXT: s_cselect_b32 s2, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s2, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB31_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: s_add_i32 s1, s1, 3
@@ -4216,8 +4244,9 @@ define <2 x float> @bitcast_f64_to_v2f32(double %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v2
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB32_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -4325,6 +4354,7 @@ define inreg <2 x float> @bitcast_f64_to_v2f32_scalar(double inreg %a, i32 inreg
; GFX11-NEXT: s_and_b32 s2, s2, exec_lo
; GFX11-NEXT: s_cselect_b32 s2, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s2, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB33_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_add_f64 v[0:1], s[0:1], 1.0
@@ -4397,8 +4427,9 @@ define double @bitcast_v2f32_to_f64(<2 x float> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v2
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_dual_add_f32 v1, 1.0, v1 :: v_dual_add_f32 v0, 1.0, v0
@@ -4508,6 +4539,7 @@ define inreg double @bitcast_v2f32_to_f64_scalar(<2 x float> inreg %a, i32 inreg
; GFX11-NEXT: s_and_b32 s2, s2, exec_lo
; GFX11-NEXT: s_cselect_b32 s2, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s2, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB35_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_add_f32_e64 v1, s1, 1.0
@@ -4592,8 +4624,9 @@ define <4 x i16> @bitcast_f64_to_v4i16(double %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v2
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB36_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -4716,6 +4749,7 @@ define inreg <4 x i16> @bitcast_f64_to_v4i16_scalar(double inreg %a, i32 inreg %
; GFX11-NEXT: s_and_b32 s2, s2, exec_lo
; GFX11-NEXT: s_cselect_b32 s2, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s2, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB37_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_add_f64 v[0:1], s[0:1], 1.0
@@ -4822,8 +4856,9 @@ define double @bitcast_v4i16_to_f64(<4 x i16> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v2
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_pk_add_u16 v1, v1, 3 op_sel_hi:[1,0]
@@ -4957,6 +4992,7 @@ define inreg double @bitcast_v4i16_to_f64_scalar(<4 x i16> inreg %a, i32 inreg %
; GFX11-NEXT: s_and_b32 s2, s2, exec_lo
; GFX11-NEXT: s_cselect_b32 s2, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s2, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB39_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_pk_add_u16 v1, s1, 3 op_sel_hi:[1,0]
@@ -5041,8 +5077,9 @@ define <4 x half> @bitcast_f64_to_v4f16(double %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v2
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB40_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -5165,6 +5202,7 @@ define inreg <4 x half> @bitcast_f64_to_v4f16_scalar(double inreg %a, i32 inreg
; GFX11-NEXT: s_and_b32 s2, s2, exec_lo
; GFX11-NEXT: s_cselect_b32 s2, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s2, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB41_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_add_f64 v[0:1], s[0:1], 1.0
@@ -5280,8 +5318,9 @@ define double @bitcast_v4f16_to_f64(<4 x half> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v2
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_pk_add_f16 v1, 0x200, v1 op_sel_hi:[0,1]
@@ -5425,6 +5464,7 @@ define inreg double @bitcast_v4f16_to_f64_scalar(<4 x half> inreg %a, i32 inreg
; GFX11-NEXT: s_and_b32 s2, s2, exec_lo
; GFX11-NEXT: s_cselect_b32 s2, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s2, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB43_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_pk_add_f16 v1, 0x200, s1 op_sel_hi:[0,1]
@@ -5518,8 +5558,9 @@ define <4 x bfloat> @bitcast_f64_to_v4bf16(double %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v2
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB44_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -5650,6 +5691,7 @@ define inreg <4 x bfloat> @bitcast_f64_to_v4bf16_scalar(double inreg %a, i32 inr
; GFX11-NEXT: s_and_b32 s2, s2, exec_lo
; GFX11-NEXT: s_cselect_b32 s2, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s2, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB45_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_add_f64 v[0:1], s[0:1], 1.0
@@ -5824,8 +5866,9 @@ define double @bitcast_v4bf16_to_f64(<4 x bfloat> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-TRUE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-TRUE16-NEXT: v_cmpx_ne_u32_e32 0, v2
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-TRUE16-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-TRUE16-NEXT: s_cbranch_execz .LBB46_2
; GFX11-TRUE16-NEXT: ; %bb.1: ; %cmp.true
@@ -5879,8 +5922,9 @@ define double @bitcast_v4bf16_to_f64(<4 x bfloat> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-FAKE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-FAKE16-NEXT: v_cmpx_ne_u32_e32 0, v2
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-FAKE16-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-FAKE16-NEXT: s_cbranch_execz .LBB46_2
; GFX11-FAKE16-NEXT: ; %bb.1: ; %cmp.true
@@ -6116,6 +6160,7 @@ define inreg double @bitcast_v4bf16_to_f64_scalar(<4 x bfloat> inreg %a, i32 inr
; GFX11-TRUE16-NEXT: s_and_b32 s2, s2, exec_lo
; GFX11-TRUE16-NEXT: s_cselect_b32 s2, 1, 0
; GFX11-TRUE16-NEXT: s_cmp_lg_u32 s2, 1
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_cbranch_scc1 .LBB47_4
; GFX11-TRUE16-NEXT: ; %bb.3: ; %cmp.true
; GFX11-TRUE16-NEXT: s_pack_lh_b32_b16 s2, 0, s0
@@ -6176,6 +6221,7 @@ define inreg double @bitcast_v4bf16_to_f64_scalar(<4 x bfloat> inreg %a, i32 inr
; GFX11-FAKE16-NEXT: s_and_b32 s2, s2, exec_lo
; GFX11-FAKE16-NEXT: s_cselect_b32 s2, 1, 0
; GFX11-FAKE16-NEXT: s_cmp_lg_u32 s2, 1
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_cbranch_scc1 .LBB47_4
; GFX11-FAKE16-NEXT: ; %bb.3: ; %cmp.true
; GFX11-FAKE16-NEXT: s_pack_lh_b32_b16 s2, 0, s0
@@ -6393,6 +6439,7 @@ define <8 x i8> @bitcast_f64_to_v8i8(double %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v1, 8, v3
; GFX11-TRUE16-NEXT: .LBB48_4: ; %end
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v0.l, v3.l
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v3.l, v8.l
; GFX11-TRUE16-NEXT: s_setpc_b64 s[30:31]
@@ -6433,6 +6480,7 @@ define <8 x i8> @bitcast_f64_to_v8i8(double %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v1, 8, v8
; GFX11-FAKE16-NEXT: .LBB48_4: ; %end
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v0, v8
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v4, v9
; GFX11-FAKE16-NEXT: s_setpc_b64 s[30:31]
@@ -6632,6 +6680,7 @@ define inreg <8 x i8> @bitcast_f64_to_v8i8_scalar(double inreg %a, i32 inreg %b)
; GFX11-NEXT: s_and_b32 s4, s4, exec_lo
; GFX11-NEXT: s_cselect_b32 s4, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s4, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB49_5
; GFX11-NEXT: ; %bb.4: ; %cmp.true
; GFX11-NEXT: v_add_f64 v[8:9], s[0:1], 1.0
@@ -6864,6 +6913,7 @@ define double @bitcast_v8i8_to_f64(<8 x i8> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-TRUE16-NEXT: ; implicit-def: $vgpr0_vgpr1
; GFX11-TRUE16-NEXT: v_cmpx_ne_u32_e32 0, v8
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB50_3
; GFX11-TRUE16-NEXT: ; %bb.1: ; %Flow
@@ -6924,6 +6974,7 @@ define double @bitcast_v8i8_to_f64(<8 x i8> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-FAKE16-NEXT: ; implicit-def: $vgpr0_vgpr1
; GFX11-FAKE16-NEXT: v_cmpx_ne_u32_e32 0, v8
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB50_3
; GFX11-FAKE16-NEXT: ; %bb.1: ; %Flow
@@ -7201,6 +7252,7 @@ define inreg double @bitcast_v8i8_to_f64_scalar(<8 x i8> inreg %a, i32 inreg %b)
; GFX11-NEXT: s_and_b32 s4, s4, exec_lo
; GFX11-NEXT: s_cselect_b32 s4, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s4, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB51_5
; GFX11-NEXT: ; %bb.4: ; %cmp.true
; GFX11-NEXT: v_mov_b32_e32 v0, 0xc0c0004
@@ -7294,8 +7346,9 @@ define <2 x float> @bitcast_v2i32_to_v2f32(<2 x i32> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v2
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_add_nc_u32_e32 v1, 3, v1
@@ -7403,6 +7456,7 @@ define inreg <2 x float> @bitcast_v2i32_to_v2f32_scalar(<2 x i32> inreg %a, i32
; GFX11-NEXT: s_and_b32 s2, s2, exec_lo
; GFX11-NEXT: s_cselect_b32 s2, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s2, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB53_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: s_add_i32 s1, s1, 3
@@ -7476,8 +7530,9 @@ define <2 x i32> @bitcast_v2f32_to_v2i32(<2 x float> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v2
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_dual_add_f32 v1, 1.0, v1 :: v_dual_add_f32 v0, 1.0, v0
@@ -7587,6 +7642,7 @@ define inreg <2 x i32> @bitcast_v2f32_to_v2i32_scalar(<2 x float> inreg %a, i32
; GFX11-NEXT: s_and_b32 s2, s2, exec_lo
; GFX11-NEXT: s_cselect_b32 s2, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s2, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB55_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_add_f32_e64 v1, s1, 1.0
@@ -7674,8 +7730,9 @@ define <4 x i16> @bitcast_v2i32_to_v4i16(<2 x i32> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v2
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_add_nc_u32_e32 v1, 3, v1
@@ -7795,6 +7852,7 @@ define inreg <4 x i16> @bitcast_v2i32_to_v4i16_scalar(<2 x i32> inreg %a, i32 in
; GFX11-NEXT: s_and_b32 s2, s2, exec_lo
; GFX11-NEXT: s_cselect_b32 s2, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s2, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB57_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: s_add_i32 s1, s1, 3
@@ -7902,8 +7960,9 @@ define <2 x i32> @bitcast_v4i16_to_v2i32(<4 x i16> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v2
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_pk_add_u16 v1, v1, 3 op_sel_hi:[1,0]
@@ -8037,6 +8096,7 @@ define inreg <2 x i32> @bitcast_v4i16_to_v2i32_scalar(<4 x i16> inreg %a, i32 in
; GFX11-NEXT: s_and_b32 s2, s2, exec_lo
; GFX11-NEXT: s_cselect_b32 s2, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s2, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB59_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_pk_add_u16 v1, s1, 3 op_sel_hi:[1,0]
@@ -8124,8 +8184,9 @@ define <4 x half> @bitcast_v2i32_to_v4f16(<2 x i32> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v2
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_add_nc_u32_e32 v1, 3, v1
@@ -8245,6 +8306,7 @@ define inreg <4 x half> @bitcast_v2i32_to_v4f16_scalar(<2 x i32> inreg %a, i32 i
; GFX11-NEXT: s_and_b32 s2, s2, exec_lo
; GFX11-NEXT: s_cselect_b32 s2, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s2, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB61_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: s_add_i32 s1, s1, 3
@@ -8361,8 +8423,9 @@ define <2 x i32> @bitcast_v4f16_to_v2i32(<4 x half> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v2
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_pk_add_f16 v1, 0x200, v1 op_sel_hi:[0,1]
@@ -8506,6 +8569,7 @@ define inreg <2 x i32> @bitcast_v4f16_to_v2i32_scalar(<4 x half> inreg %a, i32 i
; GFX11-NEXT: s_and_b32 s2, s2, exec_lo
; GFX11-NEXT: s_cselect_b32 s2, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s2, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB63_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_pk_add_f16 v1, 0x200, s1 op_sel_hi:[0,1]
@@ -8603,8 +8667,9 @@ define <4 x bfloat> @bitcast_v2i32_to_v4bf16(<2 x i32> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v2
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_add_nc_u32_e32 v1, 3, v1
@@ -8730,6 +8795,7 @@ define inreg <4 x bfloat> @bitcast_v2i32_to_v4bf16_scalar(<2 x i32> inreg %a, i3
; GFX11-NEXT: s_and_b32 s2, s2, exec_lo
; GFX11-NEXT: s_cselect_b32 s2, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s2, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB65_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: s_add_i32 s1, s1, 3
@@ -8905,8 +8971,9 @@ define <2 x i32> @bitcast_v4bf16_to_v2i32(<4 x bfloat> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-TRUE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-TRUE16-NEXT: v_cmpx_ne_u32_e32 0, v2
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-TRUE16-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-TRUE16-NEXT: s_cbranch_execz .LBB66_2
; GFX11-TRUE16-NEXT: ; %bb.1: ; %cmp.true
@@ -8960,8 +9027,9 @@ define <2 x i32> @bitcast_v4bf16_to_v2i32(<4 x bfloat> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-FAKE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-FAKE16-NEXT: v_cmpx_ne_u32_e32 0, v2
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-FAKE16-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-FAKE16-NEXT: s_cbranch_execz .LBB66_2
; GFX11-FAKE16-NEXT: ; %bb.1: ; %cmp.true
@@ -9197,6 +9265,7 @@ define inreg <2 x i32> @bitcast_v4bf16_to_v2i32_scalar(<4 x bfloat> inreg %a, i3
; GFX11-TRUE16-NEXT: s_and_b32 s2, s2, exec_lo
; GFX11-TRUE16-NEXT: s_cselect_b32 s2, 1, 0
; GFX11-TRUE16-NEXT: s_cmp_lg_u32 s2, 1
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_cbranch_scc1 .LBB67_4
; GFX11-TRUE16-NEXT: ; %bb.3: ; %cmp.true
; GFX11-TRUE16-NEXT: s_pack_lh_b32_b16 s2, 0, s0
@@ -9257,6 +9326,7 @@ define inreg <2 x i32> @bitcast_v4bf16_to_v2i32_scalar(<4 x bfloat> inreg %a, i3
; GFX11-FAKE16-NEXT: s_and_b32 s2, s2, exec_lo
; GFX11-FAKE16-NEXT: s_cselect_b32 s2, 1, 0
; GFX11-FAKE16-NEXT: s_cmp_lg_u32 s2, 1
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_cbranch_scc1 .LBB67_4
; GFX11-FAKE16-NEXT: ; %bb.3: ; %cmp.true
; GFX11-FAKE16-NEXT: s_pack_lh_b32_b16 s2, 0, s0
@@ -9479,6 +9549,7 @@ define <8 x i8> @bitcast_v2i32_to_v8i8(<2 x i32> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v1, 8, v3
; GFX11-TRUE16-NEXT: .LBB68_4: ; %end
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v0.l, v3.l
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v3.l, v8.l
; GFX11-TRUE16-NEXT: s_setpc_b64 s[30:31]
@@ -9520,6 +9591,7 @@ define <8 x i8> @bitcast_v2i32_to_v8i8(<2 x i32> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v1, 8, v8
; GFX11-FAKE16-NEXT: .LBB68_4: ; %end
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v0, v8
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v4, v9
; GFX11-FAKE16-NEXT: s_setpc_b64 s[30:31]
@@ -9709,6 +9781,7 @@ define inreg <8 x i8> @bitcast_v2i32_to_v8i8_scalar(<2 x i32> inreg %a, i32 inre
; GFX11-NEXT: s_and_b32 s8, s8, exec_lo
; GFX11-NEXT: s_cselect_b32 s8, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s8, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB69_5
; GFX11-NEXT: ; %bb.4: ; %cmp.true
; GFX11-NEXT: s_add_i32 s1, s1, 3
@@ -9937,6 +10010,7 @@ define <2 x i32> @bitcast_v8i8_to_v2i32(<8 x i8> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-TRUE16-NEXT: ; implicit-def: $vgpr0_vgpr1
; GFX11-TRUE16-NEXT: v_cmpx_ne_u32_e32 0, v8
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB70_3
; GFX11-TRUE16-NEXT: ; %bb.1: ; %Flow
@@ -9997,6 +10071,7 @@ define <2 x i32> @bitcast_v8i8_to_v2i32(<8 x i8> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-FAKE16-NEXT: ; implicit-def: $vgpr0_vgpr1
; GFX11-FAKE16-NEXT: v_cmpx_ne_u32_e32 0, v8
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB70_3
; GFX11-FAKE16-NEXT: ; %bb.1: ; %Flow
@@ -10274,6 +10349,7 @@ define inreg <2 x i32> @bitcast_v8i8_to_v2i32_scalar(<8 x i8> inreg %a, i32 inre
; GFX11-NEXT: s_and_b32 s4, s4, exec_lo
; GFX11-NEXT: s_cselect_b32 s4, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s4, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB71_5
; GFX11-NEXT: ; %bb.4: ; %cmp.true
; GFX11-NEXT: v_mov_b32_e32 v0, 0xc0c0004
@@ -10381,8 +10457,9 @@ define <4 x i16> @bitcast_v2f32_to_v4i16(<2 x float> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v2
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_dual_add_f32 v1, 1.0, v1 :: v_dual_add_f32 v0, 1.0, v0
@@ -10507,6 +10584,7 @@ define inreg <4 x i16> @bitcast_v2f32_to_v4i16_scalar(<2 x float> inreg %a, i32
; GFX11-NEXT: s_and_b32 s2, s2, exec_lo
; GFX11-NEXT: s_cselect_b32 s2, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s2, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB73_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_add_f32_e64 v1, s1, 1.0
@@ -10614,8 +10692,9 @@ define <2 x float> @bitcast_v4i16_to_v2f32(<4 x i16> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v2
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_pk_add_u16 v1, v1, 3 op_sel_hi:[1,0]
@@ -10749,6 +10828,7 @@ define inreg <2 x float> @bitcast_v4i16_to_v2f32_scalar(<4 x i16> inreg %a, i32
; GFX11-NEXT: s_and_b32 s2, s2, exec_lo
; GFX11-NEXT: s_cselect_b32 s2, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s2, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB75_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_pk_add_u16 v1, s1, 3 op_sel_hi:[1,0]
@@ -10836,8 +10916,9 @@ define <4 x half> @bitcast_v2f32_to_v4f16(<2 x float> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v2
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_dual_add_f32 v1, 1.0, v1 :: v_dual_add_f32 v0, 1.0, v0
@@ -10962,6 +11043,7 @@ define inreg <4 x half> @bitcast_v2f32_to_v4f16_scalar(<2 x float> inreg %a, i32
; GFX11-NEXT: s_and_b32 s2, s2, exec_lo
; GFX11-NEXT: s_cselect_b32 s2, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s2, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB77_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_add_f32_e64 v1, s1, 1.0
@@ -11078,8 +11160,9 @@ define <2 x float> @bitcast_v4f16_to_v2f32(<4 x half> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v2
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_pk_add_f16 v1, 0x200, v1 op_sel_hi:[0,1]
@@ -11223,6 +11306,7 @@ define inreg <2 x float> @bitcast_v4f16_to_v2f32_scalar(<4 x half> inreg %a, i32
; GFX11-NEXT: s_and_b32 s2, s2, exec_lo
; GFX11-NEXT: s_cselect_b32 s2, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s2, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB79_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_pk_add_f16 v1, 0x200, s1 op_sel_hi:[0,1]
@@ -11320,8 +11404,9 @@ define <4 x bfloat> @bitcast_v2f32_to_v4bf16(<2 x float> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v2
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_dual_add_f32 v1, 1.0, v1 :: v_dual_add_f32 v0, 1.0, v0
@@ -11454,6 +11539,7 @@ define inreg <4 x bfloat> @bitcast_v2f32_to_v4bf16_scalar(<2 x float> inreg %a,
; GFX11-NEXT: s_and_b32 s2, s2, exec_lo
; GFX11-NEXT: s_cselect_b32 s2, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s2, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB81_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_add_f32_e64 v1, s1, 1.0
@@ -11629,8 +11715,9 @@ define <2 x float> @bitcast_v4bf16_to_v2f32(<4 x bfloat> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-TRUE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-TRUE16-NEXT: v_cmpx_ne_u32_e32 0, v2
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-TRUE16-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-TRUE16-NEXT: s_cbranch_execz .LBB82_2
; GFX11-TRUE16-NEXT: ; %bb.1: ; %cmp.true
@@ -11684,8 +11771,9 @@ define <2 x float> @bitcast_v4bf16_to_v2f32(<4 x bfloat> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-FAKE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-FAKE16-NEXT: v_cmpx_ne_u32_e32 0, v2
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-FAKE16-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-FAKE16-NEXT: s_cbranch_execz .LBB82_2
; GFX11-FAKE16-NEXT: ; %bb.1: ; %cmp.true
@@ -11921,6 +12009,7 @@ define inreg <2 x float> @bitcast_v4bf16_to_v2f32_scalar(<4 x bfloat> inreg %a,
; GFX11-TRUE16-NEXT: s_and_b32 s2, s2, exec_lo
; GFX11-TRUE16-NEXT: s_cselect_b32 s2, 1, 0
; GFX11-TRUE16-NEXT: s_cmp_lg_u32 s2, 1
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_cbranch_scc1 .LBB83_4
; GFX11-TRUE16-NEXT: ; %bb.3: ; %cmp.true
; GFX11-TRUE16-NEXT: s_pack_lh_b32_b16 s2, 0, s0
@@ -11981,6 +12070,7 @@ define inreg <2 x float> @bitcast_v4bf16_to_v2f32_scalar(<4 x bfloat> inreg %a,
; GFX11-FAKE16-NEXT: s_and_b32 s2, s2, exec_lo
; GFX11-FAKE16-NEXT: s_cselect_b32 s2, 1, 0
; GFX11-FAKE16-NEXT: s_cmp_lg_u32 s2, 1
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_cbranch_scc1 .LBB83_4
; GFX11-FAKE16-NEXT: ; %bb.3: ; %cmp.true
; GFX11-FAKE16-NEXT: s_pack_lh_b32_b16 s2, 0, s0
@@ -12202,6 +12292,7 @@ define <8 x i8> @bitcast_v2f32_to_v8i8(<2 x float> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v1, 8, v3
; GFX11-TRUE16-NEXT: .LBB84_4: ; %end
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v0.l, v3.l
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v3.l, v8.l
; GFX11-TRUE16-NEXT: s_setpc_b64 s[30:31]
@@ -12242,6 +12333,7 @@ define <8 x i8> @bitcast_v2f32_to_v8i8(<2 x float> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v1, 8, v8
; GFX11-FAKE16-NEXT: .LBB84_4: ; %end
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v0, v8
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v4, v9
; GFX11-FAKE16-NEXT: s_setpc_b64 s[30:31]
@@ -12444,6 +12536,7 @@ define inreg <8 x i8> @bitcast_v2f32_to_v8i8_scalar(<2 x float> inreg %a, i32 in
; GFX11-NEXT: s_and_b32 s4, s4, exec_lo
; GFX11-NEXT: s_cselect_b32 s4, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s4, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB85_5
; GFX11-NEXT: ; %bb.4: ; %cmp.true
; GFX11-NEXT: v_add_f32_e64 v9, s1, 1.0
@@ -12677,6 +12770,7 @@ define <2 x float> @bitcast_v8i8_to_v2f32(<8 x i8> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-TRUE16-NEXT: ; implicit-def: $vgpr0_vgpr1
; GFX11-TRUE16-NEXT: v_cmpx_ne_u32_e32 0, v8
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB86_3
; GFX11-TRUE16-NEXT: ; %bb.1: ; %Flow
@@ -12737,6 +12831,7 @@ define <2 x float> @bitcast_v8i8_to_v2f32(<8 x i8> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-FAKE16-NEXT: ; implicit-def: $vgpr0_vgpr1
; GFX11-FAKE16-NEXT: v_cmpx_ne_u32_e32 0, v8
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB86_3
; GFX11-FAKE16-NEXT: ; %bb.1: ; %Flow
@@ -13014,6 +13109,7 @@ define inreg <2 x float> @bitcast_v8i8_to_v2f32_scalar(<8 x i8> inreg %a, i32 in
; GFX11-NEXT: s_and_b32 s4, s4, exec_lo
; GFX11-NEXT: s_cselect_b32 s4, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s4, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB87_5
; GFX11-NEXT: ; %bb.4: ; %cmp.true
; GFX11-NEXT: v_mov_b32_e32 v0, 0xc0c0004
@@ -13145,8 +13241,9 @@ define <4 x half> @bitcast_v4i16_to_v4f16(<4 x i16> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v2
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_pk_add_u16 v1, v1, 3 op_sel_hi:[1,0]
@@ -13291,6 +13388,7 @@ define inreg <4 x half> @bitcast_v4i16_to_v4f16_scalar(<4 x i16> inreg %a, i32 i
; GFX11-NEXT: s_and_b32 s2, s2, exec_lo
; GFX11-NEXT: s_cselect_b32 s2, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s2, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB89_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_pk_add_u16 v1, s1, 3 op_sel_hi:[1,0]
@@ -13394,8 +13492,9 @@ define <4 x i16> @bitcast_v4f16_to_v4i16(<4 x half> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v2
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_pk_add_f16 v1, 0x200, v1 op_sel_hi:[0,1]
@@ -13542,6 +13641,7 @@ define inreg <4 x i16> @bitcast_v4f16_to_v4i16_scalar(<4 x half> inreg %a, i32 i
; GFX11-NEXT: s_and_b32 s2, s2, exec_lo
; GFX11-NEXT: s_cselect_b32 s2, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s2, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB91_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_pk_add_f16 v1, 0x200, s1 op_sel_hi:[0,1]
@@ -13651,8 +13751,9 @@ define <4 x bfloat> @bitcast_v4i16_to_v4bf16(<4 x i16> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v2
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_pk_add_u16 v1, v1, 3 op_sel_hi:[1,0]
@@ -13797,6 +13898,7 @@ define inreg <4 x bfloat> @bitcast_v4i16_to_v4bf16_scalar(<4 x i16> inreg %a, i3
; GFX11-NEXT: s_and_b32 s2, s2, exec_lo
; GFX11-NEXT: s_cselect_b32 s2, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s2, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB93_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_pk_add_u16 v1, s1, 3 op_sel_hi:[1,0]
@@ -13978,8 +14080,9 @@ define <4 x i16> @bitcast_v4bf16_to_v4i16(<4 x bfloat> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-TRUE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-TRUE16-NEXT: v_cmpx_ne_u32_e32 0, v2
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-TRUE16-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-TRUE16-NEXT: s_cbranch_execz .LBB94_2
; GFX11-TRUE16-NEXT: ; %bb.1: ; %cmp.true
@@ -14036,8 +14139,9 @@ define <4 x i16> @bitcast_v4bf16_to_v4i16(<4 x bfloat> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-FAKE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-FAKE16-NEXT: v_cmpx_ne_u32_e32 0, v2
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-FAKE16-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-FAKE16-NEXT: s_cbranch_execz .LBB94_2
; GFX11-FAKE16-NEXT: ; %bb.1: ; %cmp.true
@@ -14284,6 +14388,7 @@ define inreg <4 x i16> @bitcast_v4bf16_to_v4i16_scalar(<4 x bfloat> inreg %a, i3
; GFX11-TRUE16-NEXT: s_and_b32 s2, s2, exec_lo
; GFX11-TRUE16-NEXT: s_cselect_b32 s2, 1, 0
; GFX11-TRUE16-NEXT: s_cmp_lg_u32 s2, 1
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_cbranch_scc1 .LBB95_4
; GFX11-TRUE16-NEXT: ; %bb.3: ; %cmp.true
; GFX11-TRUE16-NEXT: s_pack_lh_b32_b16 s2, 0, s1
@@ -14339,6 +14444,7 @@ define inreg <4 x i16> @bitcast_v4bf16_to_v4i16_scalar(<4 x bfloat> inreg %a, i3
; GFX11-FAKE16-NEXT: s_and_b32 s2, s2, exec_lo
; GFX11-FAKE16-NEXT: s_cselect_b32 s2, 1, 0
; GFX11-FAKE16-NEXT: s_cmp_lg_u32 s2, 1
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_cbranch_scc1 .LBB95_4
; GFX11-FAKE16-NEXT: ; %bb.3: ; %cmp.true
; GFX11-FAKE16-NEXT: s_pack_lh_b32_b16 s2, 0, s1
@@ -14582,6 +14688,7 @@ define <8 x i8> @bitcast_v4i16_to_v8i8(<4 x i16> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v1, 8, v3
; GFX11-TRUE16-NEXT: .LBB96_4: ; %end
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v0.l, v3.l
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v3.l, v8.l
; GFX11-TRUE16-NEXT: s_setpc_b64 s[30:31]
@@ -14623,6 +14730,7 @@ define <8 x i8> @bitcast_v4i16_to_v8i8(<4 x i16> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v1, 8, v8
; GFX11-FAKE16-NEXT: .LBB96_4: ; %end
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v0, v8
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v4, v9
; GFX11-FAKE16-NEXT: s_setpc_b64 s[30:31]
@@ -14841,6 +14949,7 @@ define inreg <8 x i8> @bitcast_v4i16_to_v8i8_scalar(<4 x i16> inreg %a, i32 inre
; GFX11-NEXT: s_and_b32 s4, s4, exec_lo
; GFX11-NEXT: s_cselect_b32 s4, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s4, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB97_5
; GFX11-NEXT: ; %bb.4: ; %cmp.true
; GFX11-NEXT: v_pk_add_u16 v9, s1, 3 op_sel_hi:[1,0]
@@ -15082,6 +15191,7 @@ define <4 x i16> @bitcast_v8i8_to_v4i16(<8 x i8> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-TRUE16-NEXT: ; implicit-def: $vgpr0_vgpr1
; GFX11-TRUE16-NEXT: v_cmpx_ne_u32_e32 0, v8
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB98_3
; GFX11-TRUE16-NEXT: ; %bb.1: ; %Flow
@@ -15142,6 +15252,7 @@ define <4 x i16> @bitcast_v8i8_to_v4i16(<8 x i8> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-FAKE16-NEXT: ; implicit-def: $vgpr0_vgpr1
; GFX11-FAKE16-NEXT: v_cmpx_ne_u32_e32 0, v8
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB98_3
; GFX11-FAKE16-NEXT: ; %bb.1: ; %Flow
@@ -15432,6 +15543,7 @@ define inreg <4 x i16> @bitcast_v8i8_to_v4i16_scalar(<8 x i8> inreg %a, i32 inre
; GFX11-NEXT: s_and_b32 s4, s4, exec_lo
; GFX11-NEXT: s_cselect_b32 s4, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s4, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB99_5
; GFX11-NEXT: ; %bb.4: ; %cmp.true
; GFX11-NEXT: v_mov_b32_e32 v0, 0xc0c0004
@@ -15570,8 +15682,9 @@ define <4 x bfloat> @bitcast_v4f16_to_v4bf16(<4 x half> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v2
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_pk_add_f16 v1, 0x200, v1 op_sel_hi:[0,1]
@@ -15727,6 +15840,7 @@ define inreg <4 x bfloat> @bitcast_v4f16_to_v4bf16_scalar(<4 x half> inreg %a, i
; GFX11-NEXT: s_and_b32 s2, s2, exec_lo
; GFX11-NEXT: s_cselect_b32 s2, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s2, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB101_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_pk_add_f16 v1, 0x200, s1 op_sel_hi:[0,1]
@@ -15910,8 +16024,9 @@ define <4 x half> @bitcast_v4bf16_to_v4f16(<4 x bfloat> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-TRUE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-TRUE16-NEXT: v_cmpx_ne_u32_e32 0, v2
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-TRUE16-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-TRUE16-NEXT: s_cbranch_execz .LBB102_2
; GFX11-TRUE16-NEXT: ; %bb.1: ; %cmp.true
@@ -15964,8 +16079,9 @@ define <4 x half> @bitcast_v4bf16_to_v4f16(<4 x bfloat> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-FAKE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-FAKE16-NEXT: v_cmpx_ne_u32_e32 0, v2
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-FAKE16-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-FAKE16-NEXT: s_cbranch_execz .LBB102_2
; GFX11-FAKE16-NEXT: ; %bb.1: ; %cmp.true
@@ -16214,6 +16330,7 @@ define inreg <4 x half> @bitcast_v4bf16_to_v4f16_scalar(<4 x bfloat> inreg %a, i
; GFX11-TRUE16-NEXT: s_and_b32 s2, s2, exec_lo
; GFX11-TRUE16-NEXT: s_cselect_b32 s2, 1, 0
; GFX11-TRUE16-NEXT: s_cmp_lg_u32 s2, 1
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_cbranch_scc1 .LBB103_4
; GFX11-TRUE16-NEXT: ; %bb.3: ; %cmp.true
; GFX11-TRUE16-NEXT: s_pack_lh_b32_b16 s2, 0, s1
@@ -16276,6 +16393,7 @@ define inreg <4 x half> @bitcast_v4bf16_to_v4f16_scalar(<4 x bfloat> inreg %a, i
; GFX11-FAKE16-NEXT: s_and_b32 s2, s2, exec_lo
; GFX11-FAKE16-NEXT: s_cselect_b32 s2, 1, 0
; GFX11-FAKE16-NEXT: s_cmp_lg_u32 s2, 1
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_cbranch_scc1 .LBB103_4
; GFX11-FAKE16-NEXT: ; %bb.3: ; %cmp.true
; GFX11-FAKE16-NEXT: s_pack_lh_b32_b16 s2, 0, s1
@@ -16528,6 +16646,7 @@ define <8 x i8> @bitcast_v4f16_to_v8i8(<4 x half> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v1, 8, v3
; GFX11-TRUE16-NEXT: .LBB104_4: ; %end
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v0.l, v3.l
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v3.l, v8.l
; GFX11-TRUE16-NEXT: s_setpc_b64 s[30:31]
@@ -16569,6 +16688,7 @@ define <8 x i8> @bitcast_v4f16_to_v8i8(<4 x half> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v1, 8, v8
; GFX11-FAKE16-NEXT: .LBB104_4: ; %end
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v0, v8
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v4, v9
; GFX11-FAKE16-NEXT: s_setpc_b64 s[30:31]
@@ -16797,6 +16917,7 @@ define inreg <8 x i8> @bitcast_v4f16_to_v8i8_scalar(<4 x half> inreg %a, i32 inr
; GFX11-NEXT: s_and_b32 s4, s4, exec_lo
; GFX11-NEXT: s_cselect_b32 s4, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s4, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB105_5
; GFX11-NEXT: ; %bb.4: ; %cmp.true
; GFX11-NEXT: v_pk_add_f16 v9, 0x200, s1 op_sel_hi:[0,1]
@@ -17038,6 +17159,7 @@ define <4 x half> @bitcast_v8i8_to_v4f16(<8 x i8> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-TRUE16-NEXT: ; implicit-def: $vgpr0_vgpr1
; GFX11-TRUE16-NEXT: v_cmpx_ne_u32_e32 0, v8
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB106_3
; GFX11-TRUE16-NEXT: ; %bb.1: ; %Flow
@@ -17098,6 +17220,7 @@ define <4 x half> @bitcast_v8i8_to_v4f16(<8 x i8> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-FAKE16-NEXT: ; implicit-def: $vgpr0_vgpr1
; GFX11-FAKE16-NEXT: v_cmpx_ne_u32_e32 0, v8
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB106_3
; GFX11-FAKE16-NEXT: ; %bb.1: ; %Flow
@@ -17388,6 +17511,7 @@ define inreg <4 x half> @bitcast_v8i8_to_v4f16_scalar(<8 x i8> inreg %a, i32 inr
; GFX11-NEXT: s_and_b32 s4, s4, exec_lo
; GFX11-NEXT: s_cselect_b32 s4, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s4, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB107_5
; GFX11-NEXT: ; %bb.4: ; %cmp.true
; GFX11-NEXT: v_mov_b32_e32 v0, 0xc0c0004
@@ -17725,8 +17849,8 @@ define <8 x i8> @bitcast_v4bf16_to_v8i8(<4 x bfloat> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v5, 8, v4
; GFX11-TRUE16-NEXT: .LBB108_4: ; %end
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(VALU_DEP_4)
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v0.l, v3.l
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_4)
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v3.l, v8.l
; GFX11-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -17799,6 +17923,7 @@ define <8 x i8> @bitcast_v4bf16_to_v8i8(<4 x bfloat> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v5, 8, v11
; GFX11-FAKE16-NEXT: .LBB108_4: ; %end
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v0, v8
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v4, v9
; GFX11-FAKE16-NEXT: s_setpc_b64 s[30:31]
@@ -18082,6 +18207,7 @@ define inreg <8 x i8> @bitcast_v4bf16_to_v8i8_scalar(<4 x bfloat> inreg %a, i32
; GFX11-TRUE16-NEXT: s_and_b32 s4, s4, exec_lo
; GFX11-TRUE16-NEXT: s_cselect_b32 s4, 1, 0
; GFX11-TRUE16-NEXT: s_cmp_lg_u32 s4, 1
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_cbranch_scc1 .LBB109_5
; GFX11-TRUE16-NEXT: ; %bb.4: ; %cmp.true
; GFX11-TRUE16-NEXT: s_pack_lh_b32_b16 s2, 0, s0
@@ -18169,6 +18295,7 @@ define inreg <8 x i8> @bitcast_v4bf16_to_v8i8_scalar(<4 x bfloat> inreg %a, i32
; GFX11-FAKE16-NEXT: s_and_b32 s4, s4, exec_lo
; GFX11-FAKE16-NEXT: s_cselect_b32 s4, 1, 0
; GFX11-FAKE16-NEXT: s_cmp_lg_u32 s4, 1
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_cbranch_scc1 .LBB109_5
; GFX11-FAKE16-NEXT: ; %bb.4: ; %cmp.true
; GFX11-FAKE16-NEXT: s_pack_lh_b32_b16 s2, 0, s0
@@ -18448,6 +18575,7 @@ define <4 x bfloat> @bitcast_v8i8_to_v4bf16(<8 x i8> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-TRUE16-NEXT: ; implicit-def: $vgpr0_vgpr1
; GFX11-TRUE16-NEXT: v_cmpx_ne_u32_e32 0, v8
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB110_3
; GFX11-TRUE16-NEXT: ; %bb.1: ; %Flow
@@ -18508,6 +18636,7 @@ define <4 x bfloat> @bitcast_v8i8_to_v4bf16(<8 x i8> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-FAKE16-NEXT: ; implicit-def: $vgpr0_vgpr1
; GFX11-FAKE16-NEXT: v_cmpx_ne_u32_e32 0, v8
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB110_3
; GFX11-FAKE16-NEXT: ; %bb.1: ; %Flow
@@ -18796,6 +18925,7 @@ define inreg <4 x bfloat> @bitcast_v8i8_to_v4bf16_scalar(<8 x i8> inreg %a, i32
; GFX11-NEXT: s_and_b32 s4, s4, exec_lo
; GFX11-NEXT: s_cselect_b32 s4, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s4, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB111_5
; GFX11-NEXT: ; %bb.4: ; %cmp.true
; GFX11-NEXT: v_mov_b32_e32 v0, 0xc0c0004
diff --git a/llvm/test/CodeGen/AMDGPU/amdgcn.bitcast.704bit.ll b/llvm/test/CodeGen/AMDGPU/amdgcn.bitcast.704bit.ll
index 2a5574397fad8e..a527bd4a7b5b7f 100644
--- a/llvm/test/CodeGen/AMDGPU/amdgcn.bitcast.704bit.ll
+++ b/llvm/test/CodeGen/AMDGPU/amdgcn.bitcast.704bit.ll
@@ -117,8 +117,9 @@ define <22 x float> @bitcast_v22i32_to_v22f32(<22 x i32> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v22
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB0_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -399,6 +400,7 @@ define inreg <22 x float> @bitcast_v22i32_to_v22f32_scalar(<22 x i32> inreg %a,
; GFX11-NEXT: s_and_b32 s8, s8, exec_lo
; GFX11-NEXT: s_cselect_b32 s8, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s8, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB1_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: s_add_i32 s4, s4, 3
@@ -565,8 +567,9 @@ define <22 x i32> @bitcast_v22f32_to_v22i32(<22 x float> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v22
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB2_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -1041,6 +1044,7 @@ define inreg <22 x i32> @bitcast_v22f32_to_v22i32_scalar(<22 x float> inreg %a,
; GFX11-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB3_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_add_f32_e64 v21, s57, 1.0
@@ -1229,8 +1233,9 @@ define <11 x i64> @bitcast_v22i32_to_v11i64(<22 x i32> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v22
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB4_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -1511,6 +1516,7 @@ define inreg <11 x i64> @bitcast_v22i32_to_v11i64_scalar(<22 x i32> inreg %a, i3
; GFX11-NEXT: s_and_b32 s8, s8, exec_lo
; GFX11-NEXT: s_cselect_b32 s8, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s8, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB5_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: s_add_i32 s4, s4, 3
@@ -1677,38 +1683,33 @@ define <22 x i32> @bitcast_v11i64_to_v22i32(<11 x i64> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v22
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB6_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_add_co_u32 v20, vcc_lo, v20, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v21, null, 0, v21, vcc_lo
; GFX11-NEXT: v_add_co_u32 v18, vcc_lo, v18, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v19, null, 0, v19, vcc_lo
; GFX11-NEXT: v_add_co_u32 v16, vcc_lo, v16, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v17, null, 0, v17, vcc_lo
; GFX11-NEXT: v_add_co_u32 v14, vcc_lo, v14, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v15, null, 0, v15, vcc_lo
; GFX11-NEXT: v_add_co_u32 v12, vcc_lo, v12, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v13, null, 0, v13, vcc_lo
; GFX11-NEXT: v_add_co_u32 v10, vcc_lo, v10, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v11, null, 0, v11, vcc_lo
; GFX11-NEXT: v_add_co_u32 v8, vcc_lo, v8, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v9, null, 0, v9, vcc_lo
; GFX11-NEXT: v_add_co_u32 v6, vcc_lo, v6, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v7, null, 0, v7, vcc_lo
; GFX11-NEXT: v_add_co_u32 v4, vcc_lo, v4, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v5, null, 0, v5, vcc_lo
; GFX11-NEXT: v_add_co_u32 v2, vcc_lo, v2, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX11-NEXT: v_add_co_u32 v0, vcc_lo, v0, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-NEXT: .LBB6_2: ; %end
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -1965,6 +1966,7 @@ define inreg <22 x i32> @bitcast_v11i64_to_v22i32_scalar(<11 x i64> inreg %a, i3
; GFX11-NEXT: s_and_b32 s8, s8, exec_lo
; GFX11-NEXT: s_cselect_b32 s8, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s8, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB7_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: s_add_u32 s5, s5, 3
@@ -2131,8 +2133,9 @@ define <11 x double> @bitcast_v22i32_to_v11f64(<22 x i32> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v22
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB8_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -2413,6 +2416,7 @@ define inreg <11 x double> @bitcast_v22i32_to_v11f64_scalar(<22 x i32> inreg %a,
; GFX11-NEXT: s_and_b32 s8, s8, exec_lo
; GFX11-NEXT: s_cselect_b32 s8, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s8, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB9_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: s_add_i32 s4, s4, 3
@@ -2546,8 +2550,9 @@ define <22 x i32> @bitcast_v11f64_to_v22i32(<11 x double> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v22
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB10_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -2989,6 +2994,7 @@ define inreg <22 x i32> @bitcast_v11f64_to_v22i32_scalar(<11 x double> inreg %a,
; GFX11-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB11_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_add_f64 v[20:21], s[56:57], 1.0
@@ -3604,8 +3610,9 @@ define <44 x i16> @bitcast_v22i32_to_v44i16(<22 x i32> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v51, 16, v0
; GFX11-TRUE16-NEXT: .LBB12_4: ; %end
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v0.h, v51.l
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v1.h, v50.l
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v2.h, v49.l
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v3.h, v48.l
@@ -3732,8 +3739,9 @@ define <44 x i16> @bitcast_v22i32_to_v44i16(<22 x i32> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v51, 16, v0
; GFX11-FAKE16-NEXT: .LBB12_4: ; %end
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_perm_b32 v0, v51, v0, 0x5040100
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_perm_b32 v1, v50, v1, 0x5040100
; GFX11-FAKE16-NEXT: v_perm_b32 v2, v49, v2, 0x5040100
; GFX11-FAKE16-NEXT: v_perm_b32 v3, v48, v3, 0x5040100
@@ -4406,6 +4414,7 @@ define inreg <44 x i16> @bitcast_v22i32_to_v44i16_scalar(<22 x i32> inreg %a, i3
; GFX11-NEXT: s_and_b32 s62, s62, exec_lo
; GFX11-NEXT: s_cselect_b32 s62, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s62, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB13_5
; GFX11-NEXT: ; %bb.4: ; %cmp.true
; GFX11-NEXT: s_add_i32 s4, s4, 3
@@ -5348,6 +5357,7 @@ define <22 x i32> @bitcast_v44i16_to_v22i32(<44 x i16> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v52, 16, v0
; GFX11-TRUE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-TRUE16-NEXT: v_cmpx_ne_u32_e32 0, v22
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB14_3
; GFX11-TRUE16-NEXT: ; %bb.1: ; %Flow
@@ -6362,6 +6372,7 @@ define inreg <22 x i32> @bitcast_v44i16_to_v22i32_scalar(<44 x i16> inreg %a, i3
; GFX11-NEXT: s_and_b32 s40, s40, exec_lo
; GFX11-NEXT: s_cselect_b32 s40, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s40, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB15_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_pk_add_u16 v0, s0, 3 op_sel_hi:[1,0]
@@ -6971,8 +6982,9 @@ define <44 x half> @bitcast_v22i32_to_v44f16(<22 x i32> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v51, 16, v0
; GFX11-TRUE16-NEXT: .LBB16_4: ; %end
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v0.h, v51.l
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v1.h, v50.l
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v2.h, v49.l
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v3.h, v48.l
@@ -7099,8 +7111,9 @@ define <44 x half> @bitcast_v22i32_to_v44f16(<22 x i32> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v51, 16, v0
; GFX11-FAKE16-NEXT: .LBB16_4: ; %end
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_perm_b32 v0, v51, v0, 0x5040100
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_perm_b32 v1, v50, v1, 0x5040100
; GFX11-FAKE16-NEXT: v_perm_b32 v2, v49, v2, 0x5040100
; GFX11-FAKE16-NEXT: v_perm_b32 v3, v48, v3, 0x5040100
@@ -7773,6 +7786,7 @@ define inreg <44 x half> @bitcast_v22i32_to_v44f16_scalar(<22 x i32> inreg %a, i
; GFX11-NEXT: s_and_b32 s62, s62, exec_lo
; GFX11-NEXT: s_cselect_b32 s62, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s62, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB17_5
; GFX11-NEXT: ; %bb.4: ; %cmp.true
; GFX11-NEXT: s_add_i32 s4, s4, 3
@@ -8814,6 +8828,7 @@ define <22 x i32> @bitcast_v44f16_to_v22i32(<44 x half> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v52, 16, v0
; GFX11-TRUE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-TRUE16-NEXT: v_cmpx_ne_u32_e32 0, v22
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB18_3
; GFX11-TRUE16-NEXT: ; %bb.1: ; %Flow
@@ -9898,6 +9913,7 @@ define inreg <22 x i32> @bitcast_v44f16_to_v22i32_scalar(<44 x half> inreg %a, i
; GFX11-NEXT: s_and_b32 s40, s40, exec_lo
; GFX11-NEXT: s_cselect_b32 s40, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s40, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB19_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_pk_add_f16 v0, 0x200, s0 op_sel_hi:[0,1]
@@ -10069,8 +10085,9 @@ define <11 x i64> @bitcast_v22f32_to_v11i64(<22 x float> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v22
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB20_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -10545,6 +10562,7 @@ define inreg <11 x i64> @bitcast_v22f32_to_v11i64_scalar(<22 x float> inreg %a,
; GFX11-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB21_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_add_f32_e64 v21, s57, 1.0
@@ -10733,38 +10751,33 @@ define <22 x float> @bitcast_v11i64_to_v22f32(<11 x i64> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v22
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB22_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_add_co_u32 v20, vcc_lo, v20, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v21, null, 0, v21, vcc_lo
; GFX11-NEXT: v_add_co_u32 v18, vcc_lo, v18, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v19, null, 0, v19, vcc_lo
; GFX11-NEXT: v_add_co_u32 v16, vcc_lo, v16, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v17, null, 0, v17, vcc_lo
; GFX11-NEXT: v_add_co_u32 v14, vcc_lo, v14, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v15, null, 0, v15, vcc_lo
; GFX11-NEXT: v_add_co_u32 v12, vcc_lo, v12, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v13, null, 0, v13, vcc_lo
; GFX11-NEXT: v_add_co_u32 v10, vcc_lo, v10, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v11, null, 0, v11, vcc_lo
; GFX11-NEXT: v_add_co_u32 v8, vcc_lo, v8, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v9, null, 0, v9, vcc_lo
; GFX11-NEXT: v_add_co_u32 v6, vcc_lo, v6, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v7, null, 0, v7, vcc_lo
; GFX11-NEXT: v_add_co_u32 v4, vcc_lo, v4, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v5, null, 0, v5, vcc_lo
; GFX11-NEXT: v_add_co_u32 v2, vcc_lo, v2, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX11-NEXT: v_add_co_u32 v0, vcc_lo, v0, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-NEXT: .LBB22_2: ; %end
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -11021,6 +11034,7 @@ define inreg <22 x float> @bitcast_v11i64_to_v22f32_scalar(<11 x i64> inreg %a,
; GFX11-NEXT: s_and_b32 s8, s8, exec_lo
; GFX11-NEXT: s_cselect_b32 s8, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s8, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB23_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: s_add_u32 s5, s5, 3
@@ -11187,8 +11201,9 @@ define <11 x double> @bitcast_v22f32_to_v11f64(<22 x float> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v22
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB24_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -11663,6 +11678,7 @@ define inreg <11 x double> @bitcast_v22f32_to_v11f64_scalar(<22 x float> inreg %
; GFX11-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB25_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_add_f32_e64 v21, s57, 1.0
@@ -11818,8 +11834,9 @@ define <22 x float> @bitcast_v11f64_to_v22f32(<11 x double> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v22
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB26_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -12261,6 +12278,7 @@ define inreg <22 x float> @bitcast_v11f64_to_v22f32_scalar(<11 x double> inreg %
; GFX11-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB27_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_add_f64 v[20:21], s[56:57], 1.0
@@ -12865,8 +12883,9 @@ define <44 x i16> @bitcast_v22f32_to_v44i16(<22 x float> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v51, 16, v0
; GFX11-TRUE16-NEXT: .LBB28_4: ; %end
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v0.h, v51.l
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v1.h, v50.l
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v2.h, v49.l
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v3.h, v48.l
@@ -12982,8 +13001,9 @@ define <44 x i16> @bitcast_v22f32_to_v44i16(<22 x float> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v51, 16, v0
; GFX11-FAKE16-NEXT: .LBB28_4: ; %end
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_perm_b32 v0, v51, v0, 0x5040100
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_perm_b32 v1, v50, v1, 0x5040100
; GFX11-FAKE16-NEXT: v_perm_b32 v2, v49, v2, 0x5040100
; GFX11-FAKE16-NEXT: v_perm_b32 v3, v48, v3, 0x5040100
@@ -13728,6 +13748,7 @@ define inreg <44 x i16> @bitcast_v22f32_to_v44i16_scalar(<22 x float> inreg %a,
; GFX11-TRUE16-NEXT: s_and_b32 s10, s10, exec_lo
; GFX11-TRUE16-NEXT: s_cselect_b32 s10, 1, 0
; GFX11-TRUE16-NEXT: s_cmp_lg_u32 s10, 1
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_cbranch_scc1 .LBB29_5
; GFX11-TRUE16-NEXT: ; %bb.4: ; %cmp.true
; GFX11-TRUE16-NEXT: v_add_f32_e64 v21, s4, 1.0
@@ -13888,6 +13909,7 @@ define inreg <44 x i16> @bitcast_v22f32_to_v44i16_scalar(<22 x float> inreg %a,
; GFX11-FAKE16-NEXT: s_and_b32 s10, s10, exec_lo
; GFX11-FAKE16-NEXT: s_cselect_b32 s10, 1, 0
; GFX11-FAKE16-NEXT: s_cmp_lg_u32 s10, 1
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_cbranch_scc1 .LBB29_5
; GFX11-FAKE16-NEXT: ; %bb.4: ; %cmp.true
; GFX11-FAKE16-NEXT: v_add_f32_e64 v17, s4, 1.0
@@ -14864,6 +14886,7 @@ define <22 x float> @bitcast_v44i16_to_v22f32(<44 x i16> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v52, 16, v0
; GFX11-TRUE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-TRUE16-NEXT: v_cmpx_ne_u32_e32 0, v22
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB30_3
; GFX11-TRUE16-NEXT: ; %bb.1: ; %Flow
@@ -15878,6 +15901,7 @@ define inreg <22 x float> @bitcast_v44i16_to_v22f32_scalar(<44 x i16> inreg %a,
; GFX11-NEXT: s_and_b32 s40, s40, exec_lo
; GFX11-NEXT: s_cselect_b32 s40, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s40, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB31_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_pk_add_u16 v0, s0, 3 op_sel_hi:[1,0]
@@ -16476,8 +16500,9 @@ define <44 x half> @bitcast_v22f32_to_v44f16(<22 x float> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v51, 16, v0
; GFX11-TRUE16-NEXT: .LBB32_4: ; %end
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v0.h, v51.l
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v1.h, v50.l
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v2.h, v49.l
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v3.h, v48.l
@@ -16593,8 +16618,9 @@ define <44 x half> @bitcast_v22f32_to_v44f16(<22 x float> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v51, 16, v0
; GFX11-FAKE16-NEXT: .LBB32_4: ; %end
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_perm_b32 v0, v51, v0, 0x5040100
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_perm_b32 v1, v50, v1, 0x5040100
; GFX11-FAKE16-NEXT: v_perm_b32 v2, v49, v2, 0x5040100
; GFX11-FAKE16-NEXT: v_perm_b32 v3, v48, v3, 0x5040100
@@ -17339,6 +17365,7 @@ define inreg <44 x half> @bitcast_v22f32_to_v44f16_scalar(<22 x float> inreg %a,
; GFX11-TRUE16-NEXT: s_and_b32 s10, s10, exec_lo
; GFX11-TRUE16-NEXT: s_cselect_b32 s10, 1, 0
; GFX11-TRUE16-NEXT: s_cmp_lg_u32 s10, 1
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_cbranch_scc1 .LBB33_5
; GFX11-TRUE16-NEXT: ; %bb.4: ; %cmp.true
; GFX11-TRUE16-NEXT: v_add_f32_e64 v21, s4, 1.0
@@ -17499,6 +17526,7 @@ define inreg <44 x half> @bitcast_v22f32_to_v44f16_scalar(<22 x float> inreg %a,
; GFX11-FAKE16-NEXT: s_and_b32 s10, s10, exec_lo
; GFX11-FAKE16-NEXT: s_cselect_b32 s10, 1, 0
; GFX11-FAKE16-NEXT: s_cmp_lg_u32 s10, 1
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_cbranch_scc1 .LBB33_5
; GFX11-FAKE16-NEXT: ; %bb.4: ; %cmp.true
; GFX11-FAKE16-NEXT: v_add_f32_e64 v17, s4, 1.0
@@ -18574,6 +18602,7 @@ define <22 x float> @bitcast_v44f16_to_v22f32(<44 x half> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v52, 16, v0
; GFX11-TRUE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-TRUE16-NEXT: v_cmpx_ne_u32_e32 0, v22
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB34_3
; GFX11-TRUE16-NEXT: ; %bb.1: ; %Flow
@@ -19658,6 +19687,7 @@ define inreg <22 x float> @bitcast_v44f16_to_v22f32_scalar(<44 x half> inreg %a,
; GFX11-NEXT: s_and_b32 s40, s40, exec_lo
; GFX11-NEXT: s_cselect_b32 s40, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s40, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB35_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_pk_add_f16 v0, 0x200, s0 op_sel_hi:[0,1]
@@ -19829,38 +19859,33 @@ define <11 x double> @bitcast_v11i64_to_v11f64(<11 x i64> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v22
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB36_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_add_co_u32 v0, vcc_lo, v0, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-NEXT: v_add_co_u32 v2, vcc_lo, v2, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX11-NEXT: v_add_co_u32 v4, vcc_lo, v4, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v5, null, 0, v5, vcc_lo
; GFX11-NEXT: v_add_co_u32 v6, vcc_lo, v6, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v7, null, 0, v7, vcc_lo
; GFX11-NEXT: v_add_co_u32 v8, vcc_lo, v8, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v9, null, 0, v9, vcc_lo
; GFX11-NEXT: v_add_co_u32 v10, vcc_lo, v10, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v11, null, 0, v11, vcc_lo
; GFX11-NEXT: v_add_co_u32 v12, vcc_lo, v12, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v13, null, 0, v13, vcc_lo
; GFX11-NEXT: v_add_co_u32 v14, vcc_lo, v14, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v15, null, 0, v15, vcc_lo
; GFX11-NEXT: v_add_co_u32 v16, vcc_lo, v16, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v17, null, 0, v17, vcc_lo
; GFX11-NEXT: v_add_co_u32 v18, vcc_lo, v18, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v19, null, 0, v19, vcc_lo
; GFX11-NEXT: v_add_co_u32 v20, vcc_lo, v20, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v21, null, 0, v21, vcc_lo
; GFX11-NEXT: .LBB36_2: ; %end
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -20117,6 +20142,7 @@ define inreg <11 x double> @bitcast_v11i64_to_v11f64_scalar(<11 x i64> inreg %a,
; GFX11-NEXT: s_and_b32 s8, s8, exec_lo
; GFX11-NEXT: s_cselect_b32 s8, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s8, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB37_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: s_add_u32 s0, s0, 3
@@ -20249,8 +20275,9 @@ define <11 x i64> @bitcast_v11f64_to_v11i64(<11 x double> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v22
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB38_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -20692,6 +20719,7 @@ define inreg <11 x i64> @bitcast_v11f64_to_v11i64_scalar(<11 x double> inreg %a,
; GFX11-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB39_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_add_f64 v[0:1], s[36:37], 1.0
@@ -21262,32 +21290,26 @@ define <44 x i16> @bitcast_v11i64_to_v44i16(<11 x i64> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: s_cbranch_execz .LBB40_4
; GFX11-TRUE16-NEXT: ; %bb.3: ; %cmp.true
; GFX11-TRUE16-NEXT: v_add_co_u32 v20, vcc_lo, v20, 3
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v21, null, 0, v21, vcc_lo
; GFX11-TRUE16-NEXT: v_add_co_u32 v18, vcc_lo, v18, 3
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v19, null, 0, v19, vcc_lo
; GFX11-TRUE16-NEXT: v_add_co_u32 v16, vcc_lo, v16, 3
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v17, null, 0, v17, vcc_lo
; GFX11-TRUE16-NEXT: v_add_co_u32 v14, vcc_lo, v14, 3
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v15, null, 0, v15, vcc_lo
; GFX11-TRUE16-NEXT: v_add_co_u32 v12, vcc_lo, v12, 3
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v13, null, 0, v13, vcc_lo
; GFX11-TRUE16-NEXT: v_add_co_u32 v10, vcc_lo, v10, 3
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v11, null, 0, v11, vcc_lo
; GFX11-TRUE16-NEXT: v_add_co_u32 v8, vcc_lo, v8, 3
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v9, null, 0, v9, vcc_lo
; GFX11-TRUE16-NEXT: v_add_co_u32 v6, vcc_lo, v6, 3
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v7, null, 0, v7, vcc_lo
; GFX11-TRUE16-NEXT: v_add_co_u32 v4, vcc_lo, v4, 3
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v5, null, 0, v5, vcc_lo
; GFX11-TRUE16-NEXT: v_add_co_u32 v2, vcc_lo, v2, 3
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX11-TRUE16-NEXT: v_add_co_u32 v0, vcc_lo, v0, 3
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v22, 16, v21
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v23, 16, v20
@@ -21313,8 +21335,9 @@ define <44 x i16> @bitcast_v11i64_to_v44i16(<11 x i64> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v51, 16, v0
; GFX11-TRUE16-NEXT: .LBB40_4: ; %end
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v0.h, v51.l
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v1.h, v50.l
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v2.h, v49.l
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v3.h, v48.l
@@ -21396,32 +21419,26 @@ define <44 x i16> @bitcast_v11i64_to_v44i16(<11 x i64> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: s_cbranch_execz .LBB40_4
; GFX11-FAKE16-NEXT: ; %bb.3: ; %cmp.true
; GFX11-FAKE16-NEXT: v_add_co_u32 v20, vcc_lo, v20, 3
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v21, null, 0, v21, vcc_lo
; GFX11-FAKE16-NEXT: v_add_co_u32 v18, vcc_lo, v18, 3
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v19, null, 0, v19, vcc_lo
; GFX11-FAKE16-NEXT: v_add_co_u32 v16, vcc_lo, v16, 3
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v17, null, 0, v17, vcc_lo
; GFX11-FAKE16-NEXT: v_add_co_u32 v14, vcc_lo, v14, 3
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v15, null, 0, v15, vcc_lo
; GFX11-FAKE16-NEXT: v_add_co_u32 v12, vcc_lo, v12, 3
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v13, null, 0, v13, vcc_lo
; GFX11-FAKE16-NEXT: v_add_co_u32 v10, vcc_lo, v10, 3
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v11, null, 0, v11, vcc_lo
; GFX11-FAKE16-NEXT: v_add_co_u32 v8, vcc_lo, v8, 3
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v9, null, 0, v9, vcc_lo
; GFX11-FAKE16-NEXT: v_add_co_u32 v6, vcc_lo, v6, 3
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v7, null, 0, v7, vcc_lo
; GFX11-FAKE16-NEXT: v_add_co_u32 v4, vcc_lo, v4, 3
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v5, null, 0, v5, vcc_lo
; GFX11-FAKE16-NEXT: v_add_co_u32 v2, vcc_lo, v2, 3
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX11-FAKE16-NEXT: v_add_co_u32 v0, vcc_lo, v0, 3
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v22, 16, v21
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v23, 16, v20
@@ -21447,8 +21464,9 @@ define <44 x i16> @bitcast_v11i64_to_v44i16(<11 x i64> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v51, 16, v0
; GFX11-FAKE16-NEXT: .LBB40_4: ; %end
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_perm_b32 v0, v51, v0, 0x5040100
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_perm_b32 v1, v50, v1, 0x5040100
; GFX11-FAKE16-NEXT: v_perm_b32 v2, v49, v2, 0x5040100
; GFX11-FAKE16-NEXT: v_perm_b32 v3, v48, v3, 0x5040100
@@ -22121,6 +22139,7 @@ define inreg <44 x i16> @bitcast_v11i64_to_v44i16_scalar(<11 x i64> inreg %a, i3
; GFX11-NEXT: s_and_b32 s62, s62, exec_lo
; GFX11-NEXT: s_cselect_b32 s62, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s62, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB41_5
; GFX11-NEXT: ; %bb.4: ; %cmp.true
; GFX11-NEXT: s_add_u32 s5, s5, 3
@@ -23063,6 +23082,7 @@ define <11 x i64> @bitcast_v44i16_to_v11i64(<44 x i16> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v52, 16, v0
; GFX11-TRUE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-TRUE16-NEXT: v_cmpx_ne_u32_e32 0, v22
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB42_3
; GFX11-TRUE16-NEXT: ; %bb.1: ; %Flow
@@ -24077,6 +24097,7 @@ define inreg <11 x i64> @bitcast_v44i16_to_v11i64_scalar(<44 x i16> inreg %a, i3
; GFX11-NEXT: s_and_b32 s40, s40, exec_lo
; GFX11-NEXT: s_cselect_b32 s40, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s40, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB43_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_pk_add_u16 v0, s0, 3 op_sel_hi:[1,0]
@@ -24641,32 +24662,26 @@ define <44 x half> @bitcast_v11i64_to_v44f16(<11 x i64> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: s_cbranch_execz .LBB44_4
; GFX11-TRUE16-NEXT: ; %bb.3: ; %cmp.true
; GFX11-TRUE16-NEXT: v_add_co_u32 v20, vcc_lo, v20, 3
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v21, null, 0, v21, vcc_lo
; GFX11-TRUE16-NEXT: v_add_co_u32 v18, vcc_lo, v18, 3
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v19, null, 0, v19, vcc_lo
; GFX11-TRUE16-NEXT: v_add_co_u32 v16, vcc_lo, v16, 3
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v17, null, 0, v17, vcc_lo
; GFX11-TRUE16-NEXT: v_add_co_u32 v14, vcc_lo, v14, 3
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v15, null, 0, v15, vcc_lo
; GFX11-TRUE16-NEXT: v_add_co_u32 v12, vcc_lo, v12, 3
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v13, null, 0, v13, vcc_lo
; GFX11-TRUE16-NEXT: v_add_co_u32 v10, vcc_lo, v10, 3
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v11, null, 0, v11, vcc_lo
; GFX11-TRUE16-NEXT: v_add_co_u32 v8, vcc_lo, v8, 3
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v9, null, 0, v9, vcc_lo
; GFX11-TRUE16-NEXT: v_add_co_u32 v6, vcc_lo, v6, 3
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v7, null, 0, v7, vcc_lo
; GFX11-TRUE16-NEXT: v_add_co_u32 v4, vcc_lo, v4, 3
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v5, null, 0, v5, vcc_lo
; GFX11-TRUE16-NEXT: v_add_co_u32 v2, vcc_lo, v2, 3
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX11-TRUE16-NEXT: v_add_co_u32 v0, vcc_lo, v0, 3
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v22, 16, v21
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v23, 16, v20
@@ -24692,8 +24707,9 @@ define <44 x half> @bitcast_v11i64_to_v44f16(<11 x i64> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v51, 16, v0
; GFX11-TRUE16-NEXT: .LBB44_4: ; %end
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v0.h, v51.l
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v1.h, v50.l
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v2.h, v49.l
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v3.h, v48.l
@@ -24775,32 +24791,26 @@ define <44 x half> @bitcast_v11i64_to_v44f16(<11 x i64> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: s_cbranch_execz .LBB44_4
; GFX11-FAKE16-NEXT: ; %bb.3: ; %cmp.true
; GFX11-FAKE16-NEXT: v_add_co_u32 v20, vcc_lo, v20, 3
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v21, null, 0, v21, vcc_lo
; GFX11-FAKE16-NEXT: v_add_co_u32 v18, vcc_lo, v18, 3
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v19, null, 0, v19, vcc_lo
; GFX11-FAKE16-NEXT: v_add_co_u32 v16, vcc_lo, v16, 3
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v17, null, 0, v17, vcc_lo
; GFX11-FAKE16-NEXT: v_add_co_u32 v14, vcc_lo, v14, 3
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v15, null, 0, v15, vcc_lo
; GFX11-FAKE16-NEXT: v_add_co_u32 v12, vcc_lo, v12, 3
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v13, null, 0, v13, vcc_lo
; GFX11-FAKE16-NEXT: v_add_co_u32 v10, vcc_lo, v10, 3
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v11, null, 0, v11, vcc_lo
; GFX11-FAKE16-NEXT: v_add_co_u32 v8, vcc_lo, v8, 3
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v9, null, 0, v9, vcc_lo
; GFX11-FAKE16-NEXT: v_add_co_u32 v6, vcc_lo, v6, 3
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v7, null, 0, v7, vcc_lo
; GFX11-FAKE16-NEXT: v_add_co_u32 v4, vcc_lo, v4, 3
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v5, null, 0, v5, vcc_lo
; GFX11-FAKE16-NEXT: v_add_co_u32 v2, vcc_lo, v2, 3
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX11-FAKE16-NEXT: v_add_co_u32 v0, vcc_lo, v0, 3
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v22, 16, v21
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v23, 16, v20
@@ -24826,8 +24836,9 @@ define <44 x half> @bitcast_v11i64_to_v44f16(<11 x i64> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v51, 16, v0
; GFX11-FAKE16-NEXT: .LBB44_4: ; %end
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_perm_b32 v0, v51, v0, 0x5040100
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_perm_b32 v1, v50, v1, 0x5040100
; GFX11-FAKE16-NEXT: v_perm_b32 v2, v49, v2, 0x5040100
; GFX11-FAKE16-NEXT: v_perm_b32 v3, v48, v3, 0x5040100
@@ -25500,6 +25511,7 @@ define inreg <44 x half> @bitcast_v11i64_to_v44f16_scalar(<11 x i64> inreg %a, i
; GFX11-NEXT: s_and_b32 s62, s62, exec_lo
; GFX11-NEXT: s_cselect_b32 s62, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s62, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB45_5
; GFX11-NEXT: ; %bb.4: ; %cmp.true
; GFX11-NEXT: s_add_u32 s5, s5, 3
@@ -26541,6 +26553,7 @@ define <11 x i64> @bitcast_v44f16_to_v11i64(<44 x half> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v52, 16, v0
; GFX11-TRUE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-TRUE16-NEXT: v_cmpx_ne_u32_e32 0, v22
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB46_3
; GFX11-TRUE16-NEXT: ; %bb.1: ; %Flow
@@ -27625,6 +27638,7 @@ define inreg <11 x i64> @bitcast_v44f16_to_v11i64_scalar(<44 x half> inreg %a, i
; GFX11-NEXT: s_and_b32 s40, s40, exec_lo
; GFX11-NEXT: s_cselect_b32 s40, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s40, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB47_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_pk_add_f16 v0, 0x200, s0 op_sel_hi:[0,1]
@@ -28190,8 +28204,9 @@ define <44 x i16> @bitcast_v11f64_to_v44i16(<11 x double> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v51, 16, v0
; GFX11-TRUE16-NEXT: .LBB48_4: ; %end
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v0.h, v51.l
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v1.h, v50.l
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v2.h, v49.l
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v3.h, v48.l
@@ -28307,8 +28322,9 @@ define <44 x i16> @bitcast_v11f64_to_v44i16(<11 x double> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v51, 16, v0
; GFX11-FAKE16-NEXT: .LBB48_4: ; %end
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_perm_b32 v0, v51, v0, 0x5040100
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_perm_b32 v1, v50, v1, 0x5040100
; GFX11-FAKE16-NEXT: v_perm_b32 v2, v49, v2, 0x5040100
; GFX11-FAKE16-NEXT: v_perm_b32 v3, v48, v3, 0x5040100
@@ -29020,6 +29036,7 @@ define inreg <44 x i16> @bitcast_v11f64_to_v44i16_scalar(<11 x double> inreg %a,
; GFX11-TRUE16-NEXT: s_and_b32 s10, s10, exec_lo
; GFX11-TRUE16-NEXT: s_cselect_b32 s10, 1, 0
; GFX11-TRUE16-NEXT: s_cmp_lg_u32 s10, 1
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_cbranch_scc1 .LBB49_5
; GFX11-TRUE16-NEXT: ; %bb.4: ; %cmp.true
; GFX11-TRUE16-NEXT: v_add_f64 v[20:21], s[4:5], 1.0
@@ -29169,6 +29186,7 @@ define inreg <44 x i16> @bitcast_v11f64_to_v44i16_scalar(<11 x double> inreg %a,
; GFX11-FAKE16-NEXT: s_and_b32 s10, s10, exec_lo
; GFX11-FAKE16-NEXT: s_cselect_b32 s10, 1, 0
; GFX11-FAKE16-NEXT: s_cmp_lg_u32 s10, 1
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_cbranch_scc1 .LBB49_5
; GFX11-FAKE16-NEXT: ; %bb.4: ; %cmp.true
; GFX11-FAKE16-NEXT: v_add_f64 v[17:18], s[4:5], 1.0
@@ -30134,6 +30152,7 @@ define <11 x double> @bitcast_v44i16_to_v11f64(<44 x i16> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v52, 16, v0
; GFX11-TRUE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-TRUE16-NEXT: v_cmpx_ne_u32_e32 0, v22
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB50_3
; GFX11-TRUE16-NEXT: ; %bb.1: ; %Flow
@@ -31148,6 +31167,7 @@ define inreg <11 x double> @bitcast_v44i16_to_v11f64_scalar(<44 x i16> inreg %a,
; GFX11-NEXT: s_and_b32 s40, s40, exec_lo
; GFX11-NEXT: s_cselect_b32 s40, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s40, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB51_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_pk_add_u16 v0, s0, 3 op_sel_hi:[1,0]
@@ -31713,8 +31733,9 @@ define <44 x half> @bitcast_v11f64_to_v44f16(<11 x double> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v51, 16, v0
; GFX11-TRUE16-NEXT: .LBB52_4: ; %end
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v0.h, v51.l
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v1.h, v50.l
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v2.h, v49.l
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v3.h, v48.l
@@ -31830,8 +31851,9 @@ define <44 x half> @bitcast_v11f64_to_v44f16(<11 x double> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v51, 16, v0
; GFX11-FAKE16-NEXT: .LBB52_4: ; %end
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_perm_b32 v0, v51, v0, 0x5040100
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_perm_b32 v1, v50, v1, 0x5040100
; GFX11-FAKE16-NEXT: v_perm_b32 v2, v49, v2, 0x5040100
; GFX11-FAKE16-NEXT: v_perm_b32 v3, v48, v3, 0x5040100
@@ -32543,6 +32565,7 @@ define inreg <44 x half> @bitcast_v11f64_to_v44f16_scalar(<11 x double> inreg %a
; GFX11-TRUE16-NEXT: s_and_b32 s10, s10, exec_lo
; GFX11-TRUE16-NEXT: s_cselect_b32 s10, 1, 0
; GFX11-TRUE16-NEXT: s_cmp_lg_u32 s10, 1
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_cbranch_scc1 .LBB53_5
; GFX11-TRUE16-NEXT: ; %bb.4: ; %cmp.true
; GFX11-TRUE16-NEXT: v_add_f64 v[20:21], s[4:5], 1.0
@@ -32692,6 +32715,7 @@ define inreg <44 x half> @bitcast_v11f64_to_v44f16_scalar(<11 x double> inreg %a
; GFX11-FAKE16-NEXT: s_and_b32 s10, s10, exec_lo
; GFX11-FAKE16-NEXT: s_cselect_b32 s10, 1, 0
; GFX11-FAKE16-NEXT: s_cmp_lg_u32 s10, 1
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_cbranch_scc1 .LBB53_5
; GFX11-FAKE16-NEXT: ; %bb.4: ; %cmp.true
; GFX11-FAKE16-NEXT: v_add_f64 v[17:18], s[4:5], 1.0
@@ -33756,6 +33780,7 @@ define <11 x double> @bitcast_v44f16_to_v11f64(<44 x half> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v52, 16, v0
; GFX11-TRUE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-TRUE16-NEXT: v_cmpx_ne_u32_e32 0, v22
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB54_3
; GFX11-TRUE16-NEXT: ; %bb.1: ; %Flow
@@ -34840,6 +34865,7 @@ define inreg <11 x double> @bitcast_v44f16_to_v11f64_scalar(<44 x half> inreg %a
; GFX11-NEXT: s_and_b32 s40, s40, exec_lo
; GFX11-NEXT: s_cselect_b32 s40, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s40, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB55_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_pk_add_f16 v0, 0x200, s0 op_sel_hi:[0,1]
@@ -35732,8 +35758,9 @@ define <44 x half> @bitcast_v44i16_to_v44f16(<44 x i16> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v27, 16, v0
; GFX11-TRUE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-TRUE16-NEXT: v_cmpx_ne_u32_e32 0, v22
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-TRUE16-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-TRUE16-NEXT: s_cbranch_execz .LBB56_2
; GFX11-TRUE16-NEXT: ; %bb.1: ; %cmp.true
@@ -35805,6 +35832,7 @@ define <44 x half> @bitcast_v44i16_to_v44f16(<44 x i16> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v52, 16, v21
; GFX11-TRUE16-NEXT: .LBB56_2: ; %end
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v0.h, v27.l
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v1.h, v26.l
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v2.h, v25.l
@@ -35856,8 +35884,9 @@ define <44 x half> @bitcast_v44i16_to_v44f16(<44 x i16> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v23, 16, v0
; GFX11-FAKE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-FAKE16-NEXT: v_cmpx_ne_u32_e32 0, v22
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-FAKE16-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-FAKE16-NEXT: s_cbranch_execz .LBB56_2
; GFX11-FAKE16-NEXT: ; %bb.1: ; %cmp.true
@@ -35929,6 +35958,7 @@ define <44 x half> @bitcast_v44i16_to_v44f16(<44 x i16> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v52, 16, v21
; GFX11-FAKE16-NEXT: .LBB56_2: ; %end
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_perm_b32 v0, v23, v0, 0x5040100
; GFX11-FAKE16-NEXT: v_perm_b32 v1, v24, v1, 0x5040100
; GFX11-FAKE16-NEXT: v_perm_b32 v2, v25, v2, 0x5040100
@@ -36850,6 +36880,7 @@ define inreg <44 x half> @bitcast_v44i16_to_v44f16_scalar(<44 x i16> inreg %a, i
; GFX11-TRUE16-NEXT: s_and_b32 s62, s62, exec_lo
; GFX11-TRUE16-NEXT: s_cselect_b32 s62, 1, 0
; GFX11-TRUE16-NEXT: s_cmp_lg_u32 s62, 1
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_cbranch_scc1 .LBB57_4
; GFX11-TRUE16-NEXT: ; %bb.3: ; %cmp.true
; GFX11-TRUE16-NEXT: s_pack_ll_b32_b16 s57, s61, s57
@@ -37009,6 +37040,7 @@ define inreg <44 x half> @bitcast_v44i16_to_v44f16_scalar(<44 x i16> inreg %a, i
; GFX11-FAKE16-NEXT: s_and_b32 s62, s62, exec_lo
; GFX11-FAKE16-NEXT: s_cselect_b32 s62, 1, 0
; GFX11-FAKE16-NEXT: s_cmp_lg_u32 s62, 1
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_cbranch_scc1 .LBB57_4
; GFX11-FAKE16-NEXT: ; %bb.3: ; %cmp.true
; GFX11-FAKE16-NEXT: s_pack_ll_b32_b16 s57, s61, s57
@@ -37729,8 +37761,9 @@ define <44 x i16> @bitcast_v44f16_to_v44i16(<44 x half> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v27, 16, v0
; GFX11-TRUE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-TRUE16-NEXT: v_cmpx_ne_u32_e32 0, v22
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-TRUE16-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-TRUE16-NEXT: s_cbranch_execz .LBB58_2
; GFX11-TRUE16-NEXT: ; %bb.1: ; %cmp.true
@@ -37802,6 +37835,7 @@ define <44 x i16> @bitcast_v44f16_to_v44i16(<44 x half> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v52, 16, v21
; GFX11-TRUE16-NEXT: .LBB58_2: ; %end
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v0.h, v27.l
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v1.h, v26.l
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v2.h, v25.l
@@ -37853,8 +37887,9 @@ define <44 x i16> @bitcast_v44f16_to_v44i16(<44 x half> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v23, 16, v0
; GFX11-FAKE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-FAKE16-NEXT: v_cmpx_ne_u32_e32 0, v22
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-FAKE16-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-FAKE16-NEXT: s_cbranch_execz .LBB58_2
; GFX11-FAKE16-NEXT: ; %bb.1: ; %cmp.true
@@ -37926,6 +37961,7 @@ define <44 x i16> @bitcast_v44f16_to_v44i16(<44 x half> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v52, 16, v21
; GFX11-FAKE16-NEXT: .LBB58_2: ; %end
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_perm_b32 v0, v23, v0, 0x5040100
; GFX11-FAKE16-NEXT: v_perm_b32 v1, v24, v1, 0x5040100
; GFX11-FAKE16-NEXT: v_perm_b32 v2, v25, v2, 0x5040100
@@ -38794,6 +38830,7 @@ define inreg <44 x i16> @bitcast_v44f16_to_v44i16_scalar(<44 x half> inreg %a, i
; GFX11-TRUE16-NEXT: s_and_b32 s62, s62, exec_lo
; GFX11-TRUE16-NEXT: s_cselect_b32 s62, 1, 0
; GFX11-TRUE16-NEXT: s_cmp_lg_u32 s62, 1
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_cbranch_scc1 .LBB59_4
; GFX11-TRUE16-NEXT: ; %bb.3: ; %cmp.true
; GFX11-TRUE16-NEXT: s_pack_ll_b32_b16 s57, s61, s57
@@ -38953,6 +38990,7 @@ define inreg <44 x i16> @bitcast_v44f16_to_v44i16_scalar(<44 x half> inreg %a, i
; GFX11-FAKE16-NEXT: s_and_b32 s62, s62, exec_lo
; GFX11-FAKE16-NEXT: s_cselect_b32 s62, 1, 0
; GFX11-FAKE16-NEXT: s_cmp_lg_u32 s62, 1
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_cbranch_scc1 .LBB59_4
; GFX11-FAKE16-NEXT: ; %bb.3: ; %cmp.true
; GFX11-FAKE16-NEXT: s_pack_ll_b32_b16 s57, s61, s57
diff --git a/llvm/test/CodeGen/AMDGPU/amdgcn.bitcast.768bit.ll b/llvm/test/CodeGen/AMDGPU/amdgcn.bitcast.768bit.ll
index d7b8411c8f1460..8899717f639212 100644
--- a/llvm/test/CodeGen/AMDGPU/amdgcn.bitcast.768bit.ll
+++ b/llvm/test/CodeGen/AMDGPU/amdgcn.bitcast.768bit.ll
@@ -123,8 +123,9 @@ define <24 x float> @bitcast_v24i32_to_v24f32(<24 x i32> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v24
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB0_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -427,6 +428,7 @@ define inreg <24 x float> @bitcast_v24i32_to_v24f32_scalar(<24 x i32> inreg %a,
; GFX11-NEXT: s_and_b32 s10, s10, exec_lo
; GFX11-NEXT: s_cselect_b32 s10, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s10, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB1_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: s_add_i32 s4, s4, 3
@@ -602,8 +604,9 @@ define <24 x i32> @bitcast_v24f32_to_v24i32(<24 x float> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v24
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB2_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -1093,6 +1096,7 @@ define inreg <24 x i32> @bitcast_v24f32_to_v24i32_scalar(<24 x float> inreg %a,
; GFX11-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB3_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_add_f32_e64 v23, s59, 1.0
@@ -1289,8 +1293,9 @@ define <12 x i64> @bitcast_v24i32_to_v12i64(<24 x i32> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v24
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB4_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -1593,6 +1598,7 @@ define inreg <12 x i64> @bitcast_v24i32_to_v12i64_scalar(<24 x i32> inreg %a, i3
; GFX11-NEXT: s_and_b32 s10, s10, exec_lo
; GFX11-NEXT: s_cselect_b32 s10, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s10, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB5_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: s_add_i32 s4, s4, 3
@@ -1768,38 +1774,33 @@ define <24 x i32> @bitcast_v12i64_to_v24i32(<12 x i64> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v24
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB6_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_add_co_u32 v22, vcc_lo, v22, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v23, null, 0, v23, vcc_lo
; GFX11-NEXT: v_add_co_u32 v20, vcc_lo, v20, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v21, null, 0, v21, vcc_lo
; GFX11-NEXT: v_add_co_u32 v18, vcc_lo, v18, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v19, null, 0, v19, vcc_lo
; GFX11-NEXT: v_add_co_u32 v16, vcc_lo, v16, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v17, null, 0, v17, vcc_lo
; GFX11-NEXT: v_add_co_u32 v14, vcc_lo, v14, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v15, null, 0, v15, vcc_lo
; GFX11-NEXT: v_add_co_u32 v12, vcc_lo, v12, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v13, null, 0, v13, vcc_lo
; GFX11-NEXT: v_add_co_u32 v10, vcc_lo, v10, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v11, null, 0, v11, vcc_lo
; GFX11-NEXT: v_add_co_u32 v8, vcc_lo, v8, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v9, null, 0, v9, vcc_lo
; GFX11-NEXT: v_add_co_u32 v6, vcc_lo, v6, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v7, null, 0, v7, vcc_lo
; GFX11-NEXT: v_add_co_u32 v4, vcc_lo, v4, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v5, null, 0, v5, vcc_lo
; GFX11-NEXT: v_add_co_u32 v2, vcc_lo, v2, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX11-NEXT: v_add_co_u32 v0, vcc_lo, v0, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
@@ -2078,6 +2079,7 @@ define inreg <24 x i32> @bitcast_v12i64_to_v24i32_scalar(<12 x i64> inreg %a, i3
; GFX11-NEXT: s_and_b32 s10, s10, exec_lo
; GFX11-NEXT: s_cselect_b32 s10, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s10, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB7_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: s_add_u32 s5, s5, 3
@@ -2253,8 +2255,9 @@ define <12 x double> @bitcast_v24i32_to_v12f64(<24 x i32> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v24
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB8_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -2557,6 +2560,7 @@ define inreg <12 x double> @bitcast_v24i32_to_v12f64_scalar(<24 x i32> inreg %a,
; GFX11-NEXT: s_and_b32 s10, s10, exec_lo
; GFX11-NEXT: s_cselect_b32 s10, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s10, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB9_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: s_add_i32 s4, s4, 3
@@ -2696,8 +2700,9 @@ define <24 x i32> @bitcast_v12f64_to_v24i32(<12 x double> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v24
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB10_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -3151,6 +3156,7 @@ define inreg <24 x i32> @bitcast_v12f64_to_v24i32_scalar(<12 x double> inreg %a,
; GFX11-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB11_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_add_f64 v[22:23], s[58:59], 1.0
@@ -3811,8 +3817,9 @@ define <48 x i16> @bitcast_v24i32_to_v48i16(<24 x i32> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v55, 16, v0
; GFX11-TRUE16-NEXT: .LBB12_4: ; %end
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v0.h, v55.l
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v1.h, v54.l
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v2.h, v53.l
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v3.h, v52.l
@@ -3949,8 +3956,9 @@ define <48 x i16> @bitcast_v24i32_to_v48i16(<24 x i32> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v55, 16, v0
; GFX11-FAKE16-NEXT: .LBB12_4: ; %end
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_perm_b32 v0, v55, v0, 0x5040100
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_perm_b32 v1, v54, v1, 0x5040100
; GFX11-FAKE16-NEXT: v_perm_b32 v2, v53, v2, 0x5040100
; GFX11-FAKE16-NEXT: v_perm_b32 v3, v52, v3, 0x5040100
@@ -4693,6 +4701,7 @@ define inreg <48 x i16> @bitcast_v24i32_to_v48i16_scalar(<24 x i32> inreg %a, i3
; GFX11-NEXT: s_and_b32 s74, s74, exec_lo
; GFX11-NEXT: s_cselect_b32 s74, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s74, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB13_5
; GFX11-NEXT: ; %bb.4: ; %cmp.true
; GFX11-NEXT: s_add_i32 s4, s4, 3
@@ -5735,6 +5744,7 @@ define <24 x i32> @bitcast_v48i16_to_v24i32(<48 x i16> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v64, 16, v0
; GFX11-TRUE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-TRUE16-NEXT: v_cmpx_ne_u32_e32 0, v24
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB14_3
; GFX11-TRUE16-NEXT: ; %bb.1: ; %Flow
@@ -6837,6 +6847,7 @@ define inreg <24 x i32> @bitcast_v48i16_to_v24i32_scalar(<48 x i16> inreg %a, i3
; GFX11-NEXT: s_and_b32 s40, s40, exec_lo
; GFX11-NEXT: s_cselect_b32 s40, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s40, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB15_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_pk_add_u16 v0, s0, 3 op_sel_hi:[1,0]
@@ -7492,8 +7503,9 @@ define <48 x half> @bitcast_v24i32_to_v48f16(<24 x i32> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v55, 16, v0
; GFX11-TRUE16-NEXT: .LBB16_4: ; %end
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v0.h, v55.l
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v1.h, v54.l
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v2.h, v53.l
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v3.h, v52.l
@@ -7630,8 +7642,9 @@ define <48 x half> @bitcast_v24i32_to_v48f16(<24 x i32> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v55, 16, v0
; GFX11-FAKE16-NEXT: .LBB16_4: ; %end
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_perm_b32 v0, v55, v0, 0x5040100
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_perm_b32 v1, v54, v1, 0x5040100
; GFX11-FAKE16-NEXT: v_perm_b32 v2, v53, v2, 0x5040100
; GFX11-FAKE16-NEXT: v_perm_b32 v3, v52, v3, 0x5040100
@@ -8374,6 +8387,7 @@ define inreg <48 x half> @bitcast_v24i32_to_v48f16_scalar(<24 x i32> inreg %a, i
; GFX11-NEXT: s_and_b32 s74, s74, exec_lo
; GFX11-NEXT: s_cselect_b32 s74, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s74, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB17_5
; GFX11-NEXT: ; %bb.4: ; %cmp.true
; GFX11-NEXT: s_add_i32 s4, s4, 3
@@ -9528,6 +9542,7 @@ define <24 x i32> @bitcast_v48f16_to_v24i32(<48 x half> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v64, 16, v0
; GFX11-TRUE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-TRUE16-NEXT: v_cmpx_ne_u32_e32 0, v24
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB18_3
; GFX11-TRUE16-NEXT: ; %bb.1: ; %Flow
@@ -10700,6 +10715,7 @@ define inreg <24 x i32> @bitcast_v48f16_to_v24i32_scalar(<48 x half> inreg %a, i
; GFX11-NEXT: s_and_b32 s40, s40, exec_lo
; GFX11-NEXT: s_cselect_b32 s40, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s40, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB19_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_pk_add_f16 v0, 0x200, s0 op_sel_hi:[0,1]
@@ -10879,8 +10895,9 @@ define <12 x i64> @bitcast_v24f32_to_v12i64(<24 x float> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v24
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB20_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -11370,6 +11387,7 @@ define inreg <12 x i64> @bitcast_v24f32_to_v12i64_scalar(<24 x float> inreg %a,
; GFX11-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB21_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_add_f32_e64 v23, s59, 1.0
@@ -11566,38 +11584,33 @@ define <24 x float> @bitcast_v12i64_to_v24f32(<12 x i64> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v24
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB22_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_add_co_u32 v22, vcc_lo, v22, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v23, null, 0, v23, vcc_lo
; GFX11-NEXT: v_add_co_u32 v20, vcc_lo, v20, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v21, null, 0, v21, vcc_lo
; GFX11-NEXT: v_add_co_u32 v18, vcc_lo, v18, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v19, null, 0, v19, vcc_lo
; GFX11-NEXT: v_add_co_u32 v16, vcc_lo, v16, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v17, null, 0, v17, vcc_lo
; GFX11-NEXT: v_add_co_u32 v14, vcc_lo, v14, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v15, null, 0, v15, vcc_lo
; GFX11-NEXT: v_add_co_u32 v12, vcc_lo, v12, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v13, null, 0, v13, vcc_lo
; GFX11-NEXT: v_add_co_u32 v10, vcc_lo, v10, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v11, null, 0, v11, vcc_lo
; GFX11-NEXT: v_add_co_u32 v8, vcc_lo, v8, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v9, null, 0, v9, vcc_lo
; GFX11-NEXT: v_add_co_u32 v6, vcc_lo, v6, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v7, null, 0, v7, vcc_lo
; GFX11-NEXT: v_add_co_u32 v4, vcc_lo, v4, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v5, null, 0, v5, vcc_lo
; GFX11-NEXT: v_add_co_u32 v2, vcc_lo, v2, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX11-NEXT: v_add_co_u32 v0, vcc_lo, v0, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
@@ -11876,6 +11889,7 @@ define inreg <24 x float> @bitcast_v12i64_to_v24f32_scalar(<12 x i64> inreg %a,
; GFX11-NEXT: s_and_b32 s10, s10, exec_lo
; GFX11-NEXT: s_cselect_b32 s10, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s10, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB23_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: s_add_u32 s5, s5, 3
@@ -12051,8 +12065,9 @@ define <12 x double> @bitcast_v24f32_to_v12f64(<24 x float> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v24
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB24_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -12542,6 +12557,7 @@ define inreg <12 x double> @bitcast_v24f32_to_v12f64_scalar(<24 x float> inreg %
; GFX11-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB25_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_add_f32_e64 v23, s59, 1.0
@@ -12702,8 +12718,9 @@ define <24 x float> @bitcast_v12f64_to_v24f32(<12 x double> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v24
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB26_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -13157,6 +13174,7 @@ define inreg <24 x float> @bitcast_v12f64_to_v24f32_scalar(<12 x double> inreg %
; GFX11-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB27_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_add_f64 v[22:23], s[58:59], 1.0
@@ -13805,8 +13823,9 @@ define <48 x i16> @bitcast_v24f32_to_v48i16(<24 x float> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v55, 16, v0
; GFX11-TRUE16-NEXT: .LBB28_4: ; %end
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v0.h, v55.l
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v1.h, v54.l
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v2.h, v53.l
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v3.h, v52.l
@@ -13931,8 +13950,9 @@ define <48 x i16> @bitcast_v24f32_to_v48i16(<24 x float> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v55, 16, v0
; GFX11-FAKE16-NEXT: .LBB28_4: ; %end
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_perm_b32 v0, v55, v0, 0x5040100
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_perm_b32 v1, v54, v1, 0x5040100
; GFX11-FAKE16-NEXT: v_perm_b32 v2, v53, v2, 0x5040100
; GFX11-FAKE16-NEXT: v_perm_b32 v3, v52, v3, 0x5040100
@@ -14753,6 +14773,7 @@ define inreg <48 x i16> @bitcast_v24f32_to_v48i16_scalar(<24 x float> inreg %a,
; GFX11-TRUE16-NEXT: s_and_b32 s12, s12, exec_lo
; GFX11-TRUE16-NEXT: s_cselect_b32 s12, 1, 0
; GFX11-TRUE16-NEXT: s_cmp_lg_u32 s12, 1
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_cbranch_scc1 .LBB29_5
; GFX11-TRUE16-NEXT: ; %bb.4: ; %cmp.true
; GFX11-TRUE16-NEXT: v_add_f32_e64 v23, s4, 1.0
@@ -14927,6 +14948,7 @@ define inreg <48 x i16> @bitcast_v24f32_to_v48i16_scalar(<24 x float> inreg %a,
; GFX11-FAKE16-NEXT: s_and_b32 s12, s12, exec_lo
; GFX11-FAKE16-NEXT: s_cselect_b32 s12, 1, 0
; GFX11-FAKE16-NEXT: s_cmp_lg_u32 s12, 1
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_cbranch_scc1 .LBB29_5
; GFX11-FAKE16-NEXT: ; %bb.4: ; %cmp.true
; GFX11-FAKE16-NEXT: v_add_f32_e64 v19, s4, 1.0
@@ -16006,6 +16028,7 @@ define <24 x float> @bitcast_v48i16_to_v24f32(<48 x i16> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v64, 16, v0
; GFX11-TRUE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-TRUE16-NEXT: v_cmpx_ne_u32_e32 0, v24
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB30_3
; GFX11-TRUE16-NEXT: ; %bb.1: ; %Flow
@@ -17108,6 +17131,7 @@ define inreg <24 x float> @bitcast_v48i16_to_v24f32_scalar(<48 x i16> inreg %a,
; GFX11-NEXT: s_and_b32 s40, s40, exec_lo
; GFX11-NEXT: s_cselect_b32 s40, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s40, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB31_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_pk_add_u16 v0, s0, 3 op_sel_hi:[1,0]
@@ -17751,8 +17775,9 @@ define <48 x half> @bitcast_v24f32_to_v48f16(<24 x float> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v55, 16, v0
; GFX11-TRUE16-NEXT: .LBB32_4: ; %end
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v0.h, v55.l
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v1.h, v54.l
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v2.h, v53.l
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v3.h, v52.l
@@ -17877,8 +17902,9 @@ define <48 x half> @bitcast_v24f32_to_v48f16(<24 x float> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v55, 16, v0
; GFX11-FAKE16-NEXT: .LBB32_4: ; %end
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_perm_b32 v0, v55, v0, 0x5040100
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_perm_b32 v1, v54, v1, 0x5040100
; GFX11-FAKE16-NEXT: v_perm_b32 v2, v53, v2, 0x5040100
; GFX11-FAKE16-NEXT: v_perm_b32 v3, v52, v3, 0x5040100
@@ -18699,6 +18725,7 @@ define inreg <48 x half> @bitcast_v24f32_to_v48f16_scalar(<24 x float> inreg %a,
; GFX11-TRUE16-NEXT: s_and_b32 s12, s12, exec_lo
; GFX11-TRUE16-NEXT: s_cselect_b32 s12, 1, 0
; GFX11-TRUE16-NEXT: s_cmp_lg_u32 s12, 1
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_cbranch_scc1 .LBB33_5
; GFX11-TRUE16-NEXT: ; %bb.4: ; %cmp.true
; GFX11-TRUE16-NEXT: v_add_f32_e64 v23, s4, 1.0
@@ -18873,6 +18900,7 @@ define inreg <48 x half> @bitcast_v24f32_to_v48f16_scalar(<24 x float> inreg %a,
; GFX11-FAKE16-NEXT: s_and_b32 s12, s12, exec_lo
; GFX11-FAKE16-NEXT: s_cselect_b32 s12, 1, 0
; GFX11-FAKE16-NEXT: s_cmp_lg_u32 s12, 1
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_cbranch_scc1 .LBB33_5
; GFX11-FAKE16-NEXT: ; %bb.4: ; %cmp.true
; GFX11-FAKE16-NEXT: v_add_f32_e64 v19, s4, 1.0
@@ -20064,6 +20092,7 @@ define <24 x float> @bitcast_v48f16_to_v24f32(<48 x half> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v64, 16, v0
; GFX11-TRUE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-TRUE16-NEXT: v_cmpx_ne_u32_e32 0, v24
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB34_3
; GFX11-TRUE16-NEXT: ; %bb.1: ; %Flow
@@ -21236,6 +21265,7 @@ define inreg <24 x float> @bitcast_v48f16_to_v24f32_scalar(<48 x half> inreg %a,
; GFX11-NEXT: s_and_b32 s40, s40, exec_lo
; GFX11-NEXT: s_cselect_b32 s40, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s40, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB35_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_pk_add_f16 v0, 0x200, s0 op_sel_hi:[0,1]
@@ -21415,38 +21445,33 @@ define <12 x double> @bitcast_v12i64_to_v12f64(<12 x i64> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v24
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB36_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_add_co_u32 v0, vcc_lo, v0, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-NEXT: v_add_co_u32 v2, vcc_lo, v2, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX11-NEXT: v_add_co_u32 v4, vcc_lo, v4, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v5, null, 0, v5, vcc_lo
; GFX11-NEXT: v_add_co_u32 v6, vcc_lo, v6, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v7, null, 0, v7, vcc_lo
; GFX11-NEXT: v_add_co_u32 v8, vcc_lo, v8, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v9, null, 0, v9, vcc_lo
; GFX11-NEXT: v_add_co_u32 v10, vcc_lo, v10, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v11, null, 0, v11, vcc_lo
; GFX11-NEXT: v_add_co_u32 v12, vcc_lo, v12, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v13, null, 0, v13, vcc_lo
; GFX11-NEXT: v_add_co_u32 v14, vcc_lo, v14, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v15, null, 0, v15, vcc_lo
; GFX11-NEXT: v_add_co_u32 v16, vcc_lo, v16, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v17, null, 0, v17, vcc_lo
; GFX11-NEXT: v_add_co_u32 v18, vcc_lo, v18, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v19, null, 0, v19, vcc_lo
; GFX11-NEXT: v_add_co_u32 v20, vcc_lo, v20, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v21, null, 0, v21, vcc_lo
; GFX11-NEXT: v_add_co_u32 v22, vcc_lo, v22, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v23, null, 0, v23, vcc_lo
@@ -21725,6 +21750,7 @@ define inreg <12 x double> @bitcast_v12i64_to_v12f64_scalar(<12 x i64> inreg %a,
; GFX11-NEXT: s_and_b32 s10, s10, exec_lo
; GFX11-NEXT: s_cselect_b32 s10, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s10, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB37_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: s_add_u32 s0, s0, 3
@@ -21863,8 +21889,9 @@ define <12 x i64> @bitcast_v12f64_to_v12i64(<12 x double> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v24
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB38_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -22318,6 +22345,7 @@ define inreg <12 x i64> @bitcast_v12f64_to_v12i64_scalar(<12 x double> inreg %a,
; GFX11-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB39_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_add_f64 v[0:1], s[36:37], 1.0
@@ -22929,32 +22957,26 @@ define <48 x i16> @bitcast_v12i64_to_v48i16(<12 x i64> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: s_cbranch_execz .LBB40_4
; GFX11-TRUE16-NEXT: ; %bb.3: ; %cmp.true
; GFX11-TRUE16-NEXT: v_add_co_u32 v22, vcc_lo, v22, 3
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v23, null, 0, v23, vcc_lo
; GFX11-TRUE16-NEXT: v_add_co_u32 v20, vcc_lo, v20, 3
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v21, null, 0, v21, vcc_lo
; GFX11-TRUE16-NEXT: v_add_co_u32 v18, vcc_lo, v18, 3
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v19, null, 0, v19, vcc_lo
; GFX11-TRUE16-NEXT: v_add_co_u32 v16, vcc_lo, v16, 3
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v17, null, 0, v17, vcc_lo
; GFX11-TRUE16-NEXT: v_add_co_u32 v14, vcc_lo, v14, 3
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v15, null, 0, v15, vcc_lo
; GFX11-TRUE16-NEXT: v_add_co_u32 v12, vcc_lo, v12, 3
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v13, null, 0, v13, vcc_lo
; GFX11-TRUE16-NEXT: v_add_co_u32 v10, vcc_lo, v10, 3
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v11, null, 0, v11, vcc_lo
; GFX11-TRUE16-NEXT: v_add_co_u32 v8, vcc_lo, v8, 3
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v9, null, 0, v9, vcc_lo
; GFX11-TRUE16-NEXT: v_add_co_u32 v6, vcc_lo, v6, 3
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v7, null, 0, v7, vcc_lo
; GFX11-TRUE16-NEXT: v_add_co_u32 v4, vcc_lo, v4, 3
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v5, null, 0, v5, vcc_lo
; GFX11-TRUE16-NEXT: v_add_co_u32 v2, vcc_lo, v2, 3
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX11-TRUE16-NEXT: v_add_co_u32 v0, vcc_lo, v0, 3
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
@@ -22984,8 +23006,9 @@ define <48 x i16> @bitcast_v12i64_to_v48i16(<12 x i64> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v55, 16, v0
; GFX11-TRUE16-NEXT: .LBB40_4: ; %end
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v0.h, v55.l
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v1.h, v54.l
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v2.h, v53.l
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v3.h, v52.l
@@ -23073,32 +23096,26 @@ define <48 x i16> @bitcast_v12i64_to_v48i16(<12 x i64> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: s_cbranch_execz .LBB40_4
; GFX11-FAKE16-NEXT: ; %bb.3: ; %cmp.true
; GFX11-FAKE16-NEXT: v_add_co_u32 v22, vcc_lo, v22, 3
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v23, null, 0, v23, vcc_lo
; GFX11-FAKE16-NEXT: v_add_co_u32 v20, vcc_lo, v20, 3
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v21, null, 0, v21, vcc_lo
; GFX11-FAKE16-NEXT: v_add_co_u32 v18, vcc_lo, v18, 3
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v19, null, 0, v19, vcc_lo
; GFX11-FAKE16-NEXT: v_add_co_u32 v16, vcc_lo, v16, 3
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v17, null, 0, v17, vcc_lo
; GFX11-FAKE16-NEXT: v_add_co_u32 v14, vcc_lo, v14, 3
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v15, null, 0, v15, vcc_lo
; GFX11-FAKE16-NEXT: v_add_co_u32 v12, vcc_lo, v12, 3
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v13, null, 0, v13, vcc_lo
; GFX11-FAKE16-NEXT: v_add_co_u32 v10, vcc_lo, v10, 3
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v11, null, 0, v11, vcc_lo
; GFX11-FAKE16-NEXT: v_add_co_u32 v8, vcc_lo, v8, 3
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v9, null, 0, v9, vcc_lo
; GFX11-FAKE16-NEXT: v_add_co_u32 v6, vcc_lo, v6, 3
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v7, null, 0, v7, vcc_lo
; GFX11-FAKE16-NEXT: v_add_co_u32 v4, vcc_lo, v4, 3
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v5, null, 0, v5, vcc_lo
; GFX11-FAKE16-NEXT: v_add_co_u32 v2, vcc_lo, v2, 3
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX11-FAKE16-NEXT: v_add_co_u32 v0, vcc_lo, v0, 3
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
@@ -23128,8 +23145,9 @@ define <48 x i16> @bitcast_v12i64_to_v48i16(<12 x i64> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v55, 16, v0
; GFX11-FAKE16-NEXT: .LBB40_4: ; %end
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_perm_b32 v0, v55, v0, 0x5040100
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_perm_b32 v1, v54, v1, 0x5040100
; GFX11-FAKE16-NEXT: v_perm_b32 v2, v53, v2, 0x5040100
; GFX11-FAKE16-NEXT: v_perm_b32 v3, v52, v3, 0x5040100
@@ -23872,6 +23890,7 @@ define inreg <48 x i16> @bitcast_v12i64_to_v48i16_scalar(<12 x i64> inreg %a, i3
; GFX11-NEXT: s_and_b32 s74, s74, exec_lo
; GFX11-NEXT: s_cselect_b32 s74, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s74, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB41_5
; GFX11-NEXT: ; %bb.4: ; %cmp.true
; GFX11-NEXT: s_add_u32 s5, s5, 3
@@ -24914,6 +24933,7 @@ define <12 x i64> @bitcast_v48i16_to_v12i64(<48 x i16> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v64, 16, v0
; GFX11-TRUE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-TRUE16-NEXT: v_cmpx_ne_u32_e32 0, v24
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB42_3
; GFX11-TRUE16-NEXT: ; %bb.1: ; %Flow
@@ -26016,6 +26036,7 @@ define inreg <12 x i64> @bitcast_v48i16_to_v12i64_scalar(<48 x i16> inreg %a, i3
; GFX11-NEXT: s_and_b32 s40, s40, exec_lo
; GFX11-NEXT: s_cselect_b32 s40, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s40, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB43_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_pk_add_u16 v0, s0, 3 op_sel_hi:[1,0]
@@ -26622,32 +26643,26 @@ define <48 x half> @bitcast_v12i64_to_v48f16(<12 x i64> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: s_cbranch_execz .LBB44_4
; GFX11-TRUE16-NEXT: ; %bb.3: ; %cmp.true
; GFX11-TRUE16-NEXT: v_add_co_u32 v22, vcc_lo, v22, 3
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v23, null, 0, v23, vcc_lo
; GFX11-TRUE16-NEXT: v_add_co_u32 v20, vcc_lo, v20, 3
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v21, null, 0, v21, vcc_lo
; GFX11-TRUE16-NEXT: v_add_co_u32 v18, vcc_lo, v18, 3
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v19, null, 0, v19, vcc_lo
; GFX11-TRUE16-NEXT: v_add_co_u32 v16, vcc_lo, v16, 3
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v17, null, 0, v17, vcc_lo
; GFX11-TRUE16-NEXT: v_add_co_u32 v14, vcc_lo, v14, 3
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v15, null, 0, v15, vcc_lo
; GFX11-TRUE16-NEXT: v_add_co_u32 v12, vcc_lo, v12, 3
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v13, null, 0, v13, vcc_lo
; GFX11-TRUE16-NEXT: v_add_co_u32 v10, vcc_lo, v10, 3
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v11, null, 0, v11, vcc_lo
; GFX11-TRUE16-NEXT: v_add_co_u32 v8, vcc_lo, v8, 3
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v9, null, 0, v9, vcc_lo
; GFX11-TRUE16-NEXT: v_add_co_u32 v6, vcc_lo, v6, 3
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v7, null, 0, v7, vcc_lo
; GFX11-TRUE16-NEXT: v_add_co_u32 v4, vcc_lo, v4, 3
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v5, null, 0, v5, vcc_lo
; GFX11-TRUE16-NEXT: v_add_co_u32 v2, vcc_lo, v2, 3
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX11-TRUE16-NEXT: v_add_co_u32 v0, vcc_lo, v0, 3
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
@@ -26677,8 +26692,9 @@ define <48 x half> @bitcast_v12i64_to_v48f16(<12 x i64> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v55, 16, v0
; GFX11-TRUE16-NEXT: .LBB44_4: ; %end
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v0.h, v55.l
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v1.h, v54.l
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v2.h, v53.l
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v3.h, v52.l
@@ -26766,32 +26782,26 @@ define <48 x half> @bitcast_v12i64_to_v48f16(<12 x i64> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: s_cbranch_execz .LBB44_4
; GFX11-FAKE16-NEXT: ; %bb.3: ; %cmp.true
; GFX11-FAKE16-NEXT: v_add_co_u32 v22, vcc_lo, v22, 3
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v23, null, 0, v23, vcc_lo
; GFX11-FAKE16-NEXT: v_add_co_u32 v20, vcc_lo, v20, 3
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v21, null, 0, v21, vcc_lo
; GFX11-FAKE16-NEXT: v_add_co_u32 v18, vcc_lo, v18, 3
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v19, null, 0, v19, vcc_lo
; GFX11-FAKE16-NEXT: v_add_co_u32 v16, vcc_lo, v16, 3
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v17, null, 0, v17, vcc_lo
; GFX11-FAKE16-NEXT: v_add_co_u32 v14, vcc_lo, v14, 3
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v15, null, 0, v15, vcc_lo
; GFX11-FAKE16-NEXT: v_add_co_u32 v12, vcc_lo, v12, 3
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v13, null, 0, v13, vcc_lo
; GFX11-FAKE16-NEXT: v_add_co_u32 v10, vcc_lo, v10, 3
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v11, null, 0, v11, vcc_lo
; GFX11-FAKE16-NEXT: v_add_co_u32 v8, vcc_lo, v8, 3
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v9, null, 0, v9, vcc_lo
; GFX11-FAKE16-NEXT: v_add_co_u32 v6, vcc_lo, v6, 3
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v7, null, 0, v7, vcc_lo
; GFX11-FAKE16-NEXT: v_add_co_u32 v4, vcc_lo, v4, 3
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v5, null, 0, v5, vcc_lo
; GFX11-FAKE16-NEXT: v_add_co_u32 v2, vcc_lo, v2, 3
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX11-FAKE16-NEXT: v_add_co_u32 v0, vcc_lo, v0, 3
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
@@ -26821,8 +26831,9 @@ define <48 x half> @bitcast_v12i64_to_v48f16(<12 x i64> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v55, 16, v0
; GFX11-FAKE16-NEXT: .LBB44_4: ; %end
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_perm_b32 v0, v55, v0, 0x5040100
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_perm_b32 v1, v54, v1, 0x5040100
; GFX11-FAKE16-NEXT: v_perm_b32 v2, v53, v2, 0x5040100
; GFX11-FAKE16-NEXT: v_perm_b32 v3, v52, v3, 0x5040100
@@ -27565,6 +27576,7 @@ define inreg <48 x half> @bitcast_v12i64_to_v48f16_scalar(<12 x i64> inreg %a, i
; GFX11-NEXT: s_and_b32 s74, s74, exec_lo
; GFX11-NEXT: s_cselect_b32 s74, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s74, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB45_5
; GFX11-NEXT: ; %bb.4: ; %cmp.true
; GFX11-NEXT: s_add_u32 s5, s5, 3
@@ -28719,6 +28731,7 @@ define <12 x i64> @bitcast_v48f16_to_v12i64(<48 x half> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v64, 16, v0
; GFX11-TRUE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-TRUE16-NEXT: v_cmpx_ne_u32_e32 0, v24
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB46_3
; GFX11-TRUE16-NEXT: ; %bb.1: ; %Flow
@@ -29891,6 +29904,7 @@ define inreg <12 x i64> @bitcast_v48f16_to_v12i64_scalar(<48 x half> inreg %a, i
; GFX11-NEXT: s_and_b32 s40, s40, exec_lo
; GFX11-NEXT: s_cselect_b32 s40, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s40, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB47_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_pk_add_f16 v0, 0x200, s0 op_sel_hi:[0,1]
@@ -30498,8 +30512,9 @@ define <48 x i16> @bitcast_v12f64_to_v48i16(<12 x double> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v55, 16, v0
; GFX11-TRUE16-NEXT: .LBB48_4: ; %end
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v0.h, v55.l
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v1.h, v54.l
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v2.h, v53.l
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v3.h, v52.l
@@ -30624,8 +30639,9 @@ define <48 x i16> @bitcast_v12f64_to_v48i16(<12 x double> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v55, 16, v0
; GFX11-FAKE16-NEXT: .LBB48_4: ; %end
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_perm_b32 v0, v55, v0, 0x5040100
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_perm_b32 v1, v54, v1, 0x5040100
; GFX11-FAKE16-NEXT: v_perm_b32 v2, v53, v2, 0x5040100
; GFX11-FAKE16-NEXT: v_perm_b32 v3, v52, v3, 0x5040100
@@ -31410,6 +31426,7 @@ define inreg <48 x i16> @bitcast_v12f64_to_v48i16_scalar(<12 x double> inreg %a,
; GFX11-TRUE16-NEXT: s_and_b32 s12, s12, exec_lo
; GFX11-TRUE16-NEXT: s_cselect_b32 s12, 1, 0
; GFX11-TRUE16-NEXT: s_cmp_lg_u32 s12, 1
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_cbranch_scc1 .LBB49_5
; GFX11-TRUE16-NEXT: ; %bb.4: ; %cmp.true
; GFX11-TRUE16-NEXT: v_add_f64 v[22:23], s[6:7], 1.0
@@ -31572,6 +31589,7 @@ define inreg <48 x i16> @bitcast_v12f64_to_v48i16_scalar(<12 x double> inreg %a,
; GFX11-FAKE16-NEXT: s_and_b32 s12, s12, exec_lo
; GFX11-FAKE16-NEXT: s_cselect_b32 s12, 1, 0
; GFX11-FAKE16-NEXT: s_cmp_lg_u32 s12, 1
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_cbranch_scc1 .LBB49_5
; GFX11-FAKE16-NEXT: ; %bb.4: ; %cmp.true
; GFX11-FAKE16-NEXT: v_add_f64 v[19:20], s[6:7], 1.0
@@ -32639,6 +32657,7 @@ define <12 x double> @bitcast_v48i16_to_v12f64(<48 x i16> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v64, 16, v0
; GFX11-TRUE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-TRUE16-NEXT: v_cmpx_ne_u32_e32 0, v24
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB50_3
; GFX11-TRUE16-NEXT: ; %bb.1: ; %Flow
@@ -33741,6 +33760,7 @@ define inreg <12 x double> @bitcast_v48i16_to_v12f64_scalar(<48 x i16> inreg %a,
; GFX11-NEXT: s_and_b32 s40, s40, exec_lo
; GFX11-NEXT: s_cselect_b32 s40, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s40, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB51_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_pk_add_u16 v0, s0, 3 op_sel_hi:[1,0]
@@ -34348,8 +34368,9 @@ define <48 x half> @bitcast_v12f64_to_v48f16(<12 x double> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v55, 16, v0
; GFX11-TRUE16-NEXT: .LBB52_4: ; %end
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v0.h, v55.l
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v1.h, v54.l
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v2.h, v53.l
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v3.h, v52.l
@@ -34474,8 +34495,9 @@ define <48 x half> @bitcast_v12f64_to_v48f16(<12 x double> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v55, 16, v0
; GFX11-FAKE16-NEXT: .LBB52_4: ; %end
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_perm_b32 v0, v55, v0, 0x5040100
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_perm_b32 v1, v54, v1, 0x5040100
; GFX11-FAKE16-NEXT: v_perm_b32 v2, v53, v2, 0x5040100
; GFX11-FAKE16-NEXT: v_perm_b32 v3, v52, v3, 0x5040100
@@ -35260,6 +35282,7 @@ define inreg <48 x half> @bitcast_v12f64_to_v48f16_scalar(<12 x double> inreg %a
; GFX11-TRUE16-NEXT: s_and_b32 s12, s12, exec_lo
; GFX11-TRUE16-NEXT: s_cselect_b32 s12, 1, 0
; GFX11-TRUE16-NEXT: s_cmp_lg_u32 s12, 1
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_cbranch_scc1 .LBB53_5
; GFX11-TRUE16-NEXT: ; %bb.4: ; %cmp.true
; GFX11-TRUE16-NEXT: v_add_f64 v[22:23], s[6:7], 1.0
@@ -35422,6 +35445,7 @@ define inreg <48 x half> @bitcast_v12f64_to_v48f16_scalar(<12 x double> inreg %a
; GFX11-FAKE16-NEXT: s_and_b32 s12, s12, exec_lo
; GFX11-FAKE16-NEXT: s_cselect_b32 s12, 1, 0
; GFX11-FAKE16-NEXT: s_cmp_lg_u32 s12, 1
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_cbranch_scc1 .LBB53_5
; GFX11-FAKE16-NEXT: ; %bb.4: ; %cmp.true
; GFX11-FAKE16-NEXT: v_add_f64 v[19:20], s[6:7], 1.0
@@ -36601,6 +36625,7 @@ define <12 x double> @bitcast_v48f16_to_v12f64(<48 x half> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v64, 16, v0
; GFX11-TRUE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-TRUE16-NEXT: v_cmpx_ne_u32_e32 0, v24
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB54_3
; GFX11-TRUE16-NEXT: ; %bb.1: ; %Flow
@@ -37773,6 +37798,7 @@ define inreg <12 x double> @bitcast_v48f16_to_v12f64_scalar(<48 x half> inreg %a
; GFX11-NEXT: s_and_b32 s40, s40, exec_lo
; GFX11-NEXT: s_cselect_b32 s40, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s40, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB55_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_pk_add_f16 v0, 0x200, s0 op_sel_hi:[0,1]
@@ -38768,8 +38794,9 @@ define <48 x half> @bitcast_v48i16_to_v48f16(<48 x i16> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v29, 16, v0
; GFX11-TRUE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-TRUE16-NEXT: v_cmpx_ne_u32_e32 0, v24
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-TRUE16-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-TRUE16-NEXT: s_cbranch_execz .LBB56_2
; GFX11-TRUE16-NEXT: ; %bb.1: ; %cmp.true
@@ -38847,6 +38874,7 @@ define <48 x half> @bitcast_v48i16_to_v48f16(<48 x i16> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v64, 16, v23
; GFX11-TRUE16-NEXT: .LBB56_2: ; %end
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v0.h, v29.l
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v1.h, v28.l
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v2.h, v27.l
@@ -38902,8 +38930,9 @@ define <48 x half> @bitcast_v48i16_to_v48f16(<48 x i16> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v25, 16, v0
; GFX11-FAKE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-FAKE16-NEXT: v_cmpx_ne_u32_e32 0, v24
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-FAKE16-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-FAKE16-NEXT: s_cbranch_execz .LBB56_2
; GFX11-FAKE16-NEXT: ; %bb.1: ; %cmp.true
@@ -38981,6 +39010,7 @@ define <48 x half> @bitcast_v48i16_to_v48f16(<48 x i16> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v64, 16, v23
; GFX11-FAKE16-NEXT: .LBB56_2: ; %end
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_perm_b32 v0, v25, v0, 0x5040100
; GFX11-FAKE16-NEXT: v_perm_b32 v1, v26, v1, 0x5040100
; GFX11-FAKE16-NEXT: v_perm_b32 v2, v27, v2, 0x5040100
@@ -39988,6 +40018,7 @@ define inreg <48 x half> @bitcast_v48i16_to_v48f16_scalar(<48 x i16> inreg %a, i
; GFX11-TRUE16-NEXT: s_and_b32 s74, s74, exec_lo
; GFX11-TRUE16-NEXT: s_cselect_b32 s74, 1, 0
; GFX11-TRUE16-NEXT: s_cmp_lg_u32 s74, 1
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_cbranch_scc1 .LBB57_4
; GFX11-TRUE16-NEXT: ; %bb.3: ; %cmp.true
; GFX11-TRUE16-NEXT: s_pack_ll_b32_b16 s57, s72, s57
@@ -40161,6 +40192,7 @@ define inreg <48 x half> @bitcast_v48i16_to_v48f16_scalar(<48 x i16> inreg %a, i
; GFX11-FAKE16-NEXT: s_and_b32 s74, s74, exec_lo
; GFX11-FAKE16-NEXT: s_cselect_b32 s74, 1, 0
; GFX11-FAKE16-NEXT: s_cmp_lg_u32 s74, 1
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_cbranch_scc1 .LBB57_4
; GFX11-FAKE16-NEXT: ; %bb.3: ; %cmp.true
; GFX11-FAKE16-NEXT: s_pack_ll_b32_b16 s57, s72, s57
@@ -40945,8 +40977,9 @@ define <48 x i16> @bitcast_v48f16_to_v48i16(<48 x half> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v29, 16, v0
; GFX11-TRUE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-TRUE16-NEXT: v_cmpx_ne_u32_e32 0, v24
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-TRUE16-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-TRUE16-NEXT: s_cbranch_execz .LBB58_2
; GFX11-TRUE16-NEXT: ; %bb.1: ; %cmp.true
@@ -41024,6 +41057,7 @@ define <48 x i16> @bitcast_v48f16_to_v48i16(<48 x half> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v64, 16, v23
; GFX11-TRUE16-NEXT: .LBB58_2: ; %end
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v0.h, v29.l
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v1.h, v28.l
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v2.h, v27.l
@@ -41079,8 +41113,9 @@ define <48 x i16> @bitcast_v48f16_to_v48i16(<48 x half> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v25, 16, v0
; GFX11-FAKE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-FAKE16-NEXT: v_cmpx_ne_u32_e32 0, v24
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-FAKE16-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-FAKE16-NEXT: s_cbranch_execz .LBB58_2
; GFX11-FAKE16-NEXT: ; %bb.1: ; %cmp.true
@@ -41158,6 +41193,7 @@ define <48 x i16> @bitcast_v48f16_to_v48i16(<48 x half> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v64, 16, v23
; GFX11-FAKE16-NEXT: .LBB58_2: ; %end
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_perm_b32 v0, v25, v0, 0x5040100
; GFX11-FAKE16-NEXT: v_perm_b32 v1, v26, v1, 0x5040100
; GFX11-FAKE16-NEXT: v_perm_b32 v2, v27, v2, 0x5040100
@@ -42111,6 +42147,7 @@ define inreg <48 x i16> @bitcast_v48f16_to_v48i16_scalar(<48 x half> inreg %a, i
; GFX11-TRUE16-NEXT: s_and_b32 s74, s74, exec_lo
; GFX11-TRUE16-NEXT: s_cselect_b32 s74, 1, 0
; GFX11-TRUE16-NEXT: s_cmp_lg_u32 s74, 1
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_cbranch_scc1 .LBB59_4
; GFX11-TRUE16-NEXT: ; %bb.3: ; %cmp.true
; GFX11-TRUE16-NEXT: s_pack_ll_b32_b16 s57, s72, s57
@@ -42284,6 +42321,7 @@ define inreg <48 x i16> @bitcast_v48f16_to_v48i16_scalar(<48 x half> inreg %a, i
; GFX11-FAKE16-NEXT: s_and_b32 s74, s74, exec_lo
; GFX11-FAKE16-NEXT: s_cselect_b32 s74, 1, 0
; GFX11-FAKE16-NEXT: s_cmp_lg_u32 s74, 1
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_cbranch_scc1 .LBB59_4
; GFX11-FAKE16-NEXT: ; %bb.3: ; %cmp.true
; GFX11-FAKE16-NEXT: s_pack_ll_b32_b16 s57, s72, s57
diff --git a/llvm/test/CodeGen/AMDGPU/amdgcn.bitcast.832bit.ll b/llvm/test/CodeGen/AMDGPU/amdgcn.bitcast.832bit.ll
index d070d09aa3dc39..961b7d9144cd97 100644
--- a/llvm/test/CodeGen/AMDGPU/amdgcn.bitcast.832bit.ll
+++ b/llvm/test/CodeGen/AMDGPU/amdgcn.bitcast.832bit.ll
@@ -129,8 +129,9 @@ define <26 x float> @bitcast_v26i32_to_v26f32(<26 x i32> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v26
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB0_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -455,6 +456,7 @@ define inreg <26 x float> @bitcast_v26i32_to_v26f32_scalar(<26 x i32> inreg %a,
; GFX11-NEXT: s_and_b32 s12, s12, exec_lo
; GFX11-NEXT: s_cselect_b32 s12, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s12, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB1_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: s_add_i32 s4, s4, 3
@@ -639,8 +641,9 @@ define <26 x i32> @bitcast_v26f32_to_v26i32(<26 x float> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v26
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB2_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -1145,6 +1148,7 @@ define inreg <26 x i32> @bitcast_v26f32_to_v26i32_scalar(<26 x float> inreg %a,
; GFX11-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB3_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_add_f32_e64 v25, s61, 1.0
@@ -1349,8 +1353,9 @@ define <13 x i64> @bitcast_v26i32_to_v13i64(<26 x i32> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v26
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB4_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -1675,6 +1680,7 @@ define inreg <13 x i64> @bitcast_v26i32_to_v13i64_scalar(<26 x i32> inreg %a, i3
; GFX11-NEXT: s_and_b32 s12, s12, exec_lo
; GFX11-NEXT: s_cselect_b32 s12, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s12, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB5_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: s_add_i32 s4, s4, 3
@@ -1859,43 +1865,37 @@ define <26 x i32> @bitcast_v13i64_to_v26i32(<13 x i64> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v26
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB6_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_add_co_u32 v24, vcc_lo, v24, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v25, null, 0, v25, vcc_lo
; GFX11-NEXT: v_add_co_u32 v22, vcc_lo, v22, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v23, null, 0, v23, vcc_lo
; GFX11-NEXT: v_add_co_u32 v20, vcc_lo, v20, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v21, null, 0, v21, vcc_lo
; GFX11-NEXT: v_add_co_u32 v18, vcc_lo, v18, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v19, null, 0, v19, vcc_lo
; GFX11-NEXT: v_add_co_u32 v16, vcc_lo, v16, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v17, null, 0, v17, vcc_lo
; GFX11-NEXT: v_add_co_u32 v14, vcc_lo, v14, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v15, null, 0, v15, vcc_lo
; GFX11-NEXT: v_add_co_u32 v12, vcc_lo, v12, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v13, null, 0, v13, vcc_lo
; GFX11-NEXT: v_add_co_u32 v10, vcc_lo, v10, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v11, null, 0, v11, vcc_lo
; GFX11-NEXT: v_add_co_u32 v8, vcc_lo, v8, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v9, null, 0, v9, vcc_lo
; GFX11-NEXT: v_add_co_u32 v6, vcc_lo, v6, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v7, null, 0, v7, vcc_lo
; GFX11-NEXT: v_add_co_u32 v4, vcc_lo, v4, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v5, null, 0, v5, vcc_lo
; GFX11-NEXT: v_add_co_u32 v2, vcc_lo, v2, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX11-NEXT: v_add_co_u32 v0, vcc_lo, v0, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-NEXT: .LBB6_2: ; %end
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -2192,6 +2192,7 @@ define inreg <26 x i32> @bitcast_v13i64_to_v26i32_scalar(<13 x i64> inreg %a, i3
; GFX11-NEXT: s_and_b32 s12, s12, exec_lo
; GFX11-NEXT: s_cselect_b32 s12, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s12, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB7_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: s_add_u32 s5, s5, 3
@@ -2376,8 +2377,9 @@ define <13 x double> @bitcast_v26i32_to_v13f64(<26 x i32> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v26
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB8_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -2702,6 +2704,7 @@ define inreg <13 x double> @bitcast_v26i32_to_v13f64_scalar(<26 x i32> inreg %a,
; GFX11-NEXT: s_and_b32 s12, s12, exec_lo
; GFX11-NEXT: s_cselect_b32 s12, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s12, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB9_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: s_add_i32 s4, s4, 3
@@ -2847,8 +2850,9 @@ define <26 x i32> @bitcast_v13f64_to_v26i32(<13 x double> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v26
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB10_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -3314,6 +3318,7 @@ define inreg <26 x i32> @bitcast_v13f64_to_v26i32_scalar(<13 x double> inreg %a,
; GFX11-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB11_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_add_f64 v[24:25], s[60:61], 1.0
@@ -4055,8 +4060,9 @@ define <52 x i16> @bitcast_v26i32_to_v52i16(<26 x i32> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v67, 16, v0
; GFX11-TRUE16-NEXT: .LBB12_4: ; %end
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v0.h, v67.l
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v1.h, v66.l
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v2.h, v65.l
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v3.h, v64.l
@@ -4203,8 +4209,9 @@ define <52 x i16> @bitcast_v26i32_to_v52i16(<26 x i32> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v67, 16, v0
; GFX11-FAKE16-NEXT: .LBB12_4: ; %end
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_perm_b32 v0, v67, v0, 0x5040100
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_perm_b32 v1, v66, v1, 0x5040100
; GFX11-FAKE16-NEXT: v_perm_b32 v2, v65, v2, 0x5040100
; GFX11-FAKE16-NEXT: v_perm_b32 v3, v64, v3, 0x5040100
@@ -5015,6 +5022,7 @@ define inreg <52 x i16> @bitcast_v26i32_to_v52i16_scalar(<26 x i32> inreg %a, i3
; GFX11-NEXT: s_and_b32 s78, s78, exec_lo
; GFX11-NEXT: s_cselect_b32 s78, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s78, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB13_5
; GFX11-NEXT: ; %bb.4: ; %cmp.true
; GFX11-NEXT: s_add_i32 s4, s4, 3
@@ -6160,6 +6168,7 @@ define <26 x i32> @bitcast_v52i16_to_v26i32(<52 x i16> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v68, 16, v0
; GFX11-TRUE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-TRUE16-NEXT: v_cmpx_ne_u32_e32 0, v26
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB14_3
; GFX11-TRUE16-NEXT: ; %bb.1: ; %Flow
@@ -7350,6 +7359,7 @@ define inreg <26 x i32> @bitcast_v52i16_to_v26i32_scalar(<52 x i16> inreg %a, i3
; GFX11-NEXT: s_and_b32 s40, s40, exec_lo
; GFX11-NEXT: s_cselect_b32 s40, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s40, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB15_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_pk_add_u16 v0, s0, 3 op_sel_hi:[1,0]
@@ -8087,8 +8097,9 @@ define <52 x half> @bitcast_v26i32_to_v52f16(<26 x i32> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v67, 16, v0
; GFX11-TRUE16-NEXT: .LBB16_4: ; %end
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v0.h, v67.l
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v1.h, v66.l
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v2.h, v65.l
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v3.h, v64.l
@@ -8235,8 +8246,9 @@ define <52 x half> @bitcast_v26i32_to_v52f16(<26 x i32> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v67, 16, v0
; GFX11-FAKE16-NEXT: .LBB16_4: ; %end
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_perm_b32 v0, v67, v0, 0x5040100
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_perm_b32 v1, v66, v1, 0x5040100
; GFX11-FAKE16-NEXT: v_perm_b32 v2, v65, v2, 0x5040100
; GFX11-FAKE16-NEXT: v_perm_b32 v3, v64, v3, 0x5040100
@@ -9047,6 +9059,7 @@ define inreg <52 x half> @bitcast_v26i32_to_v52f16_scalar(<26 x i32> inreg %a, i
; GFX11-NEXT: s_and_b32 s78, s78, exec_lo
; GFX11-NEXT: s_cselect_b32 s78, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s78, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB17_5
; GFX11-NEXT: ; %bb.4: ; %cmp.true
; GFX11-NEXT: s_add_i32 s4, s4, 3
@@ -10313,6 +10326,7 @@ define <26 x i32> @bitcast_v52f16_to_v26i32(<52 x half> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v68, 16, v0
; GFX11-TRUE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-TRUE16-NEXT: v_cmpx_ne_u32_e32 0, v26
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB18_3
; GFX11-TRUE16-NEXT: ; %bb.1: ; %Flow
@@ -11573,6 +11587,7 @@ define inreg <26 x i32> @bitcast_v52f16_to_v26i32_scalar(<52 x half> inreg %a, i
; GFX11-NEXT: s_and_b32 s40, s40, exec_lo
; GFX11-NEXT: s_cselect_b32 s40, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s40, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB19_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_pk_add_f16 v0, 0x200, s0 op_sel_hi:[0,1]
@@ -11760,8 +11775,9 @@ define <13 x i64> @bitcast_v26f32_to_v13i64(<26 x float> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v26
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB20_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -12266,6 +12282,7 @@ define inreg <13 x i64> @bitcast_v26f32_to_v13i64_scalar(<26 x float> inreg %a,
; GFX11-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB21_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_add_f32_e64 v25, s61, 1.0
@@ -12470,43 +12487,37 @@ define <26 x float> @bitcast_v13i64_to_v26f32(<13 x i64> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v26
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB22_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_add_co_u32 v24, vcc_lo, v24, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v25, null, 0, v25, vcc_lo
; GFX11-NEXT: v_add_co_u32 v22, vcc_lo, v22, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v23, null, 0, v23, vcc_lo
; GFX11-NEXT: v_add_co_u32 v20, vcc_lo, v20, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v21, null, 0, v21, vcc_lo
; GFX11-NEXT: v_add_co_u32 v18, vcc_lo, v18, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v19, null, 0, v19, vcc_lo
; GFX11-NEXT: v_add_co_u32 v16, vcc_lo, v16, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v17, null, 0, v17, vcc_lo
; GFX11-NEXT: v_add_co_u32 v14, vcc_lo, v14, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v15, null, 0, v15, vcc_lo
; GFX11-NEXT: v_add_co_u32 v12, vcc_lo, v12, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v13, null, 0, v13, vcc_lo
; GFX11-NEXT: v_add_co_u32 v10, vcc_lo, v10, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v11, null, 0, v11, vcc_lo
; GFX11-NEXT: v_add_co_u32 v8, vcc_lo, v8, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v9, null, 0, v9, vcc_lo
; GFX11-NEXT: v_add_co_u32 v6, vcc_lo, v6, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v7, null, 0, v7, vcc_lo
; GFX11-NEXT: v_add_co_u32 v4, vcc_lo, v4, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v5, null, 0, v5, vcc_lo
; GFX11-NEXT: v_add_co_u32 v2, vcc_lo, v2, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX11-NEXT: v_add_co_u32 v0, vcc_lo, v0, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-NEXT: .LBB22_2: ; %end
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -12803,6 +12814,7 @@ define inreg <26 x float> @bitcast_v13i64_to_v26f32_scalar(<13 x i64> inreg %a,
; GFX11-NEXT: s_and_b32 s12, s12, exec_lo
; GFX11-NEXT: s_cselect_b32 s12, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s12, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB23_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: s_add_u32 s5, s5, 3
@@ -12987,8 +12999,9 @@ define <13 x double> @bitcast_v26f32_to_v13f64(<26 x float> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v26
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB24_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -13493,6 +13506,7 @@ define inreg <13 x double> @bitcast_v26f32_to_v13f64_scalar(<26 x float> inreg %
; GFX11-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB25_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_add_f32_e64 v25, s61, 1.0
@@ -13658,8 +13672,9 @@ define <26 x float> @bitcast_v13f64_to_v26f32(<13 x double> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v26
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB26_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -14125,6 +14140,7 @@ define inreg <26 x float> @bitcast_v13f64_to_v26f32_scalar(<13 x double> inreg %
; GFX11-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB27_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_add_f64 v[24:25], s[60:61], 1.0
@@ -14853,8 +14869,9 @@ define <52 x i16> @bitcast_v26f32_to_v52i16(<26 x float> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v67, 16, v0
; GFX11-TRUE16-NEXT: .LBB28_4: ; %end
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v0.h, v67.l
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v1.h, v66.l
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v2.h, v65.l
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v3.h, v64.l
@@ -14988,8 +15005,9 @@ define <52 x i16> @bitcast_v26f32_to_v52i16(<26 x float> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v67, 16, v0
; GFX11-FAKE16-NEXT: .LBB28_4: ; %end
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_perm_b32 v0, v67, v0, 0x5040100
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_perm_b32 v1, v66, v1, 0x5040100
; GFX11-FAKE16-NEXT: v_perm_b32 v2, v65, v2, 0x5040100
; GFX11-FAKE16-NEXT: v_perm_b32 v3, v64, v3, 0x5040100
@@ -15918,6 +15936,7 @@ define inreg <52 x i16> @bitcast_v26f32_to_v52i16_scalar(<26 x float> inreg %a,
; GFX11-TRUE16-NEXT: s_and_b32 s14, s14, exec_lo
; GFX11-TRUE16-NEXT: s_cselect_b32 s14, 1, 0
; GFX11-TRUE16-NEXT: s_cmp_lg_u32 s14, 1
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_cbranch_scc1 .LBB29_5
; GFX11-TRUE16-NEXT: ; %bb.4: ; %cmp.true
; GFX11-TRUE16-NEXT: v_add_f32_e64 v25, s4, 1.0
@@ -16106,6 +16125,7 @@ define inreg <52 x i16> @bitcast_v26f32_to_v52i16_scalar(<26 x float> inreg %a,
; GFX11-FAKE16-NEXT: s_and_b32 s14, s14, exec_lo
; GFX11-FAKE16-NEXT: s_cselect_b32 s14, 1, 0
; GFX11-FAKE16-NEXT: s_cmp_lg_u32 s14, 1
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_cbranch_scc1 .LBB29_5
; GFX11-FAKE16-NEXT: ; %bb.4: ; %cmp.true
; GFX11-FAKE16-NEXT: v_add_f32_e64 v21, s4, 1.0
@@ -17291,6 +17311,7 @@ define <26 x float> @bitcast_v52i16_to_v26f32(<52 x i16> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v68, 16, v0
; GFX11-TRUE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-TRUE16-NEXT: v_cmpx_ne_u32_e32 0, v26
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB30_3
; GFX11-TRUE16-NEXT: ; %bb.1: ; %Flow
@@ -18481,6 +18502,7 @@ define inreg <26 x float> @bitcast_v52i16_to_v26f32_scalar(<52 x i16> inreg %a,
; GFX11-NEXT: s_and_b32 s40, s40, exec_lo
; GFX11-NEXT: s_cselect_b32 s40, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s40, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB31_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_pk_add_u16 v0, s0, 3 op_sel_hi:[1,0]
@@ -19205,8 +19227,9 @@ define <52 x half> @bitcast_v26f32_to_v52f16(<26 x float> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v67, 16, v0
; GFX11-TRUE16-NEXT: .LBB32_4: ; %end
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v0.h, v67.l
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v1.h, v66.l
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v2.h, v65.l
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v3.h, v64.l
@@ -19340,8 +19363,9 @@ define <52 x half> @bitcast_v26f32_to_v52f16(<26 x float> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v67, 16, v0
; GFX11-FAKE16-NEXT: .LBB32_4: ; %end
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_perm_b32 v0, v67, v0, 0x5040100
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_perm_b32 v1, v66, v1, 0x5040100
; GFX11-FAKE16-NEXT: v_perm_b32 v2, v65, v2, 0x5040100
; GFX11-FAKE16-NEXT: v_perm_b32 v3, v64, v3, 0x5040100
@@ -20270,6 +20294,7 @@ define inreg <52 x half> @bitcast_v26f32_to_v52f16_scalar(<26 x float> inreg %a,
; GFX11-TRUE16-NEXT: s_and_b32 s14, s14, exec_lo
; GFX11-TRUE16-NEXT: s_cselect_b32 s14, 1, 0
; GFX11-TRUE16-NEXT: s_cmp_lg_u32 s14, 1
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_cbranch_scc1 .LBB33_5
; GFX11-TRUE16-NEXT: ; %bb.4: ; %cmp.true
; GFX11-TRUE16-NEXT: v_add_f32_e64 v25, s4, 1.0
@@ -20458,6 +20483,7 @@ define inreg <52 x half> @bitcast_v26f32_to_v52f16_scalar(<26 x float> inreg %a,
; GFX11-FAKE16-NEXT: s_and_b32 s14, s14, exec_lo
; GFX11-FAKE16-NEXT: s_cselect_b32 s14, 1, 0
; GFX11-FAKE16-NEXT: s_cmp_lg_u32 s14, 1
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_cbranch_scc1 .LBB33_5
; GFX11-FAKE16-NEXT: ; %bb.4: ; %cmp.true
; GFX11-FAKE16-NEXT: v_add_f32_e64 v21, s4, 1.0
@@ -21764,6 +21790,7 @@ define <26 x float> @bitcast_v52f16_to_v26f32(<52 x half> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v68, 16, v0
; GFX11-TRUE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-TRUE16-NEXT: v_cmpx_ne_u32_e32 0, v26
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB34_3
; GFX11-TRUE16-NEXT: ; %bb.1: ; %Flow
@@ -23024,6 +23051,7 @@ define inreg <26 x float> @bitcast_v52f16_to_v26f32_scalar(<52 x half> inreg %a,
; GFX11-NEXT: s_and_b32 s40, s40, exec_lo
; GFX11-NEXT: s_cselect_b32 s40, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s40, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB35_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_pk_add_f16 v0, 0x200, s0 op_sel_hi:[0,1]
@@ -23211,43 +23239,37 @@ define <13 x double> @bitcast_v13i64_to_v13f64(<13 x i64> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v26
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB36_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_add_co_u32 v0, vcc_lo, v0, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-NEXT: v_add_co_u32 v2, vcc_lo, v2, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX11-NEXT: v_add_co_u32 v4, vcc_lo, v4, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v5, null, 0, v5, vcc_lo
; GFX11-NEXT: v_add_co_u32 v6, vcc_lo, v6, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v7, null, 0, v7, vcc_lo
; GFX11-NEXT: v_add_co_u32 v8, vcc_lo, v8, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v9, null, 0, v9, vcc_lo
; GFX11-NEXT: v_add_co_u32 v10, vcc_lo, v10, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v11, null, 0, v11, vcc_lo
; GFX11-NEXT: v_add_co_u32 v12, vcc_lo, v12, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v13, null, 0, v13, vcc_lo
; GFX11-NEXT: v_add_co_u32 v14, vcc_lo, v14, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v15, null, 0, v15, vcc_lo
; GFX11-NEXT: v_add_co_u32 v16, vcc_lo, v16, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v17, null, 0, v17, vcc_lo
; GFX11-NEXT: v_add_co_u32 v18, vcc_lo, v18, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v19, null, 0, v19, vcc_lo
; GFX11-NEXT: v_add_co_u32 v20, vcc_lo, v20, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v21, null, 0, v21, vcc_lo
; GFX11-NEXT: v_add_co_u32 v22, vcc_lo, v22, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v23, null, 0, v23, vcc_lo
; GFX11-NEXT: v_add_co_u32 v24, vcc_lo, v24, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v25, null, 0, v25, vcc_lo
; GFX11-NEXT: .LBB36_2: ; %end
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -23544,6 +23566,7 @@ define inreg <13 x double> @bitcast_v13i64_to_v13f64_scalar(<13 x i64> inreg %a,
; GFX11-NEXT: s_and_b32 s12, s12, exec_lo
; GFX11-NEXT: s_cselect_b32 s12, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s12, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB37_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: s_add_u32 s0, s0, 3
@@ -23688,8 +23711,9 @@ define <13 x i64> @bitcast_v13f64_to_v13i64(<13 x double> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v26
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB38_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -24155,6 +24179,7 @@ define inreg <13 x i64> @bitcast_v13f64_to_v13i64_scalar(<13 x double> inreg %a,
; GFX11-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB39_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_add_f64 v[0:1], s[36:37], 1.0
@@ -24843,37 +24868,30 @@ define <52 x i16> @bitcast_v13i64_to_v52i16(<13 x i64> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: s_cbranch_execz .LBB40_4
; GFX11-TRUE16-NEXT: ; %bb.3: ; %cmp.true
; GFX11-TRUE16-NEXT: v_add_co_u32 v24, vcc_lo, v24, 3
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v25, null, 0, v25, vcc_lo
; GFX11-TRUE16-NEXT: v_add_co_u32 v22, vcc_lo, v22, 3
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v23, null, 0, v23, vcc_lo
; GFX11-TRUE16-NEXT: v_add_co_u32 v20, vcc_lo, v20, 3
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v21, null, 0, v21, vcc_lo
; GFX11-TRUE16-NEXT: v_add_co_u32 v18, vcc_lo, v18, 3
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v19, null, 0, v19, vcc_lo
; GFX11-TRUE16-NEXT: v_add_co_u32 v16, vcc_lo, v16, 3
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v17, null, 0, v17, vcc_lo
; GFX11-TRUE16-NEXT: v_add_co_u32 v14, vcc_lo, v14, 3
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v15, null, 0, v15, vcc_lo
; GFX11-TRUE16-NEXT: v_add_co_u32 v12, vcc_lo, v12, 3
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v13, null, 0, v13, vcc_lo
; GFX11-TRUE16-NEXT: v_add_co_u32 v10, vcc_lo, v10, 3
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v11, null, 0, v11, vcc_lo
; GFX11-TRUE16-NEXT: v_add_co_u32 v8, vcc_lo, v8, 3
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v9, null, 0, v9, vcc_lo
; GFX11-TRUE16-NEXT: v_add_co_u32 v6, vcc_lo, v6, 3
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v7, null, 0, v7, vcc_lo
; GFX11-TRUE16-NEXT: v_add_co_u32 v4, vcc_lo, v4, 3
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v5, null, 0, v5, vcc_lo
; GFX11-TRUE16-NEXT: v_add_co_u32 v2, vcc_lo, v2, 3
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX11-TRUE16-NEXT: v_add_co_u32 v0, vcc_lo, v0, 3
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v26, 16, v25
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v27, 16, v24
@@ -24903,8 +24921,9 @@ define <52 x i16> @bitcast_v13i64_to_v52i16(<13 x i64> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v67, 16, v0
; GFX11-TRUE16-NEXT: .LBB40_4: ; %end
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v0.h, v67.l
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v1.h, v66.l
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v2.h, v65.l
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v3.h, v64.l
@@ -24998,37 +25017,30 @@ define <52 x i16> @bitcast_v13i64_to_v52i16(<13 x i64> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: s_cbranch_execz .LBB40_4
; GFX11-FAKE16-NEXT: ; %bb.3: ; %cmp.true
; GFX11-FAKE16-NEXT: v_add_co_u32 v24, vcc_lo, v24, 3
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v25, null, 0, v25, vcc_lo
; GFX11-FAKE16-NEXT: v_add_co_u32 v22, vcc_lo, v22, 3
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v23, null, 0, v23, vcc_lo
; GFX11-FAKE16-NEXT: v_add_co_u32 v20, vcc_lo, v20, 3
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v21, null, 0, v21, vcc_lo
; GFX11-FAKE16-NEXT: v_add_co_u32 v18, vcc_lo, v18, 3
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v19, null, 0, v19, vcc_lo
; GFX11-FAKE16-NEXT: v_add_co_u32 v16, vcc_lo, v16, 3
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v17, null, 0, v17, vcc_lo
; GFX11-FAKE16-NEXT: v_add_co_u32 v14, vcc_lo, v14, 3
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v15, null, 0, v15, vcc_lo
; GFX11-FAKE16-NEXT: v_add_co_u32 v12, vcc_lo, v12, 3
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v13, null, 0, v13, vcc_lo
; GFX11-FAKE16-NEXT: v_add_co_u32 v10, vcc_lo, v10, 3
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v11, null, 0, v11, vcc_lo
; GFX11-FAKE16-NEXT: v_add_co_u32 v8, vcc_lo, v8, 3
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v9, null, 0, v9, vcc_lo
; GFX11-FAKE16-NEXT: v_add_co_u32 v6, vcc_lo, v6, 3
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v7, null, 0, v7, vcc_lo
; GFX11-FAKE16-NEXT: v_add_co_u32 v4, vcc_lo, v4, 3
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v5, null, 0, v5, vcc_lo
; GFX11-FAKE16-NEXT: v_add_co_u32 v2, vcc_lo, v2, 3
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX11-FAKE16-NEXT: v_add_co_u32 v0, vcc_lo, v0, 3
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v26, 16, v25
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v27, 16, v24
@@ -25058,8 +25070,9 @@ define <52 x i16> @bitcast_v13i64_to_v52i16(<13 x i64> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v67, 16, v0
; GFX11-FAKE16-NEXT: .LBB40_4: ; %end
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_perm_b32 v0, v67, v0, 0x5040100
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_perm_b32 v1, v66, v1, 0x5040100
; GFX11-FAKE16-NEXT: v_perm_b32 v2, v65, v2, 0x5040100
; GFX11-FAKE16-NEXT: v_perm_b32 v3, v64, v3, 0x5040100
@@ -25870,6 +25883,7 @@ define inreg <52 x i16> @bitcast_v13i64_to_v52i16_scalar(<13 x i64> inreg %a, i3
; GFX11-NEXT: s_and_b32 s78, s78, exec_lo
; GFX11-NEXT: s_cselect_b32 s78, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s78, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB41_5
; GFX11-NEXT: ; %bb.4: ; %cmp.true
; GFX11-NEXT: s_add_u32 s5, s5, 3
@@ -27015,6 +27029,7 @@ define <13 x i64> @bitcast_v52i16_to_v13i64(<52 x i16> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v68, 16, v0
; GFX11-TRUE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-TRUE16-NEXT: v_cmpx_ne_u32_e32 0, v26
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB42_3
; GFX11-TRUE16-NEXT: ; %bb.1: ; %Flow
@@ -28205,6 +28220,7 @@ define inreg <13 x i64> @bitcast_v52i16_to_v13i64_scalar(<52 x i16> inreg %a, i3
; GFX11-NEXT: s_and_b32 s40, s40, exec_lo
; GFX11-NEXT: s_cselect_b32 s40, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s40, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB43_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_pk_add_u16 v0, s0, 3 op_sel_hi:[1,0]
@@ -28889,37 +28905,30 @@ define <52 x half> @bitcast_v13i64_to_v52f16(<13 x i64> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: s_cbranch_execz .LBB44_4
; GFX11-TRUE16-NEXT: ; %bb.3: ; %cmp.true
; GFX11-TRUE16-NEXT: v_add_co_u32 v24, vcc_lo, v24, 3
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v25, null, 0, v25, vcc_lo
; GFX11-TRUE16-NEXT: v_add_co_u32 v22, vcc_lo, v22, 3
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v23, null, 0, v23, vcc_lo
; GFX11-TRUE16-NEXT: v_add_co_u32 v20, vcc_lo, v20, 3
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v21, null, 0, v21, vcc_lo
; GFX11-TRUE16-NEXT: v_add_co_u32 v18, vcc_lo, v18, 3
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v19, null, 0, v19, vcc_lo
; GFX11-TRUE16-NEXT: v_add_co_u32 v16, vcc_lo, v16, 3
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v17, null, 0, v17, vcc_lo
; GFX11-TRUE16-NEXT: v_add_co_u32 v14, vcc_lo, v14, 3
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v15, null, 0, v15, vcc_lo
; GFX11-TRUE16-NEXT: v_add_co_u32 v12, vcc_lo, v12, 3
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v13, null, 0, v13, vcc_lo
; GFX11-TRUE16-NEXT: v_add_co_u32 v10, vcc_lo, v10, 3
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v11, null, 0, v11, vcc_lo
; GFX11-TRUE16-NEXT: v_add_co_u32 v8, vcc_lo, v8, 3
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v9, null, 0, v9, vcc_lo
; GFX11-TRUE16-NEXT: v_add_co_u32 v6, vcc_lo, v6, 3
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v7, null, 0, v7, vcc_lo
; GFX11-TRUE16-NEXT: v_add_co_u32 v4, vcc_lo, v4, 3
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v5, null, 0, v5, vcc_lo
; GFX11-TRUE16-NEXT: v_add_co_u32 v2, vcc_lo, v2, 3
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX11-TRUE16-NEXT: v_add_co_u32 v0, vcc_lo, v0, 3
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v26, 16, v25
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v27, 16, v24
@@ -28949,8 +28958,9 @@ define <52 x half> @bitcast_v13i64_to_v52f16(<13 x i64> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v67, 16, v0
; GFX11-TRUE16-NEXT: .LBB44_4: ; %end
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v0.h, v67.l
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v1.h, v66.l
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v2.h, v65.l
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v3.h, v64.l
@@ -29044,37 +29054,30 @@ define <52 x half> @bitcast_v13i64_to_v52f16(<13 x i64> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: s_cbranch_execz .LBB44_4
; GFX11-FAKE16-NEXT: ; %bb.3: ; %cmp.true
; GFX11-FAKE16-NEXT: v_add_co_u32 v24, vcc_lo, v24, 3
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v25, null, 0, v25, vcc_lo
; GFX11-FAKE16-NEXT: v_add_co_u32 v22, vcc_lo, v22, 3
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v23, null, 0, v23, vcc_lo
; GFX11-FAKE16-NEXT: v_add_co_u32 v20, vcc_lo, v20, 3
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v21, null, 0, v21, vcc_lo
; GFX11-FAKE16-NEXT: v_add_co_u32 v18, vcc_lo, v18, 3
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v19, null, 0, v19, vcc_lo
; GFX11-FAKE16-NEXT: v_add_co_u32 v16, vcc_lo, v16, 3
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v17, null, 0, v17, vcc_lo
; GFX11-FAKE16-NEXT: v_add_co_u32 v14, vcc_lo, v14, 3
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v15, null, 0, v15, vcc_lo
; GFX11-FAKE16-NEXT: v_add_co_u32 v12, vcc_lo, v12, 3
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v13, null, 0, v13, vcc_lo
; GFX11-FAKE16-NEXT: v_add_co_u32 v10, vcc_lo, v10, 3
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v11, null, 0, v11, vcc_lo
; GFX11-FAKE16-NEXT: v_add_co_u32 v8, vcc_lo, v8, 3
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v9, null, 0, v9, vcc_lo
; GFX11-FAKE16-NEXT: v_add_co_u32 v6, vcc_lo, v6, 3
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v7, null, 0, v7, vcc_lo
; GFX11-FAKE16-NEXT: v_add_co_u32 v4, vcc_lo, v4, 3
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v5, null, 0, v5, vcc_lo
; GFX11-FAKE16-NEXT: v_add_co_u32 v2, vcc_lo, v2, 3
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX11-FAKE16-NEXT: v_add_co_u32 v0, vcc_lo, v0, 3
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v26, 16, v25
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v27, 16, v24
@@ -29104,8 +29107,9 @@ define <52 x half> @bitcast_v13i64_to_v52f16(<13 x i64> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v67, 16, v0
; GFX11-FAKE16-NEXT: .LBB44_4: ; %end
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_perm_b32 v0, v67, v0, 0x5040100
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_perm_b32 v1, v66, v1, 0x5040100
; GFX11-FAKE16-NEXT: v_perm_b32 v2, v65, v2, 0x5040100
; GFX11-FAKE16-NEXT: v_perm_b32 v3, v64, v3, 0x5040100
@@ -29916,6 +29920,7 @@ define inreg <52 x half> @bitcast_v13i64_to_v52f16_scalar(<13 x i64> inreg %a, i
; GFX11-NEXT: s_and_b32 s78, s78, exec_lo
; GFX11-NEXT: s_cselect_b32 s78, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s78, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB45_5
; GFX11-NEXT: ; %bb.4: ; %cmp.true
; GFX11-NEXT: s_add_u32 s5, s5, 3
@@ -31182,6 +31187,7 @@ define <13 x i64> @bitcast_v52f16_to_v13i64(<52 x half> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v68, 16, v0
; GFX11-TRUE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-TRUE16-NEXT: v_cmpx_ne_u32_e32 0, v26
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB46_3
; GFX11-TRUE16-NEXT: ; %bb.1: ; %Flow
@@ -32442,6 +32448,7 @@ define inreg <13 x i64> @bitcast_v52f16_to_v13i64_scalar(<52 x half> inreg %a, i
; GFX11-NEXT: s_and_b32 s40, s40, exec_lo
; GFX11-NEXT: s_cselect_b32 s40, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s40, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB47_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_pk_add_f16 v0, 0x200, s0 op_sel_hi:[0,1]
@@ -33127,8 +33134,9 @@ define <52 x i16> @bitcast_v13f64_to_v52i16(<13 x double> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v67, 16, v0
; GFX11-TRUE16-NEXT: .LBB48_4: ; %end
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v0.h, v67.l
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v1.h, v66.l
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v2.h, v65.l
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v3.h, v64.l
@@ -33262,8 +33270,9 @@ define <52 x i16> @bitcast_v13f64_to_v52i16(<13 x double> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v67, 16, v0
; GFX11-FAKE16-NEXT: .LBB48_4: ; %end
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_perm_b32 v0, v67, v0, 0x5040100
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_perm_b32 v1, v66, v1, 0x5040100
; GFX11-FAKE16-NEXT: v_perm_b32 v2, v65, v2, 0x5040100
; GFX11-FAKE16-NEXT: v_perm_b32 v3, v64, v3, 0x5040100
@@ -34157,6 +34166,7 @@ define inreg <52 x i16> @bitcast_v13f64_to_v52i16_scalar(<13 x double> inreg %a,
; GFX11-TRUE16-NEXT: s_and_b32 s14, s14, exec_lo
; GFX11-TRUE16-NEXT: s_cselect_b32 s14, 1, 0
; GFX11-TRUE16-NEXT: s_cmp_lg_u32 s14, 1
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_cbranch_scc1 .LBB49_5
; GFX11-TRUE16-NEXT: ; %bb.4: ; %cmp.true
; GFX11-TRUE16-NEXT: v_add_f64 v[24:25], s[8:9], 1.0
@@ -34332,6 +34342,7 @@ define inreg <52 x i16> @bitcast_v13f64_to_v52i16_scalar(<13 x double> inreg %a,
; GFX11-FAKE16-NEXT: s_and_b32 s14, s14, exec_lo
; GFX11-FAKE16-NEXT: s_cselect_b32 s14, 1, 0
; GFX11-FAKE16-NEXT: s_cmp_lg_u32 s14, 1
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_cbranch_scc1 .LBB49_5
; GFX11-FAKE16-NEXT: ; %bb.4: ; %cmp.true
; GFX11-FAKE16-NEXT: v_add_f64 v[21:22], s[8:9], 1.0
@@ -35504,6 +35515,7 @@ define <13 x double> @bitcast_v52i16_to_v13f64(<52 x i16> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v68, 16, v0
; GFX11-TRUE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-TRUE16-NEXT: v_cmpx_ne_u32_e32 0, v26
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB50_3
; GFX11-TRUE16-NEXT: ; %bb.1: ; %Flow
@@ -36694,6 +36706,7 @@ define inreg <13 x double> @bitcast_v52i16_to_v13f64_scalar(<52 x i16> inreg %a,
; GFX11-NEXT: s_and_b32 s40, s40, exec_lo
; GFX11-NEXT: s_cselect_b32 s40, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s40, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB51_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_pk_add_u16 v0, s0, 3 op_sel_hi:[1,0]
@@ -37379,8 +37392,9 @@ define <52 x half> @bitcast_v13f64_to_v52f16(<13 x double> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v67, 16, v0
; GFX11-TRUE16-NEXT: .LBB52_4: ; %end
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v0.h, v67.l
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v1.h, v66.l
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v2.h, v65.l
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v3.h, v64.l
@@ -37514,8 +37528,9 @@ define <52 x half> @bitcast_v13f64_to_v52f16(<13 x double> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v67, 16, v0
; GFX11-FAKE16-NEXT: .LBB52_4: ; %end
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_perm_b32 v0, v67, v0, 0x5040100
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_perm_b32 v1, v66, v1, 0x5040100
; GFX11-FAKE16-NEXT: v_perm_b32 v2, v65, v2, 0x5040100
; GFX11-FAKE16-NEXT: v_perm_b32 v3, v64, v3, 0x5040100
@@ -38409,6 +38424,7 @@ define inreg <52 x half> @bitcast_v13f64_to_v52f16_scalar(<13 x double> inreg %a
; GFX11-TRUE16-NEXT: s_and_b32 s14, s14, exec_lo
; GFX11-TRUE16-NEXT: s_cselect_b32 s14, 1, 0
; GFX11-TRUE16-NEXT: s_cmp_lg_u32 s14, 1
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_cbranch_scc1 .LBB53_5
; GFX11-TRUE16-NEXT: ; %bb.4: ; %cmp.true
; GFX11-TRUE16-NEXT: v_add_f64 v[24:25], s[8:9], 1.0
@@ -38584,6 +38600,7 @@ define inreg <52 x half> @bitcast_v13f64_to_v52f16_scalar(<13 x double> inreg %a
; GFX11-FAKE16-NEXT: s_and_b32 s14, s14, exec_lo
; GFX11-FAKE16-NEXT: s_cselect_b32 s14, 1, 0
; GFX11-FAKE16-NEXT: s_cmp_lg_u32 s14, 1
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_cbranch_scc1 .LBB53_5
; GFX11-FAKE16-NEXT: ; %bb.4: ; %cmp.true
; GFX11-FAKE16-NEXT: v_add_f64 v[21:22], s[8:9], 1.0
@@ -39877,6 +39894,7 @@ define <13 x double> @bitcast_v52f16_to_v13f64(<52 x half> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v68, 16, v0
; GFX11-TRUE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-TRUE16-NEXT: v_cmpx_ne_u32_e32 0, v26
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB54_3
; GFX11-TRUE16-NEXT: ; %bb.1: ; %Flow
@@ -41137,6 +41155,7 @@ define inreg <13 x double> @bitcast_v52f16_to_v13f64_scalar(<52 x half> inreg %a
; GFX11-NEXT: s_and_b32 s40, s40, exec_lo
; GFX11-NEXT: s_cselect_b32 s40, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s40, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB55_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_pk_add_f16 v0, 0x200, s0 op_sel_hi:[0,1]
@@ -42258,8 +42277,9 @@ define <52 x half> @bitcast_v52i16_to_v52f16(<52 x i16> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v31, 16, v0
; GFX11-TRUE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-TRUE16-NEXT: v_cmpx_ne_u32_e32 0, v26
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-TRUE16-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-TRUE16-NEXT: s_cbranch_execz .LBB56_2
; GFX11-TRUE16-NEXT: ; %bb.1: ; %cmp.true
@@ -42343,6 +42363,7 @@ define <52 x half> @bitcast_v52i16_to_v52f16(<52 x i16> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v68, 16, v25
; GFX11-TRUE16-NEXT: .LBB56_2: ; %end
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v0.h, v31.l
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v1.h, v30.l
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v2.h, v29.l
@@ -42402,8 +42423,9 @@ define <52 x half> @bitcast_v52i16_to_v52f16(<52 x i16> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v27, 16, v0
; GFX11-FAKE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-FAKE16-NEXT: v_cmpx_ne_u32_e32 0, v26
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-FAKE16-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-FAKE16-NEXT: s_cbranch_execz .LBB56_2
; GFX11-FAKE16-NEXT: ; %bb.1: ; %cmp.true
@@ -42487,6 +42509,7 @@ define <52 x half> @bitcast_v52i16_to_v52f16(<52 x i16> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v68, 16, v25
; GFX11-FAKE16-NEXT: .LBB56_2: ; %end
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_perm_b32 v0, v27, v0, 0x5040100
; GFX11-FAKE16-NEXT: v_perm_b32 v1, v28, v1, 0x5040100
; GFX11-FAKE16-NEXT: v_perm_b32 v2, v29, v2, 0x5040100
@@ -43593,6 +43616,7 @@ define inreg <52 x half> @bitcast_v52i16_to_v52f16_scalar(<52 x i16> inreg %a, i
; GFX11-TRUE16-NEXT: s_and_b32 s78, s78, exec_lo
; GFX11-TRUE16-NEXT: s_cselect_b32 s78, 1, 0
; GFX11-TRUE16-NEXT: s_cmp_lg_u32 s78, 1
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_cbranch_scc1 .LBB57_4
; GFX11-TRUE16-NEXT: ; %bb.3: ; %cmp.true
; GFX11-TRUE16-NEXT: s_pack_ll_b32_b16 s57, s74, s57
@@ -43780,6 +43804,7 @@ define inreg <52 x half> @bitcast_v52i16_to_v52f16_scalar(<52 x i16> inreg %a, i
; GFX11-FAKE16-NEXT: s_and_b32 s78, s78, exec_lo
; GFX11-FAKE16-NEXT: s_cselect_b32 s78, 1, 0
; GFX11-FAKE16-NEXT: s_cmp_lg_u32 s78, 1
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_cbranch_scc1 .LBB57_4
; GFX11-FAKE16-NEXT: ; %bb.3: ; %cmp.true
; GFX11-FAKE16-NEXT: s_pack_ll_b32_b16 s57, s74, s57
@@ -44664,8 +44689,9 @@ define <52 x i16> @bitcast_v52f16_to_v52i16(<52 x half> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v31, 16, v0
; GFX11-TRUE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-TRUE16-NEXT: v_cmpx_ne_u32_e32 0, v26
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-TRUE16-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-TRUE16-NEXT: s_cbranch_execz .LBB58_2
; GFX11-TRUE16-NEXT: ; %bb.1: ; %cmp.true
@@ -44749,6 +44775,7 @@ define <52 x i16> @bitcast_v52f16_to_v52i16(<52 x half> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v68, 16, v25
; GFX11-TRUE16-NEXT: .LBB58_2: ; %end
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v0.h, v31.l
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v1.h, v30.l
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v2.h, v29.l
@@ -44808,8 +44835,9 @@ define <52 x i16> @bitcast_v52f16_to_v52i16(<52 x half> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v27, 16, v0
; GFX11-FAKE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-FAKE16-NEXT: v_cmpx_ne_u32_e32 0, v26
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-FAKE16-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-FAKE16-NEXT: s_cbranch_execz .LBB58_2
; GFX11-FAKE16-NEXT: ; %bb.1: ; %cmp.true
@@ -44893,6 +44921,7 @@ define <52 x i16> @bitcast_v52f16_to_v52i16(<52 x half> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v68, 16, v25
; GFX11-FAKE16-NEXT: .LBB58_2: ; %end
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_perm_b32 v0, v27, v0, 0x5040100
; GFX11-FAKE16-NEXT: v_perm_b32 v1, v28, v1, 0x5040100
; GFX11-FAKE16-NEXT: v_perm_b32 v2, v29, v2, 0x5040100
@@ -45932,6 +45961,7 @@ define inreg <52 x i16> @bitcast_v52f16_to_v52i16_scalar(<52 x half> inreg %a, i
; GFX11-TRUE16-NEXT: s_and_b32 s78, s78, exec_lo
; GFX11-TRUE16-NEXT: s_cselect_b32 s78, 1, 0
; GFX11-TRUE16-NEXT: s_cmp_lg_u32 s78, 1
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_cbranch_scc1 .LBB59_4
; GFX11-TRUE16-NEXT: ; %bb.3: ; %cmp.true
; GFX11-TRUE16-NEXT: s_pack_ll_b32_b16 s57, s74, s57
@@ -46119,6 +46149,7 @@ define inreg <52 x i16> @bitcast_v52f16_to_v52i16_scalar(<52 x half> inreg %a, i
; GFX11-FAKE16-NEXT: s_and_b32 s78, s78, exec_lo
; GFX11-FAKE16-NEXT: s_cselect_b32 s78, 1, 0
; GFX11-FAKE16-NEXT: s_cmp_lg_u32 s78, 1
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_cbranch_scc1 .LBB59_4
; GFX11-FAKE16-NEXT: ; %bb.3: ; %cmp.true
; GFX11-FAKE16-NEXT: s_pack_ll_b32_b16 s57, s74, s57
diff --git a/llvm/test/CodeGen/AMDGPU/amdgcn.bitcast.896bit.ll b/llvm/test/CodeGen/AMDGPU/amdgcn.bitcast.896bit.ll
index 8319358c1e3563..3d6a53d1189428 100644
--- a/llvm/test/CodeGen/AMDGPU/amdgcn.bitcast.896bit.ll
+++ b/llvm/test/CodeGen/AMDGPU/amdgcn.bitcast.896bit.ll
@@ -135,8 +135,9 @@ define <28 x float> @bitcast_v28i32_to_v28f32(<28 x i32> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v28
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB0_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -483,6 +484,7 @@ define inreg <28 x float> @bitcast_v28i32_to_v28f32_scalar(<28 x i32> inreg %a,
; GFX11-NEXT: s_and_b32 s14, s14, exec_lo
; GFX11-NEXT: s_cselect_b32 s14, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s14, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB1_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: s_add_i32 s4, s4, 3
@@ -676,8 +678,9 @@ define <28 x i32> @bitcast_v28f32_to_v28i32(<28 x float> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v28
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB2_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -1197,6 +1200,7 @@ define inreg <28 x i32> @bitcast_v28f32_to_v28i32_scalar(<28 x float> inreg %a,
; GFX11-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB3_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_add_f32_e64 v27, s63, 1.0
@@ -1409,8 +1413,9 @@ define <14 x i64> @bitcast_v28i32_to_v14i64(<28 x i32> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v28
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB4_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -1757,6 +1762,7 @@ define inreg <14 x i64> @bitcast_v28i32_to_v14i64_scalar(<28 x i32> inreg %a, i3
; GFX11-NEXT: s_and_b32 s14, s14, exec_lo
; GFX11-NEXT: s_cselect_b32 s14, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s14, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB5_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: s_add_i32 s4, s4, 3
@@ -1950,43 +1956,37 @@ define <28 x i32> @bitcast_v14i64_to_v28i32(<14 x i64> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v28
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB6_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_add_co_u32 v26, vcc_lo, v26, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v27, null, 0, v27, vcc_lo
; GFX11-NEXT: v_add_co_u32 v24, vcc_lo, v24, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v25, null, 0, v25, vcc_lo
; GFX11-NEXT: v_add_co_u32 v22, vcc_lo, v22, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v23, null, 0, v23, vcc_lo
; GFX11-NEXT: v_add_co_u32 v20, vcc_lo, v20, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v21, null, 0, v21, vcc_lo
; GFX11-NEXT: v_add_co_u32 v18, vcc_lo, v18, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v19, null, 0, v19, vcc_lo
; GFX11-NEXT: v_add_co_u32 v16, vcc_lo, v16, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v17, null, 0, v17, vcc_lo
; GFX11-NEXT: v_add_co_u32 v14, vcc_lo, v14, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v15, null, 0, v15, vcc_lo
; GFX11-NEXT: v_add_co_u32 v12, vcc_lo, v12, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v13, null, 0, v13, vcc_lo
; GFX11-NEXT: v_add_co_u32 v10, vcc_lo, v10, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v11, null, 0, v11, vcc_lo
; GFX11-NEXT: v_add_co_u32 v8, vcc_lo, v8, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v9, null, 0, v9, vcc_lo
; GFX11-NEXT: v_add_co_u32 v6, vcc_lo, v6, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v7, null, 0, v7, vcc_lo
; GFX11-NEXT: v_add_co_u32 v4, vcc_lo, v4, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v5, null, 0, v5, vcc_lo
; GFX11-NEXT: v_add_co_u32 v2, vcc_lo, v2, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX11-NEXT: v_add_co_u32 v0, vcc_lo, v0, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
@@ -2305,6 +2305,7 @@ define inreg <28 x i32> @bitcast_v14i64_to_v28i32_scalar(<14 x i64> inreg %a, i3
; GFX11-NEXT: s_and_b32 s14, s14, exec_lo
; GFX11-NEXT: s_cselect_b32 s14, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s14, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB7_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: s_add_u32 s5, s5, 3
@@ -2498,8 +2499,9 @@ define <14 x double> @bitcast_v28i32_to_v14f64(<28 x i32> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v28
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB8_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -2846,6 +2848,7 @@ define inreg <14 x double> @bitcast_v28i32_to_v14f64_scalar(<28 x i32> inreg %a,
; GFX11-NEXT: s_and_b32 s14, s14, exec_lo
; GFX11-NEXT: s_cselect_b32 s14, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s14, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB9_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: s_add_i32 s4, s4, 3
@@ -2997,8 +3000,9 @@ define <28 x i32> @bitcast_v14f64_to_v28i32(<14 x double> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v28
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB10_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -3476,6 +3480,7 @@ define inreg <28 x i32> @bitcast_v14f64_to_v28i32_scalar(<14 x double> inreg %a,
; GFX11-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB11_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_add_f64 v[26:27], s[62:63], 1.0
@@ -4292,8 +4297,9 @@ define <56 x i16> @bitcast_v28i32_to_v56i16(<28 x i32> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v71, 16, v0
; GFX11-TRUE16-NEXT: .LBB12_4: ; %end
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v0.h, v71.l
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v1.h, v70.l
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v2.h, v69.l
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v3.h, v68.l
@@ -4450,8 +4456,9 @@ define <56 x i16> @bitcast_v28i32_to_v56i16(<28 x i32> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v71, 16, v0
; GFX11-FAKE16-NEXT: .LBB12_4: ; %end
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_perm_b32 v0, v71, v0, 0x5040100
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_perm_b32 v1, v70, v1, 0x5040100
; GFX11-FAKE16-NEXT: v_perm_b32 v2, v69, v2, 0x5040100
; GFX11-FAKE16-NEXT: v_perm_b32 v3, v68, v3, 0x5040100
@@ -5341,6 +5348,7 @@ define inreg <56 x i16> @bitcast_v28i32_to_v56i16_scalar(<28 x i32> inreg %a, i3
; GFX11-NEXT: s_and_b32 s90, s90, exec_lo
; GFX11-NEXT: s_cselect_b32 s90, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s90, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB13_5
; GFX11-NEXT: ; %bb.4: ; %cmp.true
; GFX11-NEXT: s_add_i32 s4, s4, 3
@@ -6583,6 +6591,7 @@ define <28 x i32> @bitcast_v56i16_to_v28i32(<56 x i16> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v80, 16, v0
; GFX11-TRUE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-TRUE16-NEXT: v_cmpx_ne_u32_e32 0, v28
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB14_3
; GFX11-TRUE16-NEXT: ; %bb.1: ; %Flow
@@ -7938,6 +7947,7 @@ define inreg <28 x i32> @bitcast_v56i16_to_v28i32_scalar(<56 x i16> inreg %a, i3
; GFX11-NEXT: s_and_b32 s40, s40, exec_lo
; GFX11-NEXT: s_cselect_b32 s40, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s40, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB15_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_pk_add_u16 v0, s0, 3 op_sel_hi:[1,0]
@@ -8751,8 +8761,9 @@ define <56 x half> @bitcast_v28i32_to_v56f16(<28 x i32> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v71, 16, v0
; GFX11-TRUE16-NEXT: .LBB16_4: ; %end
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v0.h, v71.l
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v1.h, v70.l
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v2.h, v69.l
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v3.h, v68.l
@@ -8909,8 +8920,9 @@ define <56 x half> @bitcast_v28i32_to_v56f16(<28 x i32> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v71, 16, v0
; GFX11-FAKE16-NEXT: .LBB16_4: ; %end
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_perm_b32 v0, v71, v0, 0x5040100
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_perm_b32 v1, v70, v1, 0x5040100
; GFX11-FAKE16-NEXT: v_perm_b32 v2, v69, v2, 0x5040100
; GFX11-FAKE16-NEXT: v_perm_b32 v3, v68, v3, 0x5040100
@@ -9800,6 +9812,7 @@ define inreg <56 x half> @bitcast_v28i32_to_v56f16_scalar(<28 x i32> inreg %a, i
; GFX11-NEXT: s_and_b32 s90, s90, exec_lo
; GFX11-NEXT: s_cselect_b32 s90, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s90, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB17_5
; GFX11-NEXT: ; %bb.4: ; %cmp.true
; GFX11-NEXT: s_add_i32 s4, s4, 3
@@ -11179,6 +11192,7 @@ define <28 x i32> @bitcast_v56f16_to_v28i32(<56 x half> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v80, 16, v0
; GFX11-TRUE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-TRUE16-NEXT: v_cmpx_ne_u32_e32 0, v28
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB18_3
; GFX11-TRUE16-NEXT: ; %bb.1: ; %Flow
@@ -12546,6 +12560,7 @@ define inreg <28 x i32> @bitcast_v56f16_to_v28i32_scalar(<56 x half> inreg %a, i
; GFX11-NEXT: s_and_b32 s40, s40, exec_lo
; GFX11-NEXT: s_cselect_b32 s40, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s40, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB19_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_pk_add_f16 v0, 0x200, s0 op_sel_hi:[0,1]
@@ -12741,8 +12756,9 @@ define <14 x i64> @bitcast_v28f32_to_v14i64(<28 x float> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v28
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB20_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -13262,6 +13278,7 @@ define inreg <14 x i64> @bitcast_v28f32_to_v14i64_scalar(<28 x float> inreg %a,
; GFX11-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB21_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_add_f32_e64 v27, s63, 1.0
@@ -13474,43 +13491,37 @@ define <28 x float> @bitcast_v14i64_to_v28f32(<14 x i64> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v28
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB22_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_add_co_u32 v26, vcc_lo, v26, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v27, null, 0, v27, vcc_lo
; GFX11-NEXT: v_add_co_u32 v24, vcc_lo, v24, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v25, null, 0, v25, vcc_lo
; GFX11-NEXT: v_add_co_u32 v22, vcc_lo, v22, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v23, null, 0, v23, vcc_lo
; GFX11-NEXT: v_add_co_u32 v20, vcc_lo, v20, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v21, null, 0, v21, vcc_lo
; GFX11-NEXT: v_add_co_u32 v18, vcc_lo, v18, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v19, null, 0, v19, vcc_lo
; GFX11-NEXT: v_add_co_u32 v16, vcc_lo, v16, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v17, null, 0, v17, vcc_lo
; GFX11-NEXT: v_add_co_u32 v14, vcc_lo, v14, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v15, null, 0, v15, vcc_lo
; GFX11-NEXT: v_add_co_u32 v12, vcc_lo, v12, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v13, null, 0, v13, vcc_lo
; GFX11-NEXT: v_add_co_u32 v10, vcc_lo, v10, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v11, null, 0, v11, vcc_lo
; GFX11-NEXT: v_add_co_u32 v8, vcc_lo, v8, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v9, null, 0, v9, vcc_lo
; GFX11-NEXT: v_add_co_u32 v6, vcc_lo, v6, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v7, null, 0, v7, vcc_lo
; GFX11-NEXT: v_add_co_u32 v4, vcc_lo, v4, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v5, null, 0, v5, vcc_lo
; GFX11-NEXT: v_add_co_u32 v2, vcc_lo, v2, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX11-NEXT: v_add_co_u32 v0, vcc_lo, v0, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
@@ -13829,6 +13840,7 @@ define inreg <28 x float> @bitcast_v14i64_to_v28f32_scalar(<14 x i64> inreg %a,
; GFX11-NEXT: s_and_b32 s14, s14, exec_lo
; GFX11-NEXT: s_cselect_b32 s14, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s14, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB23_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: s_add_u32 s5, s5, 3
@@ -14022,8 +14034,9 @@ define <14 x double> @bitcast_v28f32_to_v14f64(<28 x float> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v28
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB24_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -14543,6 +14556,7 @@ define inreg <14 x double> @bitcast_v28f32_to_v14f64_scalar(<28 x float> inreg %
; GFX11-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB25_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_add_f32_e64 v27, s63, 1.0
@@ -14713,8 +14727,9 @@ define <28 x float> @bitcast_v14f64_to_v28f32(<14 x double> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v28
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB26_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -15192,6 +15207,7 @@ define inreg <28 x float> @bitcast_v14f64_to_v28f32_scalar(<14 x double> inreg %
; GFX11-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB27_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_add_f64 v[26:27], s[62:63], 1.0
@@ -15994,8 +16010,9 @@ define <56 x i16> @bitcast_v28f32_to_v56i16(<28 x float> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v71, 16, v0
; GFX11-TRUE16-NEXT: .LBB28_4: ; %end
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v0.h, v71.l
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v1.h, v70.l
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v2.h, v69.l
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v3.h, v68.l
@@ -16138,8 +16155,9 @@ define <56 x i16> @bitcast_v28f32_to_v56i16(<28 x float> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v71, 16, v0
; GFX11-FAKE16-NEXT: .LBB28_4: ; %end
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_perm_b32 v0, v71, v0, 0x5040100
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_perm_b32 v1, v70, v1, 0x5040100
; GFX11-FAKE16-NEXT: v_perm_b32 v2, v69, v2, 0x5040100
; GFX11-FAKE16-NEXT: v_perm_b32 v3, v68, v3, 0x5040100
@@ -17179,6 +17197,7 @@ define inreg <56 x i16> @bitcast_v28f32_to_v56i16_scalar(<28 x float> inreg %a,
; GFX11-TRUE16-NEXT: s_and_b32 s40, s40, exec_lo
; GFX11-TRUE16-NEXT: s_cselect_b32 s40, 1, 0
; GFX11-TRUE16-NEXT: s_cmp_lg_u32 s40, 1
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_cbranch_scc1 .LBB29_5
; GFX11-TRUE16-NEXT: ; %bb.4: ; %cmp.true
; GFX11-TRUE16-NEXT: v_add_f32_e64 v27, s4, 1.0
@@ -17381,6 +17400,7 @@ define inreg <56 x i16> @bitcast_v28f32_to_v56i16_scalar(<28 x float> inreg %a,
; GFX11-FAKE16-NEXT: s_and_b32 s40, s40, exec_lo
; GFX11-FAKE16-NEXT: s_cselect_b32 s40, 1, 0
; GFX11-FAKE16-NEXT: s_cmp_lg_u32 s40, 1
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_cbranch_scc1 .LBB29_5
; GFX11-FAKE16-NEXT: ; %bb.4: ; %cmp.true
; GFX11-FAKE16-NEXT: v_add_f32_e64 v23, s4, 1.0
@@ -18666,6 +18686,7 @@ define <28 x float> @bitcast_v56i16_to_v28f32(<56 x i16> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v80, 16, v0
; GFX11-TRUE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-TRUE16-NEXT: v_cmpx_ne_u32_e32 0, v28
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB30_3
; GFX11-TRUE16-NEXT: ; %bb.1: ; %Flow
@@ -20021,6 +20042,7 @@ define inreg <28 x float> @bitcast_v56i16_to_v28f32_scalar(<56 x i16> inreg %a,
; GFX11-NEXT: s_and_b32 s40, s40, exec_lo
; GFX11-NEXT: s_cselect_b32 s40, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s40, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB31_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_pk_add_u16 v0, s0, 3 op_sel_hi:[1,0]
@@ -20820,8 +20842,9 @@ define <56 x half> @bitcast_v28f32_to_v56f16(<28 x float> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v71, 16, v0
; GFX11-TRUE16-NEXT: .LBB32_4: ; %end
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v0.h, v71.l
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v1.h, v70.l
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v2.h, v69.l
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v3.h, v68.l
@@ -20964,8 +20987,9 @@ define <56 x half> @bitcast_v28f32_to_v56f16(<28 x float> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v71, 16, v0
; GFX11-FAKE16-NEXT: .LBB32_4: ; %end
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_perm_b32 v0, v71, v0, 0x5040100
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_perm_b32 v1, v70, v1, 0x5040100
; GFX11-FAKE16-NEXT: v_perm_b32 v2, v69, v2, 0x5040100
; GFX11-FAKE16-NEXT: v_perm_b32 v3, v68, v3, 0x5040100
@@ -22005,6 +22029,7 @@ define inreg <56 x half> @bitcast_v28f32_to_v56f16_scalar(<28 x float> inreg %a,
; GFX11-TRUE16-NEXT: s_and_b32 s40, s40, exec_lo
; GFX11-TRUE16-NEXT: s_cselect_b32 s40, 1, 0
; GFX11-TRUE16-NEXT: s_cmp_lg_u32 s40, 1
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_cbranch_scc1 .LBB33_5
; GFX11-TRUE16-NEXT: ; %bb.4: ; %cmp.true
; GFX11-TRUE16-NEXT: v_add_f32_e64 v27, s4, 1.0
@@ -22207,6 +22232,7 @@ define inreg <56 x half> @bitcast_v28f32_to_v56f16_scalar(<28 x float> inreg %a,
; GFX11-FAKE16-NEXT: s_and_b32 s40, s40, exec_lo
; GFX11-FAKE16-NEXT: s_cselect_b32 s40, 1, 0
; GFX11-FAKE16-NEXT: s_cmp_lg_u32 s40, 1
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_cbranch_scc1 .LBB33_5
; GFX11-FAKE16-NEXT: ; %bb.4: ; %cmp.true
; GFX11-FAKE16-NEXT: v_add_f32_e64 v23, s4, 1.0
@@ -23629,6 +23655,7 @@ define <28 x float> @bitcast_v56f16_to_v28f32(<56 x half> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v80, 16, v0
; GFX11-TRUE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-TRUE16-NEXT: v_cmpx_ne_u32_e32 0, v28
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB34_3
; GFX11-TRUE16-NEXT: ; %bb.1: ; %Flow
@@ -24996,6 +25023,7 @@ define inreg <28 x float> @bitcast_v56f16_to_v28f32_scalar(<56 x half> inreg %a,
; GFX11-NEXT: s_and_b32 s40, s40, exec_lo
; GFX11-NEXT: s_cselect_b32 s40, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s40, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB35_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_pk_add_f16 v0, 0x200, s0 op_sel_hi:[0,1]
@@ -25191,43 +25219,37 @@ define <14 x double> @bitcast_v14i64_to_v14f64(<14 x i64> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v28
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB36_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_add_co_u32 v0, vcc_lo, v0, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-NEXT: v_add_co_u32 v2, vcc_lo, v2, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX11-NEXT: v_add_co_u32 v4, vcc_lo, v4, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v5, null, 0, v5, vcc_lo
; GFX11-NEXT: v_add_co_u32 v6, vcc_lo, v6, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v7, null, 0, v7, vcc_lo
; GFX11-NEXT: v_add_co_u32 v8, vcc_lo, v8, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v9, null, 0, v9, vcc_lo
; GFX11-NEXT: v_add_co_u32 v10, vcc_lo, v10, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v11, null, 0, v11, vcc_lo
; GFX11-NEXT: v_add_co_u32 v12, vcc_lo, v12, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v13, null, 0, v13, vcc_lo
; GFX11-NEXT: v_add_co_u32 v14, vcc_lo, v14, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v15, null, 0, v15, vcc_lo
; GFX11-NEXT: v_add_co_u32 v16, vcc_lo, v16, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v17, null, 0, v17, vcc_lo
; GFX11-NEXT: v_add_co_u32 v18, vcc_lo, v18, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v19, null, 0, v19, vcc_lo
; GFX11-NEXT: v_add_co_u32 v20, vcc_lo, v20, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v21, null, 0, v21, vcc_lo
; GFX11-NEXT: v_add_co_u32 v22, vcc_lo, v22, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v23, null, 0, v23, vcc_lo
; GFX11-NEXT: v_add_co_u32 v24, vcc_lo, v24, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v25, null, 0, v25, vcc_lo
; GFX11-NEXT: v_add_co_u32 v26, vcc_lo, v26, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v27, null, 0, v27, vcc_lo
@@ -25546,6 +25568,7 @@ define inreg <14 x double> @bitcast_v14i64_to_v14f64_scalar(<14 x i64> inreg %a,
; GFX11-NEXT: s_and_b32 s14, s14, exec_lo
; GFX11-NEXT: s_cselect_b32 s14, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s14, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB37_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: s_add_u32 s0, s0, 3
@@ -25696,8 +25719,9 @@ define <14 x i64> @bitcast_v14f64_to_v14i64(<14 x double> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v28
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB38_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -26175,6 +26199,7 @@ define inreg <14 x i64> @bitcast_v14f64_to_v14i64_scalar(<14 x double> inreg %a,
; GFX11-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB39_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_add_f64 v[0:1], s[36:37], 1.0
@@ -26934,37 +26959,30 @@ define <56 x i16> @bitcast_v14i64_to_v56i16(<14 x i64> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: s_cbranch_execz .LBB40_4
; GFX11-TRUE16-NEXT: ; %bb.3: ; %cmp.true
; GFX11-TRUE16-NEXT: v_add_co_u32 v26, vcc_lo, v26, 3
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v27, null, 0, v27, vcc_lo
; GFX11-TRUE16-NEXT: v_add_co_u32 v24, vcc_lo, v24, 3
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v25, null, 0, v25, vcc_lo
; GFX11-TRUE16-NEXT: v_add_co_u32 v22, vcc_lo, v22, 3
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v23, null, 0, v23, vcc_lo
; GFX11-TRUE16-NEXT: v_add_co_u32 v20, vcc_lo, v20, 3
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v21, null, 0, v21, vcc_lo
; GFX11-TRUE16-NEXT: v_add_co_u32 v18, vcc_lo, v18, 3
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v19, null, 0, v19, vcc_lo
; GFX11-TRUE16-NEXT: v_add_co_u32 v16, vcc_lo, v16, 3
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v17, null, 0, v17, vcc_lo
; GFX11-TRUE16-NEXT: v_add_co_u32 v14, vcc_lo, v14, 3
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v15, null, 0, v15, vcc_lo
; GFX11-TRUE16-NEXT: v_add_co_u32 v12, vcc_lo, v12, 3
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v13, null, 0, v13, vcc_lo
; GFX11-TRUE16-NEXT: v_add_co_u32 v10, vcc_lo, v10, 3
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v11, null, 0, v11, vcc_lo
; GFX11-TRUE16-NEXT: v_add_co_u32 v8, vcc_lo, v8, 3
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v9, null, 0, v9, vcc_lo
; GFX11-TRUE16-NEXT: v_add_co_u32 v6, vcc_lo, v6, 3
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v7, null, 0, v7, vcc_lo
; GFX11-TRUE16-NEXT: v_add_co_u32 v4, vcc_lo, v4, 3
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v5, null, 0, v5, vcc_lo
; GFX11-TRUE16-NEXT: v_add_co_u32 v2, vcc_lo, v2, 3
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX11-TRUE16-NEXT: v_add_co_u32 v0, vcc_lo, v0, 3
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
@@ -26998,8 +27016,9 @@ define <56 x i16> @bitcast_v14i64_to_v56i16(<14 x i64> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v71, 16, v0
; GFX11-TRUE16-NEXT: .LBB40_4: ; %end
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v0.h, v71.l
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v1.h, v70.l
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v2.h, v69.l
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v3.h, v68.l
@@ -27099,37 +27118,30 @@ define <56 x i16> @bitcast_v14i64_to_v56i16(<14 x i64> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: s_cbranch_execz .LBB40_4
; GFX11-FAKE16-NEXT: ; %bb.3: ; %cmp.true
; GFX11-FAKE16-NEXT: v_add_co_u32 v26, vcc_lo, v26, 3
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v27, null, 0, v27, vcc_lo
; GFX11-FAKE16-NEXT: v_add_co_u32 v24, vcc_lo, v24, 3
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v25, null, 0, v25, vcc_lo
; GFX11-FAKE16-NEXT: v_add_co_u32 v22, vcc_lo, v22, 3
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v23, null, 0, v23, vcc_lo
; GFX11-FAKE16-NEXT: v_add_co_u32 v20, vcc_lo, v20, 3
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v21, null, 0, v21, vcc_lo
; GFX11-FAKE16-NEXT: v_add_co_u32 v18, vcc_lo, v18, 3
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v19, null, 0, v19, vcc_lo
; GFX11-FAKE16-NEXT: v_add_co_u32 v16, vcc_lo, v16, 3
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v17, null, 0, v17, vcc_lo
; GFX11-FAKE16-NEXT: v_add_co_u32 v14, vcc_lo, v14, 3
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v15, null, 0, v15, vcc_lo
; GFX11-FAKE16-NEXT: v_add_co_u32 v12, vcc_lo, v12, 3
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v13, null, 0, v13, vcc_lo
; GFX11-FAKE16-NEXT: v_add_co_u32 v10, vcc_lo, v10, 3
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v11, null, 0, v11, vcc_lo
; GFX11-FAKE16-NEXT: v_add_co_u32 v8, vcc_lo, v8, 3
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v9, null, 0, v9, vcc_lo
; GFX11-FAKE16-NEXT: v_add_co_u32 v6, vcc_lo, v6, 3
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v7, null, 0, v7, vcc_lo
; GFX11-FAKE16-NEXT: v_add_co_u32 v4, vcc_lo, v4, 3
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v5, null, 0, v5, vcc_lo
; GFX11-FAKE16-NEXT: v_add_co_u32 v2, vcc_lo, v2, 3
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX11-FAKE16-NEXT: v_add_co_u32 v0, vcc_lo, v0, 3
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
@@ -27163,8 +27175,9 @@ define <56 x i16> @bitcast_v14i64_to_v56i16(<14 x i64> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v71, 16, v0
; GFX11-FAKE16-NEXT: .LBB40_4: ; %end
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_perm_b32 v0, v71, v0, 0x5040100
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_perm_b32 v1, v70, v1, 0x5040100
; GFX11-FAKE16-NEXT: v_perm_b32 v2, v69, v2, 0x5040100
; GFX11-FAKE16-NEXT: v_perm_b32 v3, v68, v3, 0x5040100
@@ -28054,6 +28067,7 @@ define inreg <56 x i16> @bitcast_v14i64_to_v56i16_scalar(<14 x i64> inreg %a, i3
; GFX11-NEXT: s_and_b32 s90, s90, exec_lo
; GFX11-NEXT: s_cselect_b32 s90, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s90, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB41_5
; GFX11-NEXT: ; %bb.4: ; %cmp.true
; GFX11-NEXT: s_add_u32 s5, s5, 3
@@ -29296,6 +29310,7 @@ define <14 x i64> @bitcast_v56i16_to_v14i64(<56 x i16> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v80, 16, v0
; GFX11-TRUE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-TRUE16-NEXT: v_cmpx_ne_u32_e32 0, v28
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB42_3
; GFX11-TRUE16-NEXT: ; %bb.1: ; %Flow
@@ -30651,6 +30666,7 @@ define inreg <14 x i64> @bitcast_v56i16_to_v14i64_scalar(<56 x i16> inreg %a, i3
; GFX11-NEXT: s_and_b32 s40, s40, exec_lo
; GFX11-NEXT: s_cselect_b32 s40, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s40, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB43_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_pk_add_u16 v0, s0, 3 op_sel_hi:[1,0]
@@ -31407,37 +31423,30 @@ define <56 x half> @bitcast_v14i64_to_v56f16(<14 x i64> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: s_cbranch_execz .LBB44_4
; GFX11-TRUE16-NEXT: ; %bb.3: ; %cmp.true
; GFX11-TRUE16-NEXT: v_add_co_u32 v26, vcc_lo, v26, 3
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v27, null, 0, v27, vcc_lo
; GFX11-TRUE16-NEXT: v_add_co_u32 v24, vcc_lo, v24, 3
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v25, null, 0, v25, vcc_lo
; GFX11-TRUE16-NEXT: v_add_co_u32 v22, vcc_lo, v22, 3
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v23, null, 0, v23, vcc_lo
; GFX11-TRUE16-NEXT: v_add_co_u32 v20, vcc_lo, v20, 3
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v21, null, 0, v21, vcc_lo
; GFX11-TRUE16-NEXT: v_add_co_u32 v18, vcc_lo, v18, 3
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v19, null, 0, v19, vcc_lo
; GFX11-TRUE16-NEXT: v_add_co_u32 v16, vcc_lo, v16, 3
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v17, null, 0, v17, vcc_lo
; GFX11-TRUE16-NEXT: v_add_co_u32 v14, vcc_lo, v14, 3
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v15, null, 0, v15, vcc_lo
; GFX11-TRUE16-NEXT: v_add_co_u32 v12, vcc_lo, v12, 3
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v13, null, 0, v13, vcc_lo
; GFX11-TRUE16-NEXT: v_add_co_u32 v10, vcc_lo, v10, 3
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v11, null, 0, v11, vcc_lo
; GFX11-TRUE16-NEXT: v_add_co_u32 v8, vcc_lo, v8, 3
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v9, null, 0, v9, vcc_lo
; GFX11-TRUE16-NEXT: v_add_co_u32 v6, vcc_lo, v6, 3
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v7, null, 0, v7, vcc_lo
; GFX11-TRUE16-NEXT: v_add_co_u32 v4, vcc_lo, v4, 3
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v5, null, 0, v5, vcc_lo
; GFX11-TRUE16-NEXT: v_add_co_u32 v2, vcc_lo, v2, 3
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX11-TRUE16-NEXT: v_add_co_u32 v0, vcc_lo, v0, 3
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
@@ -31471,8 +31480,9 @@ define <56 x half> @bitcast_v14i64_to_v56f16(<14 x i64> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v71, 16, v0
; GFX11-TRUE16-NEXT: .LBB44_4: ; %end
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v0.h, v71.l
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v1.h, v70.l
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v2.h, v69.l
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v3.h, v68.l
@@ -31572,37 +31582,30 @@ define <56 x half> @bitcast_v14i64_to_v56f16(<14 x i64> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: s_cbranch_execz .LBB44_4
; GFX11-FAKE16-NEXT: ; %bb.3: ; %cmp.true
; GFX11-FAKE16-NEXT: v_add_co_u32 v26, vcc_lo, v26, 3
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v27, null, 0, v27, vcc_lo
; GFX11-FAKE16-NEXT: v_add_co_u32 v24, vcc_lo, v24, 3
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v25, null, 0, v25, vcc_lo
; GFX11-FAKE16-NEXT: v_add_co_u32 v22, vcc_lo, v22, 3
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v23, null, 0, v23, vcc_lo
; GFX11-FAKE16-NEXT: v_add_co_u32 v20, vcc_lo, v20, 3
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v21, null, 0, v21, vcc_lo
; GFX11-FAKE16-NEXT: v_add_co_u32 v18, vcc_lo, v18, 3
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v19, null, 0, v19, vcc_lo
; GFX11-FAKE16-NEXT: v_add_co_u32 v16, vcc_lo, v16, 3
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v17, null, 0, v17, vcc_lo
; GFX11-FAKE16-NEXT: v_add_co_u32 v14, vcc_lo, v14, 3
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v15, null, 0, v15, vcc_lo
; GFX11-FAKE16-NEXT: v_add_co_u32 v12, vcc_lo, v12, 3
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v13, null, 0, v13, vcc_lo
; GFX11-FAKE16-NEXT: v_add_co_u32 v10, vcc_lo, v10, 3
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v11, null, 0, v11, vcc_lo
; GFX11-FAKE16-NEXT: v_add_co_u32 v8, vcc_lo, v8, 3
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v9, null, 0, v9, vcc_lo
; GFX11-FAKE16-NEXT: v_add_co_u32 v6, vcc_lo, v6, 3
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v7, null, 0, v7, vcc_lo
; GFX11-FAKE16-NEXT: v_add_co_u32 v4, vcc_lo, v4, 3
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v5, null, 0, v5, vcc_lo
; GFX11-FAKE16-NEXT: v_add_co_u32 v2, vcc_lo, v2, 3
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX11-FAKE16-NEXT: v_add_co_u32 v0, vcc_lo, v0, 3
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
@@ -31636,8 +31639,9 @@ define <56 x half> @bitcast_v14i64_to_v56f16(<14 x i64> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v71, 16, v0
; GFX11-FAKE16-NEXT: .LBB44_4: ; %end
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_perm_b32 v0, v71, v0, 0x5040100
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_perm_b32 v1, v70, v1, 0x5040100
; GFX11-FAKE16-NEXT: v_perm_b32 v2, v69, v2, 0x5040100
; GFX11-FAKE16-NEXT: v_perm_b32 v3, v68, v3, 0x5040100
@@ -32527,6 +32531,7 @@ define inreg <56 x half> @bitcast_v14i64_to_v56f16_scalar(<14 x i64> inreg %a, i
; GFX11-NEXT: s_and_b32 s90, s90, exec_lo
; GFX11-NEXT: s_cselect_b32 s90, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s90, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB45_5
; GFX11-NEXT: ; %bb.4: ; %cmp.true
; GFX11-NEXT: s_add_u32 s5, s5, 3
@@ -33906,6 +33911,7 @@ define <14 x i64> @bitcast_v56f16_to_v14i64(<56 x half> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v80, 16, v0
; GFX11-TRUE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-TRUE16-NEXT: v_cmpx_ne_u32_e32 0, v28
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB46_3
; GFX11-TRUE16-NEXT: ; %bb.1: ; %Flow
@@ -35273,6 +35279,7 @@ define inreg <14 x i64> @bitcast_v56f16_to_v14i64_scalar(<56 x half> inreg %a, i
; GFX11-NEXT: s_and_b32 s40, s40, exec_lo
; GFX11-NEXT: s_cselect_b32 s40, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s40, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB47_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_pk_add_f16 v0, 0x200, s0 op_sel_hi:[0,1]
@@ -36030,8 +36037,9 @@ define <56 x i16> @bitcast_v14f64_to_v56i16(<14 x double> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v71, 16, v0
; GFX11-TRUE16-NEXT: .LBB48_4: ; %end
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v0.h, v71.l
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v1.h, v70.l
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v2.h, v69.l
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v3.h, v68.l
@@ -36174,8 +36182,9 @@ define <56 x i16> @bitcast_v14f64_to_v56i16(<14 x double> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v71, 16, v0
; GFX11-FAKE16-NEXT: .LBB48_4: ; %end
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_perm_b32 v0, v71, v0, 0x5040100
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_perm_b32 v1, v70, v1, 0x5040100
; GFX11-FAKE16-NEXT: v_perm_b32 v2, v69, v2, 0x5040100
; GFX11-FAKE16-NEXT: v_perm_b32 v3, v68, v3, 0x5040100
@@ -37179,6 +37188,7 @@ define inreg <56 x i16> @bitcast_v14f64_to_v56i16_scalar(<14 x double> inreg %a,
; GFX11-TRUE16-NEXT: s_and_b32 s40, s40, exec_lo
; GFX11-TRUE16-NEXT: s_cselect_b32 s40, 1, 0
; GFX11-TRUE16-NEXT: s_cmp_lg_u32 s40, 1
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_cbranch_scc1 .LBB49_5
; GFX11-TRUE16-NEXT: ; %bb.4: ; %cmp.true
; GFX11-TRUE16-NEXT: v_add_f64 v[26:27], s[10:11], 1.0
@@ -37368,6 +37378,7 @@ define inreg <56 x i16> @bitcast_v14f64_to_v56i16_scalar(<14 x double> inreg %a,
; GFX11-FAKE16-NEXT: s_and_b32 s40, s40, exec_lo
; GFX11-FAKE16-NEXT: s_cselect_b32 s40, 1, 0
; GFX11-FAKE16-NEXT: s_cmp_lg_u32 s40, 1
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_cbranch_scc1 .LBB49_5
; GFX11-FAKE16-NEXT: ; %bb.4: ; %cmp.true
; GFX11-FAKE16-NEXT: v_add_f64 v[23:24], s[10:11], 1.0
@@ -38639,6 +38650,7 @@ define <14 x double> @bitcast_v56i16_to_v14f64(<56 x i16> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v80, 16, v0
; GFX11-TRUE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-TRUE16-NEXT: v_cmpx_ne_u32_e32 0, v28
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB50_3
; GFX11-TRUE16-NEXT: ; %bb.1: ; %Flow
@@ -39994,6 +40006,7 @@ define inreg <14 x double> @bitcast_v56i16_to_v14f64_scalar(<56 x i16> inreg %a,
; GFX11-NEXT: s_and_b32 s40, s40, exec_lo
; GFX11-NEXT: s_cselect_b32 s40, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s40, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB51_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_pk_add_u16 v0, s0, 3 op_sel_hi:[1,0]
@@ -40751,8 +40764,9 @@ define <56 x half> @bitcast_v14f64_to_v56f16(<14 x double> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v71, 16, v0
; GFX11-TRUE16-NEXT: .LBB52_4: ; %end
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v0.h, v71.l
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v1.h, v70.l
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v2.h, v69.l
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v3.h, v68.l
@@ -40895,8 +40909,9 @@ define <56 x half> @bitcast_v14f64_to_v56f16(<14 x double> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v71, 16, v0
; GFX11-FAKE16-NEXT: .LBB52_4: ; %end
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_perm_b32 v0, v71, v0, 0x5040100
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_perm_b32 v1, v70, v1, 0x5040100
; GFX11-FAKE16-NEXT: v_perm_b32 v2, v69, v2, 0x5040100
; GFX11-FAKE16-NEXT: v_perm_b32 v3, v68, v3, 0x5040100
@@ -41900,6 +41915,7 @@ define inreg <56 x half> @bitcast_v14f64_to_v56f16_scalar(<14 x double> inreg %a
; GFX11-TRUE16-NEXT: s_and_b32 s40, s40, exec_lo
; GFX11-TRUE16-NEXT: s_cselect_b32 s40, 1, 0
; GFX11-TRUE16-NEXT: s_cmp_lg_u32 s40, 1
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_cbranch_scc1 .LBB53_5
; GFX11-TRUE16-NEXT: ; %bb.4: ; %cmp.true
; GFX11-TRUE16-NEXT: v_add_f64 v[26:27], s[10:11], 1.0
@@ -42089,6 +42105,7 @@ define inreg <56 x half> @bitcast_v14f64_to_v56f16_scalar(<14 x double> inreg %a
; GFX11-FAKE16-NEXT: s_and_b32 s40, s40, exec_lo
; GFX11-FAKE16-NEXT: s_cselect_b32 s40, 1, 0
; GFX11-FAKE16-NEXT: s_cmp_lg_u32 s40, 1
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_cbranch_scc1 .LBB53_5
; GFX11-FAKE16-NEXT: ; %bb.4: ; %cmp.true
; GFX11-FAKE16-NEXT: v_add_f64 v[23:24], s[10:11], 1.0
@@ -43497,6 +43514,7 @@ define <14 x double> @bitcast_v56f16_to_v14f64(<56 x half> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v80, 16, v0
; GFX11-TRUE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-TRUE16-NEXT: v_cmpx_ne_u32_e32 0, v28
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB54_3
; GFX11-TRUE16-NEXT: ; %bb.1: ; %Flow
@@ -44864,6 +44882,7 @@ define inreg <14 x double> @bitcast_v56f16_to_v14f64_scalar(<56 x half> inreg %a
; GFX11-NEXT: s_and_b32 s40, s40, exec_lo
; GFX11-NEXT: s_cselect_b32 s40, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s40, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB55_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_pk_add_f16 v0, 0x200, s0 op_sel_hi:[0,1]
@@ -46101,8 +46120,9 @@ define <56 x half> @bitcast_v56i16_to_v56f16(<56 x i16> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v33, 16, v0
; GFX11-TRUE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-TRUE16-NEXT: v_cmpx_ne_u32_e32 0, v28
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-TRUE16-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-TRUE16-NEXT: s_cbranch_execz .LBB56_2
; GFX11-TRUE16-NEXT: ; %bb.1: ; %cmp.true
@@ -46192,6 +46212,7 @@ define <56 x half> @bitcast_v56i16_to_v56f16(<56 x i16> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v80, 16, v27
; GFX11-TRUE16-NEXT: .LBB56_2: ; %end
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v0.h, v33.l
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v1.h, v32.l
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v2.h, v31.l
@@ -46255,8 +46276,9 @@ define <56 x half> @bitcast_v56i16_to_v56f16(<56 x i16> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v29, 16, v0
; GFX11-FAKE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-FAKE16-NEXT: v_cmpx_ne_u32_e32 0, v28
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-FAKE16-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-FAKE16-NEXT: s_cbranch_execz .LBB56_2
; GFX11-FAKE16-NEXT: ; %bb.1: ; %cmp.true
@@ -46346,6 +46368,7 @@ define <56 x half> @bitcast_v56i16_to_v56f16(<56 x i16> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v80, 16, v27
; GFX11-FAKE16-NEXT: .LBB56_2: ; %end
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_perm_b32 v0, v29, v0, 0x5040100
; GFX11-FAKE16-NEXT: v_perm_b32 v1, v30, v1, 0x5040100
; GFX11-FAKE16-NEXT: v_perm_b32 v2, v31, v2, 0x5040100
@@ -47565,6 +47588,7 @@ define inreg <56 x half> @bitcast_v56i16_to_v56f16_scalar(<56 x i16> inreg %a, i
; GFX11-TRUE16-NEXT: s_and_b32 s90, s90, exec_lo
; GFX11-TRUE16-NEXT: s_cselect_b32 s90, 1, 0
; GFX11-TRUE16-NEXT: s_cmp_lg_u32 s90, 1
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_cbranch_scc1 .LBB57_4
; GFX11-TRUE16-NEXT: ; %bb.3: ; %cmp.true
; GFX11-TRUE16-NEXT: s_pack_ll_b32_b16 s59, s79, s59
@@ -47766,6 +47790,7 @@ define inreg <56 x half> @bitcast_v56i16_to_v56f16_scalar(<56 x i16> inreg %a, i
; GFX11-FAKE16-NEXT: s_and_b32 s90, s90, exec_lo
; GFX11-FAKE16-NEXT: s_cselect_b32 s90, 1, 0
; GFX11-FAKE16-NEXT: s_cmp_lg_u32 s90, 1
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_cbranch_scc1 .LBB57_4
; GFX11-FAKE16-NEXT: ; %bb.3: ; %cmp.true
; GFX11-FAKE16-NEXT: s_pack_ll_b32_b16 s59, s79, s59
@@ -48736,8 +48761,9 @@ define <56 x i16> @bitcast_v56f16_to_v56i16(<56 x half> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v33, 16, v0
; GFX11-TRUE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-TRUE16-NEXT: v_cmpx_ne_u32_e32 0, v28
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-TRUE16-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-TRUE16-NEXT: s_cbranch_execz .LBB58_2
; GFX11-TRUE16-NEXT: ; %bb.1: ; %cmp.true
@@ -48827,6 +48853,7 @@ define <56 x i16> @bitcast_v56f16_to_v56i16(<56 x half> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v80, 16, v27
; GFX11-TRUE16-NEXT: .LBB58_2: ; %end
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v0.h, v33.l
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v1.h, v32.l
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v2.h, v31.l
@@ -48890,8 +48917,9 @@ define <56 x i16> @bitcast_v56f16_to_v56i16(<56 x half> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v29, 16, v0
; GFX11-FAKE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-FAKE16-NEXT: v_cmpx_ne_u32_e32 0, v28
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-FAKE16-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-FAKE16-NEXT: s_cbranch_execz .LBB58_2
; GFX11-FAKE16-NEXT: ; %bb.1: ; %cmp.true
@@ -48981,6 +49009,7 @@ define <56 x i16> @bitcast_v56f16_to_v56i16(<56 x half> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v80, 16, v27
; GFX11-FAKE16-NEXT: .LBB58_2: ; %end
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_perm_b32 v0, v29, v0, 0x5040100
; GFX11-FAKE16-NEXT: v_perm_b32 v1, v30, v1, 0x5040100
; GFX11-FAKE16-NEXT: v_perm_b32 v2, v31, v2, 0x5040100
@@ -50121,6 +50150,7 @@ define inreg <56 x i16> @bitcast_v56f16_to_v56i16_scalar(<56 x half> inreg %a, i
; GFX11-TRUE16-NEXT: s_and_b32 s90, s90, exec_lo
; GFX11-TRUE16-NEXT: s_cselect_b32 s90, 1, 0
; GFX11-TRUE16-NEXT: s_cmp_lg_u32 s90, 1
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_cbranch_scc1 .LBB59_4
; GFX11-TRUE16-NEXT: ; %bb.3: ; %cmp.true
; GFX11-TRUE16-NEXT: s_pack_ll_b32_b16 s59, s79, s59
@@ -50322,6 +50352,7 @@ define inreg <56 x i16> @bitcast_v56f16_to_v56i16_scalar(<56 x half> inreg %a, i
; GFX11-FAKE16-NEXT: s_and_b32 s90, s90, exec_lo
; GFX11-FAKE16-NEXT: s_cselect_b32 s90, 1, 0
; GFX11-FAKE16-NEXT: s_cmp_lg_u32 s90, 1
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_cbranch_scc1 .LBB59_4
; GFX11-FAKE16-NEXT: ; %bb.3: ; %cmp.true
; GFX11-FAKE16-NEXT: s_pack_ll_b32_b16 s59, s79, s59
diff --git a/llvm/test/CodeGen/AMDGPU/amdgcn.bitcast.960bit.ll b/llvm/test/CodeGen/AMDGPU/amdgcn.bitcast.960bit.ll
index deced731f537bd..bb532579259c43 100644
--- a/llvm/test/CodeGen/AMDGPU/amdgcn.bitcast.960bit.ll
+++ b/llvm/test/CodeGen/AMDGPU/amdgcn.bitcast.960bit.ll
@@ -141,8 +141,9 @@ define <30 x float> @bitcast_v30i32_to_v30f32(<30 x i32> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v30
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB0_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -511,6 +512,7 @@ define inreg <30 x float> @bitcast_v30i32_to_v30f32_scalar(<30 x i32> inreg %a,
; GFX11-NEXT: s_and_b32 s40, s40, exec_lo
; GFX11-NEXT: s_cselect_b32 s40, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s40, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB1_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: s_add_i32 s4, s4, 3
@@ -713,8 +715,9 @@ define <30 x i32> @bitcast_v30f32_to_v30i32(<30 x float> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v30
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB2_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -1263,6 +1266,7 @@ define inreg <30 x i32> @bitcast_v30f32_to_v30i32_scalar(<30 x float> inreg %a,
; GFX11-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB3_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_add_f32_e64 v29, s65, 1.0
@@ -1485,8 +1489,9 @@ define <15 x i64> @bitcast_v30i32_to_v15i64(<30 x i32> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v30
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB4_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -1855,6 +1860,7 @@ define inreg <15 x i64> @bitcast_v30i32_to_v15i64_scalar(<30 x i32> inreg %a, i3
; GFX11-NEXT: s_and_b32 s40, s40, exec_lo
; GFX11-NEXT: s_cselect_b32 s40, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s40, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB5_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: s_add_i32 s4, s4, 3
@@ -2057,48 +2063,41 @@ define <30 x i32> @bitcast_v15i64_to_v30i32(<15 x i64> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v30
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB6_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_add_co_u32 v28, vcc_lo, v28, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v29, null, 0, v29, vcc_lo
; GFX11-NEXT: v_add_co_u32 v26, vcc_lo, v26, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v27, null, 0, v27, vcc_lo
; GFX11-NEXT: v_add_co_u32 v24, vcc_lo, v24, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v25, null, 0, v25, vcc_lo
; GFX11-NEXT: v_add_co_u32 v22, vcc_lo, v22, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v23, null, 0, v23, vcc_lo
; GFX11-NEXT: v_add_co_u32 v20, vcc_lo, v20, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v21, null, 0, v21, vcc_lo
; GFX11-NEXT: v_add_co_u32 v18, vcc_lo, v18, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v19, null, 0, v19, vcc_lo
; GFX11-NEXT: v_add_co_u32 v16, vcc_lo, v16, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v17, null, 0, v17, vcc_lo
; GFX11-NEXT: v_add_co_u32 v14, vcc_lo, v14, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v15, null, 0, v15, vcc_lo
; GFX11-NEXT: v_add_co_u32 v12, vcc_lo, v12, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v13, null, 0, v13, vcc_lo
; GFX11-NEXT: v_add_co_u32 v10, vcc_lo, v10, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v11, null, 0, v11, vcc_lo
; GFX11-NEXT: v_add_co_u32 v8, vcc_lo, v8, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v9, null, 0, v9, vcc_lo
; GFX11-NEXT: v_add_co_u32 v6, vcc_lo, v6, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v7, null, 0, v7, vcc_lo
; GFX11-NEXT: v_add_co_u32 v4, vcc_lo, v4, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v5, null, 0, v5, vcc_lo
; GFX11-NEXT: v_add_co_u32 v2, vcc_lo, v2, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX11-NEXT: v_add_co_u32 v0, vcc_lo, v0, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-NEXT: .LBB6_2: ; %end
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -2435,6 +2434,7 @@ define inreg <30 x i32> @bitcast_v15i64_to_v30i32_scalar(<15 x i64> inreg %a, i3
; GFX11-NEXT: s_and_b32 s40, s40, exec_lo
; GFX11-NEXT: s_cselect_b32 s40, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s40, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB7_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: s_add_u32 s5, s5, 3
@@ -2637,8 +2637,9 @@ define <15 x double> @bitcast_v30i32_to_v15f64(<30 x i32> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v30
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB8_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -3007,6 +3008,7 @@ define inreg <15 x double> @bitcast_v30i32_to_v15f64_scalar(<30 x i32> inreg %a,
; GFX11-NEXT: s_and_b32 s40, s40, exec_lo
; GFX11-NEXT: s_cselect_b32 s40, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s40, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB9_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: s_add_i32 s4, s4, 3
@@ -3164,8 +3166,9 @@ define <30 x i32> @bitcast_v15f64_to_v30i32(<15 x double> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v30
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB10_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -3669,6 +3672,7 @@ define inreg <30 x i32> @bitcast_v15f64_to_v30i32_scalar(<15 x double> inreg %a,
; GFX11-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB11_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_add_f64 v[28:29], s[64:65], 1.0
@@ -4556,8 +4560,9 @@ define <60 x i16> @bitcast_v30i32_to_v60i16(<30 x i32> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v83, 16, v0
; GFX11-TRUE16-NEXT: .LBB12_4: ; %end
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v0.h, v83.l
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v1.h, v82.l
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v2.h, v81.l
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v3.h, v80.l
@@ -4724,8 +4729,9 @@ define <60 x i16> @bitcast_v30i32_to_v60i16(<30 x i32> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v83, 16, v0
; GFX11-FAKE16-NEXT: .LBB12_4: ; %end
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_perm_b32 v0, v83, v0, 0x5040100
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_perm_b32 v1, v82, v1, 0x5040100
; GFX11-FAKE16-NEXT: v_perm_b32 v2, v81, v2, 0x5040100
; GFX11-FAKE16-NEXT: v_perm_b32 v3, v80, v3, 0x5040100
@@ -5702,6 +5708,7 @@ define inreg <60 x i16> @bitcast_v30i32_to_v60i16_scalar(<30 x i32> inreg %a, i3
; GFX11-NEXT: s_and_b32 s94, s94, exec_lo
; GFX11-NEXT: s_cselect_b32 s94, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s94, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB13_5
; GFX11-NEXT: ; %bb.4: ; %cmp.true
; GFX11-NEXT: s_add_i32 s4, s4, 3
@@ -7045,6 +7052,7 @@ define <30 x i32> @bitcast_v60i16_to_v30i32(<60 x i16> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v84, 16, v0
; GFX11-TRUE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-TRUE16-NEXT: v_cmpx_ne_u32_e32 0, v30
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB14_3
; GFX11-TRUE16-NEXT: ; %bb.1: ; %Flow
@@ -8496,6 +8504,7 @@ define inreg <30 x i32> @bitcast_v60i16_to_v30i32_scalar(<60 x i16> inreg %a, i3
; GFX11-NEXT: s_and_b32 s40, s40, exec_lo
; GFX11-NEXT: s_cselect_b32 s40, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s40, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB15_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_pk_add_u16 v0, s0, 3 op_sel_hi:[1,0]
@@ -9379,8 +9388,9 @@ define <60 x half> @bitcast_v30i32_to_v60f16(<30 x i32> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v83, 16, v0
; GFX11-TRUE16-NEXT: .LBB16_4: ; %end
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v0.h, v83.l
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v1.h, v82.l
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v2.h, v81.l
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v3.h, v80.l
@@ -9547,8 +9557,9 @@ define <60 x half> @bitcast_v30i32_to_v60f16(<30 x i32> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v83, 16, v0
; GFX11-FAKE16-NEXT: .LBB16_4: ; %end
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_perm_b32 v0, v83, v0, 0x5040100
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_perm_b32 v1, v82, v1, 0x5040100
; GFX11-FAKE16-NEXT: v_perm_b32 v2, v81, v2, 0x5040100
; GFX11-FAKE16-NEXT: v_perm_b32 v3, v80, v3, 0x5040100
@@ -10525,6 +10536,7 @@ define inreg <60 x half> @bitcast_v30i32_to_v60f16_scalar(<30 x i32> inreg %a, i
; GFX11-NEXT: s_and_b32 s94, s94, exec_lo
; GFX11-NEXT: s_cselect_b32 s94, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s94, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB17_5
; GFX11-NEXT: ; %bb.4: ; %cmp.true
; GFX11-NEXT: s_add_i32 s4, s4, 3
@@ -12012,6 +12024,7 @@ define <30 x i32> @bitcast_v60f16_to_v30i32(<60 x half> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v84, 16, v0
; GFX11-TRUE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-TRUE16-NEXT: v_cmpx_ne_u32_e32 0, v30
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB18_3
; GFX11-TRUE16-NEXT: ; %bb.1: ; %Flow
@@ -13467,6 +13480,7 @@ define inreg <30 x i32> @bitcast_v60f16_to_v30i32_scalar(<60 x half> inreg %a, i
; GFX11-NEXT: s_and_b32 s40, s40, exec_lo
; GFX11-NEXT: s_cselect_b32 s40, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s40, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB19_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_pk_add_f16 v0, 0x200, s0 op_sel_hi:[0,1]
@@ -13670,8 +13684,9 @@ define <15 x i64> @bitcast_v30f32_to_v15i64(<30 x float> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v30
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB20_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -14220,6 +14235,7 @@ define inreg <15 x i64> @bitcast_v30f32_to_v15i64_scalar(<30 x float> inreg %a,
; GFX11-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB21_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_add_f32_e64 v29, s65, 1.0
@@ -14442,48 +14458,41 @@ define <30 x float> @bitcast_v15i64_to_v30f32(<15 x i64> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v30
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB22_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_add_co_u32 v28, vcc_lo, v28, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v29, null, 0, v29, vcc_lo
; GFX11-NEXT: v_add_co_u32 v26, vcc_lo, v26, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v27, null, 0, v27, vcc_lo
; GFX11-NEXT: v_add_co_u32 v24, vcc_lo, v24, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v25, null, 0, v25, vcc_lo
; GFX11-NEXT: v_add_co_u32 v22, vcc_lo, v22, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v23, null, 0, v23, vcc_lo
; GFX11-NEXT: v_add_co_u32 v20, vcc_lo, v20, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v21, null, 0, v21, vcc_lo
; GFX11-NEXT: v_add_co_u32 v18, vcc_lo, v18, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v19, null, 0, v19, vcc_lo
; GFX11-NEXT: v_add_co_u32 v16, vcc_lo, v16, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v17, null, 0, v17, vcc_lo
; GFX11-NEXT: v_add_co_u32 v14, vcc_lo, v14, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v15, null, 0, v15, vcc_lo
; GFX11-NEXT: v_add_co_u32 v12, vcc_lo, v12, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v13, null, 0, v13, vcc_lo
; GFX11-NEXT: v_add_co_u32 v10, vcc_lo, v10, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v11, null, 0, v11, vcc_lo
; GFX11-NEXT: v_add_co_u32 v8, vcc_lo, v8, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v9, null, 0, v9, vcc_lo
; GFX11-NEXT: v_add_co_u32 v6, vcc_lo, v6, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v7, null, 0, v7, vcc_lo
; GFX11-NEXT: v_add_co_u32 v4, vcc_lo, v4, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v5, null, 0, v5, vcc_lo
; GFX11-NEXT: v_add_co_u32 v2, vcc_lo, v2, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX11-NEXT: v_add_co_u32 v0, vcc_lo, v0, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-NEXT: .LBB22_2: ; %end
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -14820,6 +14829,7 @@ define inreg <30 x float> @bitcast_v15i64_to_v30f32_scalar(<15 x i64> inreg %a,
; GFX11-NEXT: s_and_b32 s40, s40, exec_lo
; GFX11-NEXT: s_cselect_b32 s40, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s40, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB23_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: s_add_u32 s5, s5, 3
@@ -15022,8 +15032,9 @@ define <15 x double> @bitcast_v30f32_to_v15f64(<30 x float> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v30
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB24_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -15572,6 +15583,7 @@ define inreg <15 x double> @bitcast_v30f32_to_v15f64_scalar(<30 x float> inreg %
; GFX11-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB25_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_add_f32_e64 v29, s65, 1.0
@@ -15749,8 +15761,9 @@ define <30 x float> @bitcast_v15f64_to_v30f32(<15 x double> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v30
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB26_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -16254,6 +16267,7 @@ define inreg <30 x float> @bitcast_v15f64_to_v30f32_scalar(<15 x double> inreg %
; GFX11-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB27_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_add_f64 v[28:29], s[64:65], 1.0
@@ -17126,8 +17140,9 @@ define <60 x i16> @bitcast_v30f32_to_v60i16(<30 x float> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v83, 16, v0
; GFX11-TRUE16-NEXT: .LBB28_4: ; %end
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v0.h, v83.l
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v1.h, v82.l
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v2.h, v81.l
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v3.h, v80.l
@@ -17279,8 +17294,9 @@ define <60 x i16> @bitcast_v30f32_to_v60i16(<30 x float> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v83, 16, v0
; GFX11-FAKE16-NEXT: .LBB28_4: ; %end
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_perm_b32 v0, v83, v0, 0x5040100
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_perm_b32 v1, v82, v1, 0x5040100
; GFX11-FAKE16-NEXT: v_perm_b32 v2, v81, v2, 0x5040100
; GFX11-FAKE16-NEXT: v_perm_b32 v3, v80, v3, 0x5040100
@@ -18436,6 +18452,7 @@ define inreg <60 x i16> @bitcast_v30f32_to_v60i16_scalar(<30 x float> inreg %a,
; GFX11-TRUE16-NEXT: s_and_b32 s42, s42, exec_lo
; GFX11-TRUE16-NEXT: s_cselect_b32 s42, 1, 0
; GFX11-TRUE16-NEXT: s_cmp_lg_u32 s42, 1
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_cbranch_scc1 .LBB29_5
; GFX11-TRUE16-NEXT: ; %bb.4: ; %cmp.true
; GFX11-TRUE16-NEXT: v_add_f32_e64 v29, s4, 1.0
@@ -18652,6 +18669,7 @@ define inreg <60 x i16> @bitcast_v30f32_to_v60i16_scalar(<30 x float> inreg %a,
; GFX11-FAKE16-NEXT: s_and_b32 s42, s42, exec_lo
; GFX11-FAKE16-NEXT: s_cselect_b32 s42, 1, 0
; GFX11-FAKE16-NEXT: s_cmp_lg_u32 s42, 1
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_cbranch_scc1 .LBB29_5
; GFX11-FAKE16-NEXT: ; %bb.4: ; %cmp.true
; GFX11-FAKE16-NEXT: v_add_f32_e64 v25, s4, 1.0
@@ -20041,6 +20059,7 @@ define <30 x float> @bitcast_v60i16_to_v30f32(<60 x i16> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v84, 16, v0
; GFX11-TRUE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-TRUE16-NEXT: v_cmpx_ne_u32_e32 0, v30
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB30_3
; GFX11-TRUE16-NEXT: ; %bb.1: ; %Flow
@@ -21492,6 +21511,7 @@ define inreg <30 x float> @bitcast_v60i16_to_v30f32_scalar(<60 x i16> inreg %a,
; GFX11-NEXT: s_and_b32 s40, s40, exec_lo
; GFX11-NEXT: s_cselect_b32 s40, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s40, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB31_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_pk_add_u16 v0, s0, 3 op_sel_hi:[1,0]
@@ -22360,8 +22380,9 @@ define <60 x half> @bitcast_v30f32_to_v60f16(<30 x float> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v83, 16, v0
; GFX11-TRUE16-NEXT: .LBB32_4: ; %end
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v0.h, v83.l
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v1.h, v82.l
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v2.h, v81.l
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v3.h, v80.l
@@ -22513,8 +22534,9 @@ define <60 x half> @bitcast_v30f32_to_v60f16(<30 x float> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v83, 16, v0
; GFX11-FAKE16-NEXT: .LBB32_4: ; %end
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_perm_b32 v0, v83, v0, 0x5040100
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_perm_b32 v1, v82, v1, 0x5040100
; GFX11-FAKE16-NEXT: v_perm_b32 v2, v81, v2, 0x5040100
; GFX11-FAKE16-NEXT: v_perm_b32 v3, v80, v3, 0x5040100
@@ -23670,6 +23692,7 @@ define inreg <60 x half> @bitcast_v30f32_to_v60f16_scalar(<30 x float> inreg %a,
; GFX11-TRUE16-NEXT: s_and_b32 s42, s42, exec_lo
; GFX11-TRUE16-NEXT: s_cselect_b32 s42, 1, 0
; GFX11-TRUE16-NEXT: s_cmp_lg_u32 s42, 1
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_cbranch_scc1 .LBB33_5
; GFX11-TRUE16-NEXT: ; %bb.4: ; %cmp.true
; GFX11-TRUE16-NEXT: v_add_f32_e64 v29, s4, 1.0
@@ -23886,6 +23909,7 @@ define inreg <60 x half> @bitcast_v30f32_to_v60f16_scalar(<30 x float> inreg %a,
; GFX11-FAKE16-NEXT: s_and_b32 s42, s42, exec_lo
; GFX11-FAKE16-NEXT: s_cselect_b32 s42, 1, 0
; GFX11-FAKE16-NEXT: s_cmp_lg_u32 s42, 1
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_cbranch_scc1 .LBB33_5
; GFX11-FAKE16-NEXT: ; %bb.4: ; %cmp.true
; GFX11-FAKE16-NEXT: v_add_f32_e64 v25, s4, 1.0
@@ -25419,6 +25443,7 @@ define <30 x float> @bitcast_v60f16_to_v30f32(<60 x half> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v84, 16, v0
; GFX11-TRUE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-TRUE16-NEXT: v_cmpx_ne_u32_e32 0, v30
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB34_3
; GFX11-TRUE16-NEXT: ; %bb.1: ; %Flow
@@ -26874,6 +26899,7 @@ define inreg <30 x float> @bitcast_v60f16_to_v30f32_scalar(<60 x half> inreg %a,
; GFX11-NEXT: s_and_b32 s40, s40, exec_lo
; GFX11-NEXT: s_cselect_b32 s40, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s40, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB35_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_pk_add_f16 v0, 0x200, s0 op_sel_hi:[0,1]
@@ -27077,48 +27103,41 @@ define <15 x double> @bitcast_v15i64_to_v15f64(<15 x i64> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v30
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB36_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_add_co_u32 v0, vcc_lo, v0, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-NEXT: v_add_co_u32 v2, vcc_lo, v2, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX11-NEXT: v_add_co_u32 v4, vcc_lo, v4, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v5, null, 0, v5, vcc_lo
; GFX11-NEXT: v_add_co_u32 v6, vcc_lo, v6, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v7, null, 0, v7, vcc_lo
; GFX11-NEXT: v_add_co_u32 v8, vcc_lo, v8, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v9, null, 0, v9, vcc_lo
; GFX11-NEXT: v_add_co_u32 v10, vcc_lo, v10, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v11, null, 0, v11, vcc_lo
; GFX11-NEXT: v_add_co_u32 v12, vcc_lo, v12, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v13, null, 0, v13, vcc_lo
; GFX11-NEXT: v_add_co_u32 v14, vcc_lo, v14, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v15, null, 0, v15, vcc_lo
; GFX11-NEXT: v_add_co_u32 v16, vcc_lo, v16, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v17, null, 0, v17, vcc_lo
; GFX11-NEXT: v_add_co_u32 v18, vcc_lo, v18, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v19, null, 0, v19, vcc_lo
; GFX11-NEXT: v_add_co_u32 v20, vcc_lo, v20, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v21, null, 0, v21, vcc_lo
; GFX11-NEXT: v_add_co_u32 v22, vcc_lo, v22, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v23, null, 0, v23, vcc_lo
; GFX11-NEXT: v_add_co_u32 v24, vcc_lo, v24, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v25, null, 0, v25, vcc_lo
; GFX11-NEXT: v_add_co_u32 v26, vcc_lo, v26, 3
; GFX11-NEXT: v_add_co_ci_u32_e64 v27, null, 0, v27, vcc_lo
; GFX11-NEXT: v_add_co_u32 v28, vcc_lo, v28, 3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v29, null, 0, v29, vcc_lo
; GFX11-NEXT: .LBB36_2: ; %end
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -27455,6 +27474,7 @@ define inreg <15 x double> @bitcast_v15i64_to_v15f64_scalar(<15 x i64> inreg %a,
; GFX11-NEXT: s_and_b32 s40, s40, exec_lo
; GFX11-NEXT: s_cselect_b32 s40, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s40, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB37_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: s_add_u32 s0, s0, 3
@@ -27611,8 +27631,9 @@ define <15 x i64> @bitcast_v15f64_to_v15i64(<15 x double> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v30
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB38_2
; GFX11-NEXT: ; %bb.1: ; %cmp.true
@@ -28116,6 +28137,7 @@ define inreg <15 x i64> @bitcast_v15f64_to_v15i64_scalar(<15 x double> inreg %a,
; GFX11-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB39_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_add_f64 v[0:1], s[36:37], 1.0
@@ -28942,42 +28964,34 @@ define <60 x i16> @bitcast_v15i64_to_v60i16(<15 x i64> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: s_cbranch_execz .LBB40_4
; GFX11-TRUE16-NEXT: ; %bb.3: ; %cmp.true
; GFX11-TRUE16-NEXT: v_add_co_u32 v28, vcc_lo, v28, 3
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v29, null, 0, v29, vcc_lo
; GFX11-TRUE16-NEXT: v_add_co_u32 v26, vcc_lo, v26, 3
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v27, null, 0, v27, vcc_lo
; GFX11-TRUE16-NEXT: v_add_co_u32 v24, vcc_lo, v24, 3
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v25, null, 0, v25, vcc_lo
; GFX11-TRUE16-NEXT: v_add_co_u32 v22, vcc_lo, v22, 3
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v23, null, 0, v23, vcc_lo
; GFX11-TRUE16-NEXT: v_add_co_u32 v20, vcc_lo, v20, 3
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v21, null, 0, v21, vcc_lo
; GFX11-TRUE16-NEXT: v_add_co_u32 v18, vcc_lo, v18, 3
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v19, null, 0, v19, vcc_lo
; GFX11-TRUE16-NEXT: v_add_co_u32 v16, vcc_lo, v16, 3
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v17, null, 0, v17, vcc_lo
; GFX11-TRUE16-NEXT: v_add_co_u32 v14, vcc_lo, v14, 3
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v15, null, 0, v15, vcc_lo
; GFX11-TRUE16-NEXT: v_add_co_u32 v12, vcc_lo, v12, 3
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v13, null, 0, v13, vcc_lo
; GFX11-TRUE16-NEXT: v_add_co_u32 v10, vcc_lo, v10, 3
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v11, null, 0, v11, vcc_lo
; GFX11-TRUE16-NEXT: v_add_co_u32 v8, vcc_lo, v8, 3
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v9, null, 0, v9, vcc_lo
; GFX11-TRUE16-NEXT: v_add_co_u32 v6, vcc_lo, v6, 3
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v7, null, 0, v7, vcc_lo
; GFX11-TRUE16-NEXT: v_add_co_u32 v4, vcc_lo, v4, 3
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v5, null, 0, v5, vcc_lo
; GFX11-TRUE16-NEXT: v_add_co_u32 v2, vcc_lo, v2, 3
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX11-TRUE16-NEXT: v_add_co_u32 v0, vcc_lo, v0, 3
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v30, 16, v29
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v31, 16, v28
@@ -29011,8 +29025,9 @@ define <60 x i16> @bitcast_v15i64_to_v60i16(<15 x i64> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v83, 16, v0
; GFX11-TRUE16-NEXT: .LBB40_4: ; %end
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v0.h, v83.l
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v1.h, v82.l
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v2.h, v81.l
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v3.h, v80.l
@@ -29118,42 +29133,34 @@ define <60 x i16> @bitcast_v15i64_to_v60i16(<15 x i64> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: s_cbranch_execz .LBB40_4
; GFX11-FAKE16-NEXT: ; %bb.3: ; %cmp.true
; GFX11-FAKE16-NEXT: v_add_co_u32 v28, vcc_lo, v28, 3
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v29, null, 0, v29, vcc_lo
; GFX11-FAKE16-NEXT: v_add_co_u32 v26, vcc_lo, v26, 3
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v27, null, 0, v27, vcc_lo
; GFX11-FAKE16-NEXT: v_add_co_u32 v24, vcc_lo, v24, 3
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v25, null, 0, v25, vcc_lo
; GFX11-FAKE16-NEXT: v_add_co_u32 v22, vcc_lo, v22, 3
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v23, null, 0, v23, vcc_lo
; GFX11-FAKE16-NEXT: v_add_co_u32 v20, vcc_lo, v20, 3
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v21, null, 0, v21, vcc_lo
; GFX11-FAKE16-NEXT: v_add_co_u32 v18, vcc_lo, v18, 3
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v19, null, 0, v19, vcc_lo
; GFX11-FAKE16-NEXT: v_add_co_u32 v16, vcc_lo, v16, 3
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v17, null, 0, v17, vcc_lo
; GFX11-FAKE16-NEXT: v_add_co_u32 v14, vcc_lo, v14, 3
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v15, null, 0, v15, vcc_lo
; GFX11-FAKE16-NEXT: v_add_co_u32 v12, vcc_lo, v12, 3
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v13, null, 0, v13, vcc_lo
; GFX11-FAKE16-NEXT: v_add_co_u32 v10, vcc_lo, v10, 3
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v11, null, 0, v11, vcc_lo
; GFX11-FAKE16-NEXT: v_add_co_u32 v8, vcc_lo, v8, 3
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v9, null, 0, v9, vcc_lo
; GFX11-FAKE16-NEXT: v_add_co_u32 v6, vcc_lo, v6, 3
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v7, null, 0, v7, vcc_lo
; GFX11-FAKE16-NEXT: v_add_co_u32 v4, vcc_lo, v4, 3
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v5, null, 0, v5, vcc_lo
; GFX11-FAKE16-NEXT: v_add_co_u32 v2, vcc_lo, v2, 3
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX11-FAKE16-NEXT: v_add_co_u32 v0, vcc_lo, v0, 3
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v30, 16, v29
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v31, 16, v28
@@ -29187,8 +29194,9 @@ define <60 x i16> @bitcast_v15i64_to_v60i16(<15 x i64> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v83, 16, v0
; GFX11-FAKE16-NEXT: .LBB40_4: ; %end
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_perm_b32 v0, v83, v0, 0x5040100
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_perm_b32 v1, v82, v1, 0x5040100
; GFX11-FAKE16-NEXT: v_perm_b32 v2, v81, v2, 0x5040100
; GFX11-FAKE16-NEXT: v_perm_b32 v3, v80, v3, 0x5040100
@@ -30165,6 +30173,7 @@ define inreg <60 x i16> @bitcast_v15i64_to_v60i16_scalar(<15 x i64> inreg %a, i3
; GFX11-NEXT: s_and_b32 s94, s94, exec_lo
; GFX11-NEXT: s_cselect_b32 s94, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s94, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB41_5
; GFX11-NEXT: ; %bb.4: ; %cmp.true
; GFX11-NEXT: s_add_u32 s5, s5, 3
@@ -31508,6 +31517,7 @@ define <15 x i64> @bitcast_v60i16_to_v15i64(<60 x i16> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v84, 16, v0
; GFX11-TRUE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-TRUE16-NEXT: v_cmpx_ne_u32_e32 0, v30
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB42_3
; GFX11-TRUE16-NEXT: ; %bb.1: ; %Flow
@@ -32959,6 +32969,7 @@ define inreg <15 x i64> @bitcast_v60i16_to_v15i64_scalar(<60 x i16> inreg %a, i3
; GFX11-NEXT: s_and_b32 s40, s40, exec_lo
; GFX11-NEXT: s_cselect_b32 s40, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s40, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB43_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_pk_add_u16 v0, s0, 3 op_sel_hi:[1,0]
@@ -33781,42 +33792,34 @@ define <60 x half> @bitcast_v15i64_to_v60f16(<15 x i64> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: s_cbranch_execz .LBB44_4
; GFX11-TRUE16-NEXT: ; %bb.3: ; %cmp.true
; GFX11-TRUE16-NEXT: v_add_co_u32 v28, vcc_lo, v28, 3
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v29, null, 0, v29, vcc_lo
; GFX11-TRUE16-NEXT: v_add_co_u32 v26, vcc_lo, v26, 3
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v27, null, 0, v27, vcc_lo
; GFX11-TRUE16-NEXT: v_add_co_u32 v24, vcc_lo, v24, 3
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v25, null, 0, v25, vcc_lo
; GFX11-TRUE16-NEXT: v_add_co_u32 v22, vcc_lo, v22, 3
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v23, null, 0, v23, vcc_lo
; GFX11-TRUE16-NEXT: v_add_co_u32 v20, vcc_lo, v20, 3
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v21, null, 0, v21, vcc_lo
; GFX11-TRUE16-NEXT: v_add_co_u32 v18, vcc_lo, v18, 3
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v19, null, 0, v19, vcc_lo
; GFX11-TRUE16-NEXT: v_add_co_u32 v16, vcc_lo, v16, 3
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v17, null, 0, v17, vcc_lo
; GFX11-TRUE16-NEXT: v_add_co_u32 v14, vcc_lo, v14, 3
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v15, null, 0, v15, vcc_lo
; GFX11-TRUE16-NEXT: v_add_co_u32 v12, vcc_lo, v12, 3
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v13, null, 0, v13, vcc_lo
; GFX11-TRUE16-NEXT: v_add_co_u32 v10, vcc_lo, v10, 3
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v11, null, 0, v11, vcc_lo
; GFX11-TRUE16-NEXT: v_add_co_u32 v8, vcc_lo, v8, 3
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v9, null, 0, v9, vcc_lo
; GFX11-TRUE16-NEXT: v_add_co_u32 v6, vcc_lo, v6, 3
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v7, null, 0, v7, vcc_lo
; GFX11-TRUE16-NEXT: v_add_co_u32 v4, vcc_lo, v4, 3
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v5, null, 0, v5, vcc_lo
; GFX11-TRUE16-NEXT: v_add_co_u32 v2, vcc_lo, v2, 3
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX11-TRUE16-NEXT: v_add_co_u32 v0, vcc_lo, v0, 3
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v30, 16, v29
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v31, 16, v28
@@ -33850,8 +33853,9 @@ define <60 x half> @bitcast_v15i64_to_v60f16(<15 x i64> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v83, 16, v0
; GFX11-TRUE16-NEXT: .LBB44_4: ; %end
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v0.h, v83.l
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v1.h, v82.l
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v2.h, v81.l
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v3.h, v80.l
@@ -33957,42 +33961,34 @@ define <60 x half> @bitcast_v15i64_to_v60f16(<15 x i64> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: s_cbranch_execz .LBB44_4
; GFX11-FAKE16-NEXT: ; %bb.3: ; %cmp.true
; GFX11-FAKE16-NEXT: v_add_co_u32 v28, vcc_lo, v28, 3
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v29, null, 0, v29, vcc_lo
; GFX11-FAKE16-NEXT: v_add_co_u32 v26, vcc_lo, v26, 3
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v27, null, 0, v27, vcc_lo
; GFX11-FAKE16-NEXT: v_add_co_u32 v24, vcc_lo, v24, 3
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v25, null, 0, v25, vcc_lo
; GFX11-FAKE16-NEXT: v_add_co_u32 v22, vcc_lo, v22, 3
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v23, null, 0, v23, vcc_lo
; GFX11-FAKE16-NEXT: v_add_co_u32 v20, vcc_lo, v20, 3
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v21, null, 0, v21, vcc_lo
; GFX11-FAKE16-NEXT: v_add_co_u32 v18, vcc_lo, v18, 3
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v19, null, 0, v19, vcc_lo
; GFX11-FAKE16-NEXT: v_add_co_u32 v16, vcc_lo, v16, 3
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v17, null, 0, v17, vcc_lo
; GFX11-FAKE16-NEXT: v_add_co_u32 v14, vcc_lo, v14, 3
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v15, null, 0, v15, vcc_lo
; GFX11-FAKE16-NEXT: v_add_co_u32 v12, vcc_lo, v12, 3
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v13, null, 0, v13, vcc_lo
; GFX11-FAKE16-NEXT: v_add_co_u32 v10, vcc_lo, v10, 3
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v11, null, 0, v11, vcc_lo
; GFX11-FAKE16-NEXT: v_add_co_u32 v8, vcc_lo, v8, 3
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v9, null, 0, v9, vcc_lo
; GFX11-FAKE16-NEXT: v_add_co_u32 v6, vcc_lo, v6, 3
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v7, null, 0, v7, vcc_lo
; GFX11-FAKE16-NEXT: v_add_co_u32 v4, vcc_lo, v4, 3
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v5, null, 0, v5, vcc_lo
; GFX11-FAKE16-NEXT: v_add_co_u32 v2, vcc_lo, v2, 3
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX11-FAKE16-NEXT: v_add_co_u32 v0, vcc_lo, v0, 3
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v30, 16, v29
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v31, 16, v28
@@ -34026,8 +34022,9 @@ define <60 x half> @bitcast_v15i64_to_v60f16(<15 x i64> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v83, 16, v0
; GFX11-FAKE16-NEXT: .LBB44_4: ; %end
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_perm_b32 v0, v83, v0, 0x5040100
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_perm_b32 v1, v82, v1, 0x5040100
; GFX11-FAKE16-NEXT: v_perm_b32 v2, v81, v2, 0x5040100
; GFX11-FAKE16-NEXT: v_perm_b32 v3, v80, v3, 0x5040100
@@ -35004,6 +35001,7 @@ define inreg <60 x half> @bitcast_v15i64_to_v60f16_scalar(<15 x i64> inreg %a, i
; GFX11-NEXT: s_and_b32 s94, s94, exec_lo
; GFX11-NEXT: s_cselect_b32 s94, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s94, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB45_5
; GFX11-NEXT: ; %bb.4: ; %cmp.true
; GFX11-NEXT: s_add_u32 s5, s5, 3
@@ -36491,6 +36489,7 @@ define <15 x i64> @bitcast_v60f16_to_v15i64(<60 x half> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v84, 16, v0
; GFX11-TRUE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-TRUE16-NEXT: v_cmpx_ne_u32_e32 0, v30
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB46_3
; GFX11-TRUE16-NEXT: ; %bb.1: ; %Flow
@@ -37946,6 +37945,7 @@ define inreg <15 x i64> @bitcast_v60f16_to_v15i64_scalar(<60 x half> inreg %a, i
; GFX11-NEXT: s_and_b32 s40, s40, exec_lo
; GFX11-NEXT: s_cselect_b32 s40, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s40, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB47_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_pk_add_f16 v0, 0x200, s0 op_sel_hi:[0,1]
@@ -38769,8 +38769,9 @@ define <60 x i16> @bitcast_v15f64_to_v60i16(<15 x double> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v83, 16, v0
; GFX11-TRUE16-NEXT: .LBB48_4: ; %end
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v0.h, v83.l
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v1.h, v82.l
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v2.h, v81.l
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v3.h, v80.l
@@ -38922,8 +38923,9 @@ define <60 x i16> @bitcast_v15f64_to_v60i16(<15 x double> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v83, 16, v0
; GFX11-FAKE16-NEXT: .LBB48_4: ; %end
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_perm_b32 v0, v83, v0, 0x5040100
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_perm_b32 v1, v82, v1, 0x5040100
; GFX11-FAKE16-NEXT: v_perm_b32 v2, v81, v2, 0x5040100
; GFX11-FAKE16-NEXT: v_perm_b32 v3, v80, v3, 0x5040100
@@ -40040,6 +40042,7 @@ define inreg <60 x i16> @bitcast_v15f64_to_v60i16_scalar(<15 x double> inreg %a,
; GFX11-TRUE16-NEXT: s_and_b32 s42, s42, exec_lo
; GFX11-TRUE16-NEXT: s_cselect_b32 s42, 1, 0
; GFX11-TRUE16-NEXT: s_cmp_lg_u32 s42, 1
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_cbranch_scc1 .LBB49_5
; GFX11-TRUE16-NEXT: ; %bb.4: ; %cmp.true
; GFX11-TRUE16-NEXT: v_add_f64 v[28:29], s[14:15], 1.0
@@ -40242,6 +40245,7 @@ define inreg <60 x i16> @bitcast_v15f64_to_v60i16_scalar(<15 x double> inreg %a,
; GFX11-FAKE16-NEXT: s_and_b32 s42, s42, exec_lo
; GFX11-FAKE16-NEXT: s_cselect_b32 s42, 1, 0
; GFX11-FAKE16-NEXT: s_cmp_lg_u32 s42, 1
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_cbranch_scc1 .LBB49_5
; GFX11-FAKE16-NEXT: ; %bb.4: ; %cmp.true
; GFX11-FAKE16-NEXT: v_add_f64 v[25:26], s[14:15], 1.0
@@ -41616,6 +41620,7 @@ define <15 x double> @bitcast_v60i16_to_v15f64(<60 x i16> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v84, 16, v0
; GFX11-TRUE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-TRUE16-NEXT: v_cmpx_ne_u32_e32 0, v30
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB50_3
; GFX11-TRUE16-NEXT: ; %bb.1: ; %Flow
@@ -43067,6 +43072,7 @@ define inreg <15 x double> @bitcast_v60i16_to_v15f64_scalar(<60 x i16> inreg %a,
; GFX11-NEXT: s_and_b32 s40, s40, exec_lo
; GFX11-NEXT: s_cselect_b32 s40, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s40, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB51_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_pk_add_u16 v0, s0, 3 op_sel_hi:[1,0]
@@ -43890,8 +43896,9 @@ define <60 x half> @bitcast_v15f64_to_v60f16(<15 x double> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v83, 16, v0
; GFX11-TRUE16-NEXT: .LBB52_4: ; %end
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v0.h, v83.l
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v1.h, v82.l
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v2.h, v81.l
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v3.h, v80.l
@@ -44043,8 +44050,9 @@ define <60 x half> @bitcast_v15f64_to_v60f16(<15 x double> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v83, 16, v0
; GFX11-FAKE16-NEXT: .LBB52_4: ; %end
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_perm_b32 v0, v83, v0, 0x5040100
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_perm_b32 v1, v82, v1, 0x5040100
; GFX11-FAKE16-NEXT: v_perm_b32 v2, v81, v2, 0x5040100
; GFX11-FAKE16-NEXT: v_perm_b32 v3, v80, v3, 0x5040100
@@ -45161,6 +45169,7 @@ define inreg <60 x half> @bitcast_v15f64_to_v60f16_scalar(<15 x double> inreg %a
; GFX11-TRUE16-NEXT: s_and_b32 s42, s42, exec_lo
; GFX11-TRUE16-NEXT: s_cselect_b32 s42, 1, 0
; GFX11-TRUE16-NEXT: s_cmp_lg_u32 s42, 1
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_cbranch_scc1 .LBB53_5
; GFX11-TRUE16-NEXT: ; %bb.4: ; %cmp.true
; GFX11-TRUE16-NEXT: v_add_f64 v[28:29], s[14:15], 1.0
@@ -45363,6 +45372,7 @@ define inreg <60 x half> @bitcast_v15f64_to_v60f16_scalar(<15 x double> inreg %a
; GFX11-FAKE16-NEXT: s_and_b32 s42, s42, exec_lo
; GFX11-FAKE16-NEXT: s_cselect_b32 s42, 1, 0
; GFX11-FAKE16-NEXT: s_cmp_lg_u32 s42, 1
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_cbranch_scc1 .LBB53_5
; GFX11-FAKE16-NEXT: ; %bb.4: ; %cmp.true
; GFX11-FAKE16-NEXT: v_add_f64 v[25:26], s[14:15], 1.0
@@ -46881,6 +46891,7 @@ define <15 x double> @bitcast_v60f16_to_v15f64(<60 x half> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v84, 16, v0
; GFX11-TRUE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-TRUE16-NEXT: v_cmpx_ne_u32_e32 0, v30
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB54_3
; GFX11-TRUE16-NEXT: ; %bb.1: ; %Flow
@@ -48336,6 +48347,7 @@ define inreg <15 x double> @bitcast_v60f16_to_v15f64_scalar(<60 x half> inreg %a
; GFX11-NEXT: s_and_b32 s40, s40, exec_lo
; GFX11-NEXT: s_cselect_b32 s40, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s40, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB55_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_pk_add_f16 v0, 0x200, s0 op_sel_hi:[0,1]
@@ -49688,8 +49700,9 @@ define <60 x half> @bitcast_v60i16_to_v60f16(<60 x i16> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v35, 16, v0
; GFX11-TRUE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-TRUE16-NEXT: v_cmpx_ne_u32_e32 0, v30
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-TRUE16-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-TRUE16-NEXT: s_cbranch_execz .LBB56_2
; GFX11-TRUE16-NEXT: ; %bb.1: ; %cmp.true
@@ -49785,6 +49798,7 @@ define <60 x half> @bitcast_v60i16_to_v60f16(<60 x i16> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v84, 16, v29
; GFX11-TRUE16-NEXT: .LBB56_2: ; %end
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v0.h, v35.l
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v1.h, v34.l
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v2.h, v33.l
@@ -49852,8 +49866,9 @@ define <60 x half> @bitcast_v60i16_to_v60f16(<60 x i16> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v31, 16, v0
; GFX11-FAKE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-FAKE16-NEXT: v_cmpx_ne_u32_e32 0, v30
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-FAKE16-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-FAKE16-NEXT: s_cbranch_execz .LBB56_2
; GFX11-FAKE16-NEXT: ; %bb.1: ; %cmp.true
@@ -49949,6 +49964,7 @@ define <60 x half> @bitcast_v60i16_to_v60f16(<60 x i16> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v84, 16, v29
; GFX11-FAKE16-NEXT: .LBB56_2: ; %end
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_perm_b32 v0, v31, v0, 0x5040100
; GFX11-FAKE16-NEXT: v_perm_b32 v1, v32, v1, 0x5040100
; GFX11-FAKE16-NEXT: v_perm_b32 v2, v33, v2, 0x5040100
@@ -51293,6 +51309,7 @@ define inreg <60 x half> @bitcast_v60i16_to_v60f16_scalar(<60 x i16> inreg %a, i
; GFX11-TRUE16-NEXT: s_and_b32 s94, s94, exec_lo
; GFX11-TRUE16-NEXT: s_cselect_b32 s94, 1, 0
; GFX11-TRUE16-NEXT: s_cmp_lg_u32 s94, 1
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_cbranch_scc1 .LBB57_4
; GFX11-TRUE16-NEXT: ; %bb.3: ; %cmp.true
; GFX11-TRUE16-NEXT: s_pack_ll_b32_b16 s59, s89, s59
@@ -51508,6 +51525,7 @@ define inreg <60 x half> @bitcast_v60i16_to_v60f16_scalar(<60 x i16> inreg %a, i
; GFX11-FAKE16-NEXT: s_and_b32 s94, s94, exec_lo
; GFX11-FAKE16-NEXT: s_cselect_b32 s94, 1, 0
; GFX11-FAKE16-NEXT: s_cmp_lg_u32 s94, 1
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_cbranch_scc1 .LBB57_4
; GFX11-FAKE16-NEXT: ; %bb.3: ; %cmp.true
; GFX11-FAKE16-NEXT: s_pack_ll_b32_b16 s59, s89, s59
@@ -52549,8 +52567,9 @@ define <60 x i16> @bitcast_v60f16_to_v60i16(<60 x half> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v35, 16, v0
; GFX11-TRUE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-TRUE16-NEXT: v_cmpx_ne_u32_e32 0, v30
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-TRUE16-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-TRUE16-NEXT: s_cbranch_execz .LBB58_2
; GFX11-TRUE16-NEXT: ; %bb.1: ; %cmp.true
@@ -52646,6 +52665,7 @@ define <60 x i16> @bitcast_v60f16_to_v60i16(<60 x half> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v84, 16, v29
; GFX11-TRUE16-NEXT: .LBB58_2: ; %end
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v0.h, v35.l
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v1.h, v34.l
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v2.h, v33.l
@@ -52713,8 +52733,9 @@ define <60 x i16> @bitcast_v60f16_to_v60i16(<60 x half> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v31, 16, v0
; GFX11-FAKE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-FAKE16-NEXT: v_cmpx_ne_u32_e32 0, v30
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-FAKE16-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-FAKE16-NEXT: s_cbranch_execz .LBB58_2
; GFX11-FAKE16-NEXT: ; %bb.1: ; %cmp.true
@@ -52810,6 +52831,7 @@ define <60 x i16> @bitcast_v60f16_to_v60i16(<60 x half> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v84, 16, v29
; GFX11-FAKE16-NEXT: .LBB58_2: ; %end
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_perm_b32 v0, v31, v0, 0x5040100
; GFX11-FAKE16-NEXT: v_perm_b32 v1, v32, v1, 0x5040100
; GFX11-FAKE16-NEXT: v_perm_b32 v2, v33, v2, 0x5040100
@@ -54097,6 +54119,7 @@ define inreg <60 x i16> @bitcast_v60f16_to_v60i16_scalar(<60 x half> inreg %a, i
; GFX11-TRUE16-NEXT: s_and_b32 s94, s94, exec_lo
; GFX11-TRUE16-NEXT: s_cselect_b32 s94, 1, 0
; GFX11-TRUE16-NEXT: s_cmp_lg_u32 s94, 1
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_cbranch_scc1 .LBB59_4
; GFX11-TRUE16-NEXT: ; %bb.3: ; %cmp.true
; GFX11-TRUE16-NEXT: s_pack_ll_b32_b16 s59, s89, s59
@@ -54312,6 +54335,7 @@ define inreg <60 x i16> @bitcast_v60f16_to_v60i16_scalar(<60 x half> inreg %a, i
; GFX11-FAKE16-NEXT: s_and_b32 s94, s94, exec_lo
; GFX11-FAKE16-NEXT: s_cselect_b32 s94, 1, 0
; GFX11-FAKE16-NEXT: s_cmp_lg_u32 s94, 1
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_cbranch_scc1 .LBB59_4
; GFX11-FAKE16-NEXT: ; %bb.3: ; %cmp.true
; GFX11-FAKE16-NEXT: s_pack_ll_b32_b16 s59, s89, s59
diff --git a/llvm/test/CodeGen/AMDGPU/amdgcn.bitcast.96bit.ll b/llvm/test/CodeGen/AMDGPU/amdgcn.bitcast.96bit.ll
index 374626a2578fb5..d1f642b60324d7 100644
--- a/llvm/test/CodeGen/AMDGPU/amdgcn.bitcast.96bit.ll
+++ b/llvm/test/CodeGen/AMDGPU/amdgcn.bitcast.96bit.ll
@@ -57,8 +57,9 @@ define <3 x float> @bitcast_v3i32_to_v3f32(<3 x i32> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v3
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_add_nc_u32_e32 v2, 3, v2
@@ -173,6 +174,7 @@ define inreg <3 x float> @bitcast_v3i32_to_v3f32_scalar(<3 x i32> inreg %a, i32
; GFX11-NEXT: s_and_b32 s3, s3, exec_lo
; GFX11-NEXT: s_cselect_b32 s3, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s3, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB1_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: s_add_i32 s2, s2, 3
@@ -251,8 +253,9 @@ define <3 x i32> @bitcast_v3f32_to_v3i32(<3 x float> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v3
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_dual_add_f32 v2, 1.0, v2 :: v_dual_add_f32 v1, 1.0, v1
@@ -369,6 +372,7 @@ define inreg <3 x i32> @bitcast_v3f32_to_v3i32_scalar(<3 x float> inreg %a, i32
; GFX11-NEXT: s_and_b32 s3, s3, exec_lo
; GFX11-NEXT: s_cselect_b32 s3, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s3, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB3_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_add_f32_e64 v2, s2, 1.0
@@ -599,6 +603,7 @@ define <12 x i8> @bitcast_v3i32_to_v12i8(<3 x i32> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v1, 8, v11
; GFX11-TRUE16-NEXT: .LBB4_4: ; %end
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v0.l, v11.l
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v4.l, v12.l
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v11.l, v13.l
@@ -652,6 +657,7 @@ define <12 x i8> @bitcast_v3i32_to_v12i8(<3 x i32> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v1, 8, v13
; GFX11-FAKE16-NEXT: .LBB4_4: ; %end
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v0, v13
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v4, v14
; GFX11-FAKE16-NEXT: s_setpc_b64 s[30:31]
@@ -889,6 +895,7 @@ define inreg <12 x i8> @bitcast_v3i32_to_v12i8_scalar(<3 x i32> inreg %a, i32 in
; GFX11-NEXT: s_and_b32 s5, s14, exec_lo
; GFX11-NEXT: s_cselect_b32 s5, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s5, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB5_5
; GFX11-NEXT: ; %bb.4: ; %cmp.true
; GFX11-NEXT: s_add_i32 s2, s2, 3
@@ -1185,6 +1192,7 @@ define <3 x i32> @bitcast_v12i8_to_v3i32(<12 x i8> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-TRUE16-NEXT: ; implicit-def: $vgpr0_vgpr1_vgpr2
; GFX11-TRUE16-NEXT: v_cmpx_ne_u32_e32 0, v12
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB6_3
; GFX11-TRUE16-NEXT: ; %bb.1: ; %Flow
@@ -1260,6 +1268,7 @@ define <3 x i32> @bitcast_v12i8_to_v3i32(<12 x i8> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-FAKE16-NEXT: ; implicit-def: $vgpr0_vgpr1_vgpr2
; GFX11-FAKE16-NEXT: v_cmpx_ne_u32_e32 0, v12
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB6_3
; GFX11-FAKE16-NEXT: ; %bb.1: ; %Flow
@@ -1610,6 +1619,7 @@ define inreg <3 x i32> @bitcast_v12i8_to_v3i32_scalar(<12 x i8> inreg %a, i32 in
; GFX11-NEXT: s_and_b32 s4, s4, exec_lo
; GFX11-NEXT: s_cselect_b32 s4, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s4, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB7_5
; GFX11-NEXT: ; %bb.4: ; %cmp.true
; GFX11-NEXT: v_mov_b32_e32 v0, 0xc0c0004
@@ -1746,8 +1756,9 @@ define <6 x bfloat> @bitcast_v3i32_to_v6bf16(<3 x i32> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v3
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_add_nc_u32_e32 v2, 3, v2
@@ -1889,6 +1900,7 @@ define inreg <6 x bfloat> @bitcast_v3i32_to_v6bf16_scalar(<3 x i32> inreg %a, i3
; GFX11-NEXT: s_and_b32 s3, s3, exec_lo
; GFX11-NEXT: s_cselect_b32 s3, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s3, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB9_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: s_add_i32 s2, s2, 3
@@ -2113,8 +2125,9 @@ define <3 x i32> @bitcast_v6bf16_to_v3i32(<6 x bfloat> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-TRUE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-TRUE16-NEXT: v_cmpx_ne_u32_e32 0, v3
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-TRUE16-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-TRUE16-NEXT: s_cbranch_execz .LBB10_2
; GFX11-TRUE16-NEXT: ; %bb.1: ; %cmp.true
@@ -2186,8 +2199,9 @@ define <3 x i32> @bitcast_v6bf16_to_v3i32(<6 x bfloat> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-FAKE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-FAKE16-NEXT: v_cmpx_ne_u32_e32 0, v3
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-FAKE16-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-FAKE16-NEXT: s_cbranch_execz .LBB10_2
; GFX11-FAKE16-NEXT: ; %bb.1: ; %cmp.true
@@ -2492,6 +2506,7 @@ define inreg <3 x i32> @bitcast_v6bf16_to_v3i32_scalar(<6 x bfloat> inreg %a, i3
; GFX11-TRUE16-NEXT: s_and_b32 s3, s3, exec_lo
; GFX11-TRUE16-NEXT: s_cselect_b32 s3, 1, 0
; GFX11-TRUE16-NEXT: s_cmp_lg_u32 s3, 1
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_cbranch_scc1 .LBB11_4
; GFX11-TRUE16-NEXT: ; %bb.3: ; %cmp.true
; GFX11-TRUE16-NEXT: s_pack_lh_b32_b16 s3, 0, s2
@@ -2577,6 +2592,7 @@ define inreg <3 x i32> @bitcast_v6bf16_to_v3i32_scalar(<6 x bfloat> inreg %a, i3
; GFX11-FAKE16-NEXT: s_and_b32 s3, s3, exec_lo
; GFX11-FAKE16-NEXT: s_cselect_b32 s3, 1, 0
; GFX11-FAKE16-NEXT: s_cmp_lg_u32 s3, 1
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_cbranch_scc1 .LBB11_4
; GFX11-FAKE16-NEXT: ; %bb.3: ; %cmp.true
; GFX11-FAKE16-NEXT: s_pack_lh_b32_b16 s3, 0, s2
@@ -2738,8 +2754,9 @@ define <6 x half> @bitcast_v3i32_to_v6f16(<3 x i32> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v3
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_add_nc_u32_e32 v2, 3, v2
@@ -2872,6 +2889,7 @@ define inreg <6 x half> @bitcast_v3i32_to_v6f16_scalar(<3 x i32> inreg %a, i32 i
; GFX11-NEXT: s_and_b32 s3, s3, exec_lo
; GFX11-NEXT: s_cselect_b32 s3, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s3, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB13_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: s_add_i32 s2, s2, 3
@@ -3010,8 +3028,9 @@ define <3 x i32> @bitcast_v6f16_to_v3i32(<6 x half> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v3
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_pk_add_f16 v2, 0x200, v2 op_sel_hi:[0,1]
@@ -3177,6 +3196,7 @@ define inreg <3 x i32> @bitcast_v6f16_to_v3i32_scalar(<6 x half> inreg %a, i32 i
; GFX11-NEXT: s_and_b32 s3, s3, exec_lo
; GFX11-NEXT: s_cselect_b32 s3, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s3, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB15_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_pk_add_f16 v2, 0x200, s2 op_sel_hi:[0,1]
@@ -3275,8 +3295,9 @@ define <6 x i16> @bitcast_v3i32_to_v6i16(<3 x i32> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v3
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_add_nc_u32_e32 v2, 3, v2
@@ -3409,6 +3430,7 @@ define inreg <6 x i16> @bitcast_v3i32_to_v6i16_scalar(<3 x i32> inreg %a, i32 in
; GFX11-NEXT: s_and_b32 s3, s3, exec_lo
; GFX11-NEXT: s_cselect_b32 s3, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s3, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB17_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: s_add_i32 s2, s2, 3
@@ -3534,8 +3556,9 @@ define <3 x i32> @bitcast_v6i16_to_v3i32(<6 x i16> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v3
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_pk_add_u16 v2, v2, 3 op_sel_hi:[1,0]
@@ -3688,6 +3711,7 @@ define inreg <3 x i32> @bitcast_v6i16_to_v3i32_scalar(<6 x i16> inreg %a, i32 in
; GFX11-NEXT: s_and_b32 s3, s3, exec_lo
; GFX11-NEXT: s_cselect_b32 s3, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s3, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB19_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_pk_add_u16 v2, s2, 3 op_sel_hi:[1,0]
@@ -3917,6 +3941,7 @@ define <12 x i8> @bitcast_v3f32_to_v12i8(<3 x float> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v1, 8, v11
; GFX11-TRUE16-NEXT: .LBB20_4: ; %end
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v0.l, v11.l
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v4.l, v12.l
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v11.l, v13.l
@@ -3969,6 +3994,7 @@ define <12 x i8> @bitcast_v3f32_to_v12i8(<3 x float> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v1, 8, v13
; GFX11-FAKE16-NEXT: .LBB20_4: ; %end
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v0, v13
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v4, v14
; GFX11-FAKE16-NEXT: s_setpc_b64 s[30:31]
@@ -4220,6 +4246,7 @@ define inreg <12 x i8> @bitcast_v3f32_to_v12i8_scalar(<3 x float> inreg %a, i32
; GFX11-NEXT: s_and_b32 s5, s8, exec_lo
; GFX11-NEXT: s_cselect_b32 s5, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s5, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB21_5
; GFX11-NEXT: ; %bb.4: ; %cmp.true
; GFX11-NEXT: v_add_f32_e64 v8, s2, 1.0
@@ -4522,6 +4549,7 @@ define <3 x float> @bitcast_v12i8_to_v3f32(<12 x i8> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-TRUE16-NEXT: ; implicit-def: $vgpr0_vgpr1_vgpr2
; GFX11-TRUE16-NEXT: v_cmpx_ne_u32_e32 0, v12
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB22_3
; GFX11-TRUE16-NEXT: ; %bb.1: ; %Flow
@@ -4597,6 +4625,7 @@ define <3 x float> @bitcast_v12i8_to_v3f32(<12 x i8> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-FAKE16-NEXT: ; implicit-def: $vgpr0_vgpr1_vgpr2
; GFX11-FAKE16-NEXT: v_cmpx_ne_u32_e32 0, v12
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB22_3
; GFX11-FAKE16-NEXT: ; %bb.1: ; %Flow
@@ -4947,6 +4976,7 @@ define inreg <3 x float> @bitcast_v12i8_to_v3f32_scalar(<12 x i8> inreg %a, i32
; GFX11-NEXT: s_and_b32 s4, s4, exec_lo
; GFX11-NEXT: s_cselect_b32 s4, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s4, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB23_5
; GFX11-NEXT: ; %bb.4: ; %cmp.true
; GFX11-NEXT: v_mov_b32_e32 v0, 0xc0c0004
@@ -5083,8 +5113,9 @@ define <6 x bfloat> @bitcast_v3f32_to_v6bf16(<3 x float> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v3
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_dual_add_f32 v2, 1.0, v2 :: v_dual_add_f32 v1, 1.0, v1
@@ -5237,6 +5268,7 @@ define inreg <6 x bfloat> @bitcast_v3f32_to_v6bf16_scalar(<3 x float> inreg %a,
; GFX11-NEXT: s_and_b32 s4, s4, exec_lo
; GFX11-NEXT: s_cselect_b32 s4, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s4, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB25_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_add_f32_e64 v2, s2, 1.0
@@ -5461,8 +5493,9 @@ define <3 x float> @bitcast_v6bf16_to_v3f32(<6 x bfloat> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-TRUE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-TRUE16-NEXT: v_cmpx_ne_u32_e32 0, v3
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-TRUE16-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-TRUE16-NEXT: s_cbranch_execz .LBB26_2
; GFX11-TRUE16-NEXT: ; %bb.1: ; %cmp.true
@@ -5534,8 +5567,9 @@ define <3 x float> @bitcast_v6bf16_to_v3f32(<6 x bfloat> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-FAKE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-FAKE16-NEXT: v_cmpx_ne_u32_e32 0, v3
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-FAKE16-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-FAKE16-NEXT: s_cbranch_execz .LBB26_2
; GFX11-FAKE16-NEXT: ; %bb.1: ; %cmp.true
@@ -5840,6 +5874,7 @@ define inreg <3 x float> @bitcast_v6bf16_to_v3f32_scalar(<6 x bfloat> inreg %a,
; GFX11-TRUE16-NEXT: s_and_b32 s3, s3, exec_lo
; GFX11-TRUE16-NEXT: s_cselect_b32 s3, 1, 0
; GFX11-TRUE16-NEXT: s_cmp_lg_u32 s3, 1
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_cbranch_scc1 .LBB27_4
; GFX11-TRUE16-NEXT: ; %bb.3: ; %cmp.true
; GFX11-TRUE16-NEXT: s_pack_lh_b32_b16 s3, 0, s2
@@ -5925,6 +5960,7 @@ define inreg <3 x float> @bitcast_v6bf16_to_v3f32_scalar(<6 x bfloat> inreg %a,
; GFX11-FAKE16-NEXT: s_and_b32 s3, s3, exec_lo
; GFX11-FAKE16-NEXT: s_cselect_b32 s3, 1, 0
; GFX11-FAKE16-NEXT: s_cmp_lg_u32 s3, 1
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_cbranch_scc1 .LBB27_4
; GFX11-FAKE16-NEXT: ; %bb.3: ; %cmp.true
; GFX11-FAKE16-NEXT: s_pack_lh_b32_b16 s3, 0, s2
@@ -6086,8 +6122,9 @@ define <6 x half> @bitcast_v3f32_to_v6f16(<3 x float> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v3
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_dual_add_f32 v2, 1.0, v2 :: v_dual_add_f32 v1, 1.0, v1
@@ -6228,6 +6265,7 @@ define inreg <6 x half> @bitcast_v3f32_to_v6f16_scalar(<3 x float> inreg %a, i32
; GFX11-NEXT: s_and_b32 s4, s4, exec_lo
; GFX11-NEXT: s_cselect_b32 s4, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s4, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB29_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_add_f32_e64 v2, s2, 1.0
@@ -6366,8 +6404,9 @@ define <3 x float> @bitcast_v6f16_to_v3f32(<6 x half> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v3
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_pk_add_f16 v2, 0x200, v2 op_sel_hi:[0,1]
@@ -6533,6 +6572,7 @@ define inreg <3 x float> @bitcast_v6f16_to_v3f32_scalar(<6 x half> inreg %a, i32
; GFX11-NEXT: s_and_b32 s3, s3, exec_lo
; GFX11-NEXT: s_cselect_b32 s3, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s3, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB31_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_pk_add_f16 v2, 0x200, s2 op_sel_hi:[0,1]
@@ -6631,8 +6671,9 @@ define <6 x i16> @bitcast_v3f32_to_v6i16(<3 x float> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v3
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_dual_add_f32 v2, 1.0, v2 :: v_dual_add_f32 v1, 1.0, v1
@@ -6773,6 +6814,7 @@ define inreg <6 x i16> @bitcast_v3f32_to_v6i16_scalar(<3 x float> inreg %a, i32
; GFX11-NEXT: s_and_b32 s4, s4, exec_lo
; GFX11-NEXT: s_cselect_b32 s4, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s4, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB33_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_add_f32_e64 v2, s2, 1.0
@@ -6898,8 +6940,9 @@ define <3 x float> @bitcast_v6i16_to_v3f32(<6 x i16> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v3
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_pk_add_u16 v2, v2, 3 op_sel_hi:[1,0]
@@ -7052,6 +7095,7 @@ define inreg <3 x float> @bitcast_v6i16_to_v3f32_scalar(<6 x i16> inreg %a, i32
; GFX11-NEXT: s_and_b32 s3, s3, exec_lo
; GFX11-NEXT: s_cselect_b32 s3, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s3, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB35_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_pk_add_u16 v2, s2, 3 op_sel_hi:[1,0]
@@ -7355,6 +7399,7 @@ define <6 x bfloat> @bitcast_v12i8_to_v6bf16(<12 x i8> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-TRUE16-NEXT: ; implicit-def: $vgpr0_vgpr1_vgpr2_vgpr3
; GFX11-TRUE16-NEXT: v_cmpx_ne_u32_e32 0, v12
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB36_3
; GFX11-TRUE16-NEXT: ; %bb.1: ; %Flow
@@ -7430,6 +7475,7 @@ define <6 x bfloat> @bitcast_v12i8_to_v6bf16(<12 x i8> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-FAKE16-NEXT: ; implicit-def: $vgpr0_vgpr1_vgpr2_vgpr3
; GFX11-FAKE16-NEXT: v_cmpx_ne_u32_e32 0, v12
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB36_3
; GFX11-FAKE16-NEXT: ; %bb.1: ; %Flow
@@ -7797,6 +7843,7 @@ define inreg <6 x bfloat> @bitcast_v12i8_to_v6bf16_scalar(<12 x i8> inreg %a, i3
; GFX11-NEXT: s_and_b32 s4, s4, exec_lo
; GFX11-NEXT: s_cselect_b32 s4, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s4, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB37_5
; GFX11-NEXT: ; %bb.4: ; %cmp.true
; GFX11-NEXT: v_mov_b32_e32 v0, 0xc0c0004
@@ -8243,6 +8290,7 @@ define <12 x i8> @bitcast_v6bf16_to_v12i8(<6 x bfloat> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v9, 8, v11
; GFX11-TRUE16-NEXT: .LBB38_4: ; %end
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v0.l, v13.l
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v4.l, v14.l
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v8.l, v11.l
@@ -8348,6 +8396,7 @@ define <12 x i8> @bitcast_v6bf16_to_v12i8(<6 x bfloat> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v1, 8, v0
; GFX11-FAKE16-NEXT: .LBB38_4: ; %end
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v0, v13
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v4, v14
; GFX11-FAKE16-NEXT: s_setpc_b64 s[30:31]
@@ -8724,6 +8773,7 @@ define inreg <12 x i8> @bitcast_v6bf16_to_v12i8_scalar(<6 x bfloat> inreg %a, i3
; GFX11-TRUE16-NEXT: s_and_b32 s3, s3, exec_lo
; GFX11-TRUE16-NEXT: s_cselect_b32 s3, 1, 0
; GFX11-TRUE16-NEXT: s_cmp_lg_u32 s3, 1
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_cbranch_scc1 .LBB39_5
; GFX11-TRUE16-NEXT: ; %bb.4: ; %cmp.true
; GFX11-TRUE16-NEXT: s_pack_lh_b32_b16 s3, 0, s1
@@ -8846,6 +8896,7 @@ define inreg <12 x i8> @bitcast_v6bf16_to_v12i8_scalar(<6 x bfloat> inreg %a, i3
; GFX11-FAKE16-NEXT: s_and_b32 s3, s3, exec_lo
; GFX11-FAKE16-NEXT: s_cselect_b32 s3, 1, 0
; GFX11-FAKE16-NEXT: s_cmp_lg_u32 s3, 1
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_cbranch_scc1 .LBB39_5
; GFX11-FAKE16-NEXT: ; %bb.4: ; %cmp.true
; GFX11-FAKE16-NEXT: s_pack_lh_b32_b16 s3, 0, s1
@@ -9224,6 +9275,7 @@ define <6 x half> @bitcast_v12i8_to_v6f16(<12 x i8> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-TRUE16-NEXT: ; implicit-def: $vgpr0_vgpr1_vgpr2_vgpr3
; GFX11-TRUE16-NEXT: v_cmpx_ne_u32_e32 0, v12
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB40_3
; GFX11-TRUE16-NEXT: ; %bb.1: ; %Flow
@@ -9299,6 +9351,7 @@ define <6 x half> @bitcast_v12i8_to_v6f16(<12 x i8> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-FAKE16-NEXT: ; implicit-def: $vgpr0_vgpr1_vgpr2_vgpr3
; GFX11-FAKE16-NEXT: v_cmpx_ne_u32_e32 0, v12
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB40_3
; GFX11-FAKE16-NEXT: ; %bb.1: ; %Flow
@@ -9669,6 +9722,7 @@ define inreg <6 x half> @bitcast_v12i8_to_v6f16_scalar(<12 x i8> inreg %a, i32 i
; GFX11-NEXT: s_and_b32 s4, s4, exec_lo
; GFX11-NEXT: s_cselect_b32 s4, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s4, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB41_5
; GFX11-NEXT: ; %bb.4: ; %cmp.true
; GFX11-NEXT: v_mov_b32_e32 v0, 0xc0c0004
@@ -9966,6 +10020,7 @@ define <12 x i8> @bitcast_v6f16_to_v12i8(<6 x half> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v1, 8, v13
; GFX11-TRUE16-NEXT: .LBB42_4: ; %end
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v0.l, v13.l
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v4.l, v14.l
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v8.l, v11.l
@@ -10022,6 +10077,7 @@ define <12 x i8> @bitcast_v6f16_to_v12i8(<6 x half> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v1, 8, v15
; GFX11-FAKE16-NEXT: .LBB42_4: ; %end
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v0, v15
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v4, v16
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v8, v13
@@ -10314,6 +10370,7 @@ define inreg <12 x i8> @bitcast_v6f16_to_v12i8_scalar(<6 x half> inreg %a, i32 i
; GFX11-NEXT: s_and_b32 s5, s8, exec_lo
; GFX11-NEXT: s_cselect_b32 s5, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s5, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB43_5
; GFX11-NEXT: ; %bb.4: ; %cmp.true
; GFX11-NEXT: v_pk_add_f16 v15, 0x200, s1 op_sel_hi:[0,1]
@@ -10636,6 +10693,7 @@ define <6 x i16> @bitcast_v12i8_to_v6i16(<12 x i8> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-TRUE16-NEXT: ; implicit-def: $vgpr0_vgpr1_vgpr2_vgpr3
; GFX11-TRUE16-NEXT: v_cmpx_ne_u32_e32 0, v12
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB44_3
; GFX11-TRUE16-NEXT: ; %bb.1: ; %Flow
@@ -10711,6 +10769,7 @@ define <6 x i16> @bitcast_v12i8_to_v6i16(<12 x i8> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-FAKE16-NEXT: ; implicit-def: $vgpr0_vgpr1_vgpr2_vgpr3
; GFX11-FAKE16-NEXT: v_cmpx_ne_u32_e32 0, v12
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB44_3
; GFX11-FAKE16-NEXT: ; %bb.1: ; %Flow
@@ -11081,6 +11140,7 @@ define inreg <6 x i16> @bitcast_v12i8_to_v6i16_scalar(<12 x i8> inreg %a, i32 in
; GFX11-NEXT: s_and_b32 s4, s4, exec_lo
; GFX11-NEXT: s_cselect_b32 s4, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s4, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB45_5
; GFX11-NEXT: ; %bb.4: ; %cmp.true
; GFX11-NEXT: v_mov_b32_e32 v0, 0xc0c0004
@@ -11376,6 +11436,7 @@ define <12 x i8> @bitcast_v6i16_to_v12i8(<6 x i16> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v1, 8, v11
; GFX11-TRUE16-NEXT: .LBB46_4: ; %end
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v0.l, v11.l
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v4.l, v12.l
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v11.l, v13.l
@@ -11431,6 +11492,7 @@ define <12 x i8> @bitcast_v6i16_to_v12i8(<6 x i16> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v1, 8, v13
; GFX11-FAKE16-NEXT: .LBB46_4: ; %end
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v0, v13
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v4, v14
; GFX11-FAKE16-NEXT: s_setpc_b64 s[30:31]
@@ -11707,6 +11769,7 @@ define inreg <12 x i8> @bitcast_v6i16_to_v12i8_scalar(<6 x i16> inreg %a, i32 in
; GFX11-NEXT: s_and_b32 s5, s8, exec_lo
; GFX11-NEXT: s_cselect_b32 s5, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s5, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB47_5
; GFX11-NEXT: ; %bb.4: ; %cmp.true
; GFX11-NEXT: v_pk_add_u16 v14, s1, 3 op_sel_hi:[1,0]
@@ -11964,8 +12027,9 @@ define <6 x half> @bitcast_v6bf16_to_v6f16(<6 x bfloat> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-TRUE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-TRUE16-NEXT: v_cmpx_ne_u32_e32 0, v3
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-TRUE16-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-TRUE16-NEXT: s_cbranch_execz .LBB48_2
; GFX11-TRUE16-NEXT: ; %bb.1: ; %cmp.true
@@ -12035,8 +12099,9 @@ define <6 x half> @bitcast_v6bf16_to_v6f16(<6 x bfloat> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-FAKE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-FAKE16-NEXT: v_cmpx_ne_u32_e32 0, v3
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-FAKE16-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-FAKE16-NEXT: s_cbranch_execz .LBB48_2
; GFX11-FAKE16-NEXT: ; %bb.1: ; %cmp.true
@@ -12361,6 +12426,7 @@ define inreg <6 x half> @bitcast_v6bf16_to_v6f16_scalar(<6 x bfloat> inreg %a, i
; GFX11-TRUE16-NEXT: s_and_b32 s4, s4, exec_lo
; GFX11-TRUE16-NEXT: s_cselect_b32 s4, 1, 0
; GFX11-TRUE16-NEXT: s_cmp_lg_u32 s4, 1
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_cbranch_scc1 .LBB49_4
; GFX11-TRUE16-NEXT: ; %bb.3: ; %cmp.true
; GFX11-TRUE16-NEXT: s_pack_lh_b32_b16 s3, 0, s0
@@ -12444,6 +12510,7 @@ define inreg <6 x half> @bitcast_v6bf16_to_v6f16_scalar(<6 x bfloat> inreg %a, i
; GFX11-FAKE16-NEXT: s_and_b32 s4, s4, exec_lo
; GFX11-FAKE16-NEXT: s_cselect_b32 s4, 1, 0
; GFX11-FAKE16-NEXT: s_cmp_lg_u32 s4, 1
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_cbranch_scc1 .LBB49_4
; GFX11-FAKE16-NEXT: ; %bb.3: ; %cmp.true
; GFX11-FAKE16-NEXT: s_pack_lh_b32_b16 s3, 0, s0
@@ -12651,8 +12718,9 @@ define <6 x bfloat> @bitcast_v6f16_to_v6bf16(<6 x half> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v3
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_pk_add_f16 v2, 0x200, v2 op_sel_hi:[0,1]
@@ -12838,6 +12906,7 @@ define inreg <6 x bfloat> @bitcast_v6f16_to_v6bf16_scalar(<6 x half> inreg %a, i
; GFX11-NEXT: s_and_b32 s4, s4, exec_lo
; GFX11-NEXT: s_cselect_b32 s4, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s4, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB51_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_pk_add_f16 v2, 0x200, s2 op_sel_hi:[0,1]
@@ -13073,8 +13142,9 @@ define <6 x i16> @bitcast_v6bf16_to_v6i16(<6 x bfloat> %a, i32 %b) #0 {
; GFX11-TRUE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-TRUE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-TRUE16-NEXT: v_cmpx_ne_u32_e32 0, v3
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-TRUE16-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-TRUE16-NEXT: s_cbranch_execz .LBB52_2
; GFX11-TRUE16-NEXT: ; %bb.1: ; %cmp.true
@@ -13151,8 +13221,9 @@ define <6 x i16> @bitcast_v6bf16_to_v6i16(<6 x bfloat> %a, i32 %b) #0 {
; GFX11-FAKE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-FAKE16-NEXT: s_mov_b32 s0, exec_lo
; GFX11-FAKE16-NEXT: v_cmpx_ne_u32_e32 0, v3
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-FAKE16-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-FAKE16-NEXT: s_cbranch_execz .LBB52_2
; GFX11-FAKE16-NEXT: ; %bb.1: ; %cmp.true
@@ -13475,6 +13546,7 @@ define inreg <6 x i16> @bitcast_v6bf16_to_v6i16_scalar(<6 x bfloat> inreg %a, i3
; GFX11-TRUE16-NEXT: s_and_b32 s4, s4, exec_lo
; GFX11-TRUE16-NEXT: s_cselect_b32 s4, 1, 0
; GFX11-TRUE16-NEXT: s_cmp_lg_u32 s4, 1
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_cbranch_scc1 .LBB53_4
; GFX11-TRUE16-NEXT: ; %bb.3: ; %cmp.true
; GFX11-TRUE16-NEXT: s_pack_lh_b32_b16 s3, 0, s0
@@ -13549,6 +13621,7 @@ define inreg <6 x i16> @bitcast_v6bf16_to_v6i16_scalar(<6 x bfloat> inreg %a, i3
; GFX11-FAKE16-NEXT: s_and_b32 s4, s4, exec_lo
; GFX11-FAKE16-NEXT: s_cselect_b32 s4, 1, 0
; GFX11-FAKE16-NEXT: s_cmp_lg_u32 s4, 1
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_cbranch_scc1 .LBB53_4
; GFX11-FAKE16-NEXT: ; %bb.3: ; %cmp.true
; GFX11-FAKE16-NEXT: s_pack_lh_b32_b16 s3, 0, s0
@@ -13734,8 +13807,9 @@ define <6 x bfloat> @bitcast_v6i16_to_v6bf16(<6 x i16> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v3
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_pk_add_u16 v2, v2, 3 op_sel_hi:[1,0]
@@ -13906,6 +13980,7 @@ define inreg <6 x bfloat> @bitcast_v6i16_to_v6bf16_scalar(<6 x i16> inreg %a, i3
; GFX11-NEXT: s_and_b32 s4, s4, exec_lo
; GFX11-NEXT: s_cselect_b32 s4, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s4, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB55_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_pk_add_u16 v2, s2, 3 op_sel_hi:[1,0]
@@ -14028,8 +14103,9 @@ define <6 x i16> @bitcast_v6f16_to_v6i16(<6 x half> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v3
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_pk_add_f16 v2, 0x200, v2 op_sel_hi:[0,1]
@@ -14201,6 +14277,7 @@ define inreg <6 x i16> @bitcast_v6f16_to_v6i16_scalar(<6 x half> inreg %a, i32 i
; GFX11-NEXT: s_and_b32 s4, s4, exec_lo
; GFX11-NEXT: s_cselect_b32 s4, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s4, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB57_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_pk_add_f16 v2, 0x200, s2 op_sel_hi:[0,1]
@@ -14334,8 +14411,9 @@ define <6 x half> @bitcast_v6i16_to_v6f16(<6 x i16> %a, i32 %b) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v3
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: ; %bb.1: ; %cmp.true
; GFX11-NEXT: v_pk_add_u16 v2, v2, 3 op_sel_hi:[1,0]
@@ -14505,6 +14583,7 @@ define inreg <6 x half> @bitcast_v6i16_to_v6f16_scalar(<6 x i16> inreg %a, i32 i
; GFX11-NEXT: s_and_b32 s4, s4, exec_lo
; GFX11-NEXT: s_cselect_b32 s4, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s4, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB59_4
; GFX11-NEXT: ; %bb.3: ; %cmp.true
; GFX11-NEXT: v_pk_add_u16 v2, s2, 3 op_sel_hi:[1,0]
diff --git a/llvm/test/CodeGen/AMDGPU/amdgpu-cs-chain-cc.ll b/llvm/test/CodeGen/AMDGPU/amdgpu-cs-chain-cc.ll
index 283c03d8df5bc0..ea72681840dab2 100644
--- a/llvm/test/CodeGen/AMDGPU/amdgpu-cs-chain-cc.ll
+++ b/llvm/test/CodeGen/AMDGPU/amdgpu-cs-chain-cc.ll
@@ -651,9 +651,10 @@ define amdgpu_cs_chain void @chain_to_chain_wwm(<3 x i32> inreg %a, <3 x i32> %b
; GISEL-GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GISEL-GFX11-NEXT: s_mov_b32 s3, s0
; GISEL-GFX11-NEXT: s_or_saveexec_b32 s0, -1
-; GISEL-GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
+; GISEL-GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GISEL-GFX11-NEXT: v_cndmask_b32_e64 v1, 4, 3, s0
; GISEL-GFX11-NEXT: s_mov_b32 exec_lo, s0
+; GISEL-GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GISEL-GFX11-NEXT: v_mov_b32_e32 v2, v1
; GISEL-GFX11-NEXT: ;;#ASMSTART
; GISEL-GFX11-NEXT: s_nop
@@ -690,7 +691,7 @@ define amdgpu_cs_chain void @chain_to_chain_wwm(<3 x i32> inreg %a, <3 x i32> %b
; DAGISEL-GFX11-NEXT: s_mov_b32 s3, s0
; DAGISEL-GFX11-NEXT: v_cndmask_b32_e64 v1, 4, 3, s4
; DAGISEL-GFX11-NEXT: s_mov_b32 exec_lo, s4
-; DAGISEL-GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; DAGISEL-GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; DAGISEL-GFX11-NEXT: v_mov_b32_e32 v2, v1
; DAGISEL-GFX11-NEXT: ;;#ASMSTART
; DAGISEL-GFX11-NEXT: s_nop
diff --git a/llvm/test/CodeGen/AMDGPU/amdgpu-nsa-threshold.ll b/llvm/test/CodeGen/AMDGPU/amdgpu-nsa-threshold.ll
index 5c5967111e9eb6..d68efe3d2e79bb 100644
--- a/llvm/test/CodeGen/AMDGPU/amdgpu-nsa-threshold.ll
+++ b/llvm/test/CodeGen/AMDGPU/amdgpu-nsa-threshold.ll
@@ -31,6 +31,7 @@ define amdgpu_ps <4 x float> @sample_2d_nsa2(<8 x i32> inreg %rsrc, <4 x i32> in
; FORCE-3: ; %bb.0: ; %main_body
; FORCE-3-NEXT: s_mov_b32 s12, exec_lo
; FORCE-3-NEXT: s_wqm_b32 exec_lo, exec_lo
+; FORCE-3-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; FORCE-3-NEXT: v_mov_b32_e32 v2, v0
; FORCE-3-NEXT: s_and_b32 exec_lo, exec_lo, s12
; FORCE-3-NEXT: image_sample v[0:3], v[1:2], s[0:7], s[8:11] dmask:0xf dim:SQ_RSRC_IMG_2D
@@ -41,6 +42,7 @@ define amdgpu_ps <4 x float> @sample_2d_nsa2(<8 x i32> inreg %rsrc, <4 x i32> in
; FORCE-4: ; %bb.0: ; %main_body
; FORCE-4-NEXT: s_mov_b32 s12, exec_lo
; FORCE-4-NEXT: s_wqm_b32 exec_lo, exec_lo
+; FORCE-4-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; FORCE-4-NEXT: v_mov_b32_e32 v2, v0
; FORCE-4-NEXT: s_and_b32 exec_lo, exec_lo, s12
; FORCE-4-NEXT: image_sample v[0:3], v[1:2], s[0:7], s[8:11] dmask:0xf dim:SQ_RSRC_IMG_2D
@@ -86,6 +88,7 @@ define amdgpu_ps <4 x float> @sample_3d_nsa2(<8 x i32> inreg %rsrc, <4 x i32> in
; FORCE-4: ; %bb.0: ; %main_body
; FORCE-4-NEXT: s_mov_b32 s12, exec_lo
; FORCE-4-NEXT: s_wqm_b32 exec_lo, exec_lo
+; FORCE-4-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; FORCE-4-NEXT: v_mov_b32_e32 v3, v0
; FORCE-4-NEXT: s_and_b32 exec_lo, exec_lo, s12
; FORCE-4-NEXT: image_sample v[0:3], v[1:3], s[0:7], s[8:11] dmask:0xf dim:SQ_RSRC_IMG_3D
@@ -101,6 +104,7 @@ define amdgpu_ps <4 x float> @sample_2d_nsa3(<8 x i32> inreg %rsrc, <4 x i32> in
; ATTRIB: ; %bb.0: ; %main_body
; ATTRIB-NEXT: s_mov_b32 s12, exec_lo
; ATTRIB-NEXT: s_wqm_b32 exec_lo, exec_lo
+; ATTRIB-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; ATTRIB-NEXT: v_mov_b32_e32 v2, v0
; ATTRIB-NEXT: s_and_b32 exec_lo, exec_lo, s12
; ATTRIB-NEXT: image_sample v[0:3], v[1:2], s[0:7], s[8:11] dmask:0xf dim:SQ_RSRC_IMG_2D
@@ -121,6 +125,7 @@ define amdgpu_ps <4 x float> @sample_2d_nsa3(<8 x i32> inreg %rsrc, <4 x i32> in
; FORCE-3: ; %bb.0: ; %main_body
; FORCE-3-NEXT: s_mov_b32 s12, exec_lo
; FORCE-3-NEXT: s_wqm_b32 exec_lo, exec_lo
+; FORCE-3-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; FORCE-3-NEXT: v_mov_b32_e32 v2, v0
; FORCE-3-NEXT: s_and_b32 exec_lo, exec_lo, s12
; FORCE-3-NEXT: image_sample v[0:3], v[1:2], s[0:7], s[8:11] dmask:0xf dim:SQ_RSRC_IMG_2D
@@ -131,6 +136,7 @@ define amdgpu_ps <4 x float> @sample_2d_nsa3(<8 x i32> inreg %rsrc, <4 x i32> in
; FORCE-4: ; %bb.0: ; %main_body
; FORCE-4-NEXT: s_mov_b32 s12, exec_lo
; FORCE-4-NEXT: s_wqm_b32 exec_lo, exec_lo
+; FORCE-4-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; FORCE-4-NEXT: v_mov_b32_e32 v2, v0
; FORCE-4-NEXT: s_and_b32 exec_lo, exec_lo, s12
; FORCE-4-NEXT: image_sample v[0:3], v[1:2], s[0:7], s[8:11] dmask:0xf dim:SQ_RSRC_IMG_2D
@@ -176,6 +182,7 @@ define amdgpu_ps <4 x float> @sample_3d_nsa3(<8 x i32> inreg %rsrc, <4 x i32> in
; FORCE-4: ; %bb.0: ; %main_body
; FORCE-4-NEXT: s_mov_b32 s12, exec_lo
; FORCE-4-NEXT: s_wqm_b32 exec_lo, exec_lo
+; FORCE-4-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; FORCE-4-NEXT: v_mov_b32_e32 v3, v0
; FORCE-4-NEXT: s_and_b32 exec_lo, exec_lo, s12
; FORCE-4-NEXT: image_sample v[0:3], v[1:3], s[0:7], s[8:11] dmask:0xf dim:SQ_RSRC_IMG_3D
@@ -191,6 +198,7 @@ define amdgpu_ps <4 x float> @sample_2d_nsa4(<8 x i32> inreg %rsrc, <4 x i32> in
; ATTRIB: ; %bb.0: ; %main_body
; ATTRIB-NEXT: s_mov_b32 s12, exec_lo
; ATTRIB-NEXT: s_wqm_b32 exec_lo, exec_lo
+; ATTRIB-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; ATTRIB-NEXT: v_mov_b32_e32 v2, v0
; ATTRIB-NEXT: s_and_b32 exec_lo, exec_lo, s12
; ATTRIB-NEXT: image_sample v[0:3], v[1:2], s[0:7], s[8:11] dmask:0xf dim:SQ_RSRC_IMG_2D
@@ -211,6 +219,7 @@ define amdgpu_ps <4 x float> @sample_2d_nsa4(<8 x i32> inreg %rsrc, <4 x i32> in
; FORCE-3: ; %bb.0: ; %main_body
; FORCE-3-NEXT: s_mov_b32 s12, exec_lo
; FORCE-3-NEXT: s_wqm_b32 exec_lo, exec_lo
+; FORCE-3-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; FORCE-3-NEXT: v_mov_b32_e32 v2, v0
; FORCE-3-NEXT: s_and_b32 exec_lo, exec_lo, s12
; FORCE-3-NEXT: image_sample v[0:3], v[1:2], s[0:7], s[8:11] dmask:0xf dim:SQ_RSRC_IMG_2D
@@ -221,6 +230,7 @@ define amdgpu_ps <4 x float> @sample_2d_nsa4(<8 x i32> inreg %rsrc, <4 x i32> in
; FORCE-4: ; %bb.0: ; %main_body
; FORCE-4-NEXT: s_mov_b32 s12, exec_lo
; FORCE-4-NEXT: s_wqm_b32 exec_lo, exec_lo
+; FORCE-4-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; FORCE-4-NEXT: v_mov_b32_e32 v2, v0
; FORCE-4-NEXT: s_and_b32 exec_lo, exec_lo, s12
; FORCE-4-NEXT: image_sample v[0:3], v[1:2], s[0:7], s[8:11] dmask:0xf dim:SQ_RSRC_IMG_2D
@@ -236,6 +246,7 @@ define amdgpu_ps <4 x float> @sample_3d_nsa4(<8 x i32> inreg %rsrc, <4 x i32> in
; ATTRIB: ; %bb.0: ; %main_body
; ATTRIB-NEXT: s_mov_b32 s12, exec_lo
; ATTRIB-NEXT: s_wqm_b32 exec_lo, exec_lo
+; ATTRIB-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; ATTRIB-NEXT: v_mov_b32_e32 v3, v0
; ATTRIB-NEXT: s_and_b32 exec_lo, exec_lo, s12
; ATTRIB-NEXT: image_sample v[0:3], v[1:3], s[0:7], s[8:11] dmask:0xf dim:SQ_RSRC_IMG_3D
@@ -266,6 +277,7 @@ define amdgpu_ps <4 x float> @sample_3d_nsa4(<8 x i32> inreg %rsrc, <4 x i32> in
; FORCE-4: ; %bb.0: ; %main_body
; FORCE-4-NEXT: s_mov_b32 s12, exec_lo
; FORCE-4-NEXT: s_wqm_b32 exec_lo, exec_lo
+; FORCE-4-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; FORCE-4-NEXT: v_mov_b32_e32 v3, v0
; FORCE-4-NEXT: s_and_b32 exec_lo, exec_lo, s12
; FORCE-4-NEXT: image_sample v[0:3], v[1:3], s[0:7], s[8:11] dmask:0xf dim:SQ_RSRC_IMG_3D
diff --git a/llvm/test/CodeGen/AMDGPU/arbitrary-fp-to-float-fp8-hw.ll b/llvm/test/CodeGen/AMDGPU/arbitrary-fp-to-float-fp8-hw.ll
index e289b558aecdf0..ae5dbd0ecd8a9b 100644
--- a/llvm/test/CodeGen/AMDGPU/arbitrary-fp-to-float-fp8-hw.ll
+++ b/llvm/test/CodeGen/AMDGPU/arbitrary-fp-to-float-fp8-hw.ll
@@ -683,18 +683,17 @@ define <4 x float> @v4_from_e5m3fnu(<4 x i8> %x) {
; GFX1170-NEXT: v_cndmask_b32_e64 v1, v5, v1, s1
; GFX1170-NEXT: v_or_b32_e32 v5, v3, v7
; GFX1170-NEXT: v_cmp_ne_u32_e64 s1, 0, v10
-; GFX1170-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX1170-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX1170-NEXT: v_cmp_ne_u32_e64 s3, 0, v5
; GFX1170-NEXT: v_cndmask_b32_e64 v4, 0, v4, s1
; GFX1170-NEXT: v_cmp_eq_u32_e64 s1, 7, v6
-; GFX1170-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_3) | instid1(SALU_CYCLE_1)
; GFX1170-NEXT: v_cndmask_b32_e64 v5, 0, v1, s3
; GFX1170-NEXT: v_cmp_eq_u32_e64 s3, 7, v7
; GFX1170-NEXT: v_cndmask_b32_e64 v1, v8, 0x7fc00000, s0
; GFX1170-NEXT: s_and_b32 s0, s2, s1
+; GFX1170-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX1170-NEXT: v_cndmask_b32_e64 v2, v4, 0x7fc00000, s0
; GFX1170-NEXT: s_and_b32 s0, s4, s3
-; GFX1170-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1170-NEXT: v_cndmask_b32_e64 v3, v5, 0x7fc00000, s0
; GFX1170-NEXT: s_setpc_b64 s[30:31]
;
diff --git a/llvm/test/CodeGen/AMDGPU/asyncmark-gfx12plus.ll b/llvm/test/CodeGen/AMDGPU/asyncmark-gfx12plus.ll
index d6089f85b8a4be..c331a7496223a3 100644
--- a/llvm/test/CodeGen/AMDGPU/asyncmark-gfx12plus.ll
+++ b/llvm/test/CodeGen/AMDGPU/asyncmark-gfx12plus.ll
@@ -61,7 +61,6 @@ define void @interleaved_with_wave_barrier(ptr addrspace(1) %foo, ptr addrspace(
; GISEL-NEXT: v_add_co_u32 v6, vcc_lo, 0x58, v8
; GISEL-NEXT: ; wave barrier
; GISEL-NEXT: ; asyncmark
-; GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GISEL-NEXT: v_add_co_ci_u32_e64 v7, null, 0, v9, vcc_lo
; GISEL-NEXT: v_add_nc_u32_e32 v3, 0x58, v2
; GISEL-NEXT: global_load_b32 v0, v[0:1], off offset:8
@@ -231,6 +230,7 @@ define amdgpu_kernel void @test_pipelined_loop(ptr addrspace(1) %foo, ptr addrsp
; GISEL-NEXT: s_add_co_ci_u32 s1, s1, 0
; GISEL-NEXT: s_add_co_u32 s8, s8, 4
; GISEL-NEXT: s_cmp_lt_i32 s7, s3
+; GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GISEL-NEXT: s_cbranch_scc1 .LBB1_1
; GISEL-NEXT: ; %bb.2: ; %epilog
; GISEL-NEXT: s_lshl_b32 s0, s3, 2
diff --git a/llvm/test/CodeGen/AMDGPU/atomic_optimizations_buffer.ll b/llvm/test/CodeGen/AMDGPU/atomic_optimizations_buffer.ll
index ebec9537f3f1ad..a2aa62509ba0b3 100644
--- a/llvm/test/CodeGen/AMDGPU/atomic_optimizations_buffer.ll
+++ b/llvm/test/CodeGen/AMDGPU/atomic_optimizations_buffer.ll
@@ -166,6 +166,7 @@ define amdgpu_kernel void @add_i32_constant(ptr addrspace(1) %out, ptr addrspace
; GFX11W64-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11W64-NEXT: v_mbcnt_hi_u32_b32 v0, s3, v0
; GFX11W64-NEXT: v_cmpx_eq_u32_e32 0, v0
+; GFX11W64-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11W64-NEXT: s_cbranch_execz .LBB0_2
; GFX11W64-NEXT: ; %bb.1:
; GFX11W64-NEXT: s_load_b128 s[8:11], s[4:5], 0x34
@@ -193,7 +194,7 @@ define amdgpu_kernel void @add_i32_constant(ptr addrspace(1) %out, ptr addrspace
; GFX11W32-NEXT: s_mov_b32 s0, exec_lo
; GFX11W32-NEXT: v_mbcnt_lo_u32_b32 v0, s1, 0
; GFX11W32-NEXT: ; implicit-def: $vgpr1
-; GFX11W32-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX11W32-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11W32-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX11W32-NEXT: s_cbranch_execz .LBB0_2
; GFX11W32-NEXT: ; %bb.1:
@@ -225,6 +226,7 @@ define amdgpu_kernel void @add_i32_constant(ptr addrspace(1) %out, ptr addrspace
; GFX12W64-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX12W64-NEXT: v_mbcnt_hi_u32_b32 v0, s3, v0
; GFX12W64-NEXT: v_cmpx_eq_u32_e32 0, v0
+; GFX12W64-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12W64-NEXT: s_cbranch_execz .LBB0_2
; GFX12W64-NEXT: ; %bb.1:
; GFX12W64-NEXT: s_load_b128 s[8:11], s[4:5], 0x34
@@ -254,7 +256,7 @@ define amdgpu_kernel void @add_i32_constant(ptr addrspace(1) %out, ptr addrspace
; GFX12W32-NEXT: s_mov_b32 s0, exec_lo
; GFX12W32-NEXT: v_mbcnt_lo_u32_b32 v0, s1, 0
; GFX12W32-NEXT: ; implicit-def: $vgpr1
-; GFX12W32-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX12W32-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX12W32-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX12W32-NEXT: s_cbranch_execz .LBB0_2
; GFX12W32-NEXT: ; %bb.1:
@@ -287,6 +289,7 @@ define amdgpu_kernel void @add_i32_constant(ptr addrspace(1) %out, ptr addrspace
; GFX13W64-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX13W64-NEXT: v_mbcnt_hi_u32_b32 v0, s3, v0
; GFX13W64-NEXT: v_cmpx_eq_u32_e32 0, v0
+; GFX13W64-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX13W64-NEXT: s_cbranch_execz .LBB0_2
; GFX13W64-NEXT: ; %bb.1:
; GFX13W64-NEXT: s_load_b128 s[8:11], s[4:5], 0x34 nv
@@ -314,7 +317,7 @@ define amdgpu_kernel void @add_i32_constant(ptr addrspace(1) %out, ptr addrspace
; GFX13W32-NEXT: s_mov_b32 s0, exec_lo
; GFX13W32-NEXT: v_mbcnt_lo_u32_b32 v0, s1, 0
; GFX13W32-NEXT: ; implicit-def: $vgpr1
-; GFX13W32-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX13W32-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX13W32-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX13W32-NEXT: s_cbranch_execz .LBB0_2
; GFX13W32-NEXT: ; %bb.1:
@@ -500,6 +503,7 @@ define amdgpu_kernel void @add_i32_uniform(ptr addrspace(1) %out, ptr addrspace(
; GFX11W64-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11W64-NEXT: v_mbcnt_hi_u32_b32 v0, s3, v0
; GFX11W64-NEXT: v_cmpx_eq_u32_e32 0, v0
+; GFX11W64-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11W64-NEXT: s_cbranch_execz .LBB1_2
; GFX11W64-NEXT: ; %bb.1:
; GFX11W64-NEXT: s_load_b128 s[8:11], s[4:5], 0x34
@@ -528,7 +532,7 @@ define amdgpu_kernel void @add_i32_uniform(ptr addrspace(1) %out, ptr addrspace(
; GFX11W32-NEXT: s_mov_b32 s1, exec_lo
; GFX11W32-NEXT: v_mbcnt_lo_u32_b32 v0, s2, 0
; GFX11W32-NEXT: ; implicit-def: $vgpr1
-; GFX11W32-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX11W32-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11W32-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX11W32-NEXT: s_cbranch_execz .LBB1_2
; GFX11W32-NEXT: ; %bb.1:
@@ -561,6 +565,7 @@ define amdgpu_kernel void @add_i32_uniform(ptr addrspace(1) %out, ptr addrspace(
; GFX12W64-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX12W64-NEXT: v_mbcnt_hi_u32_b32 v0, s3, v0
; GFX12W64-NEXT: v_cmpx_eq_u32_e32 0, v0
+; GFX12W64-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12W64-NEXT: s_cbranch_execz .LBB1_2
; GFX12W64-NEXT: ; %bb.1:
; GFX12W64-NEXT: s_load_b128 s[8:11], s[4:5], 0x34
@@ -591,7 +596,7 @@ define amdgpu_kernel void @add_i32_uniform(ptr addrspace(1) %out, ptr addrspace(
; GFX12W32-NEXT: s_mov_b32 s1, exec_lo
; GFX12W32-NEXT: v_mbcnt_lo_u32_b32 v0, s2, 0
; GFX12W32-NEXT: ; implicit-def: $vgpr1
-; GFX12W32-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX12W32-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX12W32-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX12W32-NEXT: s_cbranch_execz .LBB1_2
; GFX12W32-NEXT: ; %bb.1:
@@ -625,6 +630,7 @@ define amdgpu_kernel void @add_i32_uniform(ptr addrspace(1) %out, ptr addrspace(
; GFX13W64-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX13W64-NEXT: v_mbcnt_hi_u32_b32 v0, s3, v0
; GFX13W64-NEXT: v_cmpx_eq_u32_e32 0, v0
+; GFX13W64-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX13W64-NEXT: s_cbranch_execz .LBB1_2
; GFX13W64-NEXT: ; %bb.1:
; GFX13W64-NEXT: s_load_b128 s[8:11], s[4:5], 0x34 nv
@@ -653,7 +659,7 @@ define amdgpu_kernel void @add_i32_uniform(ptr addrspace(1) %out, ptr addrspace(
; GFX13W32-NEXT: s_mov_b32 s1, exec_lo
; GFX13W32-NEXT: v_mbcnt_lo_u32_b32 v0, s2, 0
; GFX13W32-NEXT: ; implicit-def: $vgpr1
-; GFX13W32-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX13W32-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX13W32-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX13W32-NEXT: s_cbranch_execz .LBB1_2
; GFX13W32-NEXT: ; %bb.1:
@@ -903,8 +909,8 @@ define amdgpu_kernel void @add_i32_varying_vdata(ptr addrspace(1) %out, ptr addr
; GFX11W64-NEXT: v_mbcnt_hi_u32_b32 v1, exec_hi, v1
; GFX11W64-NEXT: v_cmp_eq_u32_e32 vcc, 0, v1
; GFX11W64-NEXT: ; implicit-def: $vgpr1
+; GFX11W64-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11W64-NEXT: s_and_saveexec_b64 s[0:1], vcc
-; GFX11W64-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11W64-NEXT: s_xor_b64 s[0:1], exec, s[0:1]
; GFX11W64-NEXT: s_cbranch_execz .LBB2_4
; GFX11W64-NEXT: ; %bb.3:
@@ -986,8 +992,8 @@ define amdgpu_kernel void @add_i32_varying_vdata(ptr addrspace(1) %out, ptr addr
; GFX12W64-NEXT: v_mbcnt_hi_u32_b32 v1, exec_hi, v1
; GFX12W64-NEXT: v_cmp_eq_u32_e32 vcc, 0, v1
; GFX12W64-NEXT: ; implicit-def: $vgpr1
+; GFX12W64-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12W64-NEXT: s_and_saveexec_b64 s[0:1], vcc
-; GFX12W64-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12W64-NEXT: s_xor_b64 s[0:1], exec, s[0:1]
; GFX12W64-NEXT: s_cbranch_execz .LBB2_4
; GFX12W64-NEXT: ; %bb.3:
@@ -1075,8 +1081,8 @@ define amdgpu_kernel void @add_i32_varying_vdata(ptr addrspace(1) %out, ptr addr
; GFX13W64-NEXT: v_mbcnt_hi_u32_b32 v1, exec_hi, v1
; GFX13W64-NEXT: v_cmp_eq_u32_e32 vcc, 0, v1
; GFX13W64-NEXT: ; implicit-def: $vgpr1
+; GFX13W64-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX13W64-NEXT: s_and_saveexec_b64 s[0:1], vcc
-; GFX13W64-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX13W64-NEXT: s_xor_b64 s[0:1], exec, s[0:1]
; GFX13W64-NEXT: s_cbranch_execz .LBB2_4
; GFX13W64-NEXT: ; %bb.3:
@@ -1376,8 +1382,8 @@ define amdgpu_kernel void @struct_add_i32_varying_vdata(ptr addrspace(1) %out, p
; GFX11W64-NEXT: v_mbcnt_hi_u32_b32 v1, exec_hi, v1
; GFX11W64-NEXT: v_cmp_eq_u32_e32 vcc, 0, v1
; GFX11W64-NEXT: ; implicit-def: $vgpr1
+; GFX11W64-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11W64-NEXT: s_and_saveexec_b64 s[0:1], vcc
-; GFX11W64-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11W64-NEXT: s_xor_b64 s[0:1], exec, s[0:1]
; GFX11W64-NEXT: s_cbranch_execz .LBB3_4
; GFX11W64-NEXT: ; %bb.3:
@@ -1464,8 +1470,8 @@ define amdgpu_kernel void @struct_add_i32_varying_vdata(ptr addrspace(1) %out, p
; GFX12W64-NEXT: v_mbcnt_hi_u32_b32 v1, exec_hi, v1
; GFX12W64-NEXT: v_cmp_eq_u32_e32 vcc, 0, v1
; GFX12W64-NEXT: ; implicit-def: $vgpr1
+; GFX12W64-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12W64-NEXT: s_and_saveexec_b64 s[0:1], vcc
-; GFX12W64-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12W64-NEXT: s_xor_b64 s[0:1], exec, s[0:1]
; GFX12W64-NEXT: s_cbranch_execz .LBB3_4
; GFX12W64-NEXT: ; %bb.3:
@@ -1558,8 +1564,8 @@ define amdgpu_kernel void @struct_add_i32_varying_vdata(ptr addrspace(1) %out, p
; GFX13W64-NEXT: v_mbcnt_hi_u32_b32 v1, exec_hi, v1
; GFX13W64-NEXT: v_cmp_eq_u32_e32 vcc, 0, v1
; GFX13W64-NEXT: ; implicit-def: $vgpr1
+; GFX13W64-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX13W64-NEXT: s_and_saveexec_b64 s[0:1], vcc
-; GFX13W64-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX13W64-NEXT: s_xor_b64 s[0:1], exec, s[0:1]
; GFX13W64-NEXT: s_cbranch_execz .LBB3_4
; GFX13W64-NEXT: ; %bb.3:
@@ -1908,6 +1914,7 @@ define amdgpu_kernel void @sub_i32_constant(ptr addrspace(1) %out, ptr addrspace
; GFX11W64-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11W64-NEXT: v_mbcnt_hi_u32_b32 v0, s3, v0
; GFX11W64-NEXT: v_cmpx_eq_u32_e32 0, v0
+; GFX11W64-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11W64-NEXT: s_cbranch_execz .LBB5_2
; GFX11W64-NEXT: ; %bb.1:
; GFX11W64-NEXT: s_load_b128 s[8:11], s[4:5], 0x34
@@ -1936,7 +1943,7 @@ define amdgpu_kernel void @sub_i32_constant(ptr addrspace(1) %out, ptr addrspace
; GFX11W32-NEXT: s_mov_b32 s0, exec_lo
; GFX11W32-NEXT: v_mbcnt_lo_u32_b32 v0, s1, 0
; GFX11W32-NEXT: ; implicit-def: $vgpr1
-; GFX11W32-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX11W32-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11W32-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX11W32-NEXT: s_cbranch_execz .LBB5_2
; GFX11W32-NEXT: ; %bb.1:
@@ -1969,6 +1976,7 @@ define amdgpu_kernel void @sub_i32_constant(ptr addrspace(1) %out, ptr addrspace
; GFX12W64-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX12W64-NEXT: v_mbcnt_hi_u32_b32 v0, s3, v0
; GFX12W64-NEXT: v_cmpx_eq_u32_e32 0, v0
+; GFX12W64-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12W64-NEXT: s_cbranch_execz .LBB5_2
; GFX12W64-NEXT: ; %bb.1:
; GFX12W64-NEXT: s_load_b128 s[8:11], s[4:5], 0x34
@@ -1999,7 +2007,7 @@ define amdgpu_kernel void @sub_i32_constant(ptr addrspace(1) %out, ptr addrspace
; GFX12W32-NEXT: s_mov_b32 s0, exec_lo
; GFX12W32-NEXT: v_mbcnt_lo_u32_b32 v0, s1, 0
; GFX12W32-NEXT: ; implicit-def: $vgpr1
-; GFX12W32-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX12W32-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX12W32-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX12W32-NEXT: s_cbranch_execz .LBB5_2
; GFX12W32-NEXT: ; %bb.1:
@@ -2033,6 +2041,7 @@ define amdgpu_kernel void @sub_i32_constant(ptr addrspace(1) %out, ptr addrspace
; GFX13W64-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX13W64-NEXT: v_mbcnt_hi_u32_b32 v0, s3, v0
; GFX13W64-NEXT: v_cmpx_eq_u32_e32 0, v0
+; GFX13W64-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX13W64-NEXT: s_cbranch_execz .LBB5_2
; GFX13W64-NEXT: ; %bb.1:
; GFX13W64-NEXT: s_load_b128 s[8:11], s[4:5], 0x34 nv
@@ -2061,7 +2070,7 @@ define amdgpu_kernel void @sub_i32_constant(ptr addrspace(1) %out, ptr addrspace
; GFX13W32-NEXT: s_mov_b32 s0, exec_lo
; GFX13W32-NEXT: v_mbcnt_lo_u32_b32 v0, s1, 0
; GFX13W32-NEXT: ; implicit-def: $vgpr1
-; GFX13W32-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX13W32-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX13W32-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX13W32-NEXT: s_cbranch_execz .LBB5_2
; GFX13W32-NEXT: ; %bb.1:
@@ -2248,6 +2257,7 @@ define amdgpu_kernel void @sub_i32_uniform(ptr addrspace(1) %out, ptr addrspace(
; GFX11W64-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11W64-NEXT: v_mbcnt_hi_u32_b32 v0, s3, v0
; GFX11W64-NEXT: v_cmpx_eq_u32_e32 0, v0
+; GFX11W64-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11W64-NEXT: s_cbranch_execz .LBB6_2
; GFX11W64-NEXT: ; %bb.1:
; GFX11W64-NEXT: s_load_b128 s[8:11], s[4:5], 0x34
@@ -2277,7 +2287,7 @@ define amdgpu_kernel void @sub_i32_uniform(ptr addrspace(1) %out, ptr addrspace(
; GFX11W32-NEXT: s_mov_b32 s1, exec_lo
; GFX11W32-NEXT: v_mbcnt_lo_u32_b32 v0, s2, 0
; GFX11W32-NEXT: ; implicit-def: $vgpr1
-; GFX11W32-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX11W32-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11W32-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX11W32-NEXT: s_cbranch_execz .LBB6_2
; GFX11W32-NEXT: ; %bb.1:
@@ -2311,6 +2321,7 @@ define amdgpu_kernel void @sub_i32_uniform(ptr addrspace(1) %out, ptr addrspace(
; GFX12W64-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX12W64-NEXT: v_mbcnt_hi_u32_b32 v0, s3, v0
; GFX12W64-NEXT: v_cmpx_eq_u32_e32 0, v0
+; GFX12W64-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12W64-NEXT: s_cbranch_execz .LBB6_2
; GFX12W64-NEXT: ; %bb.1:
; GFX12W64-NEXT: s_load_b128 s[8:11], s[4:5], 0x34
@@ -2342,7 +2353,7 @@ define amdgpu_kernel void @sub_i32_uniform(ptr addrspace(1) %out, ptr addrspace(
; GFX12W32-NEXT: s_mov_b32 s1, exec_lo
; GFX12W32-NEXT: v_mbcnt_lo_u32_b32 v0, s2, 0
; GFX12W32-NEXT: ; implicit-def: $vgpr1
-; GFX12W32-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX12W32-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX12W32-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX12W32-NEXT: s_cbranch_execz .LBB6_2
; GFX12W32-NEXT: ; %bb.1:
@@ -2378,6 +2389,7 @@ define amdgpu_kernel void @sub_i32_uniform(ptr addrspace(1) %out, ptr addrspace(
; GFX13W64-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX13W64-NEXT: v_mbcnt_hi_u32_b32 v0, s3, v0
; GFX13W64-NEXT: v_cmpx_eq_u32_e32 0, v0
+; GFX13W64-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX13W64-NEXT: s_cbranch_execz .LBB6_2
; GFX13W64-NEXT: ; %bb.1:
; GFX13W64-NEXT: s_load_b128 s[8:11], s[4:5], 0x34 nv
@@ -2407,7 +2419,7 @@ define amdgpu_kernel void @sub_i32_uniform(ptr addrspace(1) %out, ptr addrspace(
; GFX13W32-NEXT: s_mov_b32 s1, exec_lo
; GFX13W32-NEXT: v_mbcnt_lo_u32_b32 v0, s2, 0
; GFX13W32-NEXT: ; implicit-def: $vgpr1
-; GFX13W32-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX13W32-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX13W32-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX13W32-NEXT: s_cbranch_execz .LBB6_2
; GFX13W32-NEXT: ; %bb.1:
@@ -2657,8 +2669,8 @@ define amdgpu_kernel void @sub_i32_varying_vdata(ptr addrspace(1) %out, ptr addr
; GFX11W64-NEXT: v_mbcnt_hi_u32_b32 v1, exec_hi, v1
; GFX11W64-NEXT: v_cmp_eq_u32_e32 vcc, 0, v1
; GFX11W64-NEXT: ; implicit-def: $vgpr1
+; GFX11W64-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11W64-NEXT: s_and_saveexec_b64 s[0:1], vcc
-; GFX11W64-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11W64-NEXT: s_xor_b64 s[0:1], exec, s[0:1]
; GFX11W64-NEXT: s_cbranch_execz .LBB7_4
; GFX11W64-NEXT: ; %bb.3:
@@ -2741,8 +2753,8 @@ define amdgpu_kernel void @sub_i32_varying_vdata(ptr addrspace(1) %out, ptr addr
; GFX12W64-NEXT: v_mbcnt_hi_u32_b32 v1, exec_hi, v1
; GFX12W64-NEXT: v_cmp_eq_u32_e32 vcc, 0, v1
; GFX12W64-NEXT: ; implicit-def: $vgpr1
+; GFX12W64-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12W64-NEXT: s_and_saveexec_b64 s[0:1], vcc
-; GFX12W64-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12W64-NEXT: s_xor_b64 s[0:1], exec, s[0:1]
; GFX12W64-NEXT: s_cbranch_execz .LBB7_4
; GFX12W64-NEXT: ; %bb.3:
@@ -2831,8 +2843,8 @@ define amdgpu_kernel void @sub_i32_varying_vdata(ptr addrspace(1) %out, ptr addr
; GFX13W64-NEXT: v_mbcnt_hi_u32_b32 v1, exec_hi, v1
; GFX13W64-NEXT: v_cmp_eq_u32_e32 vcc, 0, v1
; GFX13W64-NEXT: ; implicit-def: $vgpr1
+; GFX13W64-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX13W64-NEXT: s_and_saveexec_b64 s[0:1], vcc
-; GFX13W64-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX13W64-NEXT: s_xor_b64 s[0:1], exec, s[0:1]
; GFX13W64-NEXT: s_cbranch_execz .LBB7_4
; GFX13W64-NEXT: ; %bb.3:
diff --git a/llvm/test/CodeGen/AMDGPU/atomic_optimizations_dpp_lds_expected_active_lanes.ll b/llvm/test/CodeGen/AMDGPU/atomic_optimizations_dpp_lds_expected_active_lanes.ll
index dc9e0d51787465..5963756dd5263e 100644
--- a/llvm/test/CodeGen/AMDGPU/atomic_optimizations_dpp_lds_expected_active_lanes.ll
+++ b/llvm/test/CodeGen/AMDGPU/atomic_optimizations_dpp_lds_expected_active_lanes.ll
@@ -282,6 +282,7 @@ define amdgpu_kernel void @add_i32_high_active_lanes(ptr addrspace(1) %out) {
; GFX1132-NEXT: s_or_saveexec_b32 s0, -1
; GFX1132-NEXT: v_writelane_b32 v3, s1, 16
; GFX1132-NEXT: s_mov_b32 exec_lo, s0
+; GFX1132-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1132-NEXT: v_cmp_eq_u32_e32 vcc_lo, 0, v4
; GFX1132-NEXT: ; implicit-def: $vgpr4
; GFX1132-NEXT: s_and_saveexec_b32 s0, vcc_lo
@@ -336,7 +337,7 @@ define amdgpu_kernel void @add_i32_high_active_lanes(ptr addrspace(1) %out) {
; GFX1164-NEXT: v_writelane_b32 v3, s3, 32
; GFX1164-NEXT: s_waitcnt_depctr depctr_sa_sdst(0)
; GFX1164-NEXT: s_mov_b64 exec, s[0:1]
-; GFX1164-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1164-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_4) | instid1(SALU_CYCLE_1)
; GFX1164-NEXT: v_mbcnt_hi_u32_b32 v4, exec_hi, v0
; GFX1164-NEXT: v_mov_b32_e32 v0, 0
; GFX1164-NEXT: s_or_saveexec_b64 s[0:1], -1
@@ -344,6 +345,7 @@ define amdgpu_kernel void @add_i32_high_active_lanes(ptr addrspace(1) %out) {
; GFX1164-NEXT: s_mov_b64 exec, s[0:1]
; GFX1164-NEXT: v_cmp_eq_u32_e32 vcc, 0, v4
; GFX1164-NEXT: ; implicit-def: $vgpr4
+; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1164-NEXT: s_and_saveexec_b64 s[0:1], vcc
; GFX1164-NEXT: s_cbranch_execz .LBB2_2
; GFX1164-NEXT: ; %bb.1:
@@ -389,6 +391,7 @@ define amdgpu_kernel void @add_i32_high_active_lanes(ptr addrspace(1) %out) {
; GFX1232-NEXT: v_writelane_b32 v3, s1, 16
; GFX1232-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX1232-NEXT: s_mov_b32 exec_lo, s0
+; GFX1232-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1232-NEXT: v_cmp_eq_u32_e32 vcc_lo, 0, v4
; GFX1232-NEXT: ; implicit-def: $vgpr4
; GFX1232-NEXT: s_and_saveexec_b32 s0, vcc_lo
@@ -453,6 +456,7 @@ define amdgpu_kernel void @add_i32_high_active_lanes(ptr addrspace(1) %out) {
; GFX1264-NEXT: v_writelane_b32 v3, s6, 48
; GFX1264-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX1264-NEXT: s_mov_b64 exec, s[0:1]
+; GFX1264-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1264-NEXT: v_cmp_eq_u32_e32 vcc, 0, v4
; GFX1264-NEXT: ; implicit-def: $vgpr4
; GFX1264-NEXT: s_and_saveexec_b64 s[0:1], vcc
@@ -599,6 +603,7 @@ define amdgpu_kernel void @add_i32_no_metadata(ptr addrspace(1) %out) {
; GFX1132-NEXT: s_or_saveexec_b32 s0, -1
; GFX1132-NEXT: v_writelane_b32 v3, s1, 16
; GFX1132-NEXT: s_mov_b32 exec_lo, s0
+; GFX1132-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1132-NEXT: v_cmp_eq_u32_e32 vcc_lo, 0, v4
; GFX1132-NEXT: ; implicit-def: $vgpr4
; GFX1132-NEXT: s_and_saveexec_b32 s0, vcc_lo
@@ -653,7 +658,7 @@ define amdgpu_kernel void @add_i32_no_metadata(ptr addrspace(1) %out) {
; GFX1164-NEXT: v_writelane_b32 v3, s3, 32
; GFX1164-NEXT: s_waitcnt_depctr depctr_sa_sdst(0)
; GFX1164-NEXT: s_mov_b64 exec, s[0:1]
-; GFX1164-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1164-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_4) | instid1(SALU_CYCLE_1)
; GFX1164-NEXT: v_mbcnt_hi_u32_b32 v4, exec_hi, v0
; GFX1164-NEXT: v_mov_b32_e32 v0, 0
; GFX1164-NEXT: s_or_saveexec_b64 s[0:1], -1
@@ -661,6 +666,7 @@ define amdgpu_kernel void @add_i32_no_metadata(ptr addrspace(1) %out) {
; GFX1164-NEXT: s_mov_b64 exec, s[0:1]
; GFX1164-NEXT: v_cmp_eq_u32_e32 vcc, 0, v4
; GFX1164-NEXT: ; implicit-def: $vgpr4
+; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1164-NEXT: s_and_saveexec_b64 s[0:1], vcc
; GFX1164-NEXT: s_cbranch_execz .LBB3_2
; GFX1164-NEXT: ; %bb.1:
@@ -706,6 +712,7 @@ define amdgpu_kernel void @add_i32_no_metadata(ptr addrspace(1) %out) {
; GFX1232-NEXT: v_writelane_b32 v3, s1, 16
; GFX1232-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX1232-NEXT: s_mov_b32 exec_lo, s0
+; GFX1232-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1232-NEXT: v_cmp_eq_u32_e32 vcc_lo, 0, v4
; GFX1232-NEXT: ; implicit-def: $vgpr4
; GFX1232-NEXT: s_and_saveexec_b32 s0, vcc_lo
@@ -770,6 +777,7 @@ define amdgpu_kernel void @add_i32_no_metadata(ptr addrspace(1) %out) {
; GFX1264-NEXT: v_writelane_b32 v3, s6, 48
; GFX1264-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX1264-NEXT: s_mov_b64 exec, s[0:1]
+; GFX1264-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1264-NEXT: v_cmp_eq_u32_e32 vcc, 0, v4
; GFX1264-NEXT: ; implicit-def: $vgpr4
; GFX1264-NEXT: s_and_saveexec_b64 s[0:1], vcc
@@ -923,7 +931,7 @@ define amdgpu_kernel void @add_i32_global_ignores_metadata(ptr addrspace(1) %ino
; GFX1132-NEXT: v_mov_b32_dpp v3, v1 row_shr:1 row_mask:0xf bank_mask:0xf
; GFX1132-NEXT: v_readlane_b32 s3, v1, 15
; GFX1132-NEXT: s_mov_b32 exec_lo, s2
-; GFX1132-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1132-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_4) | instid1(SALU_CYCLE_1)
; GFX1132-NEXT: v_mbcnt_lo_u32_b32 v4, exec_lo, 0
; GFX1132-NEXT: v_mov_b32_e32 v0, 0
; GFX1132-NEXT: s_or_saveexec_b32 s2, -1
@@ -942,9 +950,9 @@ define amdgpu_kernel void @add_i32_global_ignores_metadata(ptr addrspace(1) %ino
; GFX1132-NEXT: buffer_gl0_inv
; GFX1132-NEXT: .LBB4_2:
; GFX1132-NEXT: s_or_b32 exec_lo, exec_lo, s2
+; GFX1132-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX1132-NEXT: v_readfirstlane_b32 s2, v4
; GFX1132-NEXT: v_mov_b32_e32 v4, v3
-; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1132-NEXT: v_add_nc_u32_e32 v4, s2, v4
; GFX1132-NEXT: s_waitcnt lgkmcnt(0)
; GFX1132-NEXT: global_store_b32 v0, v4, s[0:1]
@@ -986,7 +994,7 @@ define amdgpu_kernel void @add_i32_global_ignores_metadata(ptr addrspace(1) %ino
; GFX1164-NEXT: v_readlane_b32 s5, v1, 47
; GFX1164-NEXT: v_writelane_b32 v3, s4, 32
; GFX1164-NEXT: s_mov_b64 exec, s[2:3]
-; GFX1164-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1164-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_4) | instid1(SALU_CYCLE_1)
; GFX1164-NEXT: v_mbcnt_hi_u32_b32 v4, exec_hi, v0
; GFX1164-NEXT: v_mov_b32_e32 v0, 0
; GFX1164-NEXT: s_or_saveexec_b64 s[2:3], -1
@@ -994,6 +1002,7 @@ define amdgpu_kernel void @add_i32_global_ignores_metadata(ptr addrspace(1) %ino
; GFX1164-NEXT: s_mov_b64 exec, s[2:3]
; GFX1164-NEXT: v_cmp_eq_u32_e32 vcc, 0, v4
; GFX1164-NEXT: ; implicit-def: $vgpr4
+; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1164-NEXT: s_and_saveexec_b64 s[2:3], vcc
; GFX1164-NEXT: s_cbranch_execz .LBB4_2
; GFX1164-NEXT: ; %bb.1:
@@ -1005,9 +1014,9 @@ define amdgpu_kernel void @add_i32_global_ignores_metadata(ptr addrspace(1) %ino
; GFX1164-NEXT: buffer_gl0_inv
; GFX1164-NEXT: .LBB4_2:
; GFX1164-NEXT: s_or_b64 exec, exec, s[2:3]
+; GFX1164-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX1164-NEXT: v_readfirstlane_b32 s2, v4
; GFX1164-NEXT: v_mov_b32_e32 v4, v3
-; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1164-NEXT: v_add_nc_u32_e32 v4, s2, v4
; GFX1164-NEXT: s_waitcnt lgkmcnt(0)
; GFX1164-NEXT: global_store_b32 v0, v4, s[0:1]
@@ -1037,7 +1046,7 @@ define amdgpu_kernel void @add_i32_global_ignores_metadata(ptr addrspace(1) %ino
; GFX1232-NEXT: v_mov_b32_dpp v3, v1 row_shr:1 row_mask:0xf bank_mask:0xf
; GFX1232-NEXT: v_readlane_b32 s3, v1, 15
; GFX1232-NEXT: s_mov_b32 exec_lo, s2
-; GFX1232-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1232-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_4) | instid1(SALU_CYCLE_1)
; GFX1232-NEXT: v_mbcnt_lo_u32_b32 v4, exec_lo, 0
; GFX1232-NEXT: v_mov_b32_e32 v0, 0
; GFX1232-NEXT: s_or_saveexec_b32 s2, -1
@@ -1058,10 +1067,10 @@ define amdgpu_kernel void @add_i32_global_ignores_metadata(ptr addrspace(1) %ino
; GFX1232-NEXT: .LBB4_2:
; GFX1232-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX1232-NEXT: s_or_b32 exec_lo, exec_lo, s2
+; GFX1232-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(VALU_DEP_1)
; GFX1232-NEXT: v_readfirstlane_b32 s2, v4
; GFX1232-NEXT: v_mov_b32_e32 v4, v3
; GFX1232-NEXT: s_wait_alu depctr_va_sdst(0)
-; GFX1232-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1232-NEXT: v_add_nc_u32_e32 v4, s2, v4
; GFX1232-NEXT: s_wait_kmcnt 0x0
; GFX1232-NEXT: global_store_b32 v0, v4, s[0:1]
@@ -1112,6 +1121,7 @@ define amdgpu_kernel void @add_i32_global_ignores_metadata(ptr addrspace(1) %ino
; GFX1264-NEXT: v_writelane_b32 v3, s5, 48
; GFX1264-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX1264-NEXT: s_mov_b64 exec, s[2:3]
+; GFX1264-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1264-NEXT: v_cmp_eq_u32_e32 vcc, 0, v4
; GFX1264-NEXT: ; implicit-def: $vgpr4
; GFX1264-NEXT: s_and_saveexec_b64 s[2:3], vcc
@@ -1127,10 +1137,10 @@ define amdgpu_kernel void @add_i32_global_ignores_metadata(ptr addrspace(1) %ino
; GFX1264-NEXT: .LBB4_2:
; GFX1264-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX1264-NEXT: s_or_b64 exec, exec, s[2:3]
+; GFX1264-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(VALU_DEP_1)
; GFX1264-NEXT: v_readfirstlane_b32 s2, v4
; GFX1264-NEXT: v_mov_b32_e32 v4, v3
; GFX1264-NEXT: s_wait_alu depctr_va_sdst(0)
-; GFX1264-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1264-NEXT: v_add_nc_u32_e32 v4, s2, v4
; GFX1264-NEXT: s_wait_kmcnt 0x0
; GFX1264-NEXT: global_store_b32 v0, v4, s[0:1]
diff --git a/llvm/test/CodeGen/AMDGPU/atomic_optimizations_global_pointer.ll b/llvm/test/CodeGen/AMDGPU/atomic_optimizations_global_pointer.ll
index 3e3b5ac65eabc2..b81227a0501fa3 100644
--- a/llvm/test/CodeGen/AMDGPU/atomic_optimizations_global_pointer.ll
+++ b/llvm/test/CodeGen/AMDGPU/atomic_optimizations_global_pointer.ll
@@ -208,6 +208,7 @@ define amdgpu_kernel void @add_i32_constant(ptr addrspace(1) %out, ptr addrspace
; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164-NEXT: v_mbcnt_hi_u32_b32 v0, s7, v0
; GFX1164-NEXT: v_cmpx_eq_u32_e32 0, v0
+; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1164-NEXT: s_cbranch_execz .LBB0_2
; GFX1164-NEXT: ; %bb.1:
; GFX1164-NEXT: s_bcnt1_i32_b64 s6, s[6:7]
@@ -239,7 +240,7 @@ define amdgpu_kernel void @add_i32_constant(ptr addrspace(1) %out, ptr addrspace
; GFX1132-NEXT: s_mov_b32 s4, exec_lo
; GFX1132-NEXT: v_mbcnt_lo_u32_b32 v0, s6, 0
; GFX1132-NEXT: ; implicit-def: $vgpr1
-; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX1132-NEXT: s_cbranch_execz .LBB0_2
; GFX1132-NEXT: ; %bb.1:
@@ -275,6 +276,7 @@ define amdgpu_kernel void @add_i32_constant(ptr addrspace(1) %out, ptr addrspace
; GFX1264-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1264-NEXT: v_mbcnt_hi_u32_b32 v0, s7, v0
; GFX1264-NEXT: v_cmpx_eq_u32_e32 0, v0
+; GFX1264-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1264-NEXT: s_cbranch_execz .LBB0_2
; GFX1264-NEXT: ; %bb.1:
; GFX1264-NEXT: s_bcnt1_i32_b64 s6, s[6:7]
@@ -307,7 +309,7 @@ define amdgpu_kernel void @add_i32_constant(ptr addrspace(1) %out, ptr addrspace
; GFX1232-NEXT: s_mov_b32 s4, exec_lo
; GFX1232-NEXT: v_mbcnt_lo_u32_b32 v0, s6, 0
; GFX1232-NEXT: ; implicit-def: $vgpr1
-; GFX1232-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1232-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1232-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX1232-NEXT: s_cbranch_execz .LBB0_2
; GFX1232-NEXT: ; %bb.1:
@@ -342,6 +344,7 @@ define amdgpu_kernel void @add_i32_constant(ptr addrspace(1) %out, ptr addrspace
; GFX1364-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1364-NEXT: v_mbcnt_hi_u32_b32 v0, s7, v0
; GFX1364-NEXT: v_cmpx_eq_u32_e32 0, v0
+; GFX1364-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1364-NEXT: s_cbranch_execz .LBB0_2
; GFX1364-NEXT: ; %bb.1:
; GFX1364-NEXT: s_bcnt1_i32_b64 s6, s[6:7]
@@ -375,7 +378,7 @@ define amdgpu_kernel void @add_i32_constant(ptr addrspace(1) %out, ptr addrspace
; GFX1332-NEXT: s_mov_b32 s4, exec_lo
; GFX1332-NEXT: v_mbcnt_lo_u32_b32 v0, s6, 0
; GFX1332-NEXT: ; implicit-def: $vgpr1
-; GFX1332-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1332-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1332-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX1332-NEXT: s_cbranch_execz .LBB0_2
; GFX1332-NEXT: ; %bb.1:
@@ -594,6 +597,7 @@ define amdgpu_kernel void @add_i32_uniform(ptr addrspace(1) %out, ptr addrspace(
; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164-NEXT: v_mbcnt_hi_u32_b32 v0, s7, v0
; GFX1164-NEXT: v_cmpx_eq_u32_e32 0, v0
+; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1164-NEXT: s_cbranch_execz .LBB1_2
; GFX1164-NEXT: ; %bb.1:
; GFX1164-NEXT: s_bcnt1_i32_b64 s6, s[6:7]
@@ -628,7 +632,7 @@ define amdgpu_kernel void @add_i32_uniform(ptr addrspace(1) %out, ptr addrspace(
; GFX1132-NEXT: s_mov_b32 s5, exec_lo
; GFX1132-NEXT: v_mbcnt_lo_u32_b32 v0, s6, 0
; GFX1132-NEXT: ; implicit-def: $vgpr1
-; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX1132-NEXT: s_cbranch_execz .LBB1_2
; GFX1132-NEXT: ; %bb.1:
@@ -667,6 +671,7 @@ define amdgpu_kernel void @add_i32_uniform(ptr addrspace(1) %out, ptr addrspace(
; GFX1264-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1264-NEXT: v_mbcnt_hi_u32_b32 v0, s7, v0
; GFX1264-NEXT: v_cmpx_eq_u32_e32 0, v0
+; GFX1264-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1264-NEXT: s_cbranch_execz .LBB1_2
; GFX1264-NEXT: ; %bb.1:
; GFX1264-NEXT: s_bcnt1_i32_b64 s6, s[6:7]
@@ -702,7 +707,7 @@ define amdgpu_kernel void @add_i32_uniform(ptr addrspace(1) %out, ptr addrspace(
; GFX1232-NEXT: s_mov_b32 s5, exec_lo
; GFX1232-NEXT: v_mbcnt_lo_u32_b32 v0, s6, 0
; GFX1232-NEXT: ; implicit-def: $vgpr1
-; GFX1232-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1232-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1232-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX1232-NEXT: s_cbranch_execz .LBB1_2
; GFX1232-NEXT: ; %bb.1:
@@ -742,6 +747,7 @@ define amdgpu_kernel void @add_i32_uniform(ptr addrspace(1) %out, ptr addrspace(
; GFX1364-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1364-NEXT: v_mbcnt_hi_u32_b32 v0, s7, v0
; GFX1364-NEXT: v_cmpx_eq_u32_e32 0, v0
+; GFX1364-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1364-NEXT: s_cbranch_execz .LBB1_2
; GFX1364-NEXT: ; %bb.1:
; GFX1364-NEXT: s_bcnt1_i32_b64 s6, s[6:7]
@@ -778,7 +784,7 @@ define amdgpu_kernel void @add_i32_uniform(ptr addrspace(1) %out, ptr addrspace(
; GFX1332-NEXT: s_mov_b32 s5, exec_lo
; GFX1332-NEXT: v_mbcnt_lo_u32_b32 v0, s6, 0
; GFX1332-NEXT: ; implicit-def: $vgpr1
-; GFX1332-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1332-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1332-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX1332-NEXT: s_cbranch_execz .LBB1_2
; GFX1332-NEXT: ; %bb.1:
@@ -1060,8 +1066,8 @@ define amdgpu_kernel void @add_i32_varying(ptr addrspace(1) %out, ptr addrspace(
; GFX1164_ITERATIVE-NEXT: v_mbcnt_hi_u32_b32 v1, exec_hi, v1
; GFX1164_ITERATIVE-NEXT: v_cmp_eq_u32_e32 vcc, 0, v1
; GFX1164_ITERATIVE-NEXT: ; implicit-def: $vgpr1
+; GFX1164_ITERATIVE-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1164_ITERATIVE-NEXT: s_and_saveexec_b64 s[4:5], vcc
-; GFX1164_ITERATIVE-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164_ITERATIVE-NEXT: s_xor_b64 s[4:5], exec, s[4:5]
; GFX1164_ITERATIVE-NEXT: s_cbranch_execz .LBB2_4
; GFX1164_ITERATIVE-NEXT: ; %bb.3:
@@ -1155,8 +1161,8 @@ define amdgpu_kernel void @add_i32_varying(ptr addrspace(1) %out, ptr addrspace(
; GFX1264_ITERATIVE-NEXT: v_mbcnt_hi_u32_b32 v1, exec_hi, v1
; GFX1264_ITERATIVE-NEXT: v_cmp_eq_u32_e32 vcc, 0, v1
; GFX1264_ITERATIVE-NEXT: ; implicit-def: $vgpr1
+; GFX1264_ITERATIVE-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1264_ITERATIVE-NEXT: s_and_saveexec_b64 s[4:5], vcc
-; GFX1264_ITERATIVE-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1264_ITERATIVE-NEXT: s_xor_b64 s[4:5], exec, s[4:5]
; GFX1264_ITERATIVE-NEXT: s_cbranch_execz .LBB2_4
; GFX1264_ITERATIVE-NEXT: ; %bb.3:
@@ -1493,7 +1499,7 @@ define amdgpu_kernel void @add_i32_varying(ptr addrspace(1) %out, ptr addrspace(
; GFX1164_DPP-NEXT: v_readlane_b32 s9, v1, 63
; GFX1164_DPP-NEXT: v_writelane_b32 v3, s7, 32
; GFX1164_DPP-NEXT: s_mov_b64 exec, s[4:5]
-; GFX1164_DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1164_DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_4) | instid1(SALU_CYCLE_1)
; GFX1164_DPP-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
; GFX1164_DPP-NEXT: s_or_saveexec_b64 s[6:7], -1
; GFX1164_DPP-NEXT: s_mov_b32 s4, s9
@@ -1502,6 +1508,7 @@ define amdgpu_kernel void @add_i32_varying(ptr addrspace(1) %out, ptr addrspace(
; GFX1164_DPP-NEXT: v_cmp_eq_u32_e32 vcc, 0, v0
; GFX1164_DPP-NEXT: s_mov_b32 s6, -1
; GFX1164_DPP-NEXT: ; implicit-def: $vgpr0
+; GFX1164_DPP-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1164_DPP-NEXT: s_and_saveexec_b64 s[8:9], vcc
; GFX1164_DPP-NEXT: s_cbranch_execz .LBB2_2
; GFX1164_DPP-NEXT: ; %bb.1:
@@ -1550,11 +1557,12 @@ define amdgpu_kernel void @add_i32_varying(ptr addrspace(1) %out, ptr addrspace(
; GFX1132_DPP-NEXT: v_mov_b32_dpp v3, v1 row_shr:1 row_mask:0xf bank_mask:0xf
; GFX1132_DPP-NEXT: v_readlane_b32 s5, v1, 15
; GFX1132_DPP-NEXT: s_mov_b32 exec_lo, s4
-; GFX1132_DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_3) | instid1(VALU_DEP_2)
+; GFX1132_DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1132_DPP-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
; GFX1132_DPP-NEXT: s_or_saveexec_b32 s4, -1
; GFX1132_DPP-NEXT: v_writelane_b32 v3, s5, 16
; GFX1132_DPP-NEXT: s_mov_b32 exec_lo, s4
+; GFX1132_DPP-NEXT: s_delay_alu instid0(VALU_DEP_2) | instid1(SALU_CYCLE_1)
; GFX1132_DPP-NEXT: v_cmp_eq_u32_e32 vcc_lo, 0, v0
; GFX1132_DPP-NEXT: s_mov_b32 s4, s6
; GFX1132_DPP-NEXT: s_mov_b32 s6, -1
@@ -1626,6 +1634,7 @@ define amdgpu_kernel void @add_i32_varying(ptr addrspace(1) %out, ptr addrspace(
; GFX1264_DPP-NEXT: v_writelane_b32 v3, s8, 48
; GFX1264_DPP-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX1264_DPP-NEXT: s_mov_b64 exec, s[6:7]
+; GFX1264_DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX1264_DPP-NEXT: v_cmp_eq_u32_e32 vcc, 0, v0
; GFX1264_DPP-NEXT: s_mov_b32 s6, -1
; GFX1264_DPP-NEXT: ; implicit-def: $vgpr0
@@ -1678,11 +1687,12 @@ define amdgpu_kernel void @add_i32_varying(ptr addrspace(1) %out, ptr addrspace(
; GFX1232_DPP-NEXT: v_mov_b32_dpp v3, v1 row_shr:1 row_mask:0xf bank_mask:0xf
; GFX1232_DPP-NEXT: v_readlane_b32 s5, v1, 15
; GFX1232_DPP-NEXT: s_mov_b32 exec_lo, s4
-; GFX1232_DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_3) | instid1(VALU_DEP_2)
+; GFX1232_DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1232_DPP-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
; GFX1232_DPP-NEXT: s_or_saveexec_b32 s4, -1
; GFX1232_DPP-NEXT: v_writelane_b32 v3, s5, 16
; GFX1232_DPP-NEXT: s_mov_b32 exec_lo, s4
+; GFX1232_DPP-NEXT: s_delay_alu instid0(VALU_DEP_2) | instid1(SALU_CYCLE_1)
; GFX1232_DPP-NEXT: v_cmp_eq_u32_e32 vcc_lo, 0, v0
; GFX1232_DPP-NEXT: s_mov_b32 s4, s6
; GFX1232_DPP-NEXT: s_mov_b32 s6, -1
@@ -1747,7 +1757,7 @@ define amdgpu_kernel void @add_i32_varying(ptr addrspace(1) %out, ptr addrspace(
; GFX1364-NEXT: v_readlane_b32 s9, v1, 63
; GFX1364-NEXT: v_writelane_b32 v3, s7, 32
; GFX1364-NEXT: s_mov_b64 exec, s[4:5]
-; GFX1364-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1364-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_4) | instid1(SALU_CYCLE_1)
; GFX1364-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
; GFX1364-NEXT: s_or_saveexec_b64 s[6:7], -1
; GFX1364-NEXT: s_mov_b32 s4, s9
@@ -1756,6 +1766,7 @@ define amdgpu_kernel void @add_i32_varying(ptr addrspace(1) %out, ptr addrspace(
; GFX1364-NEXT: v_cmp_eq_u32_e32 vcc, 0, v0
; GFX1364-NEXT: s_mov_b32 s6, -1
; GFX1364-NEXT: ; implicit-def: $vgpr0
+; GFX1364-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1364-NEXT: s_and_saveexec_b64 s[8:9], vcc
; GFX1364-NEXT: s_cbranch_execz .LBB2_2
; GFX1364-NEXT: ; %bb.1:
@@ -1806,11 +1817,12 @@ define amdgpu_kernel void @add_i32_varying(ptr addrspace(1) %out, ptr addrspace(
; GFX1332-NEXT: v_mov_b32_dpp v3, v1 row_shr:1 row_mask:0xf bank_mask:0xf
; GFX1332-NEXT: v_readlane_b32 s5, v1, 15
; GFX1332-NEXT: s_mov_b32 exec_lo, s4
-; GFX1332-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_3) | instid1(VALU_DEP_2)
+; GFX1332-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1332-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
; GFX1332-NEXT: s_or_saveexec_b32 s4, -1
; GFX1332-NEXT: v_writelane_b32 v3, s5, 16
; GFX1332-NEXT: s_mov_b32 exec_lo, s4
+; GFX1332-NEXT: s_delay_alu instid0(VALU_DEP_2) | instid1(SALU_CYCLE_1)
; GFX1332-NEXT: v_cmp_eq_u32_e32 vcc_lo, 0, v0
; GFX1332-NEXT: s_mov_b32 s4, s6
; GFX1332-NEXT: s_mov_b32 s6, -1
@@ -2043,6 +2055,7 @@ define amdgpu_kernel void @add_i64_constant(ptr addrspace(1) %out, ptr addrspace
; GFX1164-NEXT: v_mbcnt_hi_u32_b32 v2, s7, v0
; GFX1164-NEXT: ; implicit-def: $vgpr0_vgpr1
; GFX1164-NEXT: v_cmpx_eq_u32_e32 0, v2
+; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1164-NEXT: s_cbranch_execz .LBB3_2
; GFX1164-NEXT: ; %bb.1:
; GFX1164-NEXT: s_bcnt1_i32_b64 s6, s[6:7]
@@ -2077,7 +2090,7 @@ define amdgpu_kernel void @add_i64_constant(ptr addrspace(1) %out, ptr addrspace
; GFX1132-NEXT: s_mov_b32 s4, exec_lo
; GFX1132-NEXT: v_mbcnt_lo_u32_b32 v2, s6, 0
; GFX1132-NEXT: ; implicit-def: $vgpr0_vgpr1
-; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132-NEXT: v_cmpx_eq_u32_e32 0, v2
; GFX1132-NEXT: s_cbranch_execz .LBB3_2
; GFX1132-NEXT: ; %bb.1:
@@ -2115,6 +2128,7 @@ define amdgpu_kernel void @add_i64_constant(ptr addrspace(1) %out, ptr addrspace
; GFX1264-NEXT: v_mbcnt_hi_u32_b32 v2, s7, v0
; GFX1264-NEXT: ; implicit-def: $vgpr0_vgpr1
; GFX1264-NEXT: v_cmpx_eq_u32_e32 0, v2
+; GFX1264-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1264-NEXT: s_cbranch_execz .LBB3_2
; GFX1264-NEXT: ; %bb.1:
; GFX1264-NEXT: s_bcnt1_i32_b64 s6, s[6:7]
@@ -2150,7 +2164,7 @@ define amdgpu_kernel void @add_i64_constant(ptr addrspace(1) %out, ptr addrspace
; GFX1232-NEXT: s_mov_b32 s4, exec_lo
; GFX1232-NEXT: v_mbcnt_lo_u32_b32 v2, s6, 0
; GFX1232-NEXT: ; implicit-def: $vgpr0_vgpr1
-; GFX1232-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1232-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1232-NEXT: v_cmpx_eq_u32_e32 0, v2
; GFX1232-NEXT: s_cbranch_execz .LBB3_2
; GFX1232-NEXT: ; %bb.1:
@@ -2187,6 +2201,7 @@ define amdgpu_kernel void @add_i64_constant(ptr addrspace(1) %out, ptr addrspace
; GFX1364-NEXT: v_mbcnt_hi_u32_b32 v2, s7, v0
; GFX1364-NEXT: ; implicit-def: $vgpr0_vgpr1
; GFX1364-NEXT: v_cmpx_eq_u32_e32 0, v2
+; GFX1364-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1364-NEXT: s_cbranch_execz .LBB3_2
; GFX1364-NEXT: ; %bb.1:
; GFX1364-NEXT: s_bcnt1_i32_b64 s6, s[6:7]
@@ -2223,7 +2238,7 @@ define amdgpu_kernel void @add_i64_constant(ptr addrspace(1) %out, ptr addrspace
; GFX1332-NEXT: s_mov_b32 s4, exec_lo
; GFX1332-NEXT: v_mbcnt_lo_u32_b32 v2, s6, 0
; GFX1332-NEXT: ; implicit-def: $vgpr0_vgpr1
-; GFX1332-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1332-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1332-NEXT: v_cmpx_eq_u32_e32 0, v2
; GFX1332-NEXT: s_cbranch_execz .LBB3_2
; GFX1332-NEXT: ; %bb.1:
@@ -2481,6 +2496,7 @@ define amdgpu_kernel void @add_i64_uniform(ptr addrspace(1) %out, ptr addrspace(
; GFX1164-NEXT: v_mbcnt_hi_u32_b32 v2, s9, v0
; GFX1164-NEXT: ; implicit-def: $vgpr0_vgpr1
; GFX1164-NEXT: v_cmpx_eq_u32_e32 0, v2
+; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1164-NEXT: s_cbranch_execz .LBB4_2
; GFX1164-NEXT: ; %bb.1:
; GFX1164-NEXT: s_bcnt1_i32_b64 s8, s[8:9]
@@ -2522,7 +2538,7 @@ define amdgpu_kernel void @add_i64_uniform(ptr addrspace(1) %out, ptr addrspace(
; GFX1132-NEXT: s_mov_b32 s6, exec_lo
; GFX1132-NEXT: v_mbcnt_lo_u32_b32 v2, s7, 0
; GFX1132-NEXT: ; implicit-def: $vgpr0_vgpr1
-; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132-NEXT: v_cmpx_eq_u32_e32 0, v2
; GFX1132-NEXT: s_cbranch_execz .LBB4_2
; GFX1132-NEXT: ; %bb.1:
@@ -2569,6 +2585,7 @@ define amdgpu_kernel void @add_i64_uniform(ptr addrspace(1) %out, ptr addrspace(
; GFX1264-NEXT: v_mbcnt_hi_u32_b32 v2, s9, v0
; GFX1264-NEXT: ; implicit-def: $vgpr0_vgpr1
; GFX1264-NEXT: v_cmpx_eq_u32_e32 0, v2
+; GFX1264-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1264-NEXT: s_cbranch_execz .LBB4_2
; GFX1264-NEXT: ; %bb.1:
; GFX1264-NEXT: s_bcnt1_i32_b64 s10, s[8:9]
@@ -2607,7 +2624,7 @@ define amdgpu_kernel void @add_i64_uniform(ptr addrspace(1) %out, ptr addrspace(
; GFX1232-NEXT: v_mbcnt_lo_u32_b32 v2, s6, 0
; GFX1232-NEXT: s_mov_b32 s8, exec_lo
; GFX1232-NEXT: ; implicit-def: $vgpr0_vgpr1
-; GFX1232-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1232-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1232-NEXT: v_cmpx_eq_u32_e32 0, v2
; GFX1232-NEXT: s_cbranch_execz .LBB4_2
; GFX1232-NEXT: ; %bb.1:
@@ -2650,6 +2667,7 @@ define amdgpu_kernel void @add_i64_uniform(ptr addrspace(1) %out, ptr addrspace(
; GFX1364-NEXT: v_mbcnt_hi_u32_b32 v2, s9, v0
; GFX1364-NEXT: ; implicit-def: $vgpr0_vgpr1
; GFX1364-NEXT: v_cmpx_eq_u32_e32 0, v2
+; GFX1364-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1364-NEXT: s_cbranch_execz .LBB4_2
; GFX1364-NEXT: ; %bb.1:
; GFX1364-NEXT: s_bcnt1_i32_b64 s10, s[8:9]
@@ -2690,7 +2708,7 @@ define amdgpu_kernel void @add_i64_uniform(ptr addrspace(1) %out, ptr addrspace(
; GFX1332-NEXT: v_mbcnt_lo_u32_b32 v2, s6, 0
; GFX1332-NEXT: s_mov_b32 s8, exec_lo
; GFX1332-NEXT: ; implicit-def: $vgpr0_vgpr1
-; GFX1332-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1332-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1332-NEXT: v_cmpx_eq_u32_e32 0, v2
; GFX1332-NEXT: s_cbranch_execz .LBB4_2
; GFX1332-NEXT: ; %bb.1:
@@ -3017,8 +3035,8 @@ define amdgpu_kernel void @add_i64_varying(ptr addrspace(1) %out, ptr addrspace(
; GFX1164_ITERATIVE-NEXT: v_mbcnt_hi_u32_b32 v2, exec_hi, v2
; GFX1164_ITERATIVE-NEXT: v_cmp_eq_u32_e32 vcc, 0, v2
; GFX1164_ITERATIVE-NEXT: ; implicit-def: $vgpr2_vgpr3
+; GFX1164_ITERATIVE-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1164_ITERATIVE-NEXT: s_and_saveexec_b64 s[4:5], vcc
-; GFX1164_ITERATIVE-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164_ITERATIVE-NEXT: s_xor_b64 s[4:5], exec, s[4:5]
; GFX1164_ITERATIVE-NEXT: s_cbranch_execz .LBB5_4
; GFX1164_ITERATIVE-NEXT: ; %bb.3:
@@ -3091,7 +3109,7 @@ define amdgpu_kernel void @add_i64_varying(ptr addrspace(1) %out, ptr addrspace(
; GFX1132_ITERATIVE-NEXT: s_waitcnt lgkmcnt(0)
; GFX1132_ITERATIVE-NEXT: v_readfirstlane_b32 s2, v2
; GFX1132_ITERATIVE-NEXT: v_readfirstlane_b32 s3, v3
-; GFX1132_ITERATIVE-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1132_ITERATIVE-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1132_ITERATIVE-NEXT: v_add_co_u32 v0, vcc_lo, s2, v0
; GFX1132_ITERATIVE-NEXT: v_add_co_ci_u32_e64 v1, null, s3, v1, vcc_lo
; GFX1132_ITERATIVE-NEXT: s_mov_b32 s3, 0x31016000
@@ -3126,8 +3144,8 @@ define amdgpu_kernel void @add_i64_varying(ptr addrspace(1) %out, ptr addrspace(
; GFX1264_ITERATIVE-NEXT: v_mbcnt_hi_u32_b32 v2, exec_hi, v2
; GFX1264_ITERATIVE-NEXT: v_cmp_eq_u32_e32 vcc, 0, v2
; GFX1264_ITERATIVE-NEXT: ; implicit-def: $vgpr2_vgpr3
+; GFX1264_ITERATIVE-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1264_ITERATIVE-NEXT: s_and_saveexec_b64 s[4:5], vcc
-; GFX1264_ITERATIVE-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1264_ITERATIVE-NEXT: s_xor_b64 s[4:5], exec, s[4:5]
; GFX1264_ITERATIVE-NEXT: s_cbranch_execz .LBB5_4
; GFX1264_ITERATIVE-NEXT: ; %bb.3:
@@ -3146,7 +3164,7 @@ define amdgpu_kernel void @add_i64_varying(ptr addrspace(1) %out, ptr addrspace(
; GFX1264_ITERATIVE-NEXT: s_wait_kmcnt 0x0
; GFX1264_ITERATIVE-NEXT: v_readfirstlane_b32 s2, v2
; GFX1264_ITERATIVE-NEXT: v_readfirstlane_b32 s3, v3
-; GFX1264_ITERATIVE-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1264_ITERATIVE-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1264_ITERATIVE-NEXT: v_add_co_u32 v0, vcc, s2, v0
; GFX1264_ITERATIVE-NEXT: v_add_co_ci_u32_e64 v1, null, s3, v1, vcc
; GFX1264_ITERATIVE-NEXT: s_mov_b32 s3, 0x31016000
@@ -3198,7 +3216,7 @@ define amdgpu_kernel void @add_i64_varying(ptr addrspace(1) %out, ptr addrspace(
; GFX1232_ITERATIVE-NEXT: s_wait_kmcnt 0x0
; GFX1232_ITERATIVE-NEXT: v_readfirstlane_b32 s2, v2
; GFX1232_ITERATIVE-NEXT: v_readfirstlane_b32 s3, v3
-; GFX1232_ITERATIVE-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1232_ITERATIVE-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1232_ITERATIVE-NEXT: v_add_co_u32 v0, vcc_lo, s2, v0
; GFX1232_ITERATIVE-NEXT: v_add_co_ci_u32_e64 v1, null, s3, v1, vcc_lo
; GFX1232_ITERATIVE-NEXT: s_mov_b32 s3, 0x31016000
@@ -3666,6 +3684,7 @@ define amdgpu_kernel void @add_i64_varying(ptr addrspace(1) %out, ptr addrspace(
; GFX1164_DPP-NEXT: s_mov_b64 s[8:9], exec
; GFX1164_DPP-NEXT: ; implicit-def: $vgpr8_vgpr9
; GFX1164_DPP-NEXT: v_cmpx_eq_u32_e32 0, v0
+; GFX1164_DPP-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1164_DPP-NEXT: s_cbranch_execz .LBB5_2
; GFX1164_DPP-NEXT: ; %bb.1:
; GFX1164_DPP-NEXT: v_mov_b32_e32 v9, s5
@@ -3753,6 +3772,7 @@ define amdgpu_kernel void @add_i64_varying(ptr addrspace(1) %out, ptr addrspace(
; GFX1132_DPP-NEXT: s_mov_b32 s8, exec_lo
; GFX1132_DPP-NEXT: ; implicit-def: $vgpr8_vgpr9
; GFX1132_DPP-NEXT: v_cmpx_eq_u32_e32 0, v0
+; GFX1132_DPP-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1132_DPP-NEXT: s_cbranch_execz .LBB5_2
; GFX1132_DPP-NEXT: ; %bb.1:
; GFX1132_DPP-NEXT: v_dual_mov_b32 v9, s5 :: v_dual_mov_b32 v8, s4
@@ -3770,7 +3790,7 @@ define amdgpu_kernel void @add_i64_varying(ptr addrspace(1) %out, ptr addrspace(
; GFX1132_DPP-NEXT: v_readfirstlane_b32 s2, v8
; GFX1132_DPP-NEXT: v_dual_mov_b32 v10, v6 :: v_dual_mov_b32 v11, v7
; GFX1132_DPP-NEXT: v_readfirstlane_b32 s3, v9
-; GFX1132_DPP-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1132_DPP-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1132_DPP-NEXT: v_add_co_u32 v8, vcc_lo, s2, v10
; GFX1132_DPP-NEXT: v_add_co_ci_u32_e64 v9, null, s3, v11, vcc_lo
; GFX1132_DPP-NEXT: s_mov_b32 s3, 0x31016000
@@ -3863,6 +3883,7 @@ define amdgpu_kernel void @add_i64_varying(ptr addrspace(1) %out, ptr addrspace(
; GFX1264_DPP-NEXT: s_mov_b64 s[8:9], exec
; GFX1264_DPP-NEXT: ; implicit-def: $vgpr6_vgpr7
; GFX1264_DPP-NEXT: v_cmpx_eq_u32_e32 0, v0
+; GFX1264_DPP-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1264_DPP-NEXT: s_cbranch_execz .LBB5_2
; GFX1264_DPP-NEXT: ; %bb.1:
; GFX1264_DPP-NEXT: v_mov_b32_e32 v7, s5
@@ -3953,6 +3974,7 @@ define amdgpu_kernel void @add_i64_varying(ptr addrspace(1) %out, ptr addrspace(
; GFX1232_DPP-NEXT: s_mov_b32 s8, exec_lo
; GFX1232_DPP-NEXT: ; implicit-def: $vgpr8_vgpr9
; GFX1232_DPP-NEXT: v_cmpx_eq_u32_e32 0, v0
+; GFX1232_DPP-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1232_DPP-NEXT: s_cbranch_execz .LBB5_2
; GFX1232_DPP-NEXT: ; %bb.1:
; GFX1232_DPP-NEXT: v_dual_mov_b32 v9, s5 :: v_dual_mov_b32 v8, s4
@@ -4058,6 +4080,7 @@ define amdgpu_kernel void @add_i64_varying(ptr addrspace(1) %out, ptr addrspace(
; GFX1364-NEXT: s_mov_b64 s[8:9], exec
; GFX1364-NEXT: ; implicit-def: $vgpr6_vgpr7
; GFX1364-NEXT: v_cmpx_eq_u32_e32 0, v0
+; GFX1364-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1364-NEXT: s_cbranch_execz .LBB5_2
; GFX1364-NEXT: ; %bb.1:
; GFX1364-NEXT: v_mov_b32_e32 v7, s5
@@ -4079,7 +4102,7 @@ define amdgpu_kernel void @add_i64_varying(ptr addrspace(1) %out, ptr addrspace(
; GFX1364-NEXT: v_mov_b32_e32 v8, v4
; GFX1364-NEXT: v_mov_b32_e32 v9, v5
; GFX1364-NEXT: v_readfirstlane_b32 s3, v7
-; GFX1364-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1364-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1364-NEXT: v_add_co_u32 v6, vcc, s2, v8
; GFX1364-NEXT: v_add_co_ci_u32_e64 v7, null, s3, v9, vcc
; GFX1364-NEXT: s_mov_b32 s3, 0x31016000
@@ -4145,6 +4168,7 @@ define amdgpu_kernel void @add_i64_varying(ptr addrspace(1) %out, ptr addrspace(
; GFX1332-NEXT: s_mov_b32 s8, exec_lo
; GFX1332-NEXT: ; implicit-def: $vgpr8_vgpr9
; GFX1332-NEXT: v_cmpx_eq_u32_e32 0, v0
+; GFX1332-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1332-NEXT: s_cbranch_execz .LBB5_2
; GFX1332-NEXT: ; %bb.1:
; GFX1332-NEXT: v_dual_mov_b32 v9, s5 :: v_dual_mov_b32 v8, s4
@@ -4164,7 +4188,7 @@ define amdgpu_kernel void @add_i64_varying(ptr addrspace(1) %out, ptr addrspace(
; GFX1332-NEXT: v_readfirstlane_b32 s2, v8
; GFX1332-NEXT: v_dual_mov_b32 v10, v6 :: v_dual_mov_b32 v11, v7
; GFX1332-NEXT: v_readfirstlane_b32 s3, v9
-; GFX1332-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1332-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1332-NEXT: v_add_co_u32 v8, vcc_lo, s2, v10
; GFX1332-NEXT: v_add_co_ci_u32_e64 v9, null, s3, v11, vcc_lo
; GFX1332-NEXT: s_mov_b32 s3, 0x31016000
@@ -4433,6 +4457,7 @@ define amdgpu_kernel void @sub_i32_constant(ptr addrspace(1) %out, ptr addrspace
; GFX1164-NEXT: v_mbcnt_hi_u32_b32 v2, s7, v0
; GFX1164-NEXT: ; implicit-def: $vgpr0
; GFX1164-NEXT: v_cmpx_eq_u32_e32 0, v2
+; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1164-NEXT: s_cbranch_execz .LBB6_4
; GFX1164-NEXT: ; %bb.1:
; GFX1164-NEXT: s_waitcnt lgkmcnt(0)
@@ -4459,9 +4484,10 @@ define amdgpu_kernel void @sub_i32_constant(ptr addrspace(1) %out, ptr addrspace
; GFX1164-NEXT: buffer_gl1_inv
; GFX1164-NEXT: buffer_gl0_inv
; GFX1164-NEXT: v_cmp_eq_u32_e32 vcc, v0, v4
+; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1164-NEXT: s_or_b64 s[10:11], vcc, s[10:11]
-; GFX1164-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164-NEXT: s_and_not1_b64 exec, exec, s[10:11]
+; GFX1164-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164-NEXT: s_cbranch_execnz .LBB6_2
; GFX1164-NEXT: ; %bb.3: ; %Flow
; GFX1164-NEXT: s_or_b64 exec, exec, s[10:11]
@@ -4485,7 +4511,7 @@ define amdgpu_kernel void @sub_i32_constant(ptr addrspace(1) %out, ptr addrspace
; GFX1132-NEXT: v_mbcnt_lo_u32_b32 v2, s6, 0
; GFX1132-NEXT: s_mov_b32 s8, exec_lo
; GFX1132-NEXT: ; implicit-def: $vgpr0
-; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132-NEXT: v_cmpx_eq_u32_e32 0, v2
; GFX1132-NEXT: s_cbranch_execz .LBB6_4
; GFX1132-NEXT: ; %bb.1:
@@ -4512,7 +4538,7 @@ define amdgpu_kernel void @sub_i32_constant(ptr addrspace(1) %out, ptr addrspace
; GFX1132-NEXT: buffer_gl0_inv
; GFX1132-NEXT: v_cmp_eq_u32_e32 vcc_lo, v0, v4
; GFX1132-NEXT: s_or_b32 s9, vcc_lo, s9
-; GFX1132-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1132-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1132-NEXT: s_and_not1_b32 exec_lo, exec_lo, s9
; GFX1132-NEXT: s_cbranch_execnz .LBB6_2
; GFX1132-NEXT: ; %bb.3: ; %Flow
@@ -4539,6 +4565,7 @@ define amdgpu_kernel void @sub_i32_constant(ptr addrspace(1) %out, ptr addrspace
; GFX1264-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1264-NEXT: v_mbcnt_hi_u32_b32 v0, s7, v0
; GFX1264-NEXT: v_cmpx_eq_u32_e32 0, v0
+; GFX1264-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1264-NEXT: s_cbranch_execz .LBB6_2
; GFX1264-NEXT: ; %bb.1:
; GFX1264-NEXT: s_bcnt1_i32_b64 s6, s[6:7]
@@ -4573,7 +4600,7 @@ define amdgpu_kernel void @sub_i32_constant(ptr addrspace(1) %out, ptr addrspace
; GFX1232-NEXT: s_mov_b32 s4, exec_lo
; GFX1232-NEXT: v_mbcnt_lo_u32_b32 v0, s6, 0
; GFX1232-NEXT: ; implicit-def: $vgpr1
-; GFX1232-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1232-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1232-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX1232-NEXT: s_cbranch_execz .LBB6_2
; GFX1232-NEXT: ; %bb.1:
@@ -4610,6 +4637,7 @@ define amdgpu_kernel void @sub_i32_constant(ptr addrspace(1) %out, ptr addrspace
; GFX1364-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1364-NEXT: v_mbcnt_hi_u32_b32 v0, s7, v0
; GFX1364-NEXT: v_cmpx_eq_u32_e32 0, v0
+; GFX1364-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1364-NEXT: s_cbranch_execz .LBB6_2
; GFX1364-NEXT: ; %bb.1:
; GFX1364-NEXT: s_bcnt1_i32_b64 s6, s[6:7]
@@ -4645,7 +4673,7 @@ define amdgpu_kernel void @sub_i32_constant(ptr addrspace(1) %out, ptr addrspace
; GFX1332-NEXT: s_mov_b32 s4, exec_lo
; GFX1332-NEXT: v_mbcnt_lo_u32_b32 v0, s6, 0
; GFX1332-NEXT: ; implicit-def: $vgpr1
-; GFX1332-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1332-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1332-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX1332-NEXT: s_cbranch_execz .LBB6_2
; GFX1332-NEXT: ; %bb.1:
@@ -4942,6 +4970,7 @@ define amdgpu_kernel void @sub_i32_uniform(ptr addrspace(1) %out, ptr addrspace(
; GFX1164-NEXT: v_mbcnt_hi_u32_b32 v2, s7, v0
; GFX1164-NEXT: ; implicit-def: $vgpr0
; GFX1164-NEXT: v_cmpx_eq_u32_e32 0, v2
+; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1164-NEXT: s_cbranch_execz .LBB7_4
; GFX1164-NEXT: ; %bb.1:
; GFX1164-NEXT: s_waitcnt lgkmcnt(0)
@@ -4968,9 +4997,10 @@ define amdgpu_kernel void @sub_i32_uniform(ptr addrspace(1) %out, ptr addrspace(
; GFX1164-NEXT: buffer_gl1_inv
; GFX1164-NEXT: buffer_gl0_inv
; GFX1164-NEXT: v_cmp_eq_u32_e32 vcc, v0, v4
+; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1164-NEXT: s_or_b64 s[10:11], vcc, s[10:11]
-; GFX1164-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164-NEXT: s_and_not1_b64 exec, exec, s[10:11]
+; GFX1164-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164-NEXT: s_cbranch_execnz .LBB7_2
; GFX1164-NEXT: ; %bb.3: ; %Flow
; GFX1164-NEXT: s_or_b64 exec, exec, s[10:11]
@@ -4996,7 +5026,7 @@ define amdgpu_kernel void @sub_i32_uniform(ptr addrspace(1) %out, ptr addrspace(
; GFX1132-NEXT: v_mbcnt_lo_u32_b32 v2, s6, 0
; GFX1132-NEXT: s_mov_b32 s9, exec_lo
; GFX1132-NEXT: ; implicit-def: $vgpr0
-; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132-NEXT: v_cmpx_eq_u32_e32 0, v2
; GFX1132-NEXT: s_cbranch_execz .LBB7_4
; GFX1132-NEXT: ; %bb.1:
@@ -5023,7 +5053,7 @@ define amdgpu_kernel void @sub_i32_uniform(ptr addrspace(1) %out, ptr addrspace(
; GFX1132-NEXT: buffer_gl0_inv
; GFX1132-NEXT: v_cmp_eq_u32_e32 vcc_lo, v0, v4
; GFX1132-NEXT: s_or_b32 s10, vcc_lo, s10
-; GFX1132-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1132-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1132-NEXT: s_and_not1_b32 exec_lo, exec_lo, s10
; GFX1132-NEXT: s_cbranch_execnz .LBB7_2
; GFX1132-NEXT: ; %bb.3: ; %Flow
@@ -5052,6 +5082,7 @@ define amdgpu_kernel void @sub_i32_uniform(ptr addrspace(1) %out, ptr addrspace(
; GFX1264-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1264-NEXT: v_mbcnt_hi_u32_b32 v0, s7, v0
; GFX1264-NEXT: v_cmpx_eq_u32_e32 0, v0
+; GFX1264-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1264-NEXT: s_cbranch_execz .LBB7_2
; GFX1264-NEXT: ; %bb.1:
; GFX1264-NEXT: s_bcnt1_i32_b64 s6, s[6:7]
@@ -5087,7 +5118,7 @@ define amdgpu_kernel void @sub_i32_uniform(ptr addrspace(1) %out, ptr addrspace(
; GFX1232-NEXT: s_mov_b32 s5, exec_lo
; GFX1232-NEXT: v_mbcnt_lo_u32_b32 v0, s6, 0
; GFX1232-NEXT: ; implicit-def: $vgpr1
-; GFX1232-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1232-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1232-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX1232-NEXT: s_cbranch_execz .LBB7_2
; GFX1232-NEXT: ; %bb.1:
@@ -5127,6 +5158,7 @@ define amdgpu_kernel void @sub_i32_uniform(ptr addrspace(1) %out, ptr addrspace(
; GFX1364-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1364-NEXT: v_mbcnt_hi_u32_b32 v0, s7, v0
; GFX1364-NEXT: v_cmpx_eq_u32_e32 0, v0
+; GFX1364-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1364-NEXT: s_cbranch_execz .LBB7_2
; GFX1364-NEXT: ; %bb.1:
; GFX1364-NEXT: s_bcnt1_i32_b64 s6, s[6:7]
@@ -5163,7 +5195,7 @@ define amdgpu_kernel void @sub_i32_uniform(ptr addrspace(1) %out, ptr addrspace(
; GFX1332-NEXT: s_mov_b32 s5, exec_lo
; GFX1332-NEXT: v_mbcnt_lo_u32_b32 v0, s6, 0
; GFX1332-NEXT: ; implicit-def: $vgpr1
-; GFX1332-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1332-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1332-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX1332-NEXT: s_cbranch_execz .LBB7_2
; GFX1332-NEXT: ; %bb.1:
@@ -5519,8 +5551,8 @@ define amdgpu_kernel void @sub_i32_varying(ptr addrspace(1) %out, ptr addrspace(
; GFX1164_ITERATIVE-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
; GFX1164_ITERATIVE-NEXT: v_cmp_eq_u32_e32 vcc, 0, v0
; GFX1164_ITERATIVE-NEXT: ; implicit-def: $vgpr0
+; GFX1164_ITERATIVE-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1164_ITERATIVE-NEXT: s_and_saveexec_b64 s[4:5], vcc
-; GFX1164_ITERATIVE-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164_ITERATIVE-NEXT: s_xor_b64 s[8:9], exec, s[4:5]
; GFX1164_ITERATIVE-NEXT: s_cbranch_execz .LBB8_6
; GFX1164_ITERATIVE-NEXT: ; %bb.3:
@@ -5546,9 +5578,10 @@ define amdgpu_kernel void @sub_i32_varying(ptr addrspace(1) %out, ptr addrspace(
; GFX1164_ITERATIVE-NEXT: buffer_gl1_inv
; GFX1164_ITERATIVE-NEXT: buffer_gl0_inv
; GFX1164_ITERATIVE-NEXT: v_cmp_eq_u32_e32 vcc, v0, v4
+; GFX1164_ITERATIVE-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1164_ITERATIVE-NEXT: s_or_b64 s[10:11], vcc, s[10:11]
-; GFX1164_ITERATIVE-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164_ITERATIVE-NEXT: s_and_not1_b64 exec, exec, s[10:11]
+; GFX1164_ITERATIVE-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164_ITERATIVE-NEXT: s_cbranch_execnz .LBB8_4
; GFX1164_ITERATIVE-NEXT: ; %bb.5: ; %Flow
; GFX1164_ITERATIVE-NEXT: s_or_b64 exec, exec, s[10:11]
@@ -5611,7 +5644,7 @@ define amdgpu_kernel void @sub_i32_varying(ptr addrspace(1) %out, ptr addrspace(
; GFX1132_ITERATIVE-NEXT: buffer_gl0_inv
; GFX1132_ITERATIVE-NEXT: v_cmp_eq_u32_e32 vcc_lo, v0, v4
; GFX1132_ITERATIVE-NEXT: s_or_b32 s10, vcc_lo, s10
-; GFX1132_ITERATIVE-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1132_ITERATIVE-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1132_ITERATIVE-NEXT: s_and_not1_b32 exec_lo, exec_lo, s10
; GFX1132_ITERATIVE-NEXT: s_cbranch_execnz .LBB8_4
; GFX1132_ITERATIVE-NEXT: ; %bb.5: ; %Flow
@@ -5651,8 +5684,8 @@ define amdgpu_kernel void @sub_i32_varying(ptr addrspace(1) %out, ptr addrspace(
; GFX1264_ITERATIVE-NEXT: v_mbcnt_hi_u32_b32 v1, exec_hi, v1
; GFX1264_ITERATIVE-NEXT: v_cmp_eq_u32_e32 vcc, 0, v1
; GFX1264_ITERATIVE-NEXT: ; implicit-def: $vgpr1
+; GFX1264_ITERATIVE-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1264_ITERATIVE-NEXT: s_and_saveexec_b64 s[4:5], vcc
-; GFX1264_ITERATIVE-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1264_ITERATIVE-NEXT: s_xor_b64 s[4:5], exec, s[4:5]
; GFX1264_ITERATIVE-NEXT: s_cbranch_execz .LBB8_4
; GFX1264_ITERATIVE-NEXT: ; %bb.3:
@@ -6066,6 +6099,7 @@ define amdgpu_kernel void @sub_i32_varying(ptr addrspace(1) %out, ptr addrspace(
; GFX1164_DPP-NEXT: s_mov_b64 s[8:9], exec
; GFX1164_DPP-NEXT: ; implicit-def: $vgpr4
; GFX1164_DPP-NEXT: v_cmpx_eq_u32_e32 0, v0
+; GFX1164_DPP-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1164_DPP-NEXT: s_cbranch_execz .LBB8_4
; GFX1164_DPP-NEXT: ; %bb.1:
; GFX1164_DPP-NEXT: s_waitcnt lgkmcnt(0)
@@ -6088,9 +6122,10 @@ define amdgpu_kernel void @sub_i32_varying(ptr addrspace(1) %out, ptr addrspace(
; GFX1164_DPP-NEXT: buffer_gl1_inv
; GFX1164_DPP-NEXT: buffer_gl0_inv
; GFX1164_DPP-NEXT: v_cmp_eq_u32_e32 vcc, v4, v6
+; GFX1164_DPP-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1164_DPP-NEXT: s_or_b64 s[10:11], vcc, s[10:11]
-; GFX1164_DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164_DPP-NEXT: s_and_not1_b64 exec, exec, s[10:11]
+; GFX1164_DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164_DPP-NEXT: s_cbranch_execnz .LBB8_2
; GFX1164_DPP-NEXT: ; %bb.3: ; %Flow
; GFX1164_DPP-NEXT: s_or_b64 exec, exec, s[10:11]
@@ -6139,6 +6174,7 @@ define amdgpu_kernel void @sub_i32_varying(ptr addrspace(1) %out, ptr addrspace(
; GFX1132_DPP-NEXT: s_mov_b32 s8, exec_lo
; GFX1132_DPP-NEXT: ; implicit-def: $vgpr4
; GFX1132_DPP-NEXT: v_cmpx_eq_u32_e32 0, v0
+; GFX1132_DPP-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1132_DPP-NEXT: s_cbranch_execz .LBB8_4
; GFX1132_DPP-NEXT: ; %bb.1:
; GFX1132_DPP-NEXT: s_waitcnt lgkmcnt(0)
@@ -6162,7 +6198,7 @@ define amdgpu_kernel void @sub_i32_varying(ptr addrspace(1) %out, ptr addrspace(
; GFX1132_DPP-NEXT: buffer_gl0_inv
; GFX1132_DPP-NEXT: v_cmp_eq_u32_e32 vcc_lo, v4, v6
; GFX1132_DPP-NEXT: s_or_b32 s10, vcc_lo, s10
-; GFX1132_DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1132_DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1132_DPP-NEXT: s_and_not1_b32 exec_lo, exec_lo, s10
; GFX1132_DPP-NEXT: s_cbranch_execnz .LBB8_2
; GFX1132_DPP-NEXT: ; %bb.3: ; %Flow
@@ -6222,6 +6258,7 @@ define amdgpu_kernel void @sub_i32_varying(ptr addrspace(1) %out, ptr addrspace(
; GFX1264_DPP-NEXT: v_writelane_b32 v3, s8, 48
; GFX1264_DPP-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX1264_DPP-NEXT: s_mov_b64 exec, s[6:7]
+; GFX1264_DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX1264_DPP-NEXT: v_cmp_eq_u32_e32 vcc, 0, v0
; GFX1264_DPP-NEXT: s_mov_b32 s6, -1
; GFX1264_DPP-NEXT: ; implicit-def: $vgpr0
@@ -6274,11 +6311,12 @@ define amdgpu_kernel void @sub_i32_varying(ptr addrspace(1) %out, ptr addrspace(
; GFX1232_DPP-NEXT: v_mov_b32_dpp v3, v1 row_shr:1 row_mask:0xf bank_mask:0xf
; GFX1232_DPP-NEXT: v_readlane_b32 s5, v1, 15
; GFX1232_DPP-NEXT: s_mov_b32 exec_lo, s4
-; GFX1232_DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_3) | instid1(VALU_DEP_2)
+; GFX1232_DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1232_DPP-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
; GFX1232_DPP-NEXT: s_or_saveexec_b32 s4, -1
; GFX1232_DPP-NEXT: v_writelane_b32 v3, s5, 16
; GFX1232_DPP-NEXT: s_mov_b32 exec_lo, s4
+; GFX1232_DPP-NEXT: s_delay_alu instid0(VALU_DEP_2) | instid1(SALU_CYCLE_1)
; GFX1232_DPP-NEXT: v_cmp_eq_u32_e32 vcc_lo, 0, v0
; GFX1232_DPP-NEXT: s_mov_b32 s4, s6
; GFX1232_DPP-NEXT: s_mov_b32 s6, -1
@@ -6343,7 +6381,7 @@ define amdgpu_kernel void @sub_i32_varying(ptr addrspace(1) %out, ptr addrspace(
; GFX1364-NEXT: v_readlane_b32 s9, v1, 63
; GFX1364-NEXT: v_writelane_b32 v3, s7, 32
; GFX1364-NEXT: s_mov_b64 exec, s[4:5]
-; GFX1364-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1364-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_4) | instid1(SALU_CYCLE_1)
; GFX1364-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
; GFX1364-NEXT: s_or_saveexec_b64 s[6:7], -1
; GFX1364-NEXT: s_mov_b32 s4, s9
@@ -6352,6 +6390,7 @@ define amdgpu_kernel void @sub_i32_varying(ptr addrspace(1) %out, ptr addrspace(
; GFX1364-NEXT: v_cmp_eq_u32_e32 vcc, 0, v0
; GFX1364-NEXT: s_mov_b32 s6, -1
; GFX1364-NEXT: ; implicit-def: $vgpr0
+; GFX1364-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1364-NEXT: s_and_saveexec_b64 s[8:9], vcc
; GFX1364-NEXT: s_cbranch_execz .LBB8_2
; GFX1364-NEXT: ; %bb.1:
@@ -6402,11 +6441,12 @@ define amdgpu_kernel void @sub_i32_varying(ptr addrspace(1) %out, ptr addrspace(
; GFX1332-NEXT: v_mov_b32_dpp v3, v1 row_shr:1 row_mask:0xf bank_mask:0xf
; GFX1332-NEXT: v_readlane_b32 s5, v1, 15
; GFX1332-NEXT: s_mov_b32 exec_lo, s4
-; GFX1332-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_3) | instid1(VALU_DEP_2)
+; GFX1332-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1332-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
; GFX1332-NEXT: s_or_saveexec_b32 s4, -1
; GFX1332-NEXT: v_writelane_b32 v3, s5, 16
; GFX1332-NEXT: s_mov_b32 exec_lo, s4
+; GFX1332-NEXT: s_delay_alu instid0(VALU_DEP_2) | instid1(SALU_CYCLE_1)
; GFX1332-NEXT: v_cmp_eq_u32_e32 vcc_lo, 0, v0
; GFX1332-NEXT: s_mov_b32 s4, s6
; GFX1332-NEXT: s_mov_b32 s6, -1
@@ -6740,6 +6780,7 @@ define amdgpu_kernel void @sub_i64_constant(ptr addrspace(1) %out, ptr addrspace
; GFX1164-NEXT: v_mbcnt_hi_u32_b32 v4, s7, v0
; GFX1164-NEXT: ; implicit-def: $vgpr0_vgpr1
; GFX1164-NEXT: v_cmpx_eq_u32_e32 0, v4
+; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1164-NEXT: s_cbranch_execz .LBB9_4
; GFX1164-NEXT: ; %bb.1:
; GFX1164-NEXT: s_waitcnt lgkmcnt(0)
@@ -6759,7 +6800,7 @@ define amdgpu_kernel void @sub_i64_constant(ptr addrspace(1) %out, ptr addrspace
; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1164-NEXT: v_mov_b32_e32 v7, v0
; GFX1164-NEXT: v_mov_b32_e32 v8, v1
-; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1164-NEXT: v_sub_co_u32 v5, vcc, v7, s12
; GFX1164-NEXT: v_subrev_co_ci_u32_e64 v6, null, 0, v8, vcc
; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_2) | instid1(VALU_DEP_4)
@@ -6773,9 +6814,10 @@ define amdgpu_kernel void @sub_i64_constant(ptr addrspace(1) %out, ptr addrspace
; GFX1164-NEXT: buffer_gl0_inv
; GFX1164-NEXT: v_cmp_eq_u64_e32 vcc, v[0:1], v[7:8]
; GFX1164-NEXT: s_waitcnt_depctr depctr_va_vcc(0)
+; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1164-NEXT: s_or_b64 s[10:11], vcc, s[10:11]
-; GFX1164-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164-NEXT: s_and_not1_b64 exec, exec, s[10:11]
+; GFX1164-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164-NEXT: s_cbranch_execnz .LBB9_2
; GFX1164-NEXT: ; %bb.3: ; %Flow
; GFX1164-NEXT: s_or_b64 exec, exec, s[10:11]
@@ -6803,7 +6845,7 @@ define amdgpu_kernel void @sub_i64_constant(ptr addrspace(1) %out, ptr addrspace
; GFX1132-NEXT: v_mbcnt_lo_u32_b32 v4, s6, 0
; GFX1132-NEXT: s_mov_b32 s8, exec_lo
; GFX1132-NEXT: ; implicit-def: $vgpr0_vgpr1
-; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132-NEXT: v_cmpx_eq_u32_e32 0, v4
; GFX1132-NEXT: s_cbranch_execz .LBB9_4
; GFX1132-NEXT: ; %bb.1:
@@ -6822,7 +6864,7 @@ define amdgpu_kernel void @sub_i64_constant(ptr addrspace(1) %out, ptr addrspace
; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132-NEXT: v_dual_mov_b32 v8, v1 :: v_dual_mov_b32 v7, v0
; GFX1132-NEXT: v_sub_co_u32 v5, vcc_lo, v7, s10
-; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
+; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX1132-NEXT: v_subrev_co_ci_u32_e64 v6, null, 0, v8, vcc_lo
; GFX1132-NEXT: v_dual_mov_b32 v2, v7 :: v_dual_mov_b32 v3, v8
; GFX1132-NEXT: v_dual_mov_b32 v0, v5 :: v_dual_mov_b32 v1, v6
@@ -6832,7 +6874,7 @@ define amdgpu_kernel void @sub_i64_constant(ptr addrspace(1) %out, ptr addrspace
; GFX1132-NEXT: buffer_gl0_inv
; GFX1132-NEXT: v_cmp_eq_u64_e32 vcc_lo, v[0:1], v[7:8]
; GFX1132-NEXT: s_or_b32 s9, vcc_lo, s9
-; GFX1132-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1132-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1132-NEXT: s_and_not1_b32 exec_lo, exec_lo, s9
; GFX1132-NEXT: s_cbranch_execnz .LBB9_2
; GFX1132-NEXT: ; %bb.3: ; %Flow
@@ -6845,7 +6887,7 @@ define amdgpu_kernel void @sub_i64_constant(ptr addrspace(1) %out, ptr addrspace
; GFX1132-NEXT: v_mul_u32_u24_e32 v0, 5, v4
; GFX1132-NEXT: v_readfirstlane_b32 s3, v1
; GFX1132-NEXT: v_mul_hi_u32_u24_e32 v1, 5, v4
-; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1132-NEXT: v_sub_co_u32 v0, vcc_lo, s2, v0
; GFX1132-NEXT: v_sub_co_ci_u32_e64 v1, null, s3, v1, vcc_lo
; GFX1132-NEXT: s_mov_b32 s3, 0x31016000
@@ -6863,6 +6905,7 @@ define amdgpu_kernel void @sub_i64_constant(ptr addrspace(1) %out, ptr addrspace
; GFX1264-NEXT: v_mbcnt_hi_u32_b32 v2, s7, v0
; GFX1264-NEXT: ; implicit-def: $vgpr0_vgpr1
; GFX1264-NEXT: v_cmpx_eq_u32_e32 0, v2
+; GFX1264-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1264-NEXT: s_cbranch_execz .LBB9_2
; GFX1264-NEXT: ; %bb.1:
; GFX1264-NEXT: s_bcnt1_i32_b64 s6, s[6:7]
@@ -6886,7 +6929,7 @@ define amdgpu_kernel void @sub_i64_constant(ptr addrspace(1) %out, ptr addrspace
; GFX1264-NEXT: v_mul_u32_u24_e32 v0, 5, v2
; GFX1264-NEXT: v_readfirstlane_b32 s3, v1
; GFX1264-NEXT: v_mul_hi_u32_u24_e32 v1, 5, v2
-; GFX1264-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1264-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1264-NEXT: v_sub_co_u32 v0, vcc, s2, v0
; GFX1264-NEXT: v_sub_co_ci_u32_e64 v1, null, s3, v1, vcc
; GFX1264-NEXT: s_mov_b32 s3, 0x31016000
@@ -6901,7 +6944,7 @@ define amdgpu_kernel void @sub_i64_constant(ptr addrspace(1) %out, ptr addrspace
; GFX1232-NEXT: s_mov_b32 s4, exec_lo
; GFX1232-NEXT: v_mbcnt_lo_u32_b32 v2, s6, 0
; GFX1232-NEXT: ; implicit-def: $vgpr0_vgpr1
-; GFX1232-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1232-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1232-NEXT: v_cmpx_eq_u32_e32 0, v2
; GFX1232-NEXT: s_cbranch_execz .LBB9_2
; GFX1232-NEXT: ; %bb.1:
@@ -6923,7 +6966,7 @@ define amdgpu_kernel void @sub_i64_constant(ptr addrspace(1) %out, ptr addrspace
; GFX1232-NEXT: v_mul_u32_u24_e32 v0, 5, v2
; GFX1232-NEXT: v_readfirstlane_b32 s3, v1
; GFX1232-NEXT: v_mul_hi_u32_u24_e32 v1, 5, v2
-; GFX1232-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1232-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1232-NEXT: v_sub_co_u32 v0, vcc_lo, s2, v0
; GFX1232-NEXT: v_sub_co_ci_u32_e64 v1, null, s3, v1, vcc_lo
; GFX1232-NEXT: s_mov_b32 s3, 0x31016000
@@ -6941,6 +6984,7 @@ define amdgpu_kernel void @sub_i64_constant(ptr addrspace(1) %out, ptr addrspace
; GFX1364-NEXT: v_mbcnt_hi_u32_b32 v2, s7, v0
; GFX1364-NEXT: ; implicit-def: $vgpr0_vgpr1
; GFX1364-NEXT: v_cmpx_eq_u32_e32 0, v2
+; GFX1364-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1364-NEXT: s_cbranch_execz .LBB9_2
; GFX1364-NEXT: ; %bb.1:
; GFX1364-NEXT: s_bcnt1_i32_b64 s6, s[6:7]
@@ -6965,7 +7009,7 @@ define amdgpu_kernel void @sub_i64_constant(ptr addrspace(1) %out, ptr addrspace
; GFX1364-NEXT: v_mul_u32_u24_e32 v0, 5, v2
; GFX1364-NEXT: v_readfirstlane_b32 s3, v1
; GFX1364-NEXT: v_mul_hi_u32_u24_e32 v1, 5, v2
-; GFX1364-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1364-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1364-NEXT: v_sub_co_u32 v0, vcc, s2, v0
; GFX1364-NEXT: v_sub_co_ci_u32_e64 v1, null, s3, v1, vcc
; GFX1364-NEXT: s_mov_b32 s3, 0x31016000
@@ -6980,7 +7024,7 @@ define amdgpu_kernel void @sub_i64_constant(ptr addrspace(1) %out, ptr addrspace
; GFX1332-NEXT: s_mov_b32 s4, exec_lo
; GFX1332-NEXT: v_mbcnt_lo_u32_b32 v2, s6, 0
; GFX1332-NEXT: ; implicit-def: $vgpr0_vgpr1
-; GFX1332-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1332-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1332-NEXT: v_cmpx_eq_u32_e32 0, v2
; GFX1332-NEXT: s_cbranch_execz .LBB9_2
; GFX1332-NEXT: ; %bb.1:
@@ -7005,7 +7049,7 @@ define amdgpu_kernel void @sub_i64_constant(ptr addrspace(1) %out, ptr addrspace
; GFX1332-NEXT: v_mul_u32_u24_e32 v0, 5, v2
; GFX1332-NEXT: v_readfirstlane_b32 s3, v1
; GFX1332-NEXT: v_mul_hi_u32_u24_e32 v1, 5, v2
-; GFX1332-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1332-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1332-NEXT: v_sub_co_u32 v0, vcc_lo, s2, v0
; GFX1332-NEXT: v_sub_co_ci_u32_e64 v1, null, s3, v1, vcc_lo
; GFX1332-NEXT: s_mov_b32 s3, 0x31016000
@@ -7344,6 +7388,7 @@ define amdgpu_kernel void @sub_i64_uniform(ptr addrspace(1) %out, ptr addrspace(
; GFX1164-NEXT: v_mbcnt_hi_u32_b32 v4, s7, v0
; GFX1164-NEXT: ; implicit-def: $vgpr0_vgpr1
; GFX1164-NEXT: v_cmpx_eq_u32_e32 0, v4
+; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1164-NEXT: s_cbranch_execz .LBB10_4
; GFX1164-NEXT: ; %bb.1:
; GFX1164-NEXT: s_waitcnt lgkmcnt(0)
@@ -7366,7 +7411,7 @@ define amdgpu_kernel void @sub_i64_uniform(ptr addrspace(1) %out, ptr addrspace(
; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1164-NEXT: v_mov_b32_e32 v7, v0
; GFX1164-NEXT: v_mov_b32_e32 v8, v1
-; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1164-NEXT: v_sub_co_u32 v5, vcc, v7, s14
; GFX1164-NEXT: v_subrev_co_ci_u32_e64 v6, null, s15, v8, vcc
; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_2) | instid1(VALU_DEP_4)
@@ -7380,9 +7425,10 @@ define amdgpu_kernel void @sub_i64_uniform(ptr addrspace(1) %out, ptr addrspace(
; GFX1164-NEXT: buffer_gl0_inv
; GFX1164-NEXT: v_cmp_eq_u64_e32 vcc, v[0:1], v[7:8]
; GFX1164-NEXT: s_waitcnt_depctr depctr_va_vcc(0)
+; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1164-NEXT: s_or_b64 s[12:13], vcc, s[12:13]
-; GFX1164-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164-NEXT: s_and_not1_b64 exec, exec, s[12:13]
+; GFX1164-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164-NEXT: s_cbranch_execnz .LBB10_2
; GFX1164-NEXT: ; %bb.3: ; %Flow
; GFX1164-NEXT: s_or_b64 exec, exec, s[12:13]
@@ -7412,7 +7458,7 @@ define amdgpu_kernel void @sub_i64_uniform(ptr addrspace(1) %out, ptr addrspace(
; GFX1132-NEXT: v_mbcnt_lo_u32_b32 v4, s6, 0
; GFX1132-NEXT: s_mov_b32 s10, exec_lo
; GFX1132-NEXT: ; implicit-def: $vgpr0_vgpr1
-; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132-NEXT: v_cmpx_eq_u32_e32 0, v4
; GFX1132-NEXT: s_cbranch_execz .LBB10_4
; GFX1132-NEXT: ; %bb.1:
@@ -7435,7 +7481,7 @@ define amdgpu_kernel void @sub_i64_uniform(ptr addrspace(1) %out, ptr addrspace(
; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132-NEXT: v_dual_mov_b32 v8, v1 :: v_dual_mov_b32 v7, v0
; GFX1132-NEXT: v_sub_co_u32 v5, vcc_lo, v7, s12
-; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
+; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX1132-NEXT: v_subrev_co_ci_u32_e64 v6, null, s13, v8, vcc_lo
; GFX1132-NEXT: v_dual_mov_b32 v2, v7 :: v_dual_mov_b32 v3, v8
; GFX1132-NEXT: v_dual_mov_b32 v0, v5 :: v_dual_mov_b32 v1, v6
@@ -7445,7 +7491,7 @@ define amdgpu_kernel void @sub_i64_uniform(ptr addrspace(1) %out, ptr addrspace(
; GFX1132-NEXT: buffer_gl0_inv
; GFX1132-NEXT: v_cmp_eq_u64_e32 vcc_lo, v[0:1], v[7:8]
; GFX1132-NEXT: s_or_b32 s11, vcc_lo, s11
-; GFX1132-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1132-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1132-NEXT: s_and_not1_b32 exec_lo, exec_lo, s11
; GFX1132-NEXT: s_cbranch_execnz .LBB10_2
; GFX1132-NEXT: ; %bb.3: ; %Flow
@@ -7479,6 +7525,7 @@ define amdgpu_kernel void @sub_i64_uniform(ptr addrspace(1) %out, ptr addrspace(
; GFX1264-NEXT: v_mbcnt_hi_u32_b32 v2, s9, v0
; GFX1264-NEXT: ; implicit-def: $vgpr0_vgpr1
; GFX1264-NEXT: v_cmpx_eq_u32_e32 0, v2
+; GFX1264-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1264-NEXT: s_cbranch_execz .LBB10_2
; GFX1264-NEXT: ; %bb.1:
; GFX1264-NEXT: s_bcnt1_i32_b64 s10, s[8:9]
@@ -7519,7 +7566,7 @@ define amdgpu_kernel void @sub_i64_uniform(ptr addrspace(1) %out, ptr addrspace(
; GFX1232-NEXT: v_mbcnt_lo_u32_b32 v2, s6, 0
; GFX1232-NEXT: s_mov_b32 s8, exec_lo
; GFX1232-NEXT: ; implicit-def: $vgpr0_vgpr1
-; GFX1232-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1232-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1232-NEXT: v_cmpx_eq_u32_e32 0, v2
; GFX1232-NEXT: s_cbranch_execz .LBB10_2
; GFX1232-NEXT: ; %bb.1:
@@ -7564,6 +7611,7 @@ define amdgpu_kernel void @sub_i64_uniform(ptr addrspace(1) %out, ptr addrspace(
; GFX1364-NEXT: v_mbcnt_hi_u32_b32 v2, s9, v0
; GFX1364-NEXT: ; implicit-def: $vgpr0_vgpr1
; GFX1364-NEXT: v_cmpx_eq_u32_e32 0, v2
+; GFX1364-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1364-NEXT: s_cbranch_execz .LBB10_2
; GFX1364-NEXT: ; %bb.1:
; GFX1364-NEXT: s_bcnt1_i32_b64 s10, s[8:9]
@@ -7606,7 +7654,7 @@ define amdgpu_kernel void @sub_i64_uniform(ptr addrspace(1) %out, ptr addrspace(
; GFX1332-NEXT: v_mbcnt_lo_u32_b32 v2, s6, 0
; GFX1332-NEXT: s_mov_b32 s8, exec_lo
; GFX1332-NEXT: ; implicit-def: $vgpr0_vgpr1
-; GFX1332-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1332-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1332-NEXT: v_cmpx_eq_u32_e32 0, v2
; GFX1332-NEXT: s_cbranch_execz .LBB10_2
; GFX1332-NEXT: ; %bb.1:
@@ -8032,8 +8080,8 @@ define amdgpu_kernel void @sub_i64_varying(ptr addrspace(1) %out, ptr addrspace(
; GFX1164_ITERATIVE-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
; GFX1164_ITERATIVE-NEXT: v_cmp_eq_u32_e32 vcc, 0, v0
; GFX1164_ITERATIVE-NEXT: ; implicit-def: $vgpr0_vgpr1
+; GFX1164_ITERATIVE-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1164_ITERATIVE-NEXT: s_and_saveexec_b64 s[4:5], vcc
-; GFX1164_ITERATIVE-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164_ITERATIVE-NEXT: s_xor_b64 s[10:11], exec, s[4:5]
; GFX1164_ITERATIVE-NEXT: s_cbranch_execz .LBB11_6
; GFX1164_ITERATIVE-NEXT: ; %bb.3:
@@ -8052,7 +8100,7 @@ define amdgpu_kernel void @sub_i64_varying(ptr addrspace(1) %out, ptr addrspace(
; GFX1164_ITERATIVE-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1164_ITERATIVE-NEXT: v_mov_b32_e32 v8, v0
; GFX1164_ITERATIVE-NEXT: v_mov_b32_e32 v9, v1
-; GFX1164_ITERATIVE-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1164_ITERATIVE-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1164_ITERATIVE-NEXT: v_sub_co_u32 v6, vcc, v8, s8
; GFX1164_ITERATIVE-NEXT: v_subrev_co_ci_u32_e64 v7, null, s9, v9, vcc
; GFX1164_ITERATIVE-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_2) | instid1(VALU_DEP_4)
@@ -8066,9 +8114,10 @@ define amdgpu_kernel void @sub_i64_varying(ptr addrspace(1) %out, ptr addrspace(
; GFX1164_ITERATIVE-NEXT: buffer_gl0_inv
; GFX1164_ITERATIVE-NEXT: v_cmp_eq_u64_e32 vcc, v[0:1], v[8:9]
; GFX1164_ITERATIVE-NEXT: s_waitcnt_depctr depctr_va_vcc(0)
+; GFX1164_ITERATIVE-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1164_ITERATIVE-NEXT: s_or_b64 s[12:13], vcc, s[12:13]
-; GFX1164_ITERATIVE-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164_ITERATIVE-NEXT: s_and_not1_b64 exec, exec, s[12:13]
+; GFX1164_ITERATIVE-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164_ITERATIVE-NEXT: s_cbranch_execnz .LBB11_4
; GFX1164_ITERATIVE-NEXT: ; %bb.5: ; %Flow
; GFX1164_ITERATIVE-NEXT: s_or_b64 exec, exec, s[12:13]
@@ -8129,7 +8178,7 @@ define amdgpu_kernel void @sub_i64_varying(ptr addrspace(1) %out, ptr addrspace(
; GFX1132_ITERATIVE-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132_ITERATIVE-NEXT: v_dual_mov_b32 v9, v1 :: v_dual_mov_b32 v8, v0
; GFX1132_ITERATIVE-NEXT: v_sub_co_u32 v6, vcc_lo, v8, s8
-; GFX1132_ITERATIVE-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
+; GFX1132_ITERATIVE-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX1132_ITERATIVE-NEXT: v_subrev_co_ci_u32_e64 v7, null, s9, v9, vcc_lo
; GFX1132_ITERATIVE-NEXT: v_dual_mov_b32 v2, v8 :: v_dual_mov_b32 v3, v9
; GFX1132_ITERATIVE-NEXT: v_dual_mov_b32 v0, v6 :: v_dual_mov_b32 v1, v7
@@ -8139,7 +8188,7 @@ define amdgpu_kernel void @sub_i64_varying(ptr addrspace(1) %out, ptr addrspace(
; GFX1132_ITERATIVE-NEXT: buffer_gl0_inv
; GFX1132_ITERATIVE-NEXT: v_cmp_eq_u64_e32 vcc_lo, v[0:1], v[8:9]
; GFX1132_ITERATIVE-NEXT: s_or_b32 s11, vcc_lo, s11
-; GFX1132_ITERATIVE-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1132_ITERATIVE-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1132_ITERATIVE-NEXT: s_and_not1_b32 exec_lo, exec_lo, s11
; GFX1132_ITERATIVE-NEXT: s_cbranch_execnz .LBB11_4
; GFX1132_ITERATIVE-NEXT: ; %bb.5: ; %Flow
@@ -8151,7 +8200,7 @@ define amdgpu_kernel void @sub_i64_varying(ptr addrspace(1) %out, ptr addrspace(
; GFX1132_ITERATIVE-NEXT: v_readfirstlane_b32 s2, v0
; GFX1132_ITERATIVE-NEXT: v_readfirstlane_b32 s3, v1
; GFX1132_ITERATIVE-NEXT: v_sub_co_u32 v0, vcc_lo, s2, v4
-; GFX1132_ITERATIVE-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1132_ITERATIVE-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX1132_ITERATIVE-NEXT: v_sub_co_ci_u32_e64 v1, null, s3, v5, vcc_lo
; GFX1132_ITERATIVE-NEXT: s_mov_b32 s3, 0x31016000
; GFX1132_ITERATIVE-NEXT: s_mov_b32 s2, -1
@@ -8185,8 +8234,8 @@ define amdgpu_kernel void @sub_i64_varying(ptr addrspace(1) %out, ptr addrspace(
; GFX1264_ITERATIVE-NEXT: v_mbcnt_hi_u32_b32 v2, exec_hi, v2
; GFX1264_ITERATIVE-NEXT: v_cmp_eq_u32_e32 vcc, 0, v2
; GFX1264_ITERATIVE-NEXT: ; implicit-def: $vgpr2_vgpr3
+; GFX1264_ITERATIVE-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1264_ITERATIVE-NEXT: s_and_saveexec_b64 s[4:5], vcc
-; GFX1264_ITERATIVE-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1264_ITERATIVE-NEXT: s_xor_b64 s[4:5], exec, s[4:5]
; GFX1264_ITERATIVE-NEXT: s_cbranch_execz .LBB11_4
; GFX1264_ITERATIVE-NEXT: ; %bb.3:
@@ -8205,7 +8254,7 @@ define amdgpu_kernel void @sub_i64_varying(ptr addrspace(1) %out, ptr addrspace(
; GFX1264_ITERATIVE-NEXT: s_wait_kmcnt 0x0
; GFX1264_ITERATIVE-NEXT: v_readfirstlane_b32 s2, v2
; GFX1264_ITERATIVE-NEXT: v_readfirstlane_b32 s3, v3
-; GFX1264_ITERATIVE-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1264_ITERATIVE-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1264_ITERATIVE-NEXT: v_sub_co_u32 v0, vcc, s2, v0
; GFX1264_ITERATIVE-NEXT: v_sub_co_ci_u32_e64 v1, null, s3, v1, vcc
; GFX1264_ITERATIVE-NEXT: s_mov_b32 s3, 0x31016000
@@ -8257,7 +8306,7 @@ define amdgpu_kernel void @sub_i64_varying(ptr addrspace(1) %out, ptr addrspace(
; GFX1232_ITERATIVE-NEXT: s_wait_kmcnt 0x0
; GFX1232_ITERATIVE-NEXT: v_readfirstlane_b32 s2, v2
; GFX1232_ITERATIVE-NEXT: v_readfirstlane_b32 s3, v3
-; GFX1232_ITERATIVE-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1232_ITERATIVE-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1232_ITERATIVE-NEXT: v_sub_co_u32 v0, vcc_lo, s2, v0
; GFX1232_ITERATIVE-NEXT: v_sub_co_ci_u32_e64 v1, null, s3, v1, vcc_lo
; GFX1232_ITERATIVE-NEXT: s_mov_b32 s3, 0x31016000
@@ -8812,6 +8861,7 @@ define amdgpu_kernel void @sub_i64_varying(ptr addrspace(1) %out, ptr addrspace(
; GFX1164_DPP-NEXT: s_mov_b64 s[10:11], exec
; GFX1164_DPP-NEXT: ; implicit-def: $vgpr8_vgpr9
; GFX1164_DPP-NEXT: v_cmpx_eq_u32_e32 0, v0
+; GFX1164_DPP-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1164_DPP-NEXT: s_cbranch_execz .LBB11_4
; GFX1164_DPP-NEXT: ; %bb.1:
; GFX1164_DPP-NEXT: s_waitcnt lgkmcnt(0)
@@ -8841,9 +8891,10 @@ define amdgpu_kernel void @sub_i64_varying(ptr addrspace(1) %out, ptr addrspace(
; GFX1164_DPP-NEXT: buffer_gl0_inv
; GFX1164_DPP-NEXT: v_cmp_eq_u64_e32 vcc, v[8:9], v[12:13]
; GFX1164_DPP-NEXT: s_waitcnt_depctr depctr_va_vcc(0)
+; GFX1164_DPP-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1164_DPP-NEXT: s_or_b64 s[12:13], vcc, s[12:13]
-; GFX1164_DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164_DPP-NEXT: s_and_not1_b64 exec, exec, s[12:13]
+; GFX1164_DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164_DPP-NEXT: s_cbranch_execnz .LBB11_2
; GFX1164_DPP-NEXT: ; %bb.3: ; %Flow
; GFX1164_DPP-NEXT: s_or_b64 exec, exec, s[12:13]
@@ -8923,6 +8974,7 @@ define amdgpu_kernel void @sub_i64_varying(ptr addrspace(1) %out, ptr addrspace(
; GFX1132_DPP-NEXT: s_mov_b32 s10, exec_lo
; GFX1132_DPP-NEXT: ; implicit-def: $vgpr8_vgpr9
; GFX1132_DPP-NEXT: v_cmpx_eq_u32_e32 0, v0
+; GFX1132_DPP-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1132_DPP-NEXT: s_cbranch_execz .LBB11_4
; GFX1132_DPP-NEXT: ; %bb.1:
; GFX1132_DPP-NEXT: s_waitcnt lgkmcnt(0)
@@ -8938,7 +8990,7 @@ define amdgpu_kernel void @sub_i64_varying(ptr addrspace(1) %out, ptr addrspace(
; GFX1132_DPP-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132_DPP-NEXT: v_dual_mov_b32 v13, v9 :: v_dual_mov_b32 v12, v8
; GFX1132_DPP-NEXT: v_sub_co_u32 v10, vcc_lo, v12, s8
-; GFX1132_DPP-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX1132_DPP-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1132_DPP-NEXT: v_subrev_co_ci_u32_e64 v11, null, s9, v13, vcc_lo
; GFX1132_DPP-NEXT: v_swap_b32 v8, v10
; GFX1132_DPP-NEXT: s_delay_alu instid0(VALU_DEP_2)
@@ -8949,7 +9001,7 @@ define amdgpu_kernel void @sub_i64_varying(ptr addrspace(1) %out, ptr addrspace(
; GFX1132_DPP-NEXT: buffer_gl0_inv
; GFX1132_DPP-NEXT: v_cmp_eq_u64_e32 vcc_lo, v[8:9], v[12:13]
; GFX1132_DPP-NEXT: s_or_b32 s11, vcc_lo, s11
-; GFX1132_DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1132_DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1132_DPP-NEXT: s_and_not1_b32 exec_lo, exec_lo, s11
; GFX1132_DPP-NEXT: s_cbranch_execnz .LBB11_2
; GFX1132_DPP-NEXT: ; %bb.3: ; %Flow
@@ -8962,7 +9014,7 @@ define amdgpu_kernel void @sub_i64_varying(ptr addrspace(1) %out, ptr addrspace(
; GFX1132_DPP-NEXT: v_dual_mov_b32 v10, v6 :: v_dual_mov_b32 v11, v7
; GFX1132_DPP-NEXT: v_readfirstlane_b32 s3, v9
; GFX1132_DPP-NEXT: v_sub_co_u32 v8, vcc_lo, s2, v10
-; GFX1132_DPP-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1132_DPP-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX1132_DPP-NEXT: v_sub_co_ci_u32_e64 v9, null, s3, v11, vcc_lo
; GFX1132_DPP-NEXT: s_mov_b32 s3, 0x31016000
; GFX1132_DPP-NEXT: s_mov_b32 s2, -1
@@ -9054,6 +9106,7 @@ define amdgpu_kernel void @sub_i64_varying(ptr addrspace(1) %out, ptr addrspace(
; GFX1264_DPP-NEXT: s_mov_b64 s[8:9], exec
; GFX1264_DPP-NEXT: ; implicit-def: $vgpr6_vgpr7
; GFX1264_DPP-NEXT: v_cmpx_eq_u32_e32 0, v0
+; GFX1264_DPP-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1264_DPP-NEXT: s_cbranch_execz .LBB11_2
; GFX1264_DPP-NEXT: ; %bb.1:
; GFX1264_DPP-NEXT: v_mov_b32_e32 v7, s5
@@ -9144,6 +9197,7 @@ define amdgpu_kernel void @sub_i64_varying(ptr addrspace(1) %out, ptr addrspace(
; GFX1232_DPP-NEXT: s_mov_b32 s8, exec_lo
; GFX1232_DPP-NEXT: ; implicit-def: $vgpr8_vgpr9
; GFX1232_DPP-NEXT: v_cmpx_eq_u32_e32 0, v0
+; GFX1232_DPP-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1232_DPP-NEXT: s_cbranch_execz .LBB11_2
; GFX1232_DPP-NEXT: ; %bb.1:
; GFX1232_DPP-NEXT: v_dual_mov_b32 v9, s5 :: v_dual_mov_b32 v8, s4
@@ -9249,6 +9303,7 @@ define amdgpu_kernel void @sub_i64_varying(ptr addrspace(1) %out, ptr addrspace(
; GFX1364-NEXT: s_mov_b64 s[8:9], exec
; GFX1364-NEXT: ; implicit-def: $vgpr6_vgpr7
; GFX1364-NEXT: v_cmpx_eq_u32_e32 0, v0
+; GFX1364-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1364-NEXT: s_cbranch_execz .LBB11_2
; GFX1364-NEXT: ; %bb.1:
; GFX1364-NEXT: v_mov_b32_e32 v7, s5
@@ -9270,7 +9325,7 @@ define amdgpu_kernel void @sub_i64_varying(ptr addrspace(1) %out, ptr addrspace(
; GFX1364-NEXT: v_mov_b32_e32 v8, v4
; GFX1364-NEXT: v_mov_b32_e32 v9, v5
; GFX1364-NEXT: v_readfirstlane_b32 s3, v7
-; GFX1364-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1364-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1364-NEXT: v_sub_co_u32 v6, vcc, s2, v8
; GFX1364-NEXT: v_sub_co_ci_u32_e64 v7, null, s3, v9, vcc
; GFX1364-NEXT: s_mov_b32 s3, 0x31016000
@@ -9336,6 +9391,7 @@ define amdgpu_kernel void @sub_i64_varying(ptr addrspace(1) %out, ptr addrspace(
; GFX1332-NEXT: s_mov_b32 s8, exec_lo
; GFX1332-NEXT: ; implicit-def: $vgpr8_vgpr9
; GFX1332-NEXT: v_cmpx_eq_u32_e32 0, v0
+; GFX1332-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1332-NEXT: s_cbranch_execz .LBB11_2
; GFX1332-NEXT: ; %bb.1:
; GFX1332-NEXT: v_dual_mov_b32 v9, s5 :: v_dual_mov_b32 v8, s4
@@ -9355,7 +9411,7 @@ define amdgpu_kernel void @sub_i64_varying(ptr addrspace(1) %out, ptr addrspace(
; GFX1332-NEXT: v_readfirstlane_b32 s2, v8
; GFX1332-NEXT: v_dual_mov_b32 v10, v6 :: v_dual_mov_b32 v11, v7
; GFX1332-NEXT: v_readfirstlane_b32 s3, v9
-; GFX1332-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1332-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1332-NEXT: v_sub_co_u32 v8, vcc_lo, s2, v10
; GFX1332-NEXT: v_sub_co_ci_u32_e64 v9, null, s3, v11, vcc_lo
; GFX1332-NEXT: s_mov_b32 s3, 0x31016000
@@ -9637,6 +9693,7 @@ define amdgpu_kernel void @uniform_or_i8(ptr addrspace(1) %result, ptr addrspace
; GFX1164-TRUE16-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
; GFX1164-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc, 0, v0
; GFX1164-TRUE16-NEXT: ; implicit-def: $vgpr0_lo16
+; GFX1164-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1164-TRUE16-NEXT: s_and_saveexec_b64 s[2:3], vcc
; GFX1164-TRUE16-NEXT: s_cbranch_execz .LBB12_4
; GFX1164-TRUE16-NEXT: ; %bb.1:
@@ -9664,20 +9721,22 @@ define amdgpu_kernel void @uniform_or_i8(ptr addrspace(1) %result, ptr addrspace
; GFX1164-TRUE16-NEXT: v_cmp_eq_u32_e64 s[0:1], v2, v1
; GFX1164-TRUE16-NEXT: v_mov_b32_e32 v1, v2
; GFX1164-TRUE16-NEXT: s_or_b64 s[12:13], s[0:1], s[12:13]
-; GFX1164-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1164-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1164-TRUE16-NEXT: s_and_not1_b64 exec, exec, s[12:13]
; GFX1164-TRUE16-NEXT: s_cbranch_execnz .LBB12_2
; GFX1164-TRUE16-NEXT: ; %bb.3: ; %atomicrmw.end
; GFX1164-TRUE16-NEXT: s_or_b64 exec, exec, s[12:13]
+; GFX1164-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164-TRUE16-NEXT: v_lshrrev_b32_e32 v0, s10, v2
; GFX1164-TRUE16-NEXT: .LBB12_4: ; %Flow
; GFX1164-TRUE16-NEXT: s_or_b64 exec, exec, s[2:3]
-; GFX1164-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_4) | instid1(VALU_DEP_1)
+; GFX1164-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX1164-TRUE16-NEXT: v_readfirstlane_b32 s0, v0
; GFX1164-TRUE16-NEXT: s_waitcnt lgkmcnt(0)
; GFX1164-TRUE16-NEXT: v_cndmask_b16 v0.l, s14, 0, vcc
; GFX1164-TRUE16-NEXT: s_mov_b32 s11, 0x31016000
; GFX1164-TRUE16-NEXT: s_mov_b32 s10, -1
+; GFX1164-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1164-TRUE16-NEXT: v_or_b16 v0.l, s0, v0.l
; GFX1164-TRUE16-NEXT: buffer_store_b8 v0, off, s[8:11], 0
; GFX1164-TRUE16-NEXT: s_endpgm
@@ -9692,6 +9751,7 @@ define amdgpu_kernel void @uniform_or_i8(ptr addrspace(1) %result, ptr addrspace
; GFX1164-FAKE16-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
; GFX1164-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc, 0, v0
; GFX1164-FAKE16-NEXT: ; implicit-def: $vgpr0
+; GFX1164-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1164-FAKE16-NEXT: s_and_saveexec_b64 s[2:3], vcc
; GFX1164-FAKE16-NEXT: s_cbranch_execz .LBB12_4
; GFX1164-FAKE16-NEXT: ; %bb.1:
@@ -9719,20 +9779,22 @@ define amdgpu_kernel void @uniform_or_i8(ptr addrspace(1) %result, ptr addrspace
; GFX1164-FAKE16-NEXT: v_cmp_eq_u32_e64 s[0:1], v2, v1
; GFX1164-FAKE16-NEXT: v_mov_b32_e32 v1, v2
; GFX1164-FAKE16-NEXT: s_or_b64 s[12:13], s[0:1], s[12:13]
-; GFX1164-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1164-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1164-FAKE16-NEXT: s_and_not1_b64 exec, exec, s[12:13]
; GFX1164-FAKE16-NEXT: s_cbranch_execnz .LBB12_2
; GFX1164-FAKE16-NEXT: ; %bb.3: ; %atomicrmw.end
; GFX1164-FAKE16-NEXT: s_or_b64 exec, exec, s[12:13]
+; GFX1164-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164-FAKE16-NEXT: v_lshrrev_b32_e32 v0, s10, v2
; GFX1164-FAKE16-NEXT: .LBB12_4: ; %Flow
; GFX1164-FAKE16-NEXT: s_or_b64 exec, exec, s[2:3]
-; GFX1164-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_4) | instid1(VALU_DEP_1)
+; GFX1164-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX1164-FAKE16-NEXT: v_readfirstlane_b32 s0, v0
; GFX1164-FAKE16-NEXT: s_waitcnt lgkmcnt(0)
; GFX1164-FAKE16-NEXT: v_cndmask_b32_e64 v0, s14, 0, vcc
; GFX1164-FAKE16-NEXT: s_mov_b32 s11, 0x31016000
; GFX1164-FAKE16-NEXT: s_mov_b32 s10, -1
+; GFX1164-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1164-FAKE16-NEXT: v_or_b32_e32 v0, s0, v0
; GFX1164-FAKE16-NEXT: buffer_store_b8 v0, off, s[8:11], 0
; GFX1164-FAKE16-NEXT: s_endpgm
@@ -9774,20 +9836,22 @@ define amdgpu_kernel void @uniform_or_i8(ptr addrspace(1) %result, ptr addrspace
; GFX1132-TRUE16-NEXT: v_cmp_eq_u32_e64 s0, v2, v1
; GFX1132-TRUE16-NEXT: v_mov_b32_e32 v1, v2
; GFX1132-TRUE16-NEXT: s_or_b32 s3, s0, s3
-; GFX1132-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1132-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1132-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s3
; GFX1132-TRUE16-NEXT: s_cbranch_execnz .LBB12_2
; GFX1132-TRUE16-NEXT: ; %bb.3: ; %atomicrmw.end
; GFX1132-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s3
+; GFX1132-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1132-TRUE16-NEXT: v_lshrrev_b32_e32 v0, s10, v2
; GFX1132-TRUE16-NEXT: .LBB12_4: ; %Flow
; GFX1132-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s2
-; GFX1132-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_4) | instid1(VALU_DEP_1)
+; GFX1132-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX1132-TRUE16-NEXT: v_readfirstlane_b32 s0, v0
; GFX1132-TRUE16-NEXT: s_waitcnt lgkmcnt(0)
; GFX1132-TRUE16-NEXT: v_cndmask_b16 v0.l, s1, 0, vcc_lo
; GFX1132-TRUE16-NEXT: s_mov_b32 s11, 0x31016000
; GFX1132-TRUE16-NEXT: s_mov_b32 s10, -1
+; GFX1132-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1132-TRUE16-NEXT: v_or_b16 v0.l, s0, v0.l
; GFX1132-TRUE16-NEXT: buffer_store_b8 v0, off, s[8:11], 0
; GFX1132-TRUE16-NEXT: s_endpgm
@@ -9829,20 +9893,22 @@ define amdgpu_kernel void @uniform_or_i8(ptr addrspace(1) %result, ptr addrspace
; GFX1132-FAKE16-NEXT: v_cmp_eq_u32_e64 s0, v2, v1
; GFX1132-FAKE16-NEXT: v_mov_b32_e32 v1, v2
; GFX1132-FAKE16-NEXT: s_or_b32 s3, s0, s3
-; GFX1132-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1132-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1132-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s3
; GFX1132-FAKE16-NEXT: s_cbranch_execnz .LBB12_2
; GFX1132-FAKE16-NEXT: ; %bb.3: ; %atomicrmw.end
; GFX1132-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s3
+; GFX1132-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1132-FAKE16-NEXT: v_lshrrev_b32_e32 v0, s10, v2
; GFX1132-FAKE16-NEXT: .LBB12_4: ; %Flow
; GFX1132-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s2
-; GFX1132-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_4) | instid1(VALU_DEP_1)
+; GFX1132-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX1132-FAKE16-NEXT: v_readfirstlane_b32 s0, v0
; GFX1132-FAKE16-NEXT: s_waitcnt lgkmcnt(0)
; GFX1132-FAKE16-NEXT: v_cndmask_b32_e64 v0, s1, 0, vcc_lo
; GFX1132-FAKE16-NEXT: s_mov_b32 s11, 0x31016000
; GFX1132-FAKE16-NEXT: s_mov_b32 s10, -1
+; GFX1132-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1132-FAKE16-NEXT: v_or_b32_e32 v0, s0, v0
; GFX1132-FAKE16-NEXT: buffer_store_b8 v0, off, s[8:11], 0
; GFX1132-FAKE16-NEXT: s_endpgm
@@ -9857,6 +9923,7 @@ define amdgpu_kernel void @uniform_or_i8(ptr addrspace(1) %result, ptr addrspace
; GFX1264-TRUE16-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
; GFX1264-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc, 0, v0
; GFX1264-TRUE16-NEXT: ; implicit-def: $vgpr0_lo16
+; GFX1264-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1264-TRUE16-NEXT: s_and_saveexec_b64 s[2:3], vcc
; GFX1264-TRUE16-NEXT: s_cbranch_execz .LBB12_4
; GFX1264-TRUE16-NEXT: ; %bb.1:
@@ -9884,15 +9951,16 @@ define amdgpu_kernel void @uniform_or_i8(ptr addrspace(1) %result, ptr addrspace
; GFX1264-TRUE16-NEXT: v_cmp_eq_u32_e64 s[0:1], v2, v1
; GFX1264-TRUE16-NEXT: v_mov_b32_e32 v1, v2
; GFX1264-TRUE16-NEXT: s_or_b64 s[12:13], s[0:1], s[12:13]
-; GFX1264-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1264-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1264-TRUE16-NEXT: s_and_not1_b64 exec, exec, s[12:13]
; GFX1264-TRUE16-NEXT: s_cbranch_execnz .LBB12_2
; GFX1264-TRUE16-NEXT: ; %bb.3: ; %atomicrmw.end
; GFX1264-TRUE16-NEXT: s_or_b64 exec, exec, s[12:13]
+; GFX1264-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1264-TRUE16-NEXT: v_lshrrev_b32_e32 v0, s10, v2
; GFX1264-TRUE16-NEXT: .LBB12_4: ; %Flow
; GFX1264-TRUE16-NEXT: s_or_b64 exec, exec, s[2:3]
-; GFX1264-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1264-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX1264-TRUE16-NEXT: v_readfirstlane_b32 s0, v0
; GFX1264-TRUE16-NEXT: s_wait_kmcnt 0x0
; GFX1264-TRUE16-NEXT: v_cndmask_b16 v0.l, s14, 0, vcc
@@ -9914,6 +9982,7 @@ define amdgpu_kernel void @uniform_or_i8(ptr addrspace(1) %result, ptr addrspace
; GFX1264-FAKE16-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
; GFX1264-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc, 0, v0
; GFX1264-FAKE16-NEXT: ; implicit-def: $vgpr0
+; GFX1264-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1264-FAKE16-NEXT: s_and_saveexec_b64 s[2:3], vcc
; GFX1264-FAKE16-NEXT: s_cbranch_execz .LBB12_4
; GFX1264-FAKE16-NEXT: ; %bb.1:
@@ -9941,15 +10010,16 @@ define amdgpu_kernel void @uniform_or_i8(ptr addrspace(1) %result, ptr addrspace
; GFX1264-FAKE16-NEXT: v_cmp_eq_u32_e64 s[0:1], v2, v1
; GFX1264-FAKE16-NEXT: v_mov_b32_e32 v1, v2
; GFX1264-FAKE16-NEXT: s_or_b64 s[12:13], s[0:1], s[12:13]
-; GFX1264-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1264-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1264-FAKE16-NEXT: s_and_not1_b64 exec, exec, s[12:13]
; GFX1264-FAKE16-NEXT: s_cbranch_execnz .LBB12_2
; GFX1264-FAKE16-NEXT: ; %bb.3: ; %atomicrmw.end
; GFX1264-FAKE16-NEXT: s_or_b64 exec, exec, s[12:13]
+; GFX1264-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1264-FAKE16-NEXT: v_lshrrev_b32_e32 v0, s10, v2
; GFX1264-FAKE16-NEXT: .LBB12_4: ; %Flow
; GFX1264-FAKE16-NEXT: s_or_b64 exec, exec, s[2:3]
-; GFX1264-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1264-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX1264-FAKE16-NEXT: v_readfirstlane_b32 s0, v0
; GFX1264-FAKE16-NEXT: s_wait_kmcnt 0x0
; GFX1264-FAKE16-NEXT: v_cndmask_b32_e64 v0, s14, 0, vcc
@@ -9998,15 +10068,16 @@ define amdgpu_kernel void @uniform_or_i8(ptr addrspace(1) %result, ptr addrspace
; GFX1232-TRUE16-NEXT: v_cmp_eq_u32_e64 s0, v2, v1
; GFX1232-TRUE16-NEXT: v_mov_b32_e32 v1, v2
; GFX1232-TRUE16-NEXT: s_or_b32 s3, s0, s3
-; GFX1232-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1232-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1232-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s3
; GFX1232-TRUE16-NEXT: s_cbranch_execnz .LBB12_2
; GFX1232-TRUE16-NEXT: ; %bb.3: ; %atomicrmw.end
; GFX1232-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s3
+; GFX1232-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1232-TRUE16-NEXT: v_lshrrev_b32_e32 v0, s10, v2
; GFX1232-TRUE16-NEXT: .LBB12_4: ; %Flow
; GFX1232-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s2
-; GFX1232-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1232-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX1232-TRUE16-NEXT: v_readfirstlane_b32 s0, v0
; GFX1232-TRUE16-NEXT: s_wait_kmcnt 0x0
; GFX1232-TRUE16-NEXT: v_cndmask_b16 v0.l, s1, 0, vcc_lo
@@ -10055,15 +10126,16 @@ define amdgpu_kernel void @uniform_or_i8(ptr addrspace(1) %result, ptr addrspace
; GFX1232-FAKE16-NEXT: v_cmp_eq_u32_e64 s0, v2, v1
; GFX1232-FAKE16-NEXT: v_mov_b32_e32 v1, v2
; GFX1232-FAKE16-NEXT: s_or_b32 s3, s0, s3
-; GFX1232-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1232-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1232-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s3
; GFX1232-FAKE16-NEXT: s_cbranch_execnz .LBB12_2
; GFX1232-FAKE16-NEXT: ; %bb.3: ; %atomicrmw.end
; GFX1232-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s3
+; GFX1232-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1232-FAKE16-NEXT: v_lshrrev_b32_e32 v0, s10, v2
; GFX1232-FAKE16-NEXT: .LBB12_4: ; %Flow
; GFX1232-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s2
-; GFX1232-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1232-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX1232-FAKE16-NEXT: v_readfirstlane_b32 s0, v0
; GFX1232-FAKE16-NEXT: s_wait_kmcnt 0x0
; GFX1232-FAKE16-NEXT: v_cndmask_b32_e64 v0, s1, 0, vcc_lo
@@ -10085,6 +10157,7 @@ define amdgpu_kernel void @uniform_or_i8(ptr addrspace(1) %result, ptr addrspace
; GFX1364-TRUE16-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
; GFX1364-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc, 0, v0
; GFX1364-TRUE16-NEXT: ; implicit-def: $vgpr0_lo16
+; GFX1364-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1364-TRUE16-NEXT: s_and_saveexec_b64 s[2:3], vcc
; GFX1364-TRUE16-NEXT: s_cbranch_execz .LBB12_4
; GFX1364-TRUE16-NEXT: ; %bb.1:
@@ -10111,20 +10184,22 @@ define amdgpu_kernel void @uniform_or_i8(ptr addrspace(1) %result, ptr addrspace
; GFX1364-TRUE16-NEXT: v_cmp_eq_u32_e64 s[0:1], v2, v1
; GFX1364-TRUE16-NEXT: v_mov_b32_e32 v1, v2
; GFX1364-TRUE16-NEXT: s_or_b64 s[10:11], s[0:1], s[10:11]
-; GFX1364-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1364-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1364-TRUE16-NEXT: s_and_not1_b64 exec, exec, s[10:11]
; GFX1364-TRUE16-NEXT: s_cbranch_execnz .LBB12_2
; GFX1364-TRUE16-NEXT: ; %bb.3: ; %atomicrmw.end
; GFX1364-TRUE16-NEXT: s_or_b64 exec, exec, s[10:11]
+; GFX1364-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1364-TRUE16-NEXT: v_lshrrev_b32_e32 v0, s13, v2
; GFX1364-TRUE16-NEXT: .LBB12_4: ; %Flow
; GFX1364-TRUE16-NEXT: s_or_b64 exec, exec, s[2:3]
-; GFX1364-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_4) | instid1(VALU_DEP_1)
+; GFX1364-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX1364-TRUE16-NEXT: v_readfirstlane_b32 s0, v0
; GFX1364-TRUE16-NEXT: s_wait_kmcnt 0x0
; GFX1364-TRUE16-NEXT: v_cndmask_b16 v0.l, s12, 0, vcc
; GFX1364-TRUE16-NEXT: s_mov_b32 s11, 0x31016000
; GFX1364-TRUE16-NEXT: s_mov_b32 s10, -1
+; GFX1364-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1364-TRUE16-NEXT: v_or_b16 v0.l, s0, v0.l
; GFX1364-TRUE16-NEXT: buffer_store_b8 v0, off, s[8:11], null
; GFX1364-TRUE16-NEXT: s_endpgm
@@ -10139,6 +10214,7 @@ define amdgpu_kernel void @uniform_or_i8(ptr addrspace(1) %result, ptr addrspace
; GFX1364-FAKE16-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
; GFX1364-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc, 0, v0
; GFX1364-FAKE16-NEXT: ; implicit-def: $vgpr0
+; GFX1364-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1364-FAKE16-NEXT: s_and_saveexec_b64 s[2:3], vcc
; GFX1364-FAKE16-NEXT: s_cbranch_execz .LBB12_4
; GFX1364-FAKE16-NEXT: ; %bb.1:
@@ -10165,20 +10241,22 @@ define amdgpu_kernel void @uniform_or_i8(ptr addrspace(1) %result, ptr addrspace
; GFX1364-FAKE16-NEXT: v_cmp_eq_u32_e64 s[0:1], v2, v1
; GFX1364-FAKE16-NEXT: v_mov_b32_e32 v1, v2
; GFX1364-FAKE16-NEXT: s_or_b64 s[10:11], s[0:1], s[10:11]
-; GFX1364-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1364-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1364-FAKE16-NEXT: s_and_not1_b64 exec, exec, s[10:11]
; GFX1364-FAKE16-NEXT: s_cbranch_execnz .LBB12_2
; GFX1364-FAKE16-NEXT: ; %bb.3: ; %atomicrmw.end
; GFX1364-FAKE16-NEXT: s_or_b64 exec, exec, s[10:11]
+; GFX1364-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1364-FAKE16-NEXT: v_lshrrev_b32_e32 v0, s13, v2
; GFX1364-FAKE16-NEXT: .LBB12_4: ; %Flow
; GFX1364-FAKE16-NEXT: s_or_b64 exec, exec, s[2:3]
-; GFX1364-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_4) | instid1(VALU_DEP_1)
+; GFX1364-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX1364-FAKE16-NEXT: v_readfirstlane_b32 s0, v0
; GFX1364-FAKE16-NEXT: s_wait_kmcnt 0x0
; GFX1364-FAKE16-NEXT: v_cndmask_b32_e64 v0, s12, 0, vcc
; GFX1364-FAKE16-NEXT: s_mov_b32 s11, 0x31016000
; GFX1364-FAKE16-NEXT: s_mov_b32 s10, -1
+; GFX1364-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1364-FAKE16-NEXT: v_or_b32_e32 v0, s0, v0
; GFX1364-FAKE16-NEXT: buffer_store_b8 v0, off, s[8:11], null
; GFX1364-FAKE16-NEXT: s_endpgm
@@ -10217,20 +10295,22 @@ define amdgpu_kernel void @uniform_or_i8(ptr addrspace(1) %result, ptr addrspace
; GFX1332-TRUE16-NEXT: v_cmp_eq_u32_e64 s0, v2, v1
; GFX1332-TRUE16-NEXT: v_mov_b32_e32 v1, v2
; GFX1332-TRUE16-NEXT: s_or_b32 s3, s0, s3
-; GFX1332-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1332-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1332-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s3
; GFX1332-TRUE16-NEXT: s_cbranch_execnz .LBB12_2
; GFX1332-TRUE16-NEXT: ; %bb.3: ; %atomicrmw.end
; GFX1332-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s3
+; GFX1332-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1332-TRUE16-NEXT: v_lshrrev_b32_e32 v0, s10, v2
; GFX1332-TRUE16-NEXT: .LBB12_4: ; %Flow
; GFX1332-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s2
-; GFX1332-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_4) | instid1(VALU_DEP_1)
+; GFX1332-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX1332-TRUE16-NEXT: v_readfirstlane_b32 s0, v0
; GFX1332-TRUE16-NEXT: s_wait_kmcnt 0x0
; GFX1332-TRUE16-NEXT: v_cndmask_b16 v0.l, s1, 0, vcc_lo
; GFX1332-TRUE16-NEXT: s_mov_b32 s11, 0x31016000
; GFX1332-TRUE16-NEXT: s_mov_b32 s10, -1
+; GFX1332-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1332-TRUE16-NEXT: v_or_b16 v0.l, s0, v0.l
; GFX1332-TRUE16-NEXT: buffer_store_b8 v0, off, s[8:11], null
; GFX1332-TRUE16-NEXT: s_endpgm
@@ -10269,20 +10349,22 @@ define amdgpu_kernel void @uniform_or_i8(ptr addrspace(1) %result, ptr addrspace
; GFX1332-FAKE16-NEXT: v_cmp_eq_u32_e64 s0, v2, v1
; GFX1332-FAKE16-NEXT: v_mov_b32_e32 v1, v2
; GFX1332-FAKE16-NEXT: s_or_b32 s3, s0, s3
-; GFX1332-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1332-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1332-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s3
; GFX1332-FAKE16-NEXT: s_cbranch_execnz .LBB12_2
; GFX1332-FAKE16-NEXT: ; %bb.3: ; %atomicrmw.end
; GFX1332-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s3
+; GFX1332-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1332-FAKE16-NEXT: v_lshrrev_b32_e32 v0, s10, v2
; GFX1332-FAKE16-NEXT: .LBB12_4: ; %Flow
; GFX1332-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s2
-; GFX1332-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_4) | instid1(VALU_DEP_1)
+; GFX1332-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX1332-FAKE16-NEXT: v_readfirstlane_b32 s0, v0
; GFX1332-FAKE16-NEXT: s_wait_kmcnt 0x0
; GFX1332-FAKE16-NEXT: v_cndmask_b32_e64 v0, s1, 0, vcc_lo
; GFX1332-FAKE16-NEXT: s_mov_b32 s11, 0x31016000
; GFX1332-FAKE16-NEXT: s_mov_b32 s10, -1
+; GFX1332-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1332-FAKE16-NEXT: v_or_b32_e32 v0, s0, v0
; GFX1332-FAKE16-NEXT: buffer_store_b8 v0, off, s[8:11], null
; GFX1332-FAKE16-NEXT: s_endpgm
@@ -10597,6 +10679,7 @@ define amdgpu_kernel void @uniform_add_i8(ptr addrspace(1) %result, ptr addrspac
; GFX1164-TRUE16-NEXT: v_mbcnt_hi_u32_b32 v4, s7, v0
; GFX1164-TRUE16-NEXT: ; implicit-def: $vgpr0_lo16
; GFX1164-TRUE16-NEXT: v_cmpx_eq_u32_e32 0, v4
+; GFX1164-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1164-TRUE16-NEXT: s_cbranch_execz .LBB13_4
; GFX1164-TRUE16-NEXT: ; %bb.1:
; GFX1164-TRUE16-NEXT: s_waitcnt lgkmcnt(0)
@@ -10625,18 +10708,19 @@ define amdgpu_kernel void @uniform_add_i8(ptr addrspace(1) %result, ptr addrspac
; GFX1164-TRUE16-NEXT: s_waitcnt_depctr depctr_va_vdst(0)
; GFX1164-TRUE16-NEXT: v_and_or_b32 v0, v1, s14, v0
; GFX1164-TRUE16-NEXT: v_mov_b32_e32 v3, v1
-; GFX1164-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2)
+; GFX1164-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_4) | instid1(VALU_DEP_2)
; GFX1164-TRUE16-NEXT: v_mov_b32_e32 v2, v0
; GFX1164-TRUE16-NEXT: buffer_atomic_cmpswap_b32 v[2:3], off, s[4:7], 0 glc
; GFX1164-TRUE16-NEXT: s_waitcnt vmcnt(0)
; GFX1164-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc, v2, v1
; GFX1164-TRUE16-NEXT: v_mov_b32_e32 v1, v2
; GFX1164-TRUE16-NEXT: s_or_b64 s[10:11], vcc, s[10:11]
-; GFX1164-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1164-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1164-TRUE16-NEXT: s_and_not1_b64 exec, exec, s[10:11]
; GFX1164-TRUE16-NEXT: s_cbranch_execnz .LBB13_2
; GFX1164-TRUE16-NEXT: ; %bb.3: ; %atomicrmw.end
; GFX1164-TRUE16-NEXT: s_or_b64 exec, exec, s[10:11]
+; GFX1164-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164-TRUE16-NEXT: v_lshrrev_b32_e32 v0, s2, v2
; GFX1164-TRUE16-NEXT: .LBB13_4: ; %Flow
; GFX1164-TRUE16-NEXT: s_or_b64 exec, exec, s[8:9]
@@ -10661,6 +10745,7 @@ define amdgpu_kernel void @uniform_add_i8(ptr addrspace(1) %result, ptr addrspac
; GFX1164-FAKE16-NEXT: v_mbcnt_hi_u32_b32 v4, s7, v0
; GFX1164-FAKE16-NEXT: ; implicit-def: $vgpr0
; GFX1164-FAKE16-NEXT: v_cmpx_eq_u32_e32 0, v4
+; GFX1164-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1164-FAKE16-NEXT: s_cbranch_execz .LBB13_4
; GFX1164-FAKE16-NEXT: ; %bb.1:
; GFX1164-FAKE16-NEXT: s_waitcnt lgkmcnt(0)
@@ -10689,18 +10774,19 @@ define amdgpu_kernel void @uniform_add_i8(ptr addrspace(1) %result, ptr addrspac
; GFX1164-FAKE16-NEXT: s_waitcnt_depctr depctr_va_vdst(0)
; GFX1164-FAKE16-NEXT: v_and_or_b32 v0, v1, s14, v0
; GFX1164-FAKE16-NEXT: v_mov_b32_e32 v3, v1
-; GFX1164-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2)
+; GFX1164-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_4) | instid1(VALU_DEP_2)
; GFX1164-FAKE16-NEXT: v_mov_b32_e32 v2, v0
; GFX1164-FAKE16-NEXT: buffer_atomic_cmpswap_b32 v[2:3], off, s[4:7], 0 glc
; GFX1164-FAKE16-NEXT: s_waitcnt vmcnt(0)
; GFX1164-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc, v2, v1
; GFX1164-FAKE16-NEXT: v_mov_b32_e32 v1, v2
; GFX1164-FAKE16-NEXT: s_or_b64 s[10:11], vcc, s[10:11]
-; GFX1164-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1164-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1164-FAKE16-NEXT: s_and_not1_b64 exec, exec, s[10:11]
; GFX1164-FAKE16-NEXT: s_cbranch_execnz .LBB13_2
; GFX1164-FAKE16-NEXT: ; %bb.3: ; %atomicrmw.end
; GFX1164-FAKE16-NEXT: s_or_b64 exec, exec, s[10:11]
+; GFX1164-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164-FAKE16-NEXT: v_lshrrev_b32_e32 v0, s2, v2
; GFX1164-FAKE16-NEXT: .LBB13_4: ; %Flow
; GFX1164-FAKE16-NEXT: s_or_b64 exec, exec, s[8:9]
@@ -10723,7 +10809,7 @@ define amdgpu_kernel void @uniform_add_i8(ptr addrspace(1) %result, ptr addrspac
; GFX1132-TRUE16-NEXT: v_mbcnt_lo_u32_b32 v4, s6, 0
; GFX1132-TRUE16-NEXT: s_mov_b32 s9, exec_lo
; GFX1132-TRUE16-NEXT: ; implicit-def: $vgpr0_lo16
-; GFX1132-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1132-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132-TRUE16-NEXT: v_cmpx_eq_u32_e32 0, v4
; GFX1132-TRUE16-NEXT: s_cbranch_execz .LBB13_4
; GFX1132-TRUE16-NEXT: ; %bb.1:
@@ -10757,11 +10843,12 @@ define amdgpu_kernel void @uniform_add_i8(ptr addrspace(1) %result, ptr addrspac
; GFX1132-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v1
; GFX1132-TRUE16-NEXT: v_mov_b32_e32 v1, v2
; GFX1132-TRUE16-NEXT: s_or_b32 s10, vcc_lo, s10
-; GFX1132-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1132-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1132-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s10
; GFX1132-TRUE16-NEXT: s_cbranch_execnz .LBB13_2
; GFX1132-TRUE16-NEXT: ; %bb.3: ; %atomicrmw.end
; GFX1132-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s10
+; GFX1132-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1132-TRUE16-NEXT: v_lshrrev_b32_e32 v0, s2, v2
; GFX1132-TRUE16-NEXT: .LBB13_4: ; %Flow
; GFX1132-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s9
@@ -10784,7 +10871,7 @@ define amdgpu_kernel void @uniform_add_i8(ptr addrspace(1) %result, ptr addrspac
; GFX1132-FAKE16-NEXT: v_mbcnt_lo_u32_b32 v4, s6, 0
; GFX1132-FAKE16-NEXT: s_mov_b32 s9, exec_lo
; GFX1132-FAKE16-NEXT: ; implicit-def: $vgpr0
-; GFX1132-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1132-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132-FAKE16-NEXT: v_cmpx_eq_u32_e32 0, v4
; GFX1132-FAKE16-NEXT: s_cbranch_execz .LBB13_4
; GFX1132-FAKE16-NEXT: ; %bb.1:
@@ -10818,11 +10905,12 @@ define amdgpu_kernel void @uniform_add_i8(ptr addrspace(1) %result, ptr addrspac
; GFX1132-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v1
; GFX1132-FAKE16-NEXT: v_mov_b32_e32 v1, v2
; GFX1132-FAKE16-NEXT: s_or_b32 s10, vcc_lo, s10
-; GFX1132-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1132-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1132-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s10
; GFX1132-FAKE16-NEXT: s_cbranch_execnz .LBB13_2
; GFX1132-FAKE16-NEXT: ; %bb.3: ; %atomicrmw.end
; GFX1132-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s10
+; GFX1132-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1132-FAKE16-NEXT: v_lshrrev_b32_e32 v0, s2, v2
; GFX1132-FAKE16-NEXT: .LBB13_4: ; %Flow
; GFX1132-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s9
@@ -10847,6 +10935,7 @@ define amdgpu_kernel void @uniform_add_i8(ptr addrspace(1) %result, ptr addrspac
; GFX1264-TRUE16-NEXT: v_mbcnt_hi_u32_b32 v4, s7, v0
; GFX1264-TRUE16-NEXT: ; implicit-def: $vgpr0_lo16
; GFX1264-TRUE16-NEXT: v_cmpx_eq_u32_e32 0, v4
+; GFX1264-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1264-TRUE16-NEXT: s_cbranch_execz .LBB13_4
; GFX1264-TRUE16-NEXT: ; %bb.1:
; GFX1264-TRUE16-NEXT: s_wait_kmcnt 0x0
@@ -10883,12 +10972,14 @@ define amdgpu_kernel void @uniform_add_i8(ptr addrspace(1) %result, ptr addrspac
; GFX1264-TRUE16-NEXT: s_wait_loadcnt 0x0
; GFX1264-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc, v2, v1
; GFX1264-TRUE16-NEXT: v_mov_b32_e32 v1, v2
+; GFX1264-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1264-TRUE16-NEXT: s_or_b64 s[10:11], vcc, s[10:11]
-; GFX1264-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1264-TRUE16-NEXT: s_and_not1_b64 exec, exec, s[10:11]
+; GFX1264-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1264-TRUE16-NEXT: s_cbranch_execnz .LBB13_2
; GFX1264-TRUE16-NEXT: ; %bb.3: ; %atomicrmw.end
; GFX1264-TRUE16-NEXT: s_or_b64 exec, exec, s[10:11]
+; GFX1264-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1264-TRUE16-NEXT: v_lshrrev_b32_e32 v0, s2, v2
; GFX1264-TRUE16-NEXT: .LBB13_4: ; %Flow
; GFX1264-TRUE16-NEXT: s_or_b64 exec, exec, s[8:9]
@@ -10914,6 +11005,7 @@ define amdgpu_kernel void @uniform_add_i8(ptr addrspace(1) %result, ptr addrspac
; GFX1264-FAKE16-NEXT: v_mbcnt_hi_u32_b32 v4, s7, v0
; GFX1264-FAKE16-NEXT: ; implicit-def: $vgpr0
; GFX1264-FAKE16-NEXT: v_cmpx_eq_u32_e32 0, v4
+; GFX1264-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1264-FAKE16-NEXT: s_cbranch_execz .LBB13_4
; GFX1264-FAKE16-NEXT: ; %bb.1:
; GFX1264-FAKE16-NEXT: s_wait_kmcnt 0x0
@@ -10950,12 +11042,14 @@ define amdgpu_kernel void @uniform_add_i8(ptr addrspace(1) %result, ptr addrspac
; GFX1264-FAKE16-NEXT: s_wait_loadcnt 0x0
; GFX1264-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc, v2, v1
; GFX1264-FAKE16-NEXT: v_mov_b32_e32 v1, v2
+; GFX1264-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1264-FAKE16-NEXT: s_or_b64 s[10:11], vcc, s[10:11]
-; GFX1264-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1264-FAKE16-NEXT: s_and_not1_b64 exec, exec, s[10:11]
+; GFX1264-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1264-FAKE16-NEXT: s_cbranch_execnz .LBB13_2
; GFX1264-FAKE16-NEXT: ; %bb.3: ; %atomicrmw.end
; GFX1264-FAKE16-NEXT: s_or_b64 exec, exec, s[10:11]
+; GFX1264-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1264-FAKE16-NEXT: v_lshrrev_b32_e32 v0, s2, v2
; GFX1264-FAKE16-NEXT: .LBB13_4: ; %Flow
; GFX1264-FAKE16-NEXT: s_or_b64 exec, exec, s[8:9]
@@ -10979,7 +11073,7 @@ define amdgpu_kernel void @uniform_add_i8(ptr addrspace(1) %result, ptr addrspac
; GFX1232-TRUE16-NEXT: v_mbcnt_lo_u32_b32 v4, s6, 0
; GFX1232-TRUE16-NEXT: s_mov_b32 s9, exec_lo
; GFX1232-TRUE16-NEXT: ; implicit-def: $vgpr0_lo16
-; GFX1232-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1232-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1232-TRUE16-NEXT: v_cmpx_eq_u32_e32 0, v4
; GFX1232-TRUE16-NEXT: s_cbranch_execz .LBB13_4
; GFX1232-TRUE16-NEXT: ; %bb.1:
@@ -11018,9 +11112,11 @@ define amdgpu_kernel void @uniform_add_i8(ptr addrspace(1) %result, ptr addrspac
; GFX1232-TRUE16-NEXT: s_or_b32 s10, vcc_lo, s10
; GFX1232-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX1232-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s10
+; GFX1232-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1232-TRUE16-NEXT: s_cbranch_execnz .LBB13_2
; GFX1232-TRUE16-NEXT: ; %bb.3: ; %atomicrmw.end
; GFX1232-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s10
+; GFX1232-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1232-TRUE16-NEXT: v_lshrrev_b32_e32 v0, s2, v2
; GFX1232-TRUE16-NEXT: .LBB13_4: ; %Flow
; GFX1232-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s9
@@ -11044,7 +11140,7 @@ define amdgpu_kernel void @uniform_add_i8(ptr addrspace(1) %result, ptr addrspac
; GFX1232-FAKE16-NEXT: v_mbcnt_lo_u32_b32 v4, s6, 0
; GFX1232-FAKE16-NEXT: s_mov_b32 s9, exec_lo
; GFX1232-FAKE16-NEXT: ; implicit-def: $vgpr0
-; GFX1232-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1232-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1232-FAKE16-NEXT: v_cmpx_eq_u32_e32 0, v4
; GFX1232-FAKE16-NEXT: s_cbranch_execz .LBB13_4
; GFX1232-FAKE16-NEXT: ; %bb.1:
@@ -11083,9 +11179,11 @@ define amdgpu_kernel void @uniform_add_i8(ptr addrspace(1) %result, ptr addrspac
; GFX1232-FAKE16-NEXT: s_or_b32 s10, vcc_lo, s10
; GFX1232-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX1232-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s10
+; GFX1232-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1232-FAKE16-NEXT: s_cbranch_execnz .LBB13_2
; GFX1232-FAKE16-NEXT: ; %bb.3: ; %atomicrmw.end
; GFX1232-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s10
+; GFX1232-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1232-FAKE16-NEXT: v_lshrrev_b32_e32 v0, s2, v2
; GFX1232-FAKE16-NEXT: .LBB13_4: ; %Flow
; GFX1232-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s9
@@ -11111,6 +11209,7 @@ define amdgpu_kernel void @uniform_add_i8(ptr addrspace(1) %result, ptr addrspac
; GFX1364-TRUE16-NEXT: v_mbcnt_hi_u32_b32 v4, s7, v0
; GFX1364-TRUE16-NEXT: ; implicit-def: $vgpr0_lo16
; GFX1364-TRUE16-NEXT: v_cmpx_eq_u32_e32 0, v4
+; GFX1364-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1364-TRUE16-NEXT: s_cbranch_execz .LBB13_4
; GFX1364-TRUE16-NEXT: ; %bb.1:
; GFX1364-TRUE16-NEXT: s_wait_kmcnt 0x0
@@ -11142,12 +11241,14 @@ define amdgpu_kernel void @uniform_add_i8(ptr addrspace(1) %result, ptr addrspac
; GFX1364-TRUE16-NEXT: s_wait_loadcnt 0x0
; GFX1364-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc, v2, v1
; GFX1364-TRUE16-NEXT: v_mov_b32_e32 v1, v2
+; GFX1364-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1364-TRUE16-NEXT: s_or_b64 s[2:3], vcc, s[2:3]
-; GFX1364-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1364-TRUE16-NEXT: s_and_not1_b64 exec, exec, s[2:3]
+; GFX1364-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1364-TRUE16-NEXT: s_cbranch_execnz .LBB13_2
; GFX1364-TRUE16-NEXT: ; %bb.3: ; %atomicrmw.end
; GFX1364-TRUE16-NEXT: s_or_b64 exec, exec, s[2:3]
+; GFX1364-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1364-TRUE16-NEXT: v_lshrrev_b32_e32 v0, s11, v2
; GFX1364-TRUE16-NEXT: .LBB13_4: ; %Flow
; GFX1364-TRUE16-NEXT: s_or_b64 exec, exec, s[8:9]
@@ -11172,6 +11273,7 @@ define amdgpu_kernel void @uniform_add_i8(ptr addrspace(1) %result, ptr addrspac
; GFX1364-FAKE16-NEXT: v_mbcnt_hi_u32_b32 v4, s7, v0
; GFX1364-FAKE16-NEXT: ; implicit-def: $vgpr0
; GFX1364-FAKE16-NEXT: v_cmpx_eq_u32_e32 0, v4
+; GFX1364-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1364-FAKE16-NEXT: s_cbranch_execz .LBB13_4
; GFX1364-FAKE16-NEXT: ; %bb.1:
; GFX1364-FAKE16-NEXT: s_wait_kmcnt 0x0
@@ -11203,12 +11305,14 @@ define amdgpu_kernel void @uniform_add_i8(ptr addrspace(1) %result, ptr addrspac
; GFX1364-FAKE16-NEXT: s_wait_loadcnt 0x0
; GFX1364-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc, v2, v1
; GFX1364-FAKE16-NEXT: v_mov_b32_e32 v1, v2
+; GFX1364-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1364-FAKE16-NEXT: s_or_b64 s[2:3], vcc, s[2:3]
-; GFX1364-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1364-FAKE16-NEXT: s_and_not1_b64 exec, exec, s[2:3]
+; GFX1364-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1364-FAKE16-NEXT: s_cbranch_execnz .LBB13_2
; GFX1364-FAKE16-NEXT: ; %bb.3: ; %atomicrmw.end
; GFX1364-FAKE16-NEXT: s_or_b64 exec, exec, s[2:3]
+; GFX1364-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1364-FAKE16-NEXT: v_lshrrev_b32_e32 v0, s11, v2
; GFX1364-FAKE16-NEXT: .LBB13_4: ; %Flow
; GFX1364-FAKE16-NEXT: s_or_b64 exec, exec, s[8:9]
@@ -11231,7 +11335,7 @@ define amdgpu_kernel void @uniform_add_i8(ptr addrspace(1) %result, ptr addrspac
; GFX1332-TRUE16-NEXT: v_mbcnt_lo_u32_b32 v4, s6, 0
; GFX1332-TRUE16-NEXT: s_mov_b32 s9, exec_lo
; GFX1332-TRUE16-NEXT: ; implicit-def: $vgpr0_lo16
-; GFX1332-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1332-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1332-TRUE16-NEXT: v_cmpx_eq_u32_e32 0, v4
; GFX1332-TRUE16-NEXT: s_cbranch_execz .LBB13_4
; GFX1332-TRUE16-NEXT: ; %bb.1:
@@ -11263,11 +11367,12 @@ define amdgpu_kernel void @uniform_add_i8(ptr addrspace(1) %result, ptr addrspac
; GFX1332-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v1
; GFX1332-TRUE16-NEXT: v_mov_b32_e32 v1, v2
; GFX1332-TRUE16-NEXT: s_or_b32 s10, vcc_lo, s10
-; GFX1332-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1332-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1332-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s10
; GFX1332-TRUE16-NEXT: s_cbranch_execnz .LBB13_2
; GFX1332-TRUE16-NEXT: ; %bb.3: ; %atomicrmw.end
; GFX1332-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s10
+; GFX1332-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1332-TRUE16-NEXT: v_lshrrev_b32_e32 v0, s2, v2
; GFX1332-TRUE16-NEXT: .LBB13_4: ; %Flow
; GFX1332-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s9
@@ -11290,7 +11395,7 @@ define amdgpu_kernel void @uniform_add_i8(ptr addrspace(1) %result, ptr addrspac
; GFX1332-FAKE16-NEXT: v_mbcnt_lo_u32_b32 v4, s6, 0
; GFX1332-FAKE16-NEXT: s_mov_b32 s9, exec_lo
; GFX1332-FAKE16-NEXT: ; implicit-def: $vgpr0
-; GFX1332-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1332-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1332-FAKE16-NEXT: v_cmpx_eq_u32_e32 0, v4
; GFX1332-FAKE16-NEXT: s_cbranch_execz .LBB13_4
; GFX1332-FAKE16-NEXT: ; %bb.1:
@@ -11322,11 +11427,12 @@ define amdgpu_kernel void @uniform_add_i8(ptr addrspace(1) %result, ptr addrspac
; GFX1332-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v1
; GFX1332-FAKE16-NEXT: v_mov_b32_e32 v1, v2
; GFX1332-FAKE16-NEXT: s_or_b32 s10, vcc_lo, s10
-; GFX1332-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1332-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1332-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s10
; GFX1332-FAKE16-NEXT: s_cbranch_execnz .LBB13_2
; GFX1332-FAKE16-NEXT: ; %bb.3: ; %atomicrmw.end
; GFX1332-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s10
+; GFX1332-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1332-FAKE16-NEXT: v_lshrrev_b32_e32 v0, s2, v2
; GFX1332-FAKE16-NEXT: .LBB13_4: ; %Flow
; GFX1332-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s9
@@ -11581,12 +11687,14 @@ define amdgpu_kernel void @uniform_xchg_i8(ptr addrspace(1) %result, ptr addrspa
; GFX1164-NEXT: s_waitcnt vmcnt(0)
; GFX1164-NEXT: v_cmp_eq_u32_e32 vcc, v2, v1
; GFX1164-NEXT: v_mov_b32_e32 v1, v2
+; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1164-NEXT: s_or_b64 s[8:9], vcc, s[8:9]
-; GFX1164-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164-NEXT: s_and_not1_b64 exec, exec, s[8:9]
+; GFX1164-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164-NEXT: s_cbranch_execnz .LBB14_1
; GFX1164-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX1164-NEXT: s_or_b64 exec, exec, s[8:9]
+; GFX1164-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164-NEXT: v_lshrrev_b32_e32 v0, s2, v2
; GFX1164-NEXT: s_mov_b32 s3, 0x31016000
; GFX1164-NEXT: s_mov_b32 s2, -1
@@ -11624,11 +11732,12 @@ define amdgpu_kernel void @uniform_xchg_i8(ptr addrspace(1) %result, ptr addrspa
; GFX1132-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v1
; GFX1132-NEXT: v_mov_b32_e32 v1, v2
; GFX1132-NEXT: s_or_b32 s9, vcc_lo, s9
-; GFX1132-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1132-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1132-NEXT: s_and_not1_b32 exec_lo, exec_lo, s9
; GFX1132-NEXT: s_cbranch_execnz .LBB14_1
; GFX1132-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX1132-NEXT: s_or_b32 exec_lo, exec_lo, s9
+; GFX1132-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1132-NEXT: v_lshrrev_b32_e32 v0, s2, v2
; GFX1132-NEXT: s_mov_b32 s3, 0x31016000
; GFX1132-NEXT: s_mov_b32 s2, -1
@@ -11666,12 +11775,14 @@ define amdgpu_kernel void @uniform_xchg_i8(ptr addrspace(1) %result, ptr addrspa
; GFX1264-NEXT: s_wait_loadcnt 0x0
; GFX1264-NEXT: v_cmp_eq_u32_e32 vcc, v2, v1
; GFX1264-NEXT: v_mov_b32_e32 v1, v2
+; GFX1264-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1264-NEXT: s_or_b64 s[8:9], vcc, s[8:9]
-; GFX1264-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1264-NEXT: s_and_not1_b64 exec, exec, s[8:9]
+; GFX1264-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1264-NEXT: s_cbranch_execnz .LBB14_1
; GFX1264-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX1264-NEXT: s_or_b64 exec, exec, s[8:9]
+; GFX1264-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1264-NEXT: v_lshrrev_b32_e32 v0, s2, v2
; GFX1264-NEXT: s_mov_b32 s3, 0x31016000
; GFX1264-NEXT: s_mov_b32 s2, -1
@@ -11711,9 +11822,11 @@ define amdgpu_kernel void @uniform_xchg_i8(ptr addrspace(1) %result, ptr addrspa
; GFX1232-NEXT: s_or_b32 s9, vcc_lo, s9
; GFX1232-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX1232-NEXT: s_and_not1_b32 exec_lo, exec_lo, s9
+; GFX1232-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1232-NEXT: s_cbranch_execnz .LBB14_1
; GFX1232-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX1232-NEXT: s_or_b32 exec_lo, exec_lo, s9
+; GFX1232-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1232-NEXT: v_lshrrev_b32_e32 v0, s2, v2
; GFX1232-NEXT: s_mov_b32 s3, 0x31016000
; GFX1232-NEXT: s_mov_b32 s2, -1
@@ -11749,12 +11862,14 @@ define amdgpu_kernel void @uniform_xchg_i8(ptr addrspace(1) %result, ptr addrspa
; GFX1364-NEXT: s_wait_loadcnt 0x0
; GFX1364-NEXT: v_cmp_eq_u32_e32 vcc, v2, v1
; GFX1364-NEXT: v_mov_b32_e32 v1, v2
+; GFX1364-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1364-NEXT: s_or_b64 s[2:3], vcc, s[2:3]
-; GFX1364-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1364-NEXT: s_and_not1_b64 exec, exec, s[2:3]
+; GFX1364-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1364-NEXT: s_cbranch_execnz .LBB14_1
; GFX1364-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX1364-NEXT: s_or_b64 exec, exec, s[2:3]
+; GFX1364-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1364-NEXT: v_lshrrev_b32_e32 v0, s8, v2
; GFX1364-NEXT: s_mov_b32 s3, 0x31016000
; GFX1364-NEXT: s_mov_b32 s2, -1
@@ -11790,11 +11905,12 @@ define amdgpu_kernel void @uniform_xchg_i8(ptr addrspace(1) %result, ptr addrspa
; GFX1332-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v1
; GFX1332-NEXT: v_mov_b32_e32 v1, v2
; GFX1332-NEXT: s_or_b32 s9, vcc_lo, s9
-; GFX1332-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1332-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1332-NEXT: s_and_not1_b32 exec_lo, exec_lo, s9
; GFX1332-NEXT: s_cbranch_execnz .LBB14_1
; GFX1332-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX1332-NEXT: s_or_b32 exec_lo, exec_lo, s9
+; GFX1332-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1332-NEXT: v_lshrrev_b32_e32 v0, s2, v2
; GFX1332-NEXT: s_mov_b32 s3, 0x31016000
; GFX1332-NEXT: s_mov_b32 s2, -1
@@ -12072,6 +12188,7 @@ define amdgpu_kernel void @uniform_or_i16(ptr addrspace(1) %result, ptr addrspac
; GFX1164-TRUE16-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
; GFX1164-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc, 0, v0
; GFX1164-TRUE16-NEXT: ; implicit-def: $vgpr0_lo16
+; GFX1164-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1164-TRUE16-NEXT: s_and_saveexec_b64 s[2:3], vcc
; GFX1164-TRUE16-NEXT: s_cbranch_execz .LBB15_4
; GFX1164-TRUE16-NEXT: ; %bb.1:
@@ -12099,20 +12216,22 @@ define amdgpu_kernel void @uniform_or_i16(ptr addrspace(1) %result, ptr addrspac
; GFX1164-TRUE16-NEXT: v_cmp_eq_u32_e64 s[0:1], v2, v1
; GFX1164-TRUE16-NEXT: v_mov_b32_e32 v1, v2
; GFX1164-TRUE16-NEXT: s_or_b64 s[12:13], s[0:1], s[12:13]
-; GFX1164-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1164-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1164-TRUE16-NEXT: s_and_not1_b64 exec, exec, s[12:13]
; GFX1164-TRUE16-NEXT: s_cbranch_execnz .LBB15_2
; GFX1164-TRUE16-NEXT: ; %bb.3: ; %atomicrmw.end
; GFX1164-TRUE16-NEXT: s_or_b64 exec, exec, s[12:13]
+; GFX1164-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164-TRUE16-NEXT: v_lshrrev_b32_e32 v0, s10, v2
; GFX1164-TRUE16-NEXT: .LBB15_4: ; %Flow
; GFX1164-TRUE16-NEXT: s_or_b64 exec, exec, s[2:3]
-; GFX1164-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_4) | instid1(VALU_DEP_1)
+; GFX1164-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX1164-TRUE16-NEXT: v_readfirstlane_b32 s0, v0
; GFX1164-TRUE16-NEXT: s_waitcnt lgkmcnt(0)
; GFX1164-TRUE16-NEXT: v_cndmask_b16 v0.l, s14, 0, vcc
; GFX1164-TRUE16-NEXT: s_mov_b32 s11, 0x31016000
; GFX1164-TRUE16-NEXT: s_mov_b32 s10, -1
+; GFX1164-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1164-TRUE16-NEXT: v_or_b16 v0.l, s0, v0.l
; GFX1164-TRUE16-NEXT: buffer_store_b16 v0, off, s[8:11], 0
; GFX1164-TRUE16-NEXT: s_endpgm
@@ -12127,6 +12246,7 @@ define amdgpu_kernel void @uniform_or_i16(ptr addrspace(1) %result, ptr addrspac
; GFX1164-FAKE16-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
; GFX1164-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc, 0, v0
; GFX1164-FAKE16-NEXT: ; implicit-def: $vgpr0
+; GFX1164-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1164-FAKE16-NEXT: s_and_saveexec_b64 s[2:3], vcc
; GFX1164-FAKE16-NEXT: s_cbranch_execz .LBB15_4
; GFX1164-FAKE16-NEXT: ; %bb.1:
@@ -12154,20 +12274,22 @@ define amdgpu_kernel void @uniform_or_i16(ptr addrspace(1) %result, ptr addrspac
; GFX1164-FAKE16-NEXT: v_cmp_eq_u32_e64 s[0:1], v2, v1
; GFX1164-FAKE16-NEXT: v_mov_b32_e32 v1, v2
; GFX1164-FAKE16-NEXT: s_or_b64 s[12:13], s[0:1], s[12:13]
-; GFX1164-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1164-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1164-FAKE16-NEXT: s_and_not1_b64 exec, exec, s[12:13]
; GFX1164-FAKE16-NEXT: s_cbranch_execnz .LBB15_2
; GFX1164-FAKE16-NEXT: ; %bb.3: ; %atomicrmw.end
; GFX1164-FAKE16-NEXT: s_or_b64 exec, exec, s[12:13]
+; GFX1164-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164-FAKE16-NEXT: v_lshrrev_b32_e32 v0, s10, v2
; GFX1164-FAKE16-NEXT: .LBB15_4: ; %Flow
; GFX1164-FAKE16-NEXT: s_or_b64 exec, exec, s[2:3]
-; GFX1164-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_4) | instid1(VALU_DEP_1)
+; GFX1164-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX1164-FAKE16-NEXT: v_readfirstlane_b32 s0, v0
; GFX1164-FAKE16-NEXT: s_waitcnt lgkmcnt(0)
; GFX1164-FAKE16-NEXT: v_cndmask_b32_e64 v0, s14, 0, vcc
; GFX1164-FAKE16-NEXT: s_mov_b32 s11, 0x31016000
; GFX1164-FAKE16-NEXT: s_mov_b32 s10, -1
+; GFX1164-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1164-FAKE16-NEXT: v_or_b32_e32 v0, s0, v0
; GFX1164-FAKE16-NEXT: buffer_store_b16 v0, off, s[8:11], 0
; GFX1164-FAKE16-NEXT: s_endpgm
@@ -12209,20 +12331,22 @@ define amdgpu_kernel void @uniform_or_i16(ptr addrspace(1) %result, ptr addrspac
; GFX1132-TRUE16-NEXT: v_cmp_eq_u32_e64 s0, v2, v1
; GFX1132-TRUE16-NEXT: v_mov_b32_e32 v1, v2
; GFX1132-TRUE16-NEXT: s_or_b32 s3, s0, s3
-; GFX1132-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1132-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1132-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s3
; GFX1132-TRUE16-NEXT: s_cbranch_execnz .LBB15_2
; GFX1132-TRUE16-NEXT: ; %bb.3: ; %atomicrmw.end
; GFX1132-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s3
+; GFX1132-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1132-TRUE16-NEXT: v_lshrrev_b32_e32 v0, s10, v2
; GFX1132-TRUE16-NEXT: .LBB15_4: ; %Flow
; GFX1132-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s2
-; GFX1132-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_4) | instid1(VALU_DEP_1)
+; GFX1132-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX1132-TRUE16-NEXT: v_readfirstlane_b32 s0, v0
; GFX1132-TRUE16-NEXT: s_waitcnt lgkmcnt(0)
; GFX1132-TRUE16-NEXT: v_cndmask_b16 v0.l, s1, 0, vcc_lo
; GFX1132-TRUE16-NEXT: s_mov_b32 s11, 0x31016000
; GFX1132-TRUE16-NEXT: s_mov_b32 s10, -1
+; GFX1132-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1132-TRUE16-NEXT: v_or_b16 v0.l, s0, v0.l
; GFX1132-TRUE16-NEXT: buffer_store_b16 v0, off, s[8:11], 0
; GFX1132-TRUE16-NEXT: s_endpgm
@@ -12264,20 +12388,22 @@ define amdgpu_kernel void @uniform_or_i16(ptr addrspace(1) %result, ptr addrspac
; GFX1132-FAKE16-NEXT: v_cmp_eq_u32_e64 s0, v2, v1
; GFX1132-FAKE16-NEXT: v_mov_b32_e32 v1, v2
; GFX1132-FAKE16-NEXT: s_or_b32 s3, s0, s3
-; GFX1132-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1132-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1132-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s3
; GFX1132-FAKE16-NEXT: s_cbranch_execnz .LBB15_2
; GFX1132-FAKE16-NEXT: ; %bb.3: ; %atomicrmw.end
; GFX1132-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s3
+; GFX1132-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1132-FAKE16-NEXT: v_lshrrev_b32_e32 v0, s10, v2
; GFX1132-FAKE16-NEXT: .LBB15_4: ; %Flow
; GFX1132-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s2
-; GFX1132-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_4) | instid1(VALU_DEP_1)
+; GFX1132-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX1132-FAKE16-NEXT: v_readfirstlane_b32 s0, v0
; GFX1132-FAKE16-NEXT: s_waitcnt lgkmcnt(0)
; GFX1132-FAKE16-NEXT: v_cndmask_b32_e64 v0, s1, 0, vcc_lo
; GFX1132-FAKE16-NEXT: s_mov_b32 s11, 0x31016000
; GFX1132-FAKE16-NEXT: s_mov_b32 s10, -1
+; GFX1132-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1132-FAKE16-NEXT: v_or_b32_e32 v0, s0, v0
; GFX1132-FAKE16-NEXT: buffer_store_b16 v0, off, s[8:11], 0
; GFX1132-FAKE16-NEXT: s_endpgm
@@ -12292,6 +12418,7 @@ define amdgpu_kernel void @uniform_or_i16(ptr addrspace(1) %result, ptr addrspac
; GFX1264-TRUE16-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
; GFX1264-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc, 0, v0
; GFX1264-TRUE16-NEXT: ; implicit-def: $vgpr0_lo16
+; GFX1264-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1264-TRUE16-NEXT: s_and_saveexec_b64 s[2:3], vcc
; GFX1264-TRUE16-NEXT: s_cbranch_execz .LBB15_4
; GFX1264-TRUE16-NEXT: ; %bb.1:
@@ -12319,15 +12446,16 @@ define amdgpu_kernel void @uniform_or_i16(ptr addrspace(1) %result, ptr addrspac
; GFX1264-TRUE16-NEXT: v_cmp_eq_u32_e64 s[0:1], v2, v1
; GFX1264-TRUE16-NEXT: v_mov_b32_e32 v1, v2
; GFX1264-TRUE16-NEXT: s_or_b64 s[12:13], s[0:1], s[12:13]
-; GFX1264-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1264-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1264-TRUE16-NEXT: s_and_not1_b64 exec, exec, s[12:13]
; GFX1264-TRUE16-NEXT: s_cbranch_execnz .LBB15_2
; GFX1264-TRUE16-NEXT: ; %bb.3: ; %atomicrmw.end
; GFX1264-TRUE16-NEXT: s_or_b64 exec, exec, s[12:13]
+; GFX1264-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1264-TRUE16-NEXT: v_lshrrev_b32_e32 v0, s10, v2
; GFX1264-TRUE16-NEXT: .LBB15_4: ; %Flow
; GFX1264-TRUE16-NEXT: s_or_b64 exec, exec, s[2:3]
-; GFX1264-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1264-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX1264-TRUE16-NEXT: v_readfirstlane_b32 s0, v0
; GFX1264-TRUE16-NEXT: s_wait_kmcnt 0x0
; GFX1264-TRUE16-NEXT: v_cndmask_b16 v0.l, s14, 0, vcc
@@ -12349,6 +12477,7 @@ define amdgpu_kernel void @uniform_or_i16(ptr addrspace(1) %result, ptr addrspac
; GFX1264-FAKE16-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
; GFX1264-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc, 0, v0
; GFX1264-FAKE16-NEXT: ; implicit-def: $vgpr0
+; GFX1264-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1264-FAKE16-NEXT: s_and_saveexec_b64 s[2:3], vcc
; GFX1264-FAKE16-NEXT: s_cbranch_execz .LBB15_4
; GFX1264-FAKE16-NEXT: ; %bb.1:
@@ -12376,15 +12505,16 @@ define amdgpu_kernel void @uniform_or_i16(ptr addrspace(1) %result, ptr addrspac
; GFX1264-FAKE16-NEXT: v_cmp_eq_u32_e64 s[0:1], v2, v1
; GFX1264-FAKE16-NEXT: v_mov_b32_e32 v1, v2
; GFX1264-FAKE16-NEXT: s_or_b64 s[12:13], s[0:1], s[12:13]
-; GFX1264-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1264-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1264-FAKE16-NEXT: s_and_not1_b64 exec, exec, s[12:13]
; GFX1264-FAKE16-NEXT: s_cbranch_execnz .LBB15_2
; GFX1264-FAKE16-NEXT: ; %bb.3: ; %atomicrmw.end
; GFX1264-FAKE16-NEXT: s_or_b64 exec, exec, s[12:13]
+; GFX1264-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1264-FAKE16-NEXT: v_lshrrev_b32_e32 v0, s10, v2
; GFX1264-FAKE16-NEXT: .LBB15_4: ; %Flow
; GFX1264-FAKE16-NEXT: s_or_b64 exec, exec, s[2:3]
-; GFX1264-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1264-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX1264-FAKE16-NEXT: v_readfirstlane_b32 s0, v0
; GFX1264-FAKE16-NEXT: s_wait_kmcnt 0x0
; GFX1264-FAKE16-NEXT: v_cndmask_b32_e64 v0, s14, 0, vcc
@@ -12433,15 +12563,16 @@ define amdgpu_kernel void @uniform_or_i16(ptr addrspace(1) %result, ptr addrspac
; GFX1232-TRUE16-NEXT: v_cmp_eq_u32_e64 s0, v2, v1
; GFX1232-TRUE16-NEXT: v_mov_b32_e32 v1, v2
; GFX1232-TRUE16-NEXT: s_or_b32 s3, s0, s3
-; GFX1232-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1232-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1232-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s3
; GFX1232-TRUE16-NEXT: s_cbranch_execnz .LBB15_2
; GFX1232-TRUE16-NEXT: ; %bb.3: ; %atomicrmw.end
; GFX1232-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s3
+; GFX1232-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1232-TRUE16-NEXT: v_lshrrev_b32_e32 v0, s10, v2
; GFX1232-TRUE16-NEXT: .LBB15_4: ; %Flow
; GFX1232-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s2
-; GFX1232-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1232-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX1232-TRUE16-NEXT: v_readfirstlane_b32 s0, v0
; GFX1232-TRUE16-NEXT: s_wait_kmcnt 0x0
; GFX1232-TRUE16-NEXT: v_cndmask_b16 v0.l, s1, 0, vcc_lo
@@ -12490,15 +12621,16 @@ define amdgpu_kernel void @uniform_or_i16(ptr addrspace(1) %result, ptr addrspac
; GFX1232-FAKE16-NEXT: v_cmp_eq_u32_e64 s0, v2, v1
; GFX1232-FAKE16-NEXT: v_mov_b32_e32 v1, v2
; GFX1232-FAKE16-NEXT: s_or_b32 s3, s0, s3
-; GFX1232-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1232-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1232-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s3
; GFX1232-FAKE16-NEXT: s_cbranch_execnz .LBB15_2
; GFX1232-FAKE16-NEXT: ; %bb.3: ; %atomicrmw.end
; GFX1232-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s3
+; GFX1232-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1232-FAKE16-NEXT: v_lshrrev_b32_e32 v0, s10, v2
; GFX1232-FAKE16-NEXT: .LBB15_4: ; %Flow
; GFX1232-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s2
-; GFX1232-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1232-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX1232-FAKE16-NEXT: v_readfirstlane_b32 s0, v0
; GFX1232-FAKE16-NEXT: s_wait_kmcnt 0x0
; GFX1232-FAKE16-NEXT: v_cndmask_b32_e64 v0, s1, 0, vcc_lo
@@ -12520,6 +12652,7 @@ define amdgpu_kernel void @uniform_or_i16(ptr addrspace(1) %result, ptr addrspac
; GFX1364-TRUE16-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
; GFX1364-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc, 0, v0
; GFX1364-TRUE16-NEXT: ; implicit-def: $vgpr0_lo16
+; GFX1364-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1364-TRUE16-NEXT: s_and_saveexec_b64 s[2:3], vcc
; GFX1364-TRUE16-NEXT: s_cbranch_execz .LBB15_4
; GFX1364-TRUE16-NEXT: ; %bb.1:
@@ -12546,20 +12679,22 @@ define amdgpu_kernel void @uniform_or_i16(ptr addrspace(1) %result, ptr addrspac
; GFX1364-TRUE16-NEXT: v_cmp_eq_u32_e64 s[0:1], v2, v1
; GFX1364-TRUE16-NEXT: v_mov_b32_e32 v1, v2
; GFX1364-TRUE16-NEXT: s_or_b64 s[10:11], s[0:1], s[10:11]
-; GFX1364-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1364-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1364-TRUE16-NEXT: s_and_not1_b64 exec, exec, s[10:11]
; GFX1364-TRUE16-NEXT: s_cbranch_execnz .LBB15_2
; GFX1364-TRUE16-NEXT: ; %bb.3: ; %atomicrmw.end
; GFX1364-TRUE16-NEXT: s_or_b64 exec, exec, s[10:11]
+; GFX1364-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1364-TRUE16-NEXT: v_lshrrev_b32_e32 v0, s13, v2
; GFX1364-TRUE16-NEXT: .LBB15_4: ; %Flow
; GFX1364-TRUE16-NEXT: s_or_b64 exec, exec, s[2:3]
-; GFX1364-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_4) | instid1(VALU_DEP_1)
+; GFX1364-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX1364-TRUE16-NEXT: v_readfirstlane_b32 s0, v0
; GFX1364-TRUE16-NEXT: s_wait_kmcnt 0x0
; GFX1364-TRUE16-NEXT: v_cndmask_b16 v0.l, s12, 0, vcc
; GFX1364-TRUE16-NEXT: s_mov_b32 s11, 0x31016000
; GFX1364-TRUE16-NEXT: s_mov_b32 s10, -1
+; GFX1364-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1364-TRUE16-NEXT: v_or_b16 v0.l, s0, v0.l
; GFX1364-TRUE16-NEXT: buffer_store_b16 v0, off, s[8:11], null
; GFX1364-TRUE16-NEXT: s_endpgm
@@ -12574,6 +12709,7 @@ define amdgpu_kernel void @uniform_or_i16(ptr addrspace(1) %result, ptr addrspac
; GFX1364-FAKE16-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
; GFX1364-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc, 0, v0
; GFX1364-FAKE16-NEXT: ; implicit-def: $vgpr0
+; GFX1364-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1364-FAKE16-NEXT: s_and_saveexec_b64 s[2:3], vcc
; GFX1364-FAKE16-NEXT: s_cbranch_execz .LBB15_4
; GFX1364-FAKE16-NEXT: ; %bb.1:
@@ -12600,20 +12736,22 @@ define amdgpu_kernel void @uniform_or_i16(ptr addrspace(1) %result, ptr addrspac
; GFX1364-FAKE16-NEXT: v_cmp_eq_u32_e64 s[0:1], v2, v1
; GFX1364-FAKE16-NEXT: v_mov_b32_e32 v1, v2
; GFX1364-FAKE16-NEXT: s_or_b64 s[10:11], s[0:1], s[10:11]
-; GFX1364-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1364-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1364-FAKE16-NEXT: s_and_not1_b64 exec, exec, s[10:11]
; GFX1364-FAKE16-NEXT: s_cbranch_execnz .LBB15_2
; GFX1364-FAKE16-NEXT: ; %bb.3: ; %atomicrmw.end
; GFX1364-FAKE16-NEXT: s_or_b64 exec, exec, s[10:11]
+; GFX1364-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1364-FAKE16-NEXT: v_lshrrev_b32_e32 v0, s13, v2
; GFX1364-FAKE16-NEXT: .LBB15_4: ; %Flow
; GFX1364-FAKE16-NEXT: s_or_b64 exec, exec, s[2:3]
-; GFX1364-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_4) | instid1(VALU_DEP_1)
+; GFX1364-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX1364-FAKE16-NEXT: v_readfirstlane_b32 s0, v0
; GFX1364-FAKE16-NEXT: s_wait_kmcnt 0x0
; GFX1364-FAKE16-NEXT: v_cndmask_b32_e64 v0, s12, 0, vcc
; GFX1364-FAKE16-NEXT: s_mov_b32 s11, 0x31016000
; GFX1364-FAKE16-NEXT: s_mov_b32 s10, -1
+; GFX1364-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1364-FAKE16-NEXT: v_or_b32_e32 v0, s0, v0
; GFX1364-FAKE16-NEXT: buffer_store_b16 v0, off, s[8:11], null
; GFX1364-FAKE16-NEXT: s_endpgm
@@ -12652,20 +12790,22 @@ define amdgpu_kernel void @uniform_or_i16(ptr addrspace(1) %result, ptr addrspac
; GFX1332-TRUE16-NEXT: v_cmp_eq_u32_e64 s0, v2, v1
; GFX1332-TRUE16-NEXT: v_mov_b32_e32 v1, v2
; GFX1332-TRUE16-NEXT: s_or_b32 s3, s0, s3
-; GFX1332-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1332-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1332-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s3
; GFX1332-TRUE16-NEXT: s_cbranch_execnz .LBB15_2
; GFX1332-TRUE16-NEXT: ; %bb.3: ; %atomicrmw.end
; GFX1332-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s3
+; GFX1332-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1332-TRUE16-NEXT: v_lshrrev_b32_e32 v0, s10, v2
; GFX1332-TRUE16-NEXT: .LBB15_4: ; %Flow
; GFX1332-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s2
-; GFX1332-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_4) | instid1(VALU_DEP_1)
+; GFX1332-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX1332-TRUE16-NEXT: v_readfirstlane_b32 s0, v0
; GFX1332-TRUE16-NEXT: s_wait_kmcnt 0x0
; GFX1332-TRUE16-NEXT: v_cndmask_b16 v0.l, s1, 0, vcc_lo
; GFX1332-TRUE16-NEXT: s_mov_b32 s11, 0x31016000
; GFX1332-TRUE16-NEXT: s_mov_b32 s10, -1
+; GFX1332-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1332-TRUE16-NEXT: v_or_b16 v0.l, s0, v0.l
; GFX1332-TRUE16-NEXT: buffer_store_b16 v0, off, s[8:11], null
; GFX1332-TRUE16-NEXT: s_endpgm
@@ -12704,20 +12844,22 @@ define amdgpu_kernel void @uniform_or_i16(ptr addrspace(1) %result, ptr addrspac
; GFX1332-FAKE16-NEXT: v_cmp_eq_u32_e64 s0, v2, v1
; GFX1332-FAKE16-NEXT: v_mov_b32_e32 v1, v2
; GFX1332-FAKE16-NEXT: s_or_b32 s3, s0, s3
-; GFX1332-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1332-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1332-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s3
; GFX1332-FAKE16-NEXT: s_cbranch_execnz .LBB15_2
; GFX1332-FAKE16-NEXT: ; %bb.3: ; %atomicrmw.end
; GFX1332-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s3
+; GFX1332-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1332-FAKE16-NEXT: v_lshrrev_b32_e32 v0, s10, v2
; GFX1332-FAKE16-NEXT: .LBB15_4: ; %Flow
; GFX1332-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s2
-; GFX1332-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_4) | instid1(VALU_DEP_1)
+; GFX1332-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX1332-FAKE16-NEXT: v_readfirstlane_b32 s0, v0
; GFX1332-FAKE16-NEXT: s_wait_kmcnt 0x0
; GFX1332-FAKE16-NEXT: v_cndmask_b32_e64 v0, s1, 0, vcc_lo
; GFX1332-FAKE16-NEXT: s_mov_b32 s11, 0x31016000
; GFX1332-FAKE16-NEXT: s_mov_b32 s10, -1
+; GFX1332-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1332-FAKE16-NEXT: v_or_b32_e32 v0, s0, v0
; GFX1332-FAKE16-NEXT: buffer_store_b16 v0, off, s[8:11], null
; GFX1332-FAKE16-NEXT: s_endpgm
@@ -13032,6 +13174,7 @@ define amdgpu_kernel void @uniform_add_i16(ptr addrspace(1) %result, ptr addrspa
; GFX1164-TRUE16-NEXT: v_mbcnt_hi_u32_b32 v4, s7, v0
; GFX1164-TRUE16-NEXT: ; implicit-def: $vgpr0_lo16
; GFX1164-TRUE16-NEXT: v_cmpx_eq_u32_e32 0, v4
+; GFX1164-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1164-TRUE16-NEXT: s_cbranch_execz .LBB16_4
; GFX1164-TRUE16-NEXT: ; %bb.1:
; GFX1164-TRUE16-NEXT: s_waitcnt lgkmcnt(0)
@@ -13060,18 +13203,19 @@ define amdgpu_kernel void @uniform_add_i16(ptr addrspace(1) %result, ptr addrspa
; GFX1164-TRUE16-NEXT: s_waitcnt_depctr depctr_va_vdst(0)
; GFX1164-TRUE16-NEXT: v_and_or_b32 v0, v1, s14, v0
; GFX1164-TRUE16-NEXT: v_mov_b32_e32 v3, v1
-; GFX1164-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2)
+; GFX1164-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_4) | instid1(VALU_DEP_2)
; GFX1164-TRUE16-NEXT: v_mov_b32_e32 v2, v0
; GFX1164-TRUE16-NEXT: buffer_atomic_cmpswap_b32 v[2:3], off, s[4:7], 0 glc
; GFX1164-TRUE16-NEXT: s_waitcnt vmcnt(0)
; GFX1164-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc, v2, v1
; GFX1164-TRUE16-NEXT: v_mov_b32_e32 v1, v2
; GFX1164-TRUE16-NEXT: s_or_b64 s[10:11], vcc, s[10:11]
-; GFX1164-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1164-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1164-TRUE16-NEXT: s_and_not1_b64 exec, exec, s[10:11]
; GFX1164-TRUE16-NEXT: s_cbranch_execnz .LBB16_2
; GFX1164-TRUE16-NEXT: ; %bb.3: ; %atomicrmw.end
; GFX1164-TRUE16-NEXT: s_or_b64 exec, exec, s[10:11]
+; GFX1164-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164-TRUE16-NEXT: v_lshrrev_b32_e32 v0, s2, v2
; GFX1164-TRUE16-NEXT: .LBB16_4: ; %Flow
; GFX1164-TRUE16-NEXT: s_or_b64 exec, exec, s[8:9]
@@ -13096,6 +13240,7 @@ define amdgpu_kernel void @uniform_add_i16(ptr addrspace(1) %result, ptr addrspa
; GFX1164-FAKE16-NEXT: v_mbcnt_hi_u32_b32 v4, s7, v0
; GFX1164-FAKE16-NEXT: ; implicit-def: $vgpr0
; GFX1164-FAKE16-NEXT: v_cmpx_eq_u32_e32 0, v4
+; GFX1164-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1164-FAKE16-NEXT: s_cbranch_execz .LBB16_4
; GFX1164-FAKE16-NEXT: ; %bb.1:
; GFX1164-FAKE16-NEXT: s_waitcnt lgkmcnt(0)
@@ -13124,18 +13269,19 @@ define amdgpu_kernel void @uniform_add_i16(ptr addrspace(1) %result, ptr addrspa
; GFX1164-FAKE16-NEXT: s_waitcnt_depctr depctr_va_vdst(0)
; GFX1164-FAKE16-NEXT: v_and_or_b32 v0, v1, s14, v0
; GFX1164-FAKE16-NEXT: v_mov_b32_e32 v3, v1
-; GFX1164-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2)
+; GFX1164-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_4) | instid1(VALU_DEP_2)
; GFX1164-FAKE16-NEXT: v_mov_b32_e32 v2, v0
; GFX1164-FAKE16-NEXT: buffer_atomic_cmpswap_b32 v[2:3], off, s[4:7], 0 glc
; GFX1164-FAKE16-NEXT: s_waitcnt vmcnt(0)
; GFX1164-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc, v2, v1
; GFX1164-FAKE16-NEXT: v_mov_b32_e32 v1, v2
; GFX1164-FAKE16-NEXT: s_or_b64 s[10:11], vcc, s[10:11]
-; GFX1164-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1164-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1164-FAKE16-NEXT: s_and_not1_b64 exec, exec, s[10:11]
; GFX1164-FAKE16-NEXT: s_cbranch_execnz .LBB16_2
; GFX1164-FAKE16-NEXT: ; %bb.3: ; %atomicrmw.end
; GFX1164-FAKE16-NEXT: s_or_b64 exec, exec, s[10:11]
+; GFX1164-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164-FAKE16-NEXT: v_lshrrev_b32_e32 v0, s2, v2
; GFX1164-FAKE16-NEXT: .LBB16_4: ; %Flow
; GFX1164-FAKE16-NEXT: s_or_b64 exec, exec, s[8:9]
@@ -13158,7 +13304,7 @@ define amdgpu_kernel void @uniform_add_i16(ptr addrspace(1) %result, ptr addrspa
; GFX1132-TRUE16-NEXT: v_mbcnt_lo_u32_b32 v4, s6, 0
; GFX1132-TRUE16-NEXT: s_mov_b32 s9, exec_lo
; GFX1132-TRUE16-NEXT: ; implicit-def: $vgpr0_lo16
-; GFX1132-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1132-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132-TRUE16-NEXT: v_cmpx_eq_u32_e32 0, v4
; GFX1132-TRUE16-NEXT: s_cbranch_execz .LBB16_4
; GFX1132-TRUE16-NEXT: ; %bb.1:
@@ -13192,11 +13338,12 @@ define amdgpu_kernel void @uniform_add_i16(ptr addrspace(1) %result, ptr addrspa
; GFX1132-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v1
; GFX1132-TRUE16-NEXT: v_mov_b32_e32 v1, v2
; GFX1132-TRUE16-NEXT: s_or_b32 s10, vcc_lo, s10
-; GFX1132-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1132-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1132-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s10
; GFX1132-TRUE16-NEXT: s_cbranch_execnz .LBB16_2
; GFX1132-TRUE16-NEXT: ; %bb.3: ; %atomicrmw.end
; GFX1132-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s10
+; GFX1132-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1132-TRUE16-NEXT: v_lshrrev_b32_e32 v0, s2, v2
; GFX1132-TRUE16-NEXT: .LBB16_4: ; %Flow
; GFX1132-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s9
@@ -13219,7 +13366,7 @@ define amdgpu_kernel void @uniform_add_i16(ptr addrspace(1) %result, ptr addrspa
; GFX1132-FAKE16-NEXT: v_mbcnt_lo_u32_b32 v4, s6, 0
; GFX1132-FAKE16-NEXT: s_mov_b32 s9, exec_lo
; GFX1132-FAKE16-NEXT: ; implicit-def: $vgpr0
-; GFX1132-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1132-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132-FAKE16-NEXT: v_cmpx_eq_u32_e32 0, v4
; GFX1132-FAKE16-NEXT: s_cbranch_execz .LBB16_4
; GFX1132-FAKE16-NEXT: ; %bb.1:
@@ -13253,11 +13400,12 @@ define amdgpu_kernel void @uniform_add_i16(ptr addrspace(1) %result, ptr addrspa
; GFX1132-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v1
; GFX1132-FAKE16-NEXT: v_mov_b32_e32 v1, v2
; GFX1132-FAKE16-NEXT: s_or_b32 s10, vcc_lo, s10
-; GFX1132-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1132-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1132-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s10
; GFX1132-FAKE16-NEXT: s_cbranch_execnz .LBB16_2
; GFX1132-FAKE16-NEXT: ; %bb.3: ; %atomicrmw.end
; GFX1132-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s10
+; GFX1132-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1132-FAKE16-NEXT: v_lshrrev_b32_e32 v0, s2, v2
; GFX1132-FAKE16-NEXT: .LBB16_4: ; %Flow
; GFX1132-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s9
@@ -13282,6 +13430,7 @@ define amdgpu_kernel void @uniform_add_i16(ptr addrspace(1) %result, ptr addrspa
; GFX1264-TRUE16-NEXT: v_mbcnt_hi_u32_b32 v4, s7, v0
; GFX1264-TRUE16-NEXT: ; implicit-def: $vgpr0_lo16
; GFX1264-TRUE16-NEXT: v_cmpx_eq_u32_e32 0, v4
+; GFX1264-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1264-TRUE16-NEXT: s_cbranch_execz .LBB16_4
; GFX1264-TRUE16-NEXT: ; %bb.1:
; GFX1264-TRUE16-NEXT: s_wait_kmcnt 0x0
@@ -13318,12 +13467,14 @@ define amdgpu_kernel void @uniform_add_i16(ptr addrspace(1) %result, ptr addrspa
; GFX1264-TRUE16-NEXT: s_wait_loadcnt 0x0
; GFX1264-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc, v2, v1
; GFX1264-TRUE16-NEXT: v_mov_b32_e32 v1, v2
+; GFX1264-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1264-TRUE16-NEXT: s_or_b64 s[10:11], vcc, s[10:11]
-; GFX1264-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1264-TRUE16-NEXT: s_and_not1_b64 exec, exec, s[10:11]
+; GFX1264-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1264-TRUE16-NEXT: s_cbranch_execnz .LBB16_2
; GFX1264-TRUE16-NEXT: ; %bb.3: ; %atomicrmw.end
; GFX1264-TRUE16-NEXT: s_or_b64 exec, exec, s[10:11]
+; GFX1264-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1264-TRUE16-NEXT: v_lshrrev_b32_e32 v0, s2, v2
; GFX1264-TRUE16-NEXT: .LBB16_4: ; %Flow
; GFX1264-TRUE16-NEXT: s_or_b64 exec, exec, s[8:9]
@@ -13349,6 +13500,7 @@ define amdgpu_kernel void @uniform_add_i16(ptr addrspace(1) %result, ptr addrspa
; GFX1264-FAKE16-NEXT: v_mbcnt_hi_u32_b32 v4, s7, v0
; GFX1264-FAKE16-NEXT: ; implicit-def: $vgpr0
; GFX1264-FAKE16-NEXT: v_cmpx_eq_u32_e32 0, v4
+; GFX1264-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1264-FAKE16-NEXT: s_cbranch_execz .LBB16_4
; GFX1264-FAKE16-NEXT: ; %bb.1:
; GFX1264-FAKE16-NEXT: s_wait_kmcnt 0x0
@@ -13385,12 +13537,14 @@ define amdgpu_kernel void @uniform_add_i16(ptr addrspace(1) %result, ptr addrspa
; GFX1264-FAKE16-NEXT: s_wait_loadcnt 0x0
; GFX1264-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc, v2, v1
; GFX1264-FAKE16-NEXT: v_mov_b32_e32 v1, v2
+; GFX1264-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1264-FAKE16-NEXT: s_or_b64 s[10:11], vcc, s[10:11]
-; GFX1264-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1264-FAKE16-NEXT: s_and_not1_b64 exec, exec, s[10:11]
+; GFX1264-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1264-FAKE16-NEXT: s_cbranch_execnz .LBB16_2
; GFX1264-FAKE16-NEXT: ; %bb.3: ; %atomicrmw.end
; GFX1264-FAKE16-NEXT: s_or_b64 exec, exec, s[10:11]
+; GFX1264-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1264-FAKE16-NEXT: v_lshrrev_b32_e32 v0, s2, v2
; GFX1264-FAKE16-NEXT: .LBB16_4: ; %Flow
; GFX1264-FAKE16-NEXT: s_or_b64 exec, exec, s[8:9]
@@ -13414,7 +13568,7 @@ define amdgpu_kernel void @uniform_add_i16(ptr addrspace(1) %result, ptr addrspa
; GFX1232-TRUE16-NEXT: v_mbcnt_lo_u32_b32 v4, s6, 0
; GFX1232-TRUE16-NEXT: s_mov_b32 s9, exec_lo
; GFX1232-TRUE16-NEXT: ; implicit-def: $vgpr0_lo16
-; GFX1232-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1232-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1232-TRUE16-NEXT: v_cmpx_eq_u32_e32 0, v4
; GFX1232-TRUE16-NEXT: s_cbranch_execz .LBB16_4
; GFX1232-TRUE16-NEXT: ; %bb.1:
@@ -13453,9 +13607,11 @@ define amdgpu_kernel void @uniform_add_i16(ptr addrspace(1) %result, ptr addrspa
; GFX1232-TRUE16-NEXT: s_or_b32 s10, vcc_lo, s10
; GFX1232-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX1232-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s10
+; GFX1232-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1232-TRUE16-NEXT: s_cbranch_execnz .LBB16_2
; GFX1232-TRUE16-NEXT: ; %bb.3: ; %atomicrmw.end
; GFX1232-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s10
+; GFX1232-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1232-TRUE16-NEXT: v_lshrrev_b32_e32 v0, s2, v2
; GFX1232-TRUE16-NEXT: .LBB16_4: ; %Flow
; GFX1232-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s9
@@ -13479,7 +13635,7 @@ define amdgpu_kernel void @uniform_add_i16(ptr addrspace(1) %result, ptr addrspa
; GFX1232-FAKE16-NEXT: v_mbcnt_lo_u32_b32 v4, s6, 0
; GFX1232-FAKE16-NEXT: s_mov_b32 s9, exec_lo
; GFX1232-FAKE16-NEXT: ; implicit-def: $vgpr0
-; GFX1232-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1232-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1232-FAKE16-NEXT: v_cmpx_eq_u32_e32 0, v4
; GFX1232-FAKE16-NEXT: s_cbranch_execz .LBB16_4
; GFX1232-FAKE16-NEXT: ; %bb.1:
@@ -13518,9 +13674,11 @@ define amdgpu_kernel void @uniform_add_i16(ptr addrspace(1) %result, ptr addrspa
; GFX1232-FAKE16-NEXT: s_or_b32 s10, vcc_lo, s10
; GFX1232-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX1232-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s10
+; GFX1232-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1232-FAKE16-NEXT: s_cbranch_execnz .LBB16_2
; GFX1232-FAKE16-NEXT: ; %bb.3: ; %atomicrmw.end
; GFX1232-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s10
+; GFX1232-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1232-FAKE16-NEXT: v_lshrrev_b32_e32 v0, s2, v2
; GFX1232-FAKE16-NEXT: .LBB16_4: ; %Flow
; GFX1232-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s9
@@ -13546,6 +13704,7 @@ define amdgpu_kernel void @uniform_add_i16(ptr addrspace(1) %result, ptr addrspa
; GFX1364-TRUE16-NEXT: v_mbcnt_hi_u32_b32 v4, s7, v0
; GFX1364-TRUE16-NEXT: ; implicit-def: $vgpr0_lo16
; GFX1364-TRUE16-NEXT: v_cmpx_eq_u32_e32 0, v4
+; GFX1364-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1364-TRUE16-NEXT: s_cbranch_execz .LBB16_4
; GFX1364-TRUE16-NEXT: ; %bb.1:
; GFX1364-TRUE16-NEXT: s_wait_kmcnt 0x0
@@ -13577,12 +13736,14 @@ define amdgpu_kernel void @uniform_add_i16(ptr addrspace(1) %result, ptr addrspa
; GFX1364-TRUE16-NEXT: s_wait_loadcnt 0x0
; GFX1364-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc, v2, v1
; GFX1364-TRUE16-NEXT: v_mov_b32_e32 v1, v2
+; GFX1364-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1364-TRUE16-NEXT: s_or_b64 s[2:3], vcc, s[2:3]
-; GFX1364-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1364-TRUE16-NEXT: s_and_not1_b64 exec, exec, s[2:3]
+; GFX1364-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1364-TRUE16-NEXT: s_cbranch_execnz .LBB16_2
; GFX1364-TRUE16-NEXT: ; %bb.3: ; %atomicrmw.end
; GFX1364-TRUE16-NEXT: s_or_b64 exec, exec, s[2:3]
+; GFX1364-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1364-TRUE16-NEXT: v_lshrrev_b32_e32 v0, s11, v2
; GFX1364-TRUE16-NEXT: .LBB16_4: ; %Flow
; GFX1364-TRUE16-NEXT: s_or_b64 exec, exec, s[8:9]
@@ -13607,6 +13768,7 @@ define amdgpu_kernel void @uniform_add_i16(ptr addrspace(1) %result, ptr addrspa
; GFX1364-FAKE16-NEXT: v_mbcnt_hi_u32_b32 v4, s7, v0
; GFX1364-FAKE16-NEXT: ; implicit-def: $vgpr0
; GFX1364-FAKE16-NEXT: v_cmpx_eq_u32_e32 0, v4
+; GFX1364-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1364-FAKE16-NEXT: s_cbranch_execz .LBB16_4
; GFX1364-FAKE16-NEXT: ; %bb.1:
; GFX1364-FAKE16-NEXT: s_wait_kmcnt 0x0
@@ -13638,12 +13800,14 @@ define amdgpu_kernel void @uniform_add_i16(ptr addrspace(1) %result, ptr addrspa
; GFX1364-FAKE16-NEXT: s_wait_loadcnt 0x0
; GFX1364-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc, v2, v1
; GFX1364-FAKE16-NEXT: v_mov_b32_e32 v1, v2
+; GFX1364-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1364-FAKE16-NEXT: s_or_b64 s[2:3], vcc, s[2:3]
-; GFX1364-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1364-FAKE16-NEXT: s_and_not1_b64 exec, exec, s[2:3]
+; GFX1364-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1364-FAKE16-NEXT: s_cbranch_execnz .LBB16_2
; GFX1364-FAKE16-NEXT: ; %bb.3: ; %atomicrmw.end
; GFX1364-FAKE16-NEXT: s_or_b64 exec, exec, s[2:3]
+; GFX1364-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1364-FAKE16-NEXT: v_lshrrev_b32_e32 v0, s11, v2
; GFX1364-FAKE16-NEXT: .LBB16_4: ; %Flow
; GFX1364-FAKE16-NEXT: s_or_b64 exec, exec, s[8:9]
@@ -13666,7 +13830,7 @@ define amdgpu_kernel void @uniform_add_i16(ptr addrspace(1) %result, ptr addrspa
; GFX1332-TRUE16-NEXT: v_mbcnt_lo_u32_b32 v4, s6, 0
; GFX1332-TRUE16-NEXT: s_mov_b32 s9, exec_lo
; GFX1332-TRUE16-NEXT: ; implicit-def: $vgpr0_lo16
-; GFX1332-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1332-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1332-TRUE16-NEXT: v_cmpx_eq_u32_e32 0, v4
; GFX1332-TRUE16-NEXT: s_cbranch_execz .LBB16_4
; GFX1332-TRUE16-NEXT: ; %bb.1:
@@ -13698,11 +13862,12 @@ define amdgpu_kernel void @uniform_add_i16(ptr addrspace(1) %result, ptr addrspa
; GFX1332-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v1
; GFX1332-TRUE16-NEXT: v_mov_b32_e32 v1, v2
; GFX1332-TRUE16-NEXT: s_or_b32 s10, vcc_lo, s10
-; GFX1332-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1332-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1332-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s10
; GFX1332-TRUE16-NEXT: s_cbranch_execnz .LBB16_2
; GFX1332-TRUE16-NEXT: ; %bb.3: ; %atomicrmw.end
; GFX1332-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s10
+; GFX1332-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1332-TRUE16-NEXT: v_lshrrev_b32_e32 v0, s2, v2
; GFX1332-TRUE16-NEXT: .LBB16_4: ; %Flow
; GFX1332-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s9
@@ -13725,7 +13890,7 @@ define amdgpu_kernel void @uniform_add_i16(ptr addrspace(1) %result, ptr addrspa
; GFX1332-FAKE16-NEXT: v_mbcnt_lo_u32_b32 v4, s6, 0
; GFX1332-FAKE16-NEXT: s_mov_b32 s9, exec_lo
; GFX1332-FAKE16-NEXT: ; implicit-def: $vgpr0
-; GFX1332-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1332-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1332-FAKE16-NEXT: v_cmpx_eq_u32_e32 0, v4
; GFX1332-FAKE16-NEXT: s_cbranch_execz .LBB16_4
; GFX1332-FAKE16-NEXT: ; %bb.1:
@@ -13757,11 +13922,12 @@ define amdgpu_kernel void @uniform_add_i16(ptr addrspace(1) %result, ptr addrspa
; GFX1332-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v1
; GFX1332-FAKE16-NEXT: v_mov_b32_e32 v1, v2
; GFX1332-FAKE16-NEXT: s_or_b32 s10, vcc_lo, s10
-; GFX1332-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1332-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1332-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s10
; GFX1332-FAKE16-NEXT: s_cbranch_execnz .LBB16_2
; GFX1332-FAKE16-NEXT: ; %bb.3: ; %atomicrmw.end
; GFX1332-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s10
+; GFX1332-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1332-FAKE16-NEXT: v_lshrrev_b32_e32 v0, s2, v2
; GFX1332-FAKE16-NEXT: .LBB16_4: ; %Flow
; GFX1332-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s9
@@ -14016,12 +14182,14 @@ define amdgpu_kernel void @uniform_xchg_i16(ptr addrspace(1) %result, ptr addrsp
; GFX1164-NEXT: s_waitcnt vmcnt(0)
; GFX1164-NEXT: v_cmp_eq_u32_e32 vcc, v2, v1
; GFX1164-NEXT: v_mov_b32_e32 v1, v2
+; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1164-NEXT: s_or_b64 s[8:9], vcc, s[8:9]
-; GFX1164-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164-NEXT: s_and_not1_b64 exec, exec, s[8:9]
+; GFX1164-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164-NEXT: s_cbranch_execnz .LBB17_1
; GFX1164-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX1164-NEXT: s_or_b64 exec, exec, s[8:9]
+; GFX1164-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164-NEXT: v_lshrrev_b32_e32 v0, s2, v2
; GFX1164-NEXT: s_mov_b32 s3, 0x31016000
; GFX1164-NEXT: s_mov_b32 s2, -1
@@ -14059,11 +14227,12 @@ define amdgpu_kernel void @uniform_xchg_i16(ptr addrspace(1) %result, ptr addrsp
; GFX1132-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v1
; GFX1132-NEXT: v_mov_b32_e32 v1, v2
; GFX1132-NEXT: s_or_b32 s9, vcc_lo, s9
-; GFX1132-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1132-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1132-NEXT: s_and_not1_b32 exec_lo, exec_lo, s9
; GFX1132-NEXT: s_cbranch_execnz .LBB17_1
; GFX1132-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX1132-NEXT: s_or_b32 exec_lo, exec_lo, s9
+; GFX1132-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1132-NEXT: v_lshrrev_b32_e32 v0, s2, v2
; GFX1132-NEXT: s_mov_b32 s3, 0x31016000
; GFX1132-NEXT: s_mov_b32 s2, -1
@@ -14101,12 +14270,14 @@ define amdgpu_kernel void @uniform_xchg_i16(ptr addrspace(1) %result, ptr addrsp
; GFX1264-NEXT: s_wait_loadcnt 0x0
; GFX1264-NEXT: v_cmp_eq_u32_e32 vcc, v2, v1
; GFX1264-NEXT: v_mov_b32_e32 v1, v2
+; GFX1264-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1264-NEXT: s_or_b64 s[8:9], vcc, s[8:9]
-; GFX1264-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1264-NEXT: s_and_not1_b64 exec, exec, s[8:9]
+; GFX1264-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1264-NEXT: s_cbranch_execnz .LBB17_1
; GFX1264-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX1264-NEXT: s_or_b64 exec, exec, s[8:9]
+; GFX1264-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1264-NEXT: v_lshrrev_b32_e32 v0, s2, v2
; GFX1264-NEXT: s_mov_b32 s3, 0x31016000
; GFX1264-NEXT: s_mov_b32 s2, -1
@@ -14146,9 +14317,11 @@ define amdgpu_kernel void @uniform_xchg_i16(ptr addrspace(1) %result, ptr addrsp
; GFX1232-NEXT: s_or_b32 s9, vcc_lo, s9
; GFX1232-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX1232-NEXT: s_and_not1_b32 exec_lo, exec_lo, s9
+; GFX1232-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1232-NEXT: s_cbranch_execnz .LBB17_1
; GFX1232-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX1232-NEXT: s_or_b32 exec_lo, exec_lo, s9
+; GFX1232-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1232-NEXT: v_lshrrev_b32_e32 v0, s2, v2
; GFX1232-NEXT: s_mov_b32 s3, 0x31016000
; GFX1232-NEXT: s_mov_b32 s2, -1
@@ -14184,12 +14357,14 @@ define amdgpu_kernel void @uniform_xchg_i16(ptr addrspace(1) %result, ptr addrsp
; GFX1364-NEXT: s_wait_loadcnt 0x0
; GFX1364-NEXT: v_cmp_eq_u32_e32 vcc, v2, v1
; GFX1364-NEXT: v_mov_b32_e32 v1, v2
+; GFX1364-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1364-NEXT: s_or_b64 s[2:3], vcc, s[2:3]
-; GFX1364-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1364-NEXT: s_and_not1_b64 exec, exec, s[2:3]
+; GFX1364-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1364-NEXT: s_cbranch_execnz .LBB17_1
; GFX1364-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX1364-NEXT: s_or_b64 exec, exec, s[2:3]
+; GFX1364-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1364-NEXT: v_lshrrev_b32_e32 v0, s8, v2
; GFX1364-NEXT: s_mov_b32 s3, 0x31016000
; GFX1364-NEXT: s_mov_b32 s2, -1
@@ -14225,11 +14400,12 @@ define amdgpu_kernel void @uniform_xchg_i16(ptr addrspace(1) %result, ptr addrsp
; GFX1332-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v1
; GFX1332-NEXT: v_mov_b32_e32 v1, v2
; GFX1332-NEXT: s_or_b32 s9, vcc_lo, s9
-; GFX1332-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1332-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1332-NEXT: s_and_not1_b32 exec_lo, exec_lo, s9
; GFX1332-NEXT: s_cbranch_execnz .LBB17_1
; GFX1332-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX1332-NEXT: s_or_b32 exec_lo, exec_lo, s9
+; GFX1332-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1332-NEXT: v_lshrrev_b32_e32 v0, s2, v2
; GFX1332-NEXT: s_mov_b32 s3, 0x31016000
; GFX1332-NEXT: s_mov_b32 s2, -1
@@ -14490,12 +14666,14 @@ define amdgpu_kernel void @uniform_fadd_f16(ptr addrspace(1) %result, ptr addrsp
; GFX1164-TRUE16-NEXT: s_waitcnt vmcnt(0)
; GFX1164-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc, v2, v1
; GFX1164-TRUE16-NEXT: v_mov_b32_e32 v1, v2
+; GFX1164-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1164-TRUE16-NEXT: s_or_b64 s[8:9], vcc, s[8:9]
-; GFX1164-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164-TRUE16-NEXT: s_and_not1_b64 exec, exec, s[8:9]
+; GFX1164-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164-TRUE16-NEXT: s_cbranch_execnz .LBB18_1
; GFX1164-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX1164-TRUE16-NEXT: s_or_b64 exec, exec, s[8:9]
+; GFX1164-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164-TRUE16-NEXT: v_lshrrev_b32_e32 v0, s2, v2
; GFX1164-TRUE16-NEXT: s_mov_b32 s3, 0x31016000
; GFX1164-TRUE16-NEXT: s_mov_b32 s2, -1
@@ -14538,12 +14716,14 @@ define amdgpu_kernel void @uniform_fadd_f16(ptr addrspace(1) %result, ptr addrsp
; GFX1164-FAKE16-NEXT: s_waitcnt vmcnt(0)
; GFX1164-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc, v2, v1
; GFX1164-FAKE16-NEXT: v_mov_b32_e32 v1, v2
+; GFX1164-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1164-FAKE16-NEXT: s_or_b64 s[8:9], vcc, s[8:9]
-; GFX1164-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164-FAKE16-NEXT: s_and_not1_b64 exec, exec, s[8:9]
+; GFX1164-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164-FAKE16-NEXT: s_cbranch_execnz .LBB18_1
; GFX1164-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX1164-FAKE16-NEXT: s_or_b64 exec, exec, s[8:9]
+; GFX1164-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164-FAKE16-NEXT: v_lshrrev_b32_e32 v0, s2, v2
; GFX1164-FAKE16-NEXT: s_mov_b32 s3, 0x31016000
; GFX1164-FAKE16-NEXT: s_mov_b32 s2, -1
@@ -14586,11 +14766,12 @@ define amdgpu_kernel void @uniform_fadd_f16(ptr addrspace(1) %result, ptr addrsp
; GFX1132-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v1
; GFX1132-TRUE16-NEXT: v_mov_b32_e32 v1, v2
; GFX1132-TRUE16-NEXT: s_or_b32 s9, vcc_lo, s9
-; GFX1132-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1132-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1132-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s9
; GFX1132-TRUE16-NEXT: s_cbranch_execnz .LBB18_1
; GFX1132-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX1132-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s9
+; GFX1132-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1132-TRUE16-NEXT: v_lshrrev_b32_e32 v0, s2, v2
; GFX1132-TRUE16-NEXT: s_mov_b32 s3, 0x31016000
; GFX1132-TRUE16-NEXT: s_mov_b32 s2, -1
@@ -14633,11 +14814,12 @@ define amdgpu_kernel void @uniform_fadd_f16(ptr addrspace(1) %result, ptr addrsp
; GFX1132-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v1
; GFX1132-FAKE16-NEXT: v_mov_b32_e32 v1, v2
; GFX1132-FAKE16-NEXT: s_or_b32 s9, vcc_lo, s9
-; GFX1132-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1132-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1132-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s9
; GFX1132-FAKE16-NEXT: s_cbranch_execnz .LBB18_1
; GFX1132-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX1132-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s9
+; GFX1132-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1132-FAKE16-NEXT: v_lshrrev_b32_e32 v0, s2, v2
; GFX1132-FAKE16-NEXT: s_mov_b32 s3, 0x31016000
; GFX1132-FAKE16-NEXT: s_mov_b32 s2, -1
@@ -14680,12 +14862,14 @@ define amdgpu_kernel void @uniform_fadd_f16(ptr addrspace(1) %result, ptr addrsp
; GFX1264-TRUE16-NEXT: s_wait_loadcnt 0x0
; GFX1264-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc, v2, v1
; GFX1264-TRUE16-NEXT: v_mov_b32_e32 v1, v2
+; GFX1264-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1264-TRUE16-NEXT: s_or_b64 s[8:9], vcc, s[8:9]
-; GFX1264-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1264-TRUE16-NEXT: s_and_not1_b64 exec, exec, s[8:9]
+; GFX1264-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1264-TRUE16-NEXT: s_cbranch_execnz .LBB18_1
; GFX1264-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX1264-TRUE16-NEXT: s_or_b64 exec, exec, s[8:9]
+; GFX1264-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1264-TRUE16-NEXT: v_lshrrev_b32_e32 v0, s2, v2
; GFX1264-TRUE16-NEXT: s_mov_b32 s3, 0x31016000
; GFX1264-TRUE16-NEXT: s_mov_b32 s2, -1
@@ -14728,12 +14912,14 @@ define amdgpu_kernel void @uniform_fadd_f16(ptr addrspace(1) %result, ptr addrsp
; GFX1264-FAKE16-NEXT: s_wait_loadcnt 0x0
; GFX1264-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc, v2, v1
; GFX1264-FAKE16-NEXT: v_mov_b32_e32 v1, v2
+; GFX1264-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1264-FAKE16-NEXT: s_or_b64 s[8:9], vcc, s[8:9]
-; GFX1264-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1264-FAKE16-NEXT: s_and_not1_b64 exec, exec, s[8:9]
+; GFX1264-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1264-FAKE16-NEXT: s_cbranch_execnz .LBB18_1
; GFX1264-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX1264-FAKE16-NEXT: s_or_b64 exec, exec, s[8:9]
+; GFX1264-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1264-FAKE16-NEXT: v_lshrrev_b32_e32 v0, s2, v2
; GFX1264-FAKE16-NEXT: s_mov_b32 s3, 0x31016000
; GFX1264-FAKE16-NEXT: s_mov_b32 s2, -1
@@ -14778,9 +14964,11 @@ define amdgpu_kernel void @uniform_fadd_f16(ptr addrspace(1) %result, ptr addrsp
; GFX1232-TRUE16-NEXT: s_or_b32 s9, vcc_lo, s9
; GFX1232-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX1232-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s9
+; GFX1232-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1232-TRUE16-NEXT: s_cbranch_execnz .LBB18_1
; GFX1232-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX1232-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s9
+; GFX1232-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1232-TRUE16-NEXT: v_lshrrev_b32_e32 v0, s2, v2
; GFX1232-TRUE16-NEXT: s_mov_b32 s3, 0x31016000
; GFX1232-TRUE16-NEXT: s_mov_b32 s2, -1
@@ -14825,9 +15013,11 @@ define amdgpu_kernel void @uniform_fadd_f16(ptr addrspace(1) %result, ptr addrsp
; GFX1232-FAKE16-NEXT: s_or_b32 s9, vcc_lo, s9
; GFX1232-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX1232-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s9
+; GFX1232-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1232-FAKE16-NEXT: s_cbranch_execnz .LBB18_1
; GFX1232-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX1232-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s9
+; GFX1232-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1232-FAKE16-NEXT: v_lshrrev_b32_e32 v0, s2, v2
; GFX1232-FAKE16-NEXT: s_mov_b32 s3, 0x31016000
; GFX1232-FAKE16-NEXT: s_mov_b32 s2, -1
@@ -14868,12 +15058,14 @@ define amdgpu_kernel void @uniform_fadd_f16(ptr addrspace(1) %result, ptr addrsp
; GFX1364-TRUE16-NEXT: s_wait_loadcnt 0x0
; GFX1364-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc, v2, v1
; GFX1364-TRUE16-NEXT: v_mov_b32_e32 v1, v2
+; GFX1364-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1364-TRUE16-NEXT: s_or_b64 s[2:3], vcc, s[2:3]
-; GFX1364-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1364-TRUE16-NEXT: s_and_not1_b64 exec, exec, s[2:3]
+; GFX1364-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1364-TRUE16-NEXT: s_cbranch_execnz .LBB18_1
; GFX1364-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX1364-TRUE16-NEXT: s_or_b64 exec, exec, s[2:3]
+; GFX1364-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1364-TRUE16-NEXT: v_lshrrev_b32_e32 v0, s9, v2
; GFX1364-TRUE16-NEXT: s_mov_b32 s3, 0x31016000
; GFX1364-TRUE16-NEXT: s_mov_b32 s2, -1
@@ -14914,12 +15106,14 @@ define amdgpu_kernel void @uniform_fadd_f16(ptr addrspace(1) %result, ptr addrsp
; GFX1364-FAKE16-NEXT: s_wait_loadcnt 0x0
; GFX1364-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc, v2, v1
; GFX1364-FAKE16-NEXT: v_mov_b32_e32 v1, v2
+; GFX1364-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1364-FAKE16-NEXT: s_or_b64 s[2:3], vcc, s[2:3]
-; GFX1364-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1364-FAKE16-NEXT: s_and_not1_b64 exec, exec, s[2:3]
+; GFX1364-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1364-FAKE16-NEXT: s_cbranch_execnz .LBB18_1
; GFX1364-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX1364-FAKE16-NEXT: s_or_b64 exec, exec, s[2:3]
+; GFX1364-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1364-FAKE16-NEXT: v_lshrrev_b32_e32 v0, s9, v2
; GFX1364-FAKE16-NEXT: s_mov_b32 s3, 0x31016000
; GFX1364-FAKE16-NEXT: s_mov_b32 s2, -1
@@ -14960,11 +15154,12 @@ define amdgpu_kernel void @uniform_fadd_f16(ptr addrspace(1) %result, ptr addrsp
; GFX1332-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v1
; GFX1332-TRUE16-NEXT: v_mov_b32_e32 v1, v2
; GFX1332-TRUE16-NEXT: s_or_b32 s9, vcc_lo, s9
-; GFX1332-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1332-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1332-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s9
; GFX1332-TRUE16-NEXT: s_cbranch_execnz .LBB18_1
; GFX1332-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX1332-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s9
+; GFX1332-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1332-TRUE16-NEXT: v_lshrrev_b32_e32 v0, s2, v2
; GFX1332-TRUE16-NEXT: s_mov_b32 s3, 0x31016000
; GFX1332-TRUE16-NEXT: s_mov_b32 s2, -1
@@ -15005,11 +15200,12 @@ define amdgpu_kernel void @uniform_fadd_f16(ptr addrspace(1) %result, ptr addrsp
; GFX1332-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v1
; GFX1332-FAKE16-NEXT: v_mov_b32_e32 v1, v2
; GFX1332-FAKE16-NEXT: s_or_b32 s9, vcc_lo, s9
-; GFX1332-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1332-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1332-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s9
; GFX1332-FAKE16-NEXT: s_cbranch_execnz .LBB18_1
; GFX1332-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX1332-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s9
+; GFX1332-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1332-FAKE16-NEXT: v_lshrrev_b32_e32 v0, s2, v2
; GFX1332-FAKE16-NEXT: s_mov_b32 s3, 0x31016000
; GFX1332-FAKE16-NEXT: s_mov_b32 s2, -1
@@ -15309,12 +15505,14 @@ define amdgpu_kernel void @uniform_fadd_bf16(ptr addrspace(1) %result, ptr addrs
; GFX1164-NEXT: v_cmp_eq_u32_e32 vcc, v2, v1
; GFX1164-NEXT: v_mov_b32_e32 v1, v2
; GFX1164-NEXT: s_waitcnt_depctr depctr_va_vcc(0)
+; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1164-NEXT: s_or_b64 s[8:9], vcc, s[8:9]
-; GFX1164-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164-NEXT: s_and_not1_b64 exec, exec, s[8:9]
+; GFX1164-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164-NEXT: s_cbranch_execnz .LBB19_1
; GFX1164-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX1164-NEXT: s_or_b64 exec, exec, s[8:9]
+; GFX1164-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164-NEXT: v_lshrrev_b32_e32 v0, s2, v2
; GFX1164-NEXT: s_mov_b32 s3, 0x31016000
; GFX1164-NEXT: s_mov_b32 s2, -1
@@ -15366,11 +15564,12 @@ define amdgpu_kernel void @uniform_fadd_bf16(ptr addrspace(1) %result, ptr addrs
; GFX1132-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v1
; GFX1132-NEXT: v_mov_b32_e32 v1, v2
; GFX1132-NEXT: s_or_b32 s8, vcc_lo, s8
-; GFX1132-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1132-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1132-NEXT: s_and_not1_b32 exec_lo, exec_lo, s8
; GFX1132-NEXT: s_cbranch_execnz .LBB19_1
; GFX1132-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX1132-NEXT: s_or_b32 exec_lo, exec_lo, s8
+; GFX1132-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1132-NEXT: v_lshrrev_b32_e32 v0, s2, v2
; GFX1132-NEXT: s_mov_b32 s3, 0x31016000
; GFX1132-NEXT: s_mov_b32 s2, -1
@@ -15422,12 +15621,14 @@ define amdgpu_kernel void @uniform_fadd_bf16(ptr addrspace(1) %result, ptr addrs
; GFX1264-NEXT: s_wait_loadcnt 0x0
; GFX1264-NEXT: v_cmp_eq_u32_e32 vcc, v2, v1
; GFX1264-NEXT: v_mov_b32_e32 v1, v2
+; GFX1264-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1264-NEXT: s_or_b64 s[8:9], vcc, s[8:9]
-; GFX1264-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1264-NEXT: s_and_not1_b64 exec, exec, s[8:9]
+; GFX1264-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1264-NEXT: s_cbranch_execnz .LBB19_1
; GFX1264-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX1264-NEXT: s_or_b64 exec, exec, s[8:9]
+; GFX1264-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1264-NEXT: v_lshrrev_b32_e32 v0, s2, v2
; GFX1264-NEXT: s_mov_b32 s3, 0x31016000
; GFX1264-NEXT: s_mov_b32 s2, -1
@@ -15481,9 +15682,11 @@ define amdgpu_kernel void @uniform_fadd_bf16(ptr addrspace(1) %result, ptr addrs
; GFX1232-NEXT: s_or_b32 s8, vcc_lo, s8
; GFX1232-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX1232-NEXT: s_and_not1_b32 exec_lo, exec_lo, s8
+; GFX1232-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1232-NEXT: s_cbranch_execnz .LBB19_1
; GFX1232-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX1232-NEXT: s_or_b32 exec_lo, exec_lo, s8
+; GFX1232-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1232-NEXT: v_lshrrev_b32_e32 v0, s2, v2
; GFX1232-NEXT: s_mov_b32 s3, 0x31016000
; GFX1232-NEXT: s_mov_b32 s2, -1
@@ -15524,12 +15727,14 @@ define amdgpu_kernel void @uniform_fadd_bf16(ptr addrspace(1) %result, ptr addrs
; GFX1364-NEXT: s_wait_loadcnt 0x0
; GFX1364-NEXT: v_cmp_eq_u32_e32 vcc, v2, v1
; GFX1364-NEXT: v_mov_b32_e32 v1, v2
+; GFX1364-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1364-NEXT: s_or_b64 s[2:3], vcc, s[2:3]
-; GFX1364-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1364-NEXT: s_and_not1_b64 exec, exec, s[2:3]
+; GFX1364-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1364-NEXT: s_cbranch_execnz .LBB19_1
; GFX1364-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX1364-NEXT: s_or_b64 exec, exec, s[2:3]
+; GFX1364-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1364-NEXT: v_lshrrev_b32_e32 v0, s9, v2
; GFX1364-NEXT: s_mov_b32 s3, 0x31016000
; GFX1364-NEXT: s_mov_b32 s2, -1
@@ -15570,11 +15775,12 @@ define amdgpu_kernel void @uniform_fadd_bf16(ptr addrspace(1) %result, ptr addrs
; GFX1332-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v1
; GFX1332-NEXT: v_mov_b32_e32 v1, v2
; GFX1332-NEXT: s_or_b32 s9, vcc_lo, s9
-; GFX1332-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1332-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1332-NEXT: s_and_not1_b32 exec_lo, exec_lo, s9
; GFX1332-NEXT: s_cbranch_execnz .LBB19_1
; GFX1332-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX1332-NEXT: s_or_b32 exec_lo, exec_lo, s9
+; GFX1332-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1332-NEXT: v_lshrrev_b32_e32 v0, s2, v2
; GFX1332-NEXT: s_mov_b32 s3, 0x31016000
; GFX1332-NEXT: s_mov_b32 s2, -1
@@ -15795,9 +16001,10 @@ define amdgpu_kernel void @uniform_fadd_v2f16(ptr addrspace(1) %result, ptr addr
; GFX1164-NEXT: s_waitcnt vmcnt(0)
; GFX1164-NEXT: v_cmp_eq_u32_e32 vcc, v2, v1
; GFX1164-NEXT: v_mov_b32_e32 v1, v2
+; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1164-NEXT: s_or_b64 s[8:9], vcc, s[8:9]
-; GFX1164-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164-NEXT: s_and_not1_b64 exec, exec, s[8:9]
+; GFX1164-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164-NEXT: s_cbranch_execnz .LBB20_1
; GFX1164-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX1164-NEXT: s_or_b64 exec, exec, s[8:9]
@@ -15830,7 +16037,7 @@ define amdgpu_kernel void @uniform_fadd_v2f16(ptr addrspace(1) %result, ptr addr
; GFX1132-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v1
; GFX1132-NEXT: v_mov_b32_e32 v1, v2
; GFX1132-NEXT: s_or_b32 s9, vcc_lo, s9
-; GFX1132-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1132-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1132-NEXT: s_and_not1_b32 exec_lo, exec_lo, s9
; GFX1132-NEXT: s_cbranch_execnz .LBB20_1
; GFX1132-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -15864,9 +16071,10 @@ define amdgpu_kernel void @uniform_fadd_v2f16(ptr addrspace(1) %result, ptr addr
; GFX1264-NEXT: s_wait_loadcnt 0x0
; GFX1264-NEXT: v_cmp_eq_u32_e32 vcc, v2, v1
; GFX1264-NEXT: v_mov_b32_e32 v1, v2
+; GFX1264-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1264-NEXT: s_or_b64 s[8:9], vcc, s[8:9]
-; GFX1264-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1264-NEXT: s_and_not1_b64 exec, exec, s[8:9]
+; GFX1264-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1264-NEXT: s_cbranch_execnz .LBB20_1
; GFX1264-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX1264-NEXT: s_or_b64 exec, exec, s[8:9]
@@ -15901,6 +16109,7 @@ define amdgpu_kernel void @uniform_fadd_v2f16(ptr addrspace(1) %result, ptr addr
; GFX1232-NEXT: s_or_b32 s9, vcc_lo, s9
; GFX1232-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX1232-NEXT: s_and_not1_b32 exec_lo, exec_lo, s9
+; GFX1232-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1232-NEXT: s_cbranch_execnz .LBB20_1
; GFX1232-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX1232-NEXT: s_or_b32 exec_lo, exec_lo, s9
@@ -15933,9 +16142,10 @@ define amdgpu_kernel void @uniform_fadd_v2f16(ptr addrspace(1) %result, ptr addr
; GFX1364-NEXT: s_wait_loadcnt 0x0
; GFX1364-NEXT: v_cmp_eq_u32_e32 vcc, v2, v1
; GFX1364-NEXT: v_mov_b32_e32 v1, v2
+; GFX1364-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1364-NEXT: s_or_b64 s[8:9], vcc, s[8:9]
-; GFX1364-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1364-NEXT: s_and_not1_b64 exec, exec, s[8:9]
+; GFX1364-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1364-NEXT: s_cbranch_execnz .LBB20_1
; GFX1364-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX1364-NEXT: s_or_b64 exec, exec, s[8:9]
@@ -15968,7 +16178,7 @@ define amdgpu_kernel void @uniform_fadd_v2f16(ptr addrspace(1) %result, ptr addr
; GFX1332-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v1
; GFX1332-NEXT: v_mov_b32_e32 v1, v2
; GFX1332-NEXT: s_or_b32 s9, vcc_lo, s9
-; GFX1332-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1332-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1332-NEXT: s_and_not1_b32 exec_lo, exec_lo, s9
; GFX1332-NEXT: s_cbranch_execnz .LBB20_1
; GFX1332-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -16289,9 +16499,10 @@ define amdgpu_kernel void @uniform_fadd_v2bf16(ptr addrspace(1) %result, ptr add
; GFX1164-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc, v2, v1
; GFX1164-TRUE16-NEXT: v_mov_b32_e32 v1, v2
; GFX1164-TRUE16-NEXT: s_waitcnt_depctr depctr_va_vcc(0)
+; GFX1164-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1164-TRUE16-NEXT: s_or_b64 s[8:9], vcc, s[8:9]
-; GFX1164-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164-TRUE16-NEXT: s_and_not1_b64 exec, exec, s[8:9]
+; GFX1164-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164-TRUE16-NEXT: s_cbranch_execnz .LBB21_1
; GFX1164-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX1164-TRUE16-NEXT: s_set_inst_prefetch_distance 0x2
@@ -16350,9 +16561,10 @@ define amdgpu_kernel void @uniform_fadd_v2bf16(ptr addrspace(1) %result, ptr add
; GFX1164-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc, v2, v1
; GFX1164-FAKE16-NEXT: v_mov_b32_e32 v1, v2
; GFX1164-FAKE16-NEXT: s_waitcnt_depctr depctr_va_vcc(0)
+; GFX1164-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1164-FAKE16-NEXT: s_or_b64 s[2:3], vcc, s[2:3]
-; GFX1164-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164-FAKE16-NEXT: s_and_not1_b64 exec, exec, s[2:3]
+; GFX1164-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164-FAKE16-NEXT: s_cbranch_execnz .LBB21_1
; GFX1164-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX1164-FAKE16-NEXT: s_set_inst_prefetch_distance 0x2
@@ -16411,7 +16623,7 @@ define amdgpu_kernel void @uniform_fadd_v2bf16(ptr addrspace(1) %result, ptr add
; GFX1132-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v1
; GFX1132-TRUE16-NEXT: v_mov_b32_e32 v1, v2
; GFX1132-TRUE16-NEXT: s_or_b32 s8, vcc_lo, s8
-; GFX1132-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1132-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1132-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s8
; GFX1132-TRUE16-NEXT: s_cbranch_execnz .LBB21_1
; GFX1132-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -16457,7 +16669,7 @@ define amdgpu_kernel void @uniform_fadd_v2bf16(ptr addrspace(1) %result, ptr add
; GFX1132-FAKE16-NEXT: v_add3_u32 v3, v3, v0, 0x7fff
; GFX1132-FAKE16-NEXT: v_add3_u32 v4, v4, v2, 0x7fff
; GFX1132-FAKE16-NEXT: v_cmp_u_f32_e64 s0, v0, v0
-; GFX1132-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX1132-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_4)
; GFX1132-FAKE16-NEXT: v_cndmask_b32_e32 v2, v4, v6, vcc_lo
; GFX1132-FAKE16-NEXT: v_cndmask_b32_e64 v0, v3, v5, s0
; GFX1132-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
@@ -16468,7 +16680,7 @@ define amdgpu_kernel void @uniform_fadd_v2bf16(ptr addrspace(1) %result, ptr add
; GFX1132-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v1
; GFX1132-FAKE16-NEXT: v_mov_b32_e32 v1, v2
; GFX1132-FAKE16-NEXT: s_or_b32 s1, vcc_lo, s1
-; GFX1132-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1132-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1132-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
; GFX1132-FAKE16-NEXT: s_cbranch_execnz .LBB21_1
; GFX1132-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -16522,14 +16734,14 @@ define amdgpu_kernel void @uniform_fadd_v2bf16(ptr addrspace(1) %result, ptr add
; GFX1264-TRUE16-NEXT: v_lshrrev_b32_e32 v0, 16, v2
; GFX1264-TRUE16-NEXT: v_mov_b16_e32 v0.h, v3.l
; GFX1264-TRUE16-NEXT: v_mov_b32_e32 v3, v1
-; GFX1264-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2)
+; GFX1264-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_4) | instid1(VALU_DEP_2)
; GFX1264-TRUE16-NEXT: v_mov_b32_e32 v2, v0
; GFX1264-TRUE16-NEXT: buffer_atomic_cmpswap_b32 v[2:3], off, s[4:7], null th:TH_ATOMIC_RETURN scope:SCOPE_SYS
; GFX1264-TRUE16-NEXT: s_wait_loadcnt 0x0
; GFX1264-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc, v2, v1
; GFX1264-TRUE16-NEXT: v_mov_b32_e32 v1, v2
; GFX1264-TRUE16-NEXT: s_or_b64 s[8:9], vcc, s[8:9]
-; GFX1264-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1264-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1264-TRUE16-NEXT: s_and_not1_b64 exec, exec, s[8:9]
; GFX1264-TRUE16-NEXT: s_cbranch_execnz .LBB21_1
; GFX1264-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -16573,20 +16785,20 @@ define amdgpu_kernel void @uniform_fadd_v2bf16(ptr addrspace(1) %result, ptr add
; GFX1264-FAKE16-NEXT: v_add3_u32 v4, v4, v2, 0x7fff
; GFX1264-FAKE16-NEXT: v_cmp_u_f32_e64 s[0:1], v0, v0
; GFX1264-FAKE16-NEXT: s_wait_alu depctr_va_vcc(0)
-; GFX1264-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_2)
+; GFX1264-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_2) | instid1(VALU_DEP_1)
; GFX1264-FAKE16-NEXT: v_cndmask_b32_e32 v2, v4, v6, vcc
; GFX1264-FAKE16-NEXT: s_wait_alu depctr_va_sdst(0)
; GFX1264-FAKE16-NEXT: v_cndmask_b32_e64 v0, v3, v5, s[0:1]
-; GFX1264-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX1264-FAKE16-NEXT: v_perm_b32 v0, v2, v0, 0x7060302
; GFX1264-FAKE16-NEXT: v_mov_b32_e32 v3, v1
+; GFX1264-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_4) | instid1(VALU_DEP_2)
; GFX1264-FAKE16-NEXT: v_mov_b32_e32 v2, v0
; GFX1264-FAKE16-NEXT: buffer_atomic_cmpswap_b32 v[2:3], off, s[4:7], null th:TH_ATOMIC_RETURN scope:SCOPE_SYS
; GFX1264-FAKE16-NEXT: s_wait_loadcnt 0x0
; GFX1264-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc, v2, v1
; GFX1264-FAKE16-NEXT: v_mov_b32_e32 v1, v2
; GFX1264-FAKE16-NEXT: s_or_b64 s[2:3], vcc, s[2:3]
-; GFX1264-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1264-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1264-FAKE16-NEXT: s_and_not1_b64 exec, exec, s[2:3]
; GFX1264-FAKE16-NEXT: s_cbranch_execnz .LBB21_1
; GFX1264-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -16647,6 +16859,7 @@ define amdgpu_kernel void @uniform_fadd_v2bf16(ptr addrspace(1) %result, ptr add
; GFX1232-TRUE16-NEXT: s_or_b32 s8, vcc_lo, s8
; GFX1232-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX1232-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s8
+; GFX1232-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1232-TRUE16-NEXT: s_cbranch_execnz .LBB21_1
; GFX1232-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX1232-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s8
@@ -16689,12 +16902,12 @@ define amdgpu_kernel void @uniform_fadd_v2bf16(ptr addrspace(1) %result, ptr add
; GFX1232-FAKE16-NEXT: v_add3_u32 v4, v4, v2, 0x7fff
; GFX1232-FAKE16-NEXT: v_cmp_u_f32_e64 s0, v0, v0
; GFX1232-FAKE16-NEXT: s_wait_alu depctr_va_vcc(0)
-; GFX1232-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_2)
+; GFX1232-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_2) | instid1(VALU_DEP_1)
; GFX1232-FAKE16-NEXT: v_cndmask_b32_e32 v2, v4, v6, vcc_lo
; GFX1232-FAKE16-NEXT: s_wait_alu depctr_va_sdst(0)
; GFX1232-FAKE16-NEXT: v_cndmask_b32_e64 v0, v3, v5, s0
-; GFX1232-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1232-FAKE16-NEXT: v_perm_b32 v0, v2, v0, 0x7060302
+; GFX1232-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1232-FAKE16-NEXT: v_dual_mov_b32 v3, v1 :: v_dual_mov_b32 v2, v0
; GFX1232-FAKE16-NEXT: buffer_atomic_cmpswap_b32 v[2:3], off, s[4:7], null th:TH_ATOMIC_RETURN scope:SCOPE_SYS
; GFX1232-FAKE16-NEXT: s_wait_loadcnt 0x0
@@ -16703,6 +16916,7 @@ define amdgpu_kernel void @uniform_fadd_v2bf16(ptr addrspace(1) %result, ptr add
; GFX1232-FAKE16-NEXT: s_or_b32 s1, vcc_lo, s1
; GFX1232-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX1232-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
+; GFX1232-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1232-FAKE16-NEXT: s_cbranch_execnz .LBB21_1
; GFX1232-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX1232-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s1
@@ -16735,9 +16949,10 @@ define amdgpu_kernel void @uniform_fadd_v2bf16(ptr addrspace(1) %result, ptr add
; GFX1364-NEXT: s_wait_loadcnt 0x0
; GFX1364-NEXT: v_cmp_eq_u32_e32 vcc, v2, v1
; GFX1364-NEXT: v_mov_b32_e32 v1, v2
+; GFX1364-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1364-NEXT: s_or_b64 s[8:9], vcc, s[8:9]
-; GFX1364-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1364-NEXT: s_and_not1_b64 exec, exec, s[8:9]
+; GFX1364-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1364-NEXT: s_cbranch_execnz .LBB21_1
; GFX1364-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX1364-NEXT: s_or_b64 exec, exec, s[8:9]
@@ -16770,7 +16985,7 @@ define amdgpu_kernel void @uniform_fadd_v2bf16(ptr addrspace(1) %result, ptr add
; GFX1332-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v1
; GFX1332-NEXT: v_mov_b32_e32 v1, v2
; GFX1332-NEXT: s_or_b32 s9, vcc_lo, s9
-; GFX1332-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1332-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1332-NEXT: s_and_not1_b32 exec_lo, exec_lo, s9
; GFX1332-NEXT: s_cbranch_execnz .LBB21_1
; GFX1332-NEXT: ; %bb.2: ; %atomicrmw.end
diff --git a/llvm/test/CodeGen/AMDGPU/atomic_optimizations_local_pointer.ll b/llvm/test/CodeGen/AMDGPU/atomic_optimizations_local_pointer.ll
index c424df950c5c0b..db3f853848b4eb 100644
--- a/llvm/test/CodeGen/AMDGPU/atomic_optimizations_local_pointer.ll
+++ b/llvm/test/CodeGen/AMDGPU/atomic_optimizations_local_pointer.ll
@@ -173,6 +173,7 @@ define amdgpu_kernel void @add_i32_constant(ptr addrspace(1) %out) {
; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164-NEXT: v_mbcnt_hi_u32_b32 v0, s3, v0
; GFX1164-NEXT: v_cmpx_eq_u32_e32 0, v0
+; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1164-NEXT: s_cbranch_execz .LBB0_2
; GFX1164-NEXT: ; %bb.1:
; GFX1164-NEXT: s_bcnt1_i32_b64 s2, s[2:3]
@@ -200,7 +201,7 @@ define amdgpu_kernel void @add_i32_constant(ptr addrspace(1) %out) {
; GFX1132-NEXT: s_mov_b32 s0, exec_lo
; GFX1132-NEXT: v_mbcnt_lo_u32_b32 v0, s1, 0
; GFX1132-NEXT: ; implicit-def: $vgpr1
-; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX1132-NEXT: s_cbranch_execz .LBB0_2
; GFX1132-NEXT: ; %bb.1:
@@ -231,6 +232,7 @@ define amdgpu_kernel void @add_i32_constant(ptr addrspace(1) %out) {
; GFX1364-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1364-NEXT: v_mbcnt_hi_u32_b32 v0, s3, v0
; GFX1364-NEXT: v_cmpx_eq_u32_e32 0, v0
+; GFX1364-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1364-NEXT: s_cbranch_execz .LBB0_2
; GFX1364-NEXT: ; %bb.1:
; GFX1364-NEXT: s_bcnt1_i32_b64 s2, s[2:3]
@@ -258,7 +260,7 @@ define amdgpu_kernel void @add_i32_constant(ptr addrspace(1) %out) {
; GFX1332-NEXT: s_mov_b32 s0, exec_lo
; GFX1332-NEXT: v_mbcnt_lo_u32_b32 v0, s1, 0
; GFX1332-NEXT: ; implicit-def: $vgpr1
-; GFX1332-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1332-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1332-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX1332-NEXT: s_cbranch_execz .LBB0_2
; GFX1332-NEXT: ; %bb.1:
@@ -449,6 +451,7 @@ define amdgpu_kernel void @add_i32_uniform(ptr addrspace(1) %out, i32 %additive)
; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164-NEXT: v_mbcnt_hi_u32_b32 v0, s3, v0
; GFX1164-NEXT: v_cmpx_eq_u32_e32 0, v0
+; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1164-NEXT: s_cbranch_execz .LBB1_2
; GFX1164-NEXT: ; %bb.1:
; GFX1164-NEXT: s_bcnt1_i32_b64 s2, s[2:3]
@@ -479,7 +482,7 @@ define amdgpu_kernel void @add_i32_uniform(ptr addrspace(1) %out, i32 %additive)
; GFX1132-NEXT: s_mov_b32 s1, exec_lo
; GFX1132-NEXT: v_mbcnt_lo_u32_b32 v0, s2, 0
; GFX1132-NEXT: ; implicit-def: $vgpr1
-; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX1132-NEXT: s_cbranch_execz .LBB1_2
; GFX1132-NEXT: ; %bb.1:
@@ -512,6 +515,7 @@ define amdgpu_kernel void @add_i32_uniform(ptr addrspace(1) %out, i32 %additive)
; GFX1364-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1364-NEXT: v_mbcnt_hi_u32_b32 v0, s3, v0
; GFX1364-NEXT: v_cmpx_eq_u32_e32 0, v0
+; GFX1364-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1364-NEXT: s_cbranch_execz .LBB1_2
; GFX1364-NEXT: ; %bb.1:
; GFX1364-NEXT: s_bcnt1_i32_b64 s2, s[2:3]
@@ -542,7 +546,7 @@ define amdgpu_kernel void @add_i32_uniform(ptr addrspace(1) %out, i32 %additive)
; GFX1332-NEXT: s_mov_b32 s1, exec_lo
; GFX1332-NEXT: v_mbcnt_lo_u32_b32 v0, s2, 0
; GFX1332-NEXT: ; implicit-def: $vgpr1
-; GFX1332-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1332-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1332-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX1332-NEXT: s_cbranch_execz .LBB1_2
; GFX1332-NEXT: ; %bb.1:
@@ -793,8 +797,8 @@ define amdgpu_kernel void @add_i32_varying(ptr addrspace(1) %out) {
; GFX1164_ITERATIVE-NEXT: v_mbcnt_hi_u32_b32 v1, exec_hi, v1
; GFX1164_ITERATIVE-NEXT: v_cmp_eq_u32_e32 vcc, 0, v1
; GFX1164_ITERATIVE-NEXT: ; implicit-def: $vgpr1
+; GFX1164_ITERATIVE-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1164_ITERATIVE-NEXT: s_and_saveexec_b64 s[0:1], vcc
-; GFX1164_ITERATIVE-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164_ITERATIVE-NEXT: s_xor_b64 s[0:1], exec, s[0:1]
; GFX1164_ITERATIVE-NEXT: s_cbranch_execz .LBB2_4
; GFX1164_ITERATIVE-NEXT: ; %bb.3:
@@ -1085,7 +1089,7 @@ define amdgpu_kernel void @add_i32_varying(ptr addrspace(1) %out) {
; GFX1164_DPP-NEXT: v_writelane_b32 v3, s3, 32
; GFX1164_DPP-NEXT: s_waitcnt_depctr depctr_sa_sdst(0)
; GFX1164_DPP-NEXT: s_mov_b64 exec, s[0:1]
-; GFX1164_DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1164_DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_4) | instid1(SALU_CYCLE_1)
; GFX1164_DPP-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
; GFX1164_DPP-NEXT: v_mov_b32_e32 v4, 0
; GFX1164_DPP-NEXT: s_or_saveexec_b64 s[0:1], -1
@@ -1094,6 +1098,7 @@ define amdgpu_kernel void @add_i32_varying(ptr addrspace(1) %out) {
; GFX1164_DPP-NEXT: v_cmp_eq_u32_e32 vcc, 0, v0
; GFX1164_DPP-NEXT: s_mov_b32 s2, -1
; GFX1164_DPP-NEXT: ; implicit-def: $vgpr0
+; GFX1164_DPP-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1164_DPP-NEXT: s_and_saveexec_b64 s[0:1], vcc
; GFX1164_DPP-NEXT: s_cbranch_execz .LBB2_2
; GFX1164_DPP-NEXT: ; %bb.1:
@@ -1139,6 +1144,7 @@ define amdgpu_kernel void @add_i32_varying(ptr addrspace(1) %out) {
; GFX1132_DPP-NEXT: s_or_saveexec_b32 s0, -1
; GFX1132_DPP-NEXT: v_writelane_b32 v3, s1, 16
; GFX1132_DPP-NEXT: s_mov_b32 exec_lo, s0
+; GFX1132_DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1132_DPP-NEXT: v_cmp_eq_u32_e32 vcc_lo, 0, v0
; GFX1132_DPP-NEXT: s_mov_b32 s0, s2
; GFX1132_DPP-NEXT: s_mov_b32 s2, -1
@@ -1195,7 +1201,7 @@ define amdgpu_kernel void @add_i32_varying(ptr addrspace(1) %out) {
; GFX1364-NEXT: v_readlane_b32 s6, v1, 63
; GFX1364-NEXT: v_writelane_b32 v3, s3, 32
; GFX1364-NEXT: s_mov_b64 exec, s[0:1]
-; GFX1364-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1364-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_4) | instid1(SALU_CYCLE_1)
; GFX1364-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
; GFX1364-NEXT: v_mov_b32_e32 v4, 0
; GFX1364-NEXT: s_or_saveexec_b64 s[0:1], -1
@@ -1204,6 +1210,7 @@ define amdgpu_kernel void @add_i32_varying(ptr addrspace(1) %out) {
; GFX1364-NEXT: v_cmp_eq_u32_e32 vcc, 0, v0
; GFX1364-NEXT: s_mov_b32 s2, -1
; GFX1364-NEXT: ; implicit-def: $vgpr0
+; GFX1364-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1364-NEXT: s_and_saveexec_b64 s[0:1], vcc
; GFX1364-NEXT: s_cbranch_execz .LBB2_2
; GFX1364-NEXT: ; %bb.1:
@@ -1248,6 +1255,7 @@ define amdgpu_kernel void @add_i32_varying(ptr addrspace(1) %out) {
; GFX1332-NEXT: s_or_saveexec_b32 s0, -1
; GFX1332-NEXT: v_writelane_b32 v3, s1, 16
; GFX1332-NEXT: s_mov_b32 exec_lo, s0
+; GFX1332-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1332-NEXT: v_cmp_eq_u32_e32 vcc_lo, 0, v0
; GFX1332-NEXT: s_mov_b32 s0, s2
; GFX1332-NEXT: s_mov_b32 s2, -1
@@ -1438,6 +1446,7 @@ define amdgpu_kernel void @add_i32_varying_nouse() {
; GFX1164_ITERATIVE-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164_ITERATIVE-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
; GFX1164_ITERATIVE-NEXT: v_cmpx_eq_u32_e32 0, v0
+; GFX1164_ITERATIVE-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1164_ITERATIVE-NEXT: s_xor_b64 s[0:1], exec, s[0:1]
; GFX1164_ITERATIVE-NEXT: s_cbranch_execz .LBB3_4
; GFX1164_ITERATIVE-NEXT: ; %bb.3:
@@ -1466,9 +1475,10 @@ define amdgpu_kernel void @add_i32_varying_nouse() {
; GFX1132_ITERATIVE-NEXT: ; %bb.2: ; %ComputeEnd
; GFX1132_ITERATIVE-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
; GFX1132_ITERATIVE-NEXT: s_mov_b32 s1, exec_lo
-; GFX1132_ITERATIVE-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1132_ITERATIVE-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132_ITERATIVE-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX1132_ITERATIVE-NEXT: s_xor_b32 s1, exec_lo, s1
+; GFX1132_ITERATIVE-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1132_ITERATIVE-NEXT: s_cbranch_execz .LBB3_4
; GFX1132_ITERATIVE-NEXT: ; %bb.3:
; GFX1132_ITERATIVE-NEXT: v_dual_mov_b32 v0, 0 :: v_dual_mov_b32 v1, s0
@@ -1633,7 +1643,7 @@ define amdgpu_kernel void @add_i32_varying_nouse() {
; GFX1164_DPP-NEXT: v_mov_b32_e32 v0, 0
; GFX1164_DPP-NEXT: v_mov_b32_e32 v3, v1
; GFX1164_DPP-NEXT: s_mov_b64 s[0:1], exec
-; GFX1164_DPP-NEXT: s_delay_alu instid0(VALU_DEP_3)
+; GFX1164_DPP-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164_DPP-NEXT: v_cmpx_eq_u32_e32 0, v4
; GFX1164_DPP-NEXT: s_cbranch_execz .LBB3_2
; GFX1164_DPP-NEXT: ; %bb.1:
@@ -1663,7 +1673,7 @@ define amdgpu_kernel void @add_i32_varying_nouse() {
; GFX1132_DPP-NEXT: v_mbcnt_lo_u32_b32 v4, exec_lo, 0
; GFX1132_DPP-NEXT: v_dual_mov_b32 v0, 0 :: v_dual_mov_b32 v3, v1
; GFX1132_DPP-NEXT: s_mov_b32 s0, exec_lo
-; GFX1132_DPP-NEXT: s_delay_alu instid0(VALU_DEP_2)
+; GFX1132_DPP-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132_DPP-NEXT: v_cmpx_eq_u32_e32 0, v4
; GFX1132_DPP-NEXT: s_cbranch_execz .LBB3_2
; GFX1132_DPP-NEXT: ; %bb.1:
@@ -1703,6 +1713,7 @@ define amdgpu_kernel void @add_i32_varying_nouse() {
; GFX1364-NEXT: v_mov_b32_e32 v3, v1
; GFX1364-NEXT: s_mov_b64 s[0:1], exec
; GFX1364-NEXT: v_cmpx_eq_u32_e32 0, v4
+; GFX1364-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1364-NEXT: s_cbranch_execz .LBB3_2
; GFX1364-NEXT: ; %bb.1:
; GFX1364-NEXT: ds_add_u32 v0, v3
@@ -1731,7 +1742,7 @@ define amdgpu_kernel void @add_i32_varying_nouse() {
; GFX1332-NEXT: v_mbcnt_lo_u32_b32 v4, exec_lo, 0
; GFX1332-NEXT: v_dual_mov_b32 v0, 0 :: v_dual_mov_b32 v3, v1
; GFX1332-NEXT: s_mov_b32 s0, exec_lo
-; GFX1332-NEXT: s_delay_alu instid0(VALU_DEP_2)
+; GFX1332-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1332-NEXT: v_cmpx_eq_u32_e32 0, v4
; GFX1332-NEXT: s_cbranch_execz .LBB3_2
; GFX1332-NEXT: ; %bb.1:
@@ -1912,6 +1923,7 @@ define amdgpu_kernel void @add_i64_constant(ptr addrspace(1) %out) {
; GFX1164-NEXT: v_mbcnt_hi_u32_b32 v2, s3, v0
; GFX1164-NEXT: ; implicit-def: $vgpr0_vgpr1
; GFX1164-NEXT: v_cmpx_eq_u32_e32 0, v2
+; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1164-NEXT: s_cbranch_execz .LBB4_2
; GFX1164-NEXT: ; %bb.1:
; GFX1164-NEXT: s_bcnt1_i32_b64 s2, s[2:3]
@@ -1941,7 +1953,7 @@ define amdgpu_kernel void @add_i64_constant(ptr addrspace(1) %out) {
; GFX1132-NEXT: s_mov_b32 s0, exec_lo
; GFX1132-NEXT: v_mbcnt_lo_u32_b32 v2, s1, 0
; GFX1132-NEXT: ; implicit-def: $vgpr0_vgpr1
-; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132-NEXT: v_cmpx_eq_u32_e32 0, v2
; GFX1132-NEXT: s_cbranch_execz .LBB4_2
; GFX1132-NEXT: ; %bb.1:
@@ -1975,6 +1987,7 @@ define amdgpu_kernel void @add_i64_constant(ptr addrspace(1) %out) {
; GFX1364-NEXT: v_mbcnt_hi_u32_b32 v2, s3, v0
; GFX1364-NEXT: ; implicit-def: $vgpr0_vgpr1
; GFX1364-NEXT: v_cmpx_eq_u32_e32 0, v2
+; GFX1364-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1364-NEXT: s_cbranch_execz .LBB4_2
; GFX1364-NEXT: ; %bb.1:
; GFX1364-NEXT: s_bcnt1_i32_b64 s2, s[2:3]
@@ -2004,7 +2017,7 @@ define amdgpu_kernel void @add_i64_constant(ptr addrspace(1) %out) {
; GFX1332-NEXT: s_mov_b32 s0, exec_lo
; GFX1332-NEXT: v_mbcnt_lo_u32_b32 v2, s1, 0
; GFX1332-NEXT: ; implicit-def: $vgpr0_vgpr1
-; GFX1332-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1332-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1332-NEXT: v_cmpx_eq_u32_e32 0, v2
; GFX1332-NEXT: s_cbranch_execz .LBB4_2
; GFX1332-NEXT: ; %bb.1:
@@ -2234,6 +2247,7 @@ define amdgpu_kernel void @add_i64_uniform(ptr addrspace(1) %out, i64 %additive)
; GFX1164-NEXT: v_mbcnt_hi_u32_b32 v2, s7, v0
; GFX1164-NEXT: ; implicit-def: $vgpr0_vgpr1
; GFX1164-NEXT: v_cmpx_eq_u32_e32 0, v2
+; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1164-NEXT: s_cbranch_execz .LBB5_2
; GFX1164-NEXT: ; %bb.1:
; GFX1164-NEXT: s_bcnt1_i32_b64 s6, s[6:7]
@@ -2250,10 +2264,10 @@ define amdgpu_kernel void @add_i64_uniform(ptr addrspace(1) %out, i64 %additive)
; GFX1164-NEXT: buffer_gl0_inv
; GFX1164-NEXT: .LBB5_2:
; GFX1164-NEXT: s_or_b64 exec, exec, s[4:5]
+; GFX1164-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(VALU_DEP_1)
; GFX1164-NEXT: v_readfirstlane_b32 s5, v1
; GFX1164-NEXT: v_readfirstlane_b32 s4, v0
; GFX1164-NEXT: s_waitcnt lgkmcnt(0)
-; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1164-NEXT: v_mad_u64_u32 v[0:1], null, s2, v2, s[4:5]
; GFX1164-NEXT: s_mov_b32 s2, -1
; GFX1164-NEXT: v_mad_u64_u32 v[3:4], null, s3, v2, v[1:2]
@@ -2269,7 +2283,7 @@ define amdgpu_kernel void @add_i64_uniform(ptr addrspace(1) %out, i64 %additive)
; GFX1132-NEXT: s_mov_b32 s4, exec_lo
; GFX1132-NEXT: v_mbcnt_lo_u32_b32 v2, s6, 0
; GFX1132-NEXT: ; implicit-def: $vgpr0_vgpr1
-; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132-NEXT: v_cmpx_eq_u32_e32 0, v2
; GFX1132-NEXT: s_cbranch_execz .LBB5_2
; GFX1132-NEXT: ; %bb.1:
@@ -2287,10 +2301,10 @@ define amdgpu_kernel void @add_i64_uniform(ptr addrspace(1) %out, i64 %additive)
; GFX1132-NEXT: buffer_gl0_inv
; GFX1132-NEXT: .LBB5_2:
; GFX1132-NEXT: s_or_b32 exec_lo, exec_lo, s4
+; GFX1132-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(VALU_DEP_1)
; GFX1132-NEXT: v_readfirstlane_b32 s5, v1
; GFX1132-NEXT: v_readfirstlane_b32 s4, v0
; GFX1132-NEXT: s_waitcnt lgkmcnt(0)
-; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1132-NEXT: v_mad_u64_u32 v[0:1], null, s2, v2, s[4:5]
; GFX1132-NEXT: s_mov_b32 s2, -1
; GFX1132-NEXT: v_mad_u64_u32 v[3:4], null, s3, v2, v[1:2]
@@ -2310,6 +2324,7 @@ define amdgpu_kernel void @add_i64_uniform(ptr addrspace(1) %out, i64 %additive)
; GFX1364-NEXT: v_mbcnt_hi_u32_b32 v2, s7, v0
; GFX1364-NEXT: ; implicit-def: $vgpr0_vgpr1
; GFX1364-NEXT: v_cmpx_eq_u32_e32 0, v2
+; GFX1364-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1364-NEXT: s_cbranch_execz .LBB5_2
; GFX1364-NEXT: ; %bb.1:
; GFX1364-NEXT: s_bcnt1_i32_b64 s8, s[6:7]
@@ -2324,10 +2339,10 @@ define amdgpu_kernel void @add_i64_uniform(ptr addrspace(1) %out, i64 %additive)
; GFX1364-NEXT: global_inv scope:SCOPE_SE
; GFX1364-NEXT: .LBB5_2:
; GFX1364-NEXT: s_or_b64 exec, exec, s[4:5]
+; GFX1364-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(VALU_DEP_1)
; GFX1364-NEXT: v_readfirstlane_b32 s5, v1
; GFX1364-NEXT: v_readfirstlane_b32 s4, v0
; GFX1364-NEXT: s_wait_kmcnt 0x0
-; GFX1364-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1364-NEXT: v_mad_co_u64_u32 v[0:1], null, s2, v2, s[4:5]
; GFX1364-NEXT: s_mov_b32 s2, -1
; GFX1364-NEXT: v_mad_co_u64_u32 v[1:2], null, s3, v2, v[1:2]
@@ -2343,7 +2358,7 @@ define amdgpu_kernel void @add_i64_uniform(ptr addrspace(1) %out, i64 %additive)
; GFX1332-NEXT: v_mbcnt_lo_u32_b32 v2, s6, 0
; GFX1332-NEXT: s_mov_b32 s7, exec_lo
; GFX1332-NEXT: ; implicit-def: $vgpr0_vgpr1
-; GFX1332-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1332-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1332-NEXT: v_cmpx_eq_u32_e32 0, v2
; GFX1332-NEXT: s_cbranch_execz .LBB5_2
; GFX1332-NEXT: ; %bb.1:
@@ -2358,10 +2373,10 @@ define amdgpu_kernel void @add_i64_uniform(ptr addrspace(1) %out, i64 %additive)
; GFX1332-NEXT: global_inv scope:SCOPE_SE
; GFX1332-NEXT: .LBB5_2:
; GFX1332-NEXT: s_or_b32 exec_lo, exec_lo, s7
+; GFX1332-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(VALU_DEP_1)
; GFX1332-NEXT: v_readfirstlane_b32 s5, v1
; GFX1332-NEXT: v_readfirstlane_b32 s4, v0
; GFX1332-NEXT: s_wait_kmcnt 0x0
-; GFX1332-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1332-NEXT: v_mad_co_u64_u32 v[0:1], null, s2, v2, s[4:5]
; GFX1332-NEXT: s_mov_b32 s2, -1
; GFX1332-NEXT: v_mad_co_u64_u32 v[1:2], null, s3, v2, v[1:2]
@@ -2640,8 +2655,8 @@ define amdgpu_kernel void @add_i64_varying(ptr addrspace(1) %out) {
; GFX1164_ITERATIVE-NEXT: v_mbcnt_hi_u32_b32 v2, exec_hi, v2
; GFX1164_ITERATIVE-NEXT: v_cmp_eq_u32_e32 vcc, 0, v2
; GFX1164_ITERATIVE-NEXT: ; implicit-def: $vgpr2_vgpr3
+; GFX1164_ITERATIVE-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1164_ITERATIVE-NEXT: s_and_saveexec_b64 s[2:3], vcc
-; GFX1164_ITERATIVE-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164_ITERATIVE-NEXT: s_xor_b64 s[2:3], exec, s[2:3]
; GFX1164_ITERATIVE-NEXT: s_cbranch_execz .LBB6_4
; GFX1164_ITERATIVE-NEXT: ; %bb.3:
@@ -2704,7 +2719,7 @@ define amdgpu_kernel void @add_i64_varying(ptr addrspace(1) %out) {
; GFX1132_ITERATIVE-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
; GFX1132_ITERATIVE-NEXT: v_readfirstlane_b32 s2, v2
; GFX1132_ITERATIVE-NEXT: v_readfirstlane_b32 s3, v3
-; GFX1132_ITERATIVE-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1132_ITERATIVE-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1132_ITERATIVE-NEXT: v_add_co_u32 v0, vcc_lo, s2, v0
; GFX1132_ITERATIVE-NEXT: v_add_co_ci_u32_e64 v1, null, s3, v1, vcc_lo
; GFX1132_ITERATIVE-NEXT: s_mov_b32 s3, 0x31016000
@@ -3141,6 +3156,7 @@ define amdgpu_kernel void @add_i64_varying(ptr addrspace(1) %out) {
; GFX1164_DPP-NEXT: v_writelane_b32 v6, s8, 48
; GFX1164_DPP-NEXT: v_writelane_b32 v7, s9, 48
; GFX1164_DPP-NEXT: s_mov_b64 exec, s[6:7]
+; GFX1164_DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(VALU_DEP_1)
; GFX1164_DPP-NEXT: v_cmp_eq_u32_e32 vcc, 0, v8
; GFX1164_DPP-NEXT: s_mov_b32 s2, -1
; GFX1164_DPP-NEXT: ; implicit-def: $vgpr8_vgpr9
@@ -3221,6 +3237,7 @@ define amdgpu_kernel void @add_i64_varying(ptr addrspace(1) %out) {
; GFX1132_DPP-NEXT: v_writelane_b32 v6, s3, 16
; GFX1132_DPP-NEXT: v_writelane_b32 v7, s6, 16
; GFX1132_DPP-NEXT: s_mov_b32 exec_lo, s2
+; GFX1132_DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1132_DPP-NEXT: v_cmp_eq_u32_e32 vcc_lo, 0, v8
; GFX1132_DPP-NEXT: s_mov_b32 s2, -1
; GFX1132_DPP-NEXT: ; implicit-def: $vgpr8_vgpr9
@@ -3237,7 +3254,7 @@ define amdgpu_kernel void @add_i64_varying(ptr addrspace(1) %out) {
; GFX1132_DPP-NEXT: v_readfirstlane_b32 s3, v8
; GFX1132_DPP-NEXT: v_dual_mov_b32 v10, v6 :: v_dual_mov_b32 v11, v7
; GFX1132_DPP-NEXT: v_readfirstlane_b32 s4, v9
-; GFX1132_DPP-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1132_DPP-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1132_DPP-NEXT: v_add_co_u32 v8, vcc_lo, s3, v10
; GFX1132_DPP-NEXT: v_add_co_ci_u32_e64 v9, null, s4, v11, vcc_lo
; GFX1132_DPP-NEXT: s_mov_b32 s3, 0x31016000
@@ -3321,6 +3338,7 @@ define amdgpu_kernel void @add_i64_varying(ptr addrspace(1) %out) {
; GFX1364-NEXT: v_writelane_b32 v5, s8, 48
; GFX1364-NEXT: v_writelane_b32 v6, s9, 48
; GFX1364-NEXT: s_mov_b64 exec, s[6:7]
+; GFX1364-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX1364-NEXT: v_cmp_eq_u32_e32 vcc, 0, v7
; GFX1364-NEXT: s_mov_b32 s2, -1
; GFX1364-NEXT: ; implicit-def: $vgpr7_vgpr8
@@ -3339,7 +3357,7 @@ define amdgpu_kernel void @add_i64_varying(ptr addrspace(1) %out) {
; GFX1364-NEXT: v_mov_b32_e32 v9, v5
; GFX1364-NEXT: v_mov_b32_e32 v10, v6
; GFX1364-NEXT: v_readfirstlane_b32 s4, v8
-; GFX1364-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1364-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1364-NEXT: v_add_co_u32 v7, vcc, s3, v9
; GFX1364-NEXT: v_add_co_ci_u32_e64 v8, null, s4, v10, vcc
; GFX1364-NEXT: s_mov_b32 s3, 0x31016000
@@ -3399,6 +3417,7 @@ define amdgpu_kernel void @add_i64_varying(ptr addrspace(1) %out) {
; GFX1332-NEXT: v_writelane_b32 v6, s3, 16
; GFX1332-NEXT: v_writelane_b32 v7, s6, 16
; GFX1332-NEXT: s_mov_b32 exec_lo, s2
+; GFX1332-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1332-NEXT: v_cmp_eq_u32_e32 vcc_lo, 0, v8
; GFX1332-NEXT: s_mov_b32 s2, -1
; GFX1332-NEXT: ; implicit-def: $vgpr8_vgpr9
@@ -3415,7 +3434,7 @@ define amdgpu_kernel void @add_i64_varying(ptr addrspace(1) %out) {
; GFX1332-NEXT: v_readfirstlane_b32 s3, v8
; GFX1332-NEXT: v_dual_mov_b32 v10, v6 :: v_dual_mov_b32 v11, v7
; GFX1332-NEXT: v_readfirstlane_b32 s4, v9
-; GFX1332-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1332-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1332-NEXT: v_add_co_u32 v8, vcc_lo, s3, v10
; GFX1332-NEXT: v_add_co_ci_u32_e64 v9, null, s4, v11, vcc_lo
; GFX1332-NEXT: s_mov_b32 s3, 0x31016000
@@ -3615,6 +3634,7 @@ define amdgpu_kernel void @add_i64_varying_nouse() {
; GFX1164_ITERATIVE-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164_ITERATIVE-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
; GFX1164_ITERATIVE-NEXT: v_cmpx_eq_u32_e32 0, v0
+; GFX1164_ITERATIVE-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1164_ITERATIVE-NEXT: s_xor_b64 s[2:3], exec, s[2:3]
; GFX1164_ITERATIVE-NEXT: s_cbranch_execz .LBB7_4
; GFX1164_ITERATIVE-NEXT: ; %bb.3:
@@ -3647,9 +3667,10 @@ define amdgpu_kernel void @add_i64_varying_nouse() {
; GFX1132_ITERATIVE-NEXT: ; %bb.2: ; %ComputeEnd
; GFX1132_ITERATIVE-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
; GFX1132_ITERATIVE-NEXT: s_mov_b32 s2, exec_lo
-; GFX1132_ITERATIVE-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1132_ITERATIVE-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132_ITERATIVE-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX1132_ITERATIVE-NEXT: s_xor_b32 s2, exec_lo, s2
+; GFX1132_ITERATIVE-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1132_ITERATIVE-NEXT: s_cbranch_execz .LBB7_4
; GFX1132_ITERATIVE-NEXT: ; %bb.3:
; GFX1132_ITERATIVE-NEXT: v_dual_mov_b32 v2, 0 :: v_dual_mov_b32 v1, s1
@@ -3948,6 +3969,7 @@ define amdgpu_kernel void @add_i64_varying_nouse() {
; GFX1164_DPP-NEXT: v_mov_b32_e32 v6, v3
; GFX1164_DPP-NEXT: s_mov_b64 s[0:1], exec
; GFX1164_DPP-NEXT: v_cmpx_eq_u32_e32 0, v7
+; GFX1164_DPP-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1164_DPP-NEXT: s_cbranch_execz .LBB7_2
; GFX1164_DPP-NEXT: ; %bb.1:
; GFX1164_DPP-NEXT: ds_add_u64 v0, v[5:6]
@@ -3998,6 +4020,7 @@ define amdgpu_kernel void @add_i64_varying_nouse() {
; GFX1132_DPP-NEXT: v_mov_b32_e32 v6, v3
; GFX1132_DPP-NEXT: s_mov_b32 s0, exec_lo
; GFX1132_DPP-NEXT: v_cmpx_eq_u32_e32 0, v7
+; GFX1132_DPP-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1132_DPP-NEXT: s_cbranch_execz .LBB7_2
; GFX1132_DPP-NEXT: ; %bb.1:
; GFX1132_DPP-NEXT: ds_add_u64 v0, v[5:6]
@@ -4045,19 +4068,19 @@ define amdgpu_kernel void @add_i64_varying_nouse() {
; GFX1364-NEXT: v_add_co_ci_u32_e64 v1, null, v1, v4, vcc
; GFX1364-NEXT: v_permlane64_b32 v4, v1
; GFX1364-NEXT: s_mov_b64 exec, s[0:1]
-; GFX1364-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(VALU_DEP_1)
+; GFX1364-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_4) | instid1(SALU_CYCLE_1)
; GFX1364-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
; GFX1364-NEXT: s_or_saveexec_b64 s[0:1], -1
; GFX1364-NEXT: v_add_co_u32 v2, vcc, v2, v3
; GFX1364-NEXT: v_add_co_ci_u32_e64 v3, null, v1, v4, vcc
; GFX1364-NEXT: s_mov_b64 exec, s[0:1]
-; GFX1364-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1364-NEXT: v_mbcnt_hi_u32_b32 v7, exec_hi, v0
; GFX1364-NEXT: v_mov_b32_e32 v0, 0
; GFX1364-NEXT: v_mov_b32_e32 v5, v2
; GFX1364-NEXT: v_mov_b32_e32 v6, v3
; GFX1364-NEXT: s_mov_b64 s[0:1], exec
; GFX1364-NEXT: v_cmpx_eq_u32_e32 0, v7
+; GFX1364-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1364-NEXT: s_cbranch_execz .LBB7_2
; GFX1364-NEXT: ; %bb.1:
; GFX1364-NEXT: ds_add_u64 v0, v[5:6]
@@ -4107,6 +4130,7 @@ define amdgpu_kernel void @add_i64_varying_nouse() {
; GFX1332-NEXT: v_mov_b32_e32 v6, v3
; GFX1332-NEXT: s_mov_b32 s0, exec_lo
; GFX1332-NEXT: v_cmpx_eq_u32_e32 0, v7
+; GFX1332-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1332-NEXT: s_cbranch_execz .LBB7_2
; GFX1332-NEXT: ; %bb.1:
; GFX1332-NEXT: ds_add_u64 v0, v[5:6]
@@ -4276,6 +4300,7 @@ define amdgpu_kernel void @sub_i32_constant(ptr addrspace(1) %out) {
; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164-NEXT: v_mbcnt_hi_u32_b32 v0, s3, v0
; GFX1164-NEXT: v_cmpx_eq_u32_e32 0, v0
+; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1164-NEXT: s_cbranch_execz .LBB8_2
; GFX1164-NEXT: ; %bb.1:
; GFX1164-NEXT: s_bcnt1_i32_b64 s2, s[2:3]
@@ -4305,7 +4330,7 @@ define amdgpu_kernel void @sub_i32_constant(ptr addrspace(1) %out) {
; GFX1132-NEXT: s_mov_b32 s0, exec_lo
; GFX1132-NEXT: v_mbcnt_lo_u32_b32 v0, s1, 0
; GFX1132-NEXT: ; implicit-def: $vgpr1
-; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX1132-NEXT: s_cbranch_execz .LBB8_2
; GFX1132-NEXT: ; %bb.1:
@@ -4338,6 +4363,7 @@ define amdgpu_kernel void @sub_i32_constant(ptr addrspace(1) %out) {
; GFX1364-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1364-NEXT: v_mbcnt_hi_u32_b32 v0, s3, v0
; GFX1364-NEXT: v_cmpx_eq_u32_e32 0, v0
+; GFX1364-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1364-NEXT: s_cbranch_execz .LBB8_2
; GFX1364-NEXT: ; %bb.1:
; GFX1364-NEXT: s_bcnt1_i32_b64 s2, s[2:3]
@@ -4367,7 +4393,7 @@ define amdgpu_kernel void @sub_i32_constant(ptr addrspace(1) %out) {
; GFX1332-NEXT: s_mov_b32 s0, exec_lo
; GFX1332-NEXT: v_mbcnt_lo_u32_b32 v0, s1, 0
; GFX1332-NEXT: ; implicit-def: $vgpr1
-; GFX1332-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1332-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1332-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX1332-NEXT: s_cbranch_execz .LBB8_2
; GFX1332-NEXT: ; %bb.1:
@@ -4562,6 +4588,7 @@ define amdgpu_kernel void @sub_i32_uniform(ptr addrspace(1) %out, i32 %subitive)
; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164-NEXT: v_mbcnt_hi_u32_b32 v0, s3, v0
; GFX1164-NEXT: v_cmpx_eq_u32_e32 0, v0
+; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1164-NEXT: s_cbranch_execz .LBB9_2
; GFX1164-NEXT: ; %bb.1:
; GFX1164-NEXT: s_bcnt1_i32_b64 s2, s[2:3]
@@ -4592,7 +4619,7 @@ define amdgpu_kernel void @sub_i32_uniform(ptr addrspace(1) %out, i32 %subitive)
; GFX1132-NEXT: s_mov_b32 s1, exec_lo
; GFX1132-NEXT: v_mbcnt_lo_u32_b32 v0, s2, 0
; GFX1132-NEXT: ; implicit-def: $vgpr1
-; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX1132-NEXT: s_cbranch_execz .LBB9_2
; GFX1132-NEXT: ; %bb.1:
@@ -4626,6 +4653,7 @@ define amdgpu_kernel void @sub_i32_uniform(ptr addrspace(1) %out, i32 %subitive)
; GFX1364-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1364-NEXT: v_mbcnt_hi_u32_b32 v0, s3, v0
; GFX1364-NEXT: v_cmpx_eq_u32_e32 0, v0
+; GFX1364-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1364-NEXT: s_cbranch_execz .LBB9_2
; GFX1364-NEXT: ; %bb.1:
; GFX1364-NEXT: s_bcnt1_i32_b64 s2, s[2:3]
@@ -4656,7 +4684,7 @@ define amdgpu_kernel void @sub_i32_uniform(ptr addrspace(1) %out, i32 %subitive)
; GFX1332-NEXT: s_mov_b32 s1, exec_lo
; GFX1332-NEXT: v_mbcnt_lo_u32_b32 v0, s2, 0
; GFX1332-NEXT: ; implicit-def: $vgpr1
-; GFX1332-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1332-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1332-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX1332-NEXT: s_cbranch_execz .LBB9_2
; GFX1332-NEXT: ; %bb.1:
@@ -4908,8 +4936,8 @@ define amdgpu_kernel void @sub_i32_varying(ptr addrspace(1) %out) {
; GFX1164_ITERATIVE-NEXT: v_mbcnt_hi_u32_b32 v1, exec_hi, v1
; GFX1164_ITERATIVE-NEXT: v_cmp_eq_u32_e32 vcc, 0, v1
; GFX1164_ITERATIVE-NEXT: ; implicit-def: $vgpr1
+; GFX1164_ITERATIVE-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1164_ITERATIVE-NEXT: s_and_saveexec_b64 s[0:1], vcc
-; GFX1164_ITERATIVE-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164_ITERATIVE-NEXT: s_xor_b64 s[0:1], exec, s[0:1]
; GFX1164_ITERATIVE-NEXT: s_cbranch_execz .LBB10_4
; GFX1164_ITERATIVE-NEXT: ; %bb.3:
@@ -5200,7 +5228,7 @@ define amdgpu_kernel void @sub_i32_varying(ptr addrspace(1) %out) {
; GFX1164_DPP-NEXT: v_writelane_b32 v3, s3, 32
; GFX1164_DPP-NEXT: s_waitcnt_depctr depctr_sa_sdst(0)
; GFX1164_DPP-NEXT: s_mov_b64 exec, s[0:1]
-; GFX1164_DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1164_DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_4) | instid1(SALU_CYCLE_1)
; GFX1164_DPP-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
; GFX1164_DPP-NEXT: v_mov_b32_e32 v4, 0
; GFX1164_DPP-NEXT: s_or_saveexec_b64 s[0:1], -1
@@ -5209,6 +5237,7 @@ define amdgpu_kernel void @sub_i32_varying(ptr addrspace(1) %out) {
; GFX1164_DPP-NEXT: v_cmp_eq_u32_e32 vcc, 0, v0
; GFX1164_DPP-NEXT: s_mov_b32 s2, -1
; GFX1164_DPP-NEXT: ; implicit-def: $vgpr0
+; GFX1164_DPP-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1164_DPP-NEXT: s_and_saveexec_b64 s[0:1], vcc
; GFX1164_DPP-NEXT: s_cbranch_execz .LBB10_2
; GFX1164_DPP-NEXT: ; %bb.1:
@@ -5254,6 +5283,7 @@ define amdgpu_kernel void @sub_i32_varying(ptr addrspace(1) %out) {
; GFX1132_DPP-NEXT: s_or_saveexec_b32 s0, -1
; GFX1132_DPP-NEXT: v_writelane_b32 v3, s1, 16
; GFX1132_DPP-NEXT: s_mov_b32 exec_lo, s0
+; GFX1132_DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1132_DPP-NEXT: v_cmp_eq_u32_e32 vcc_lo, 0, v0
; GFX1132_DPP-NEXT: s_mov_b32 s0, s2
; GFX1132_DPP-NEXT: s_mov_b32 s2, -1
@@ -5310,7 +5340,7 @@ define amdgpu_kernel void @sub_i32_varying(ptr addrspace(1) %out) {
; GFX1364-NEXT: v_readlane_b32 s6, v1, 63
; GFX1364-NEXT: v_writelane_b32 v3, s3, 32
; GFX1364-NEXT: s_mov_b64 exec, s[0:1]
-; GFX1364-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1364-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_4) | instid1(SALU_CYCLE_1)
; GFX1364-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
; GFX1364-NEXT: v_mov_b32_e32 v4, 0
; GFX1364-NEXT: s_or_saveexec_b64 s[0:1], -1
@@ -5319,6 +5349,7 @@ define amdgpu_kernel void @sub_i32_varying(ptr addrspace(1) %out) {
; GFX1364-NEXT: v_cmp_eq_u32_e32 vcc, 0, v0
; GFX1364-NEXT: s_mov_b32 s2, -1
; GFX1364-NEXT: ; implicit-def: $vgpr0
+; GFX1364-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1364-NEXT: s_and_saveexec_b64 s[0:1], vcc
; GFX1364-NEXT: s_cbranch_execz .LBB10_2
; GFX1364-NEXT: ; %bb.1:
@@ -5363,6 +5394,7 @@ define amdgpu_kernel void @sub_i32_varying(ptr addrspace(1) %out) {
; GFX1332-NEXT: s_or_saveexec_b32 s0, -1
; GFX1332-NEXT: v_writelane_b32 v3, s1, 16
; GFX1332-NEXT: s_mov_b32 exec_lo, s0
+; GFX1332-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1332-NEXT: v_cmp_eq_u32_e32 vcc_lo, 0, v0
; GFX1332-NEXT: s_mov_b32 s0, s2
; GFX1332-NEXT: s_mov_b32 s2, -1
@@ -5553,6 +5585,7 @@ define amdgpu_kernel void @sub_i32_varying_nouse() {
; GFX1164_ITERATIVE-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164_ITERATIVE-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
; GFX1164_ITERATIVE-NEXT: v_cmpx_eq_u32_e32 0, v0
+; GFX1164_ITERATIVE-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1164_ITERATIVE-NEXT: s_xor_b64 s[0:1], exec, s[0:1]
; GFX1164_ITERATIVE-NEXT: s_cbranch_execz .LBB11_4
; GFX1164_ITERATIVE-NEXT: ; %bb.3:
@@ -5581,9 +5614,10 @@ define amdgpu_kernel void @sub_i32_varying_nouse() {
; GFX1132_ITERATIVE-NEXT: ; %bb.2: ; %ComputeEnd
; GFX1132_ITERATIVE-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
; GFX1132_ITERATIVE-NEXT: s_mov_b32 s1, exec_lo
-; GFX1132_ITERATIVE-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1132_ITERATIVE-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132_ITERATIVE-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX1132_ITERATIVE-NEXT: s_xor_b32 s1, exec_lo, s1
+; GFX1132_ITERATIVE-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1132_ITERATIVE-NEXT: s_cbranch_execz .LBB11_4
; GFX1132_ITERATIVE-NEXT: ; %bb.3:
; GFX1132_ITERATIVE-NEXT: v_dual_mov_b32 v0, 0 :: v_dual_mov_b32 v1, s0
@@ -5748,7 +5782,7 @@ define amdgpu_kernel void @sub_i32_varying_nouse() {
; GFX1164_DPP-NEXT: v_mov_b32_e32 v0, 0
; GFX1164_DPP-NEXT: v_mov_b32_e32 v3, v1
; GFX1164_DPP-NEXT: s_mov_b64 s[0:1], exec
-; GFX1164_DPP-NEXT: s_delay_alu instid0(VALU_DEP_3)
+; GFX1164_DPP-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164_DPP-NEXT: v_cmpx_eq_u32_e32 0, v4
; GFX1164_DPP-NEXT: s_cbranch_execz .LBB11_2
; GFX1164_DPP-NEXT: ; %bb.1:
@@ -5778,7 +5812,7 @@ define amdgpu_kernel void @sub_i32_varying_nouse() {
; GFX1132_DPP-NEXT: v_mbcnt_lo_u32_b32 v4, exec_lo, 0
; GFX1132_DPP-NEXT: v_dual_mov_b32 v0, 0 :: v_dual_mov_b32 v3, v1
; GFX1132_DPP-NEXT: s_mov_b32 s0, exec_lo
-; GFX1132_DPP-NEXT: s_delay_alu instid0(VALU_DEP_2)
+; GFX1132_DPP-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132_DPP-NEXT: v_cmpx_eq_u32_e32 0, v4
; GFX1132_DPP-NEXT: s_cbranch_execz .LBB11_2
; GFX1132_DPP-NEXT: ; %bb.1:
@@ -5818,6 +5852,7 @@ define amdgpu_kernel void @sub_i32_varying_nouse() {
; GFX1364-NEXT: v_mov_b32_e32 v3, v1
; GFX1364-NEXT: s_mov_b64 s[0:1], exec
; GFX1364-NEXT: v_cmpx_eq_u32_e32 0, v4
+; GFX1364-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1364-NEXT: s_cbranch_execz .LBB11_2
; GFX1364-NEXT: ; %bb.1:
; GFX1364-NEXT: ds_sub_u32 v0, v3
@@ -5846,7 +5881,7 @@ define amdgpu_kernel void @sub_i32_varying_nouse() {
; GFX1332-NEXT: v_mbcnt_lo_u32_b32 v4, exec_lo, 0
; GFX1332-NEXT: v_dual_mov_b32 v0, 0 :: v_dual_mov_b32 v3, v1
; GFX1332-NEXT: s_mov_b32 s0, exec_lo
-; GFX1332-NEXT: s_delay_alu instid0(VALU_DEP_2)
+; GFX1332-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1332-NEXT: v_cmpx_eq_u32_e32 0, v4
; GFX1332-NEXT: s_cbranch_execz .LBB11_2
; GFX1332-NEXT: ; %bb.1:
@@ -6034,6 +6069,7 @@ define amdgpu_kernel void @sub_i64_constant(ptr addrspace(1) %out) {
; GFX1164-NEXT: v_mbcnt_hi_u32_b32 v2, s3, v0
; GFX1164-NEXT: ; implicit-def: $vgpr0_vgpr1
; GFX1164-NEXT: v_cmpx_eq_u32_e32 0, v2
+; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1164-NEXT: s_cbranch_execz .LBB12_2
; GFX1164-NEXT: ; %bb.1:
; GFX1164-NEXT: s_bcnt1_i32_b64 s2, s[2:3]
@@ -6066,7 +6102,7 @@ define amdgpu_kernel void @sub_i64_constant(ptr addrspace(1) %out) {
; GFX1132-NEXT: s_mov_b32 s0, exec_lo
; GFX1132-NEXT: v_mbcnt_lo_u32_b32 v2, s1, 0
; GFX1132-NEXT: ; implicit-def: $vgpr0_vgpr1
-; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132-NEXT: v_cmpx_eq_u32_e32 0, v2
; GFX1132-NEXT: s_cbranch_execz .LBB12_2
; GFX1132-NEXT: ; %bb.1:
@@ -6085,7 +6121,7 @@ define amdgpu_kernel void @sub_i64_constant(ptr addrspace(1) %out) {
; GFX1132-NEXT: v_mul_u32_u24_e32 v0, 5, v2
; GFX1132-NEXT: v_readfirstlane_b32 s3, v1
; GFX1132-NEXT: v_mul_hi_u32_u24_e32 v1, 5, v2
-; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1132-NEXT: v_sub_co_u32 v0, vcc_lo, s2, v0
; GFX1132-NEXT: v_sub_co_ci_u32_e64 v1, null, s3, v1, vcc_lo
; GFX1132-NEXT: s_mov_b32 s3, 0x31016000
@@ -6103,6 +6139,7 @@ define amdgpu_kernel void @sub_i64_constant(ptr addrspace(1) %out) {
; GFX1364-NEXT: v_mbcnt_hi_u32_b32 v2, s3, v0
; GFX1364-NEXT: ; implicit-def: $vgpr0_vgpr1
; GFX1364-NEXT: v_cmpx_eq_u32_e32 0, v2
+; GFX1364-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1364-NEXT: s_cbranch_execz .LBB12_2
; GFX1364-NEXT: ; %bb.1:
; GFX1364-NEXT: s_bcnt1_i32_b64 s2, s[2:3]
@@ -6120,7 +6157,7 @@ define amdgpu_kernel void @sub_i64_constant(ptr addrspace(1) %out) {
; GFX1364-NEXT: v_mul_u32_u24_e32 v0, 5, v2
; GFX1364-NEXT: v_readfirstlane_b32 s3, v1
; GFX1364-NEXT: v_mul_hi_u32_u24_e32 v1, 5, v2
-; GFX1364-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1364-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1364-NEXT: v_sub_co_u32 v0, vcc, s2, v0
; GFX1364-NEXT: v_sub_co_ci_u32_e64 v1, null, s3, v1, vcc
; GFX1364-NEXT: s_mov_b32 s3, 0x31016000
@@ -6135,7 +6172,7 @@ define amdgpu_kernel void @sub_i64_constant(ptr addrspace(1) %out) {
; GFX1332-NEXT: s_mov_b32 s0, exec_lo
; GFX1332-NEXT: v_mbcnt_lo_u32_b32 v2, s1, 0
; GFX1332-NEXT: ; implicit-def: $vgpr0_vgpr1
-; GFX1332-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1332-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1332-NEXT: v_cmpx_eq_u32_e32 0, v2
; GFX1332-NEXT: s_cbranch_execz .LBB12_2
; GFX1332-NEXT: ; %bb.1:
@@ -6154,7 +6191,7 @@ define amdgpu_kernel void @sub_i64_constant(ptr addrspace(1) %out) {
; GFX1332-NEXT: v_mul_u32_u24_e32 v0, 5, v2
; GFX1332-NEXT: v_readfirstlane_b32 s3, v1
; GFX1332-NEXT: v_mul_hi_u32_u24_e32 v1, 5, v2
-; GFX1332-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1332-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1332-NEXT: v_sub_co_u32 v0, vcc_lo, s2, v0
; GFX1332-NEXT: v_sub_co_ci_u32_e64 v1, null, s3, v1, vcc_lo
; GFX1332-NEXT: s_mov_b32 s3, 0x31016000
@@ -6374,6 +6411,7 @@ define amdgpu_kernel void @sub_i64_uniform(ptr addrspace(1) %out, i64 %subitive)
; GFX1164-NEXT: v_mbcnt_hi_u32_b32 v2, s7, v0
; GFX1164-NEXT: ; implicit-def: $vgpr0_vgpr1
; GFX1164-NEXT: v_cmpx_eq_u32_e32 0, v2
+; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1164-NEXT: s_cbranch_execz .LBB13_2
; GFX1164-NEXT: ; %bb.1:
; GFX1164-NEXT: s_bcnt1_i32_b64 s6, s[6:7]
@@ -6410,7 +6448,7 @@ define amdgpu_kernel void @sub_i64_uniform(ptr addrspace(1) %out, i64 %subitive)
; GFX1132-NEXT: s_mov_b32 s4, exec_lo
; GFX1132-NEXT: v_mbcnt_lo_u32_b32 v2, s6, 0
; GFX1132-NEXT: ; implicit-def: $vgpr0_vgpr1
-; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132-NEXT: v_cmpx_eq_u32_e32 0, v2
; GFX1132-NEXT: s_cbranch_execz .LBB13_2
; GFX1132-NEXT: ; %bb.1:
@@ -6452,6 +6490,7 @@ define amdgpu_kernel void @sub_i64_uniform(ptr addrspace(1) %out, i64 %subitive)
; GFX1364-NEXT: v_mbcnt_hi_u32_b32 v2, s7, v0
; GFX1364-NEXT: ; implicit-def: $vgpr0_vgpr1
; GFX1364-NEXT: v_cmpx_eq_u32_e32 0, v2
+; GFX1364-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1364-NEXT: s_cbranch_execz .LBB13_2
; GFX1364-NEXT: ; %bb.1:
; GFX1364-NEXT: s_bcnt1_i32_b64 s8, s[6:7]
@@ -6487,7 +6526,7 @@ define amdgpu_kernel void @sub_i64_uniform(ptr addrspace(1) %out, i64 %subitive)
; GFX1332-NEXT: v_mbcnt_lo_u32_b32 v2, s6, 0
; GFX1332-NEXT: s_mov_b32 s7, exec_lo
; GFX1332-NEXT: ; implicit-def: $vgpr0_vgpr1
-; GFX1332-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1332-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1332-NEXT: v_cmpx_eq_u32_e32 0, v2
; GFX1332-NEXT: s_cbranch_execz .LBB13_2
; GFX1332-NEXT: ; %bb.1:
@@ -6786,8 +6825,8 @@ define amdgpu_kernel void @sub_i64_varying(ptr addrspace(1) %out) {
; GFX1164_ITERATIVE-NEXT: v_mbcnt_hi_u32_b32 v2, exec_hi, v2
; GFX1164_ITERATIVE-NEXT: v_cmp_eq_u32_e32 vcc, 0, v2
; GFX1164_ITERATIVE-NEXT: ; implicit-def: $vgpr2_vgpr3
+; GFX1164_ITERATIVE-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1164_ITERATIVE-NEXT: s_and_saveexec_b64 s[2:3], vcc
-; GFX1164_ITERATIVE-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164_ITERATIVE-NEXT: s_xor_b64 s[2:3], exec, s[2:3]
; GFX1164_ITERATIVE-NEXT: s_cbranch_execz .LBB14_4
; GFX1164_ITERATIVE-NEXT: ; %bb.3:
@@ -6850,7 +6889,7 @@ define amdgpu_kernel void @sub_i64_varying(ptr addrspace(1) %out) {
; GFX1132_ITERATIVE-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
; GFX1132_ITERATIVE-NEXT: v_readfirstlane_b32 s2, v2
; GFX1132_ITERATIVE-NEXT: v_readfirstlane_b32 s3, v3
-; GFX1132_ITERATIVE-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1132_ITERATIVE-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1132_ITERATIVE-NEXT: v_sub_co_u32 v0, vcc_lo, s2, v0
; GFX1132_ITERATIVE-NEXT: v_sub_co_ci_u32_e64 v1, null, s3, v1, vcc_lo
; GFX1132_ITERATIVE-NEXT: s_mov_b32 s3, 0x31016000
@@ -7287,6 +7326,7 @@ define amdgpu_kernel void @sub_i64_varying(ptr addrspace(1) %out) {
; GFX1164_DPP-NEXT: v_writelane_b32 v6, s8, 48
; GFX1164_DPP-NEXT: v_writelane_b32 v7, s9, 48
; GFX1164_DPP-NEXT: s_mov_b64 exec, s[6:7]
+; GFX1164_DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(VALU_DEP_1)
; GFX1164_DPP-NEXT: v_cmp_eq_u32_e32 vcc, 0, v8
; GFX1164_DPP-NEXT: s_mov_b32 s2, -1
; GFX1164_DPP-NEXT: ; implicit-def: $vgpr8_vgpr9
@@ -7367,6 +7407,7 @@ define amdgpu_kernel void @sub_i64_varying(ptr addrspace(1) %out) {
; GFX1132_DPP-NEXT: v_writelane_b32 v6, s3, 16
; GFX1132_DPP-NEXT: v_writelane_b32 v7, s6, 16
; GFX1132_DPP-NEXT: s_mov_b32 exec_lo, s2
+; GFX1132_DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1132_DPP-NEXT: v_cmp_eq_u32_e32 vcc_lo, 0, v8
; GFX1132_DPP-NEXT: s_mov_b32 s2, -1
; GFX1132_DPP-NEXT: ; implicit-def: $vgpr8_vgpr9
@@ -7383,7 +7424,7 @@ define amdgpu_kernel void @sub_i64_varying(ptr addrspace(1) %out) {
; GFX1132_DPP-NEXT: v_readfirstlane_b32 s3, v8
; GFX1132_DPP-NEXT: v_dual_mov_b32 v10, v6 :: v_dual_mov_b32 v11, v7
; GFX1132_DPP-NEXT: v_readfirstlane_b32 s4, v9
-; GFX1132_DPP-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1132_DPP-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1132_DPP-NEXT: v_sub_co_u32 v8, vcc_lo, s3, v10
; GFX1132_DPP-NEXT: v_sub_co_ci_u32_e64 v9, null, s4, v11, vcc_lo
; GFX1132_DPP-NEXT: s_mov_b32 s3, 0x31016000
@@ -7467,6 +7508,7 @@ define amdgpu_kernel void @sub_i64_varying(ptr addrspace(1) %out) {
; GFX1364-NEXT: v_writelane_b32 v5, s8, 48
; GFX1364-NEXT: v_writelane_b32 v6, s9, 48
; GFX1364-NEXT: s_mov_b64 exec, s[6:7]
+; GFX1364-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX1364-NEXT: v_cmp_eq_u32_e32 vcc, 0, v7
; GFX1364-NEXT: s_mov_b32 s2, -1
; GFX1364-NEXT: ; implicit-def: $vgpr7_vgpr8
@@ -7485,7 +7527,7 @@ define amdgpu_kernel void @sub_i64_varying(ptr addrspace(1) %out) {
; GFX1364-NEXT: v_mov_b32_e32 v9, v5
; GFX1364-NEXT: v_mov_b32_e32 v10, v6
; GFX1364-NEXT: v_readfirstlane_b32 s4, v8
-; GFX1364-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1364-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1364-NEXT: v_sub_co_u32 v7, vcc, s3, v9
; GFX1364-NEXT: v_sub_co_ci_u32_e64 v8, null, s4, v10, vcc
; GFX1364-NEXT: s_mov_b32 s3, 0x31016000
@@ -7545,6 +7587,7 @@ define amdgpu_kernel void @sub_i64_varying(ptr addrspace(1) %out) {
; GFX1332-NEXT: v_writelane_b32 v6, s3, 16
; GFX1332-NEXT: v_writelane_b32 v7, s6, 16
; GFX1332-NEXT: s_mov_b32 exec_lo, s2
+; GFX1332-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1332-NEXT: v_cmp_eq_u32_e32 vcc_lo, 0, v8
; GFX1332-NEXT: s_mov_b32 s2, -1
; GFX1332-NEXT: ; implicit-def: $vgpr8_vgpr9
@@ -7561,7 +7604,7 @@ define amdgpu_kernel void @sub_i64_varying(ptr addrspace(1) %out) {
; GFX1332-NEXT: v_readfirstlane_b32 s3, v8
; GFX1332-NEXT: v_dual_mov_b32 v10, v6 :: v_dual_mov_b32 v11, v7
; GFX1332-NEXT: v_readfirstlane_b32 s4, v9
-; GFX1332-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1332-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1332-NEXT: v_sub_co_u32 v8, vcc_lo, s3, v10
; GFX1332-NEXT: v_sub_co_ci_u32_e64 v9, null, s4, v11, vcc_lo
; GFX1332-NEXT: s_mov_b32 s3, 0x31016000
@@ -7799,8 +7842,8 @@ define amdgpu_kernel void @and_i32_varying(ptr addrspace(1) %out) {
; GFX1164_ITERATIVE-NEXT: v_mbcnt_hi_u32_b32 v1, exec_hi, v1
; GFX1164_ITERATIVE-NEXT: v_cmp_eq_u32_e32 vcc, 0, v1
; GFX1164_ITERATIVE-NEXT: ; implicit-def: $vgpr1
+; GFX1164_ITERATIVE-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1164_ITERATIVE-NEXT: s_and_saveexec_b64 s[0:1], vcc
-; GFX1164_ITERATIVE-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164_ITERATIVE-NEXT: s_xor_b64 s[0:1], exec, s[0:1]
; GFX1164_ITERATIVE-NEXT: s_cbranch_execz .LBB15_4
; GFX1164_ITERATIVE-NEXT: ; %bb.3:
@@ -8091,14 +8134,16 @@ define amdgpu_kernel void @and_i32_varying(ptr addrspace(1) %out) {
; GFX1164_DPP-NEXT: v_writelane_b32 v3, s3, 32
; GFX1164_DPP-NEXT: s_waitcnt_depctr depctr_sa_sdst(0)
; GFX1164_DPP-NEXT: s_mov_b64 exec, s[0:1]
-; GFX1164_DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_3) | instid1(VALU_DEP_2)
+; GFX1164_DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164_DPP-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
; GFX1164_DPP-NEXT: s_or_saveexec_b64 s[0:1], -1
; GFX1164_DPP-NEXT: v_writelane_b32 v3, s2, 48
; GFX1164_DPP-NEXT: s_mov_b64 exec, s[0:1]
+; GFX1164_DPP-NEXT: s_delay_alu instid0(VALU_DEP_2) | instid1(SALU_CYCLE_1)
; GFX1164_DPP-NEXT: v_cmp_eq_u32_e32 vcc, 0, v0
; GFX1164_DPP-NEXT: s_mov_b32 s2, -1
; GFX1164_DPP-NEXT: ; implicit-def: $vgpr0
+; GFX1164_DPP-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1164_DPP-NEXT: s_and_saveexec_b64 s[0:1], vcc
; GFX1164_DPP-NEXT: s_cbranch_execz .LBB15_2
; GFX1164_DPP-NEXT: ; %bb.1:
@@ -8144,7 +8189,7 @@ define amdgpu_kernel void @and_i32_varying(ptr addrspace(1) %out) {
; GFX1132_DPP-NEXT: s_or_saveexec_b32 s0, -1
; GFX1132_DPP-NEXT: v_writelane_b32 v3, s1, 16
; GFX1132_DPP-NEXT: s_mov_b32 exec_lo, s0
-; GFX1132_DPP-NEXT: s_delay_alu instid0(VALU_DEP_2)
+; GFX1132_DPP-NEXT: s_delay_alu instid0(VALU_DEP_2) | instid1(SALU_CYCLE_1)
; GFX1132_DPP-NEXT: v_cmp_eq_u32_e32 vcc_lo, 0, v0
; GFX1132_DPP-NEXT: s_mov_b32 s0, s2
; GFX1132_DPP-NEXT: s_mov_b32 s2, -1
@@ -8217,14 +8262,16 @@ define amdgpu_kernel void @and_i32_varying(ptr addrspace(1) %out) {
; GFX1364-NEXT: v_readlane_b32 s6, v1, 63
; GFX1364-NEXT: v_writelane_b32 v3, s3, 32
; GFX1364-NEXT: s_mov_b64 exec, s[0:1]
-; GFX1364-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_3) | instid1(VALU_DEP_2)
+; GFX1364-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1364-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
; GFX1364-NEXT: s_or_saveexec_b64 s[0:1], -1
; GFX1364-NEXT: v_writelane_b32 v3, s2, 48
; GFX1364-NEXT: s_mov_b64 exec, s[0:1]
+; GFX1364-NEXT: s_delay_alu instid0(VALU_DEP_2) | instid1(SALU_CYCLE_1)
; GFX1364-NEXT: v_cmp_eq_u32_e32 vcc, 0, v0
; GFX1364-NEXT: s_mov_b32 s2, -1
; GFX1364-NEXT: ; implicit-def: $vgpr0
+; GFX1364-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1364-NEXT: s_and_saveexec_b64 s[0:1], vcc
; GFX1364-NEXT: s_cbranch_execz .LBB15_2
; GFX1364-NEXT: ; %bb.1:
@@ -8281,7 +8328,7 @@ define amdgpu_kernel void @and_i32_varying(ptr addrspace(1) %out) {
; GFX1332-NEXT: s_or_saveexec_b32 s0, -1
; GFX1332-NEXT: v_writelane_b32 v1, s1, 16
; GFX1332-NEXT: s_mov_b32 exec_lo, s0
-; GFX1332-NEXT: s_delay_alu instid0(VALU_DEP_2)
+; GFX1332-NEXT: s_delay_alu instid0(VALU_DEP_2) | instid1(SALU_CYCLE_1)
; GFX1332-NEXT: v_cmp_eq_u32_e32 vcc_lo, 0, v0
; GFX1332-NEXT: s_mov_b32 s0, s2
; GFX1332-NEXT: s_mov_b32 s2, -1
@@ -8567,8 +8614,8 @@ define amdgpu_kernel void @and_i64_varying(ptr addrspace(1) %out) {
; GFX1164_ITERATIVE-NEXT: v_mbcnt_hi_u32_b32 v2, exec_hi, v2
; GFX1164_ITERATIVE-NEXT: v_cmp_eq_u32_e32 vcc, 0, v2
; GFX1164_ITERATIVE-NEXT: ; implicit-def: $vgpr2_vgpr3
+; GFX1164_ITERATIVE-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1164_ITERATIVE-NEXT: s_and_saveexec_b64 s[2:3], vcc
-; GFX1164_ITERATIVE-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164_ITERATIVE-NEXT: s_xor_b64 s[2:3], exec, s[2:3]
; GFX1164_ITERATIVE-NEXT: s_cbranch_execz .LBB16_4
; GFX1164_ITERATIVE-NEXT: ; %bb.3:
@@ -8962,6 +9009,7 @@ define amdgpu_kernel void @and_i64_varying(ptr addrspace(1) %out) {
; GFX1164_DPP-NEXT: v_writelane_b32 v6, s8, 48
; GFX1164_DPP-NEXT: v_writelane_b32 v5, s9, 48
; GFX1164_DPP-NEXT: s_mov_b64 exec, s[6:7]
+; GFX1164_DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX1164_DPP-NEXT: v_cmp_eq_u32_e32 vcc, 0, v7
; GFX1164_DPP-NEXT: s_mov_b32 s2, -1
; GFX1164_DPP-NEXT: ; implicit-def: $vgpr7_vgpr8
@@ -9029,6 +9077,7 @@ define amdgpu_kernel void @and_i64_varying(ptr addrspace(1) %out) {
; GFX1132_DPP-NEXT: v_writelane_b32 v6, s3, 16
; GFX1132_DPP-NEXT: v_writelane_b32 v5, s6, 16
; GFX1132_DPP-NEXT: s_mov_b32 exec_lo, s2
+; GFX1132_DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1132_DPP-NEXT: v_cmp_eq_u32_e32 vcc_lo, 0, v7
; GFX1132_DPP-NEXT: s_mov_b32 s2, -1
; GFX1132_DPP-NEXT: ; implicit-def: $vgpr7_vgpr8
@@ -9116,6 +9165,7 @@ define amdgpu_kernel void @and_i64_varying(ptr addrspace(1) %out) {
; GFX1364-NEXT: v_writelane_b32 v6, s8, 48
; GFX1364-NEXT: v_writelane_b32 v5, s9, 48
; GFX1364-NEXT: s_mov_b64 exec, s[6:7]
+; GFX1364-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX1364-NEXT: v_cmp_eq_u32_e32 vcc, 0, v7
; GFX1364-NEXT: s_mov_b32 s2, -1
; GFX1364-NEXT: ; implicit-def: $vgpr7_vgpr8
@@ -9183,6 +9233,7 @@ define amdgpu_kernel void @and_i64_varying(ptr addrspace(1) %out) {
; GFX1332-NEXT: v_writelane_b32 v6, s3, 16
; GFX1332-NEXT: v_writelane_b32 v5, s6, 16
; GFX1332-NEXT: s_mov_b32 exec_lo, s2
+; GFX1332-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1332-NEXT: v_cmp_eq_u32_e32 vcc_lo, 0, v7
; GFX1332-NEXT: s_mov_b32 s2, -1
; GFX1332-NEXT: ; implicit-def: $vgpr7_vgpr8
@@ -9437,8 +9488,8 @@ define amdgpu_kernel void @or_i32_varying(ptr addrspace(1) %out) {
; GFX1164_ITERATIVE-NEXT: v_mbcnt_hi_u32_b32 v1, exec_hi, v1
; GFX1164_ITERATIVE-NEXT: v_cmp_eq_u32_e32 vcc, 0, v1
; GFX1164_ITERATIVE-NEXT: ; implicit-def: $vgpr1
+; GFX1164_ITERATIVE-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1164_ITERATIVE-NEXT: s_and_saveexec_b64 s[0:1], vcc
-; GFX1164_ITERATIVE-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164_ITERATIVE-NEXT: s_xor_b64 s[0:1], exec, s[0:1]
; GFX1164_ITERATIVE-NEXT: s_cbranch_execz .LBB17_4
; GFX1164_ITERATIVE-NEXT: ; %bb.3:
@@ -9729,7 +9780,7 @@ define amdgpu_kernel void @or_i32_varying(ptr addrspace(1) %out) {
; GFX1164_DPP-NEXT: v_writelane_b32 v3, s3, 32
; GFX1164_DPP-NEXT: s_waitcnt_depctr depctr_sa_sdst(0)
; GFX1164_DPP-NEXT: s_mov_b64 exec, s[0:1]
-; GFX1164_DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1164_DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_4) | instid1(SALU_CYCLE_1)
; GFX1164_DPP-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
; GFX1164_DPP-NEXT: v_mov_b32_e32 v4, 0
; GFX1164_DPP-NEXT: s_or_saveexec_b64 s[0:1], -1
@@ -9738,6 +9789,7 @@ define amdgpu_kernel void @or_i32_varying(ptr addrspace(1) %out) {
; GFX1164_DPP-NEXT: v_cmp_eq_u32_e32 vcc, 0, v0
; GFX1164_DPP-NEXT: s_mov_b32 s2, -1
; GFX1164_DPP-NEXT: ; implicit-def: $vgpr0
+; GFX1164_DPP-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1164_DPP-NEXT: s_and_saveexec_b64 s[0:1], vcc
; GFX1164_DPP-NEXT: s_cbranch_execz .LBB17_2
; GFX1164_DPP-NEXT: ; %bb.1:
@@ -9783,6 +9835,7 @@ define amdgpu_kernel void @or_i32_varying(ptr addrspace(1) %out) {
; GFX1132_DPP-NEXT: s_or_saveexec_b32 s0, -1
; GFX1132_DPP-NEXT: v_writelane_b32 v3, s1, 16
; GFX1132_DPP-NEXT: s_mov_b32 exec_lo, s0
+; GFX1132_DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1132_DPP-NEXT: v_cmp_eq_u32_e32 vcc_lo, 0, v0
; GFX1132_DPP-NEXT: s_mov_b32 s0, s2
; GFX1132_DPP-NEXT: s_mov_b32 s2, -1
@@ -9839,7 +9892,7 @@ define amdgpu_kernel void @or_i32_varying(ptr addrspace(1) %out) {
; GFX1364-NEXT: v_readlane_b32 s6, v1, 63
; GFX1364-NEXT: v_writelane_b32 v3, s3, 32
; GFX1364-NEXT: s_mov_b64 exec, s[0:1]
-; GFX1364-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1364-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_4) | instid1(SALU_CYCLE_1)
; GFX1364-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
; GFX1364-NEXT: v_mov_b32_e32 v4, 0
; GFX1364-NEXT: s_or_saveexec_b64 s[0:1], -1
@@ -9848,6 +9901,7 @@ define amdgpu_kernel void @or_i32_varying(ptr addrspace(1) %out) {
; GFX1364-NEXT: v_cmp_eq_u32_e32 vcc, 0, v0
; GFX1364-NEXT: s_mov_b32 s2, -1
; GFX1364-NEXT: ; implicit-def: $vgpr0
+; GFX1364-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1364-NEXT: s_and_saveexec_b64 s[0:1], vcc
; GFX1364-NEXT: s_cbranch_execz .LBB17_2
; GFX1364-NEXT: ; %bb.1:
@@ -9892,6 +9946,7 @@ define amdgpu_kernel void @or_i32_varying(ptr addrspace(1) %out) {
; GFX1332-NEXT: s_or_saveexec_b32 s0, -1
; GFX1332-NEXT: v_writelane_b32 v3, s1, 16
; GFX1332-NEXT: s_mov_b32 exec_lo, s0
+; GFX1332-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1332-NEXT: v_cmp_eq_u32_e32 vcc_lo, 0, v0
; GFX1332-NEXT: s_mov_b32 s0, s2
; GFX1332-NEXT: s_mov_b32 s2, -1
@@ -10177,8 +10232,8 @@ define amdgpu_kernel void @or_i64_varying(ptr addrspace(1) %out) {
; GFX1164_ITERATIVE-NEXT: v_mbcnt_hi_u32_b32 v2, exec_hi, v2
; GFX1164_ITERATIVE-NEXT: v_cmp_eq_u32_e32 vcc, 0, v2
; GFX1164_ITERATIVE-NEXT: ; implicit-def: $vgpr2_vgpr3
+; GFX1164_ITERATIVE-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1164_ITERATIVE-NEXT: s_and_saveexec_b64 s[2:3], vcc
-; GFX1164_ITERATIVE-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164_ITERATIVE-NEXT: s_xor_b64 s[2:3], exec, s[2:3]
; GFX1164_ITERATIVE-NEXT: s_cbranch_execz .LBB18_4
; GFX1164_ITERATIVE-NEXT: ; %bb.3:
@@ -10572,6 +10627,7 @@ define amdgpu_kernel void @or_i64_varying(ptr addrspace(1) %out) {
; GFX1164_DPP-NEXT: v_writelane_b32 v6, s8, 48
; GFX1164_DPP-NEXT: v_writelane_b32 v5, s9, 48
; GFX1164_DPP-NEXT: s_mov_b64 exec, s[6:7]
+; GFX1164_DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX1164_DPP-NEXT: v_cmp_eq_u32_e32 vcc, 0, v7
; GFX1164_DPP-NEXT: s_mov_b32 s2, -1
; GFX1164_DPP-NEXT: ; implicit-def: $vgpr7_vgpr8
@@ -10639,6 +10695,7 @@ define amdgpu_kernel void @or_i64_varying(ptr addrspace(1) %out) {
; GFX1132_DPP-NEXT: v_writelane_b32 v6, s3, 16
; GFX1132_DPP-NEXT: v_writelane_b32 v5, s6, 16
; GFX1132_DPP-NEXT: s_mov_b32 exec_lo, s2
+; GFX1132_DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1132_DPP-NEXT: v_cmp_eq_u32_e32 vcc_lo, 0, v7
; GFX1132_DPP-NEXT: s_mov_b32 s2, -1
; GFX1132_DPP-NEXT: ; implicit-def: $vgpr7_vgpr8
@@ -10726,6 +10783,7 @@ define amdgpu_kernel void @or_i64_varying(ptr addrspace(1) %out) {
; GFX1364-NEXT: v_writelane_b32 v6, s8, 48
; GFX1364-NEXT: v_writelane_b32 v5, s9, 48
; GFX1364-NEXT: s_mov_b64 exec, s[6:7]
+; GFX1364-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX1364-NEXT: v_cmp_eq_u32_e32 vcc, 0, v7
; GFX1364-NEXT: s_mov_b32 s2, -1
; GFX1364-NEXT: ; implicit-def: $vgpr7_vgpr8
@@ -10793,6 +10851,7 @@ define amdgpu_kernel void @or_i64_varying(ptr addrspace(1) %out) {
; GFX1332-NEXT: v_writelane_b32 v6, s3, 16
; GFX1332-NEXT: v_writelane_b32 v5, s6, 16
; GFX1332-NEXT: s_mov_b32 exec_lo, s2
+; GFX1332-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1332-NEXT: v_cmp_eq_u32_e32 vcc_lo, 0, v7
; GFX1332-NEXT: s_mov_b32 s2, -1
; GFX1332-NEXT: ; implicit-def: $vgpr7_vgpr8
@@ -11047,8 +11106,8 @@ define amdgpu_kernel void @xor_i32_varying(ptr addrspace(1) %out) {
; GFX1164_ITERATIVE-NEXT: v_mbcnt_hi_u32_b32 v1, exec_hi, v1
; GFX1164_ITERATIVE-NEXT: v_cmp_eq_u32_e32 vcc, 0, v1
; GFX1164_ITERATIVE-NEXT: ; implicit-def: $vgpr1
+; GFX1164_ITERATIVE-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1164_ITERATIVE-NEXT: s_and_saveexec_b64 s[0:1], vcc
-; GFX1164_ITERATIVE-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164_ITERATIVE-NEXT: s_xor_b64 s[0:1], exec, s[0:1]
; GFX1164_ITERATIVE-NEXT: s_cbranch_execz .LBB19_4
; GFX1164_ITERATIVE-NEXT: ; %bb.3:
@@ -11339,7 +11398,7 @@ define amdgpu_kernel void @xor_i32_varying(ptr addrspace(1) %out) {
; GFX1164_DPP-NEXT: v_writelane_b32 v3, s3, 32
; GFX1164_DPP-NEXT: s_waitcnt_depctr depctr_sa_sdst(0)
; GFX1164_DPP-NEXT: s_mov_b64 exec, s[0:1]
-; GFX1164_DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1164_DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_4) | instid1(SALU_CYCLE_1)
; GFX1164_DPP-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
; GFX1164_DPP-NEXT: v_mov_b32_e32 v4, 0
; GFX1164_DPP-NEXT: s_or_saveexec_b64 s[0:1], -1
@@ -11348,6 +11407,7 @@ define amdgpu_kernel void @xor_i32_varying(ptr addrspace(1) %out) {
; GFX1164_DPP-NEXT: v_cmp_eq_u32_e32 vcc, 0, v0
; GFX1164_DPP-NEXT: s_mov_b32 s2, -1
; GFX1164_DPP-NEXT: ; implicit-def: $vgpr0
+; GFX1164_DPP-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1164_DPP-NEXT: s_and_saveexec_b64 s[0:1], vcc
; GFX1164_DPP-NEXT: s_cbranch_execz .LBB19_2
; GFX1164_DPP-NEXT: ; %bb.1:
@@ -11393,6 +11453,7 @@ define amdgpu_kernel void @xor_i32_varying(ptr addrspace(1) %out) {
; GFX1132_DPP-NEXT: s_or_saveexec_b32 s0, -1
; GFX1132_DPP-NEXT: v_writelane_b32 v3, s1, 16
; GFX1132_DPP-NEXT: s_mov_b32 exec_lo, s0
+; GFX1132_DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1132_DPP-NEXT: v_cmp_eq_u32_e32 vcc_lo, 0, v0
; GFX1132_DPP-NEXT: s_mov_b32 s0, s2
; GFX1132_DPP-NEXT: s_mov_b32 s2, -1
@@ -11449,7 +11510,7 @@ define amdgpu_kernel void @xor_i32_varying(ptr addrspace(1) %out) {
; GFX1364-NEXT: v_readlane_b32 s6, v1, 63
; GFX1364-NEXT: v_writelane_b32 v3, s3, 32
; GFX1364-NEXT: s_mov_b64 exec, s[0:1]
-; GFX1364-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1364-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_4) | instid1(SALU_CYCLE_1)
; GFX1364-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
; GFX1364-NEXT: v_mov_b32_e32 v4, 0
; GFX1364-NEXT: s_or_saveexec_b64 s[0:1], -1
@@ -11458,6 +11519,7 @@ define amdgpu_kernel void @xor_i32_varying(ptr addrspace(1) %out) {
; GFX1364-NEXT: v_cmp_eq_u32_e32 vcc, 0, v0
; GFX1364-NEXT: s_mov_b32 s2, -1
; GFX1364-NEXT: ; implicit-def: $vgpr0
+; GFX1364-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1364-NEXT: s_and_saveexec_b64 s[0:1], vcc
; GFX1364-NEXT: s_cbranch_execz .LBB19_2
; GFX1364-NEXT: ; %bb.1:
@@ -11502,6 +11564,7 @@ define amdgpu_kernel void @xor_i32_varying(ptr addrspace(1) %out) {
; GFX1332-NEXT: s_or_saveexec_b32 s0, -1
; GFX1332-NEXT: v_writelane_b32 v3, s1, 16
; GFX1332-NEXT: s_mov_b32 exec_lo, s0
+; GFX1332-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1332-NEXT: v_cmp_eq_u32_e32 vcc_lo, 0, v0
; GFX1332-NEXT: s_mov_b32 s0, s2
; GFX1332-NEXT: s_mov_b32 s2, -1
@@ -11787,8 +11850,8 @@ define amdgpu_kernel void @xor_i64_varying(ptr addrspace(1) %out) {
; GFX1164_ITERATIVE-NEXT: v_mbcnt_hi_u32_b32 v2, exec_hi, v2
; GFX1164_ITERATIVE-NEXT: v_cmp_eq_u32_e32 vcc, 0, v2
; GFX1164_ITERATIVE-NEXT: ; implicit-def: $vgpr2_vgpr3
+; GFX1164_ITERATIVE-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1164_ITERATIVE-NEXT: s_and_saveexec_b64 s[2:3], vcc
-; GFX1164_ITERATIVE-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164_ITERATIVE-NEXT: s_xor_b64 s[2:3], exec, s[2:3]
; GFX1164_ITERATIVE-NEXT: s_cbranch_execz .LBB20_4
; GFX1164_ITERATIVE-NEXT: ; %bb.3:
@@ -12182,6 +12245,7 @@ define amdgpu_kernel void @xor_i64_varying(ptr addrspace(1) %out) {
; GFX1164_DPP-NEXT: v_writelane_b32 v6, s8, 48
; GFX1164_DPP-NEXT: v_writelane_b32 v5, s9, 48
; GFX1164_DPP-NEXT: s_mov_b64 exec, s[6:7]
+; GFX1164_DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX1164_DPP-NEXT: v_cmp_eq_u32_e32 vcc, 0, v7
; GFX1164_DPP-NEXT: s_mov_b32 s2, -1
; GFX1164_DPP-NEXT: ; implicit-def: $vgpr7_vgpr8
@@ -12249,6 +12313,7 @@ define amdgpu_kernel void @xor_i64_varying(ptr addrspace(1) %out) {
; GFX1132_DPP-NEXT: v_writelane_b32 v6, s3, 16
; GFX1132_DPP-NEXT: v_writelane_b32 v5, s6, 16
; GFX1132_DPP-NEXT: s_mov_b32 exec_lo, s2
+; GFX1132_DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1132_DPP-NEXT: v_cmp_eq_u32_e32 vcc_lo, 0, v7
; GFX1132_DPP-NEXT: s_mov_b32 s2, -1
; GFX1132_DPP-NEXT: ; implicit-def: $vgpr7_vgpr8
@@ -12336,6 +12401,7 @@ define amdgpu_kernel void @xor_i64_varying(ptr addrspace(1) %out) {
; GFX1364-NEXT: v_writelane_b32 v6, s8, 48
; GFX1364-NEXT: v_writelane_b32 v5, s9, 48
; GFX1364-NEXT: s_mov_b64 exec, s[6:7]
+; GFX1364-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX1364-NEXT: v_cmp_eq_u32_e32 vcc, 0, v7
; GFX1364-NEXT: s_mov_b32 s2, -1
; GFX1364-NEXT: ; implicit-def: $vgpr7_vgpr8
@@ -12403,6 +12469,7 @@ define amdgpu_kernel void @xor_i64_varying(ptr addrspace(1) %out) {
; GFX1332-NEXT: v_writelane_b32 v6, s3, 16
; GFX1332-NEXT: v_writelane_b32 v5, s6, 16
; GFX1332-NEXT: s_mov_b32 exec_lo, s2
+; GFX1332-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1332-NEXT: v_cmp_eq_u32_e32 vcc_lo, 0, v7
; GFX1332-NEXT: s_mov_b32 s2, -1
; GFX1332-NEXT: ; implicit-def: $vgpr7_vgpr8
@@ -12657,8 +12724,8 @@ define amdgpu_kernel void @max_i32_varying(ptr addrspace(1) %out) {
; GFX1164_ITERATIVE-NEXT: v_mbcnt_hi_u32_b32 v1, exec_hi, v1
; GFX1164_ITERATIVE-NEXT: v_cmp_eq_u32_e32 vcc, 0, v1
; GFX1164_ITERATIVE-NEXT: ; implicit-def: $vgpr1
+; GFX1164_ITERATIVE-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1164_ITERATIVE-NEXT: s_and_saveexec_b64 s[0:1], vcc
-; GFX1164_ITERATIVE-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164_ITERATIVE-NEXT: s_xor_b64 s[0:1], exec, s[0:1]
; GFX1164_ITERATIVE-NEXT: s_cbranch_execz .LBB21_4
; GFX1164_ITERATIVE-NEXT: ; %bb.3:
@@ -12949,14 +13016,16 @@ define amdgpu_kernel void @max_i32_varying(ptr addrspace(1) %out) {
; GFX1164_DPP-NEXT: v_writelane_b32 v3, s3, 32
; GFX1164_DPP-NEXT: s_waitcnt_depctr depctr_sa_sdst(0)
; GFX1164_DPP-NEXT: s_mov_b64 exec, s[0:1]
-; GFX1164_DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_3) | instid1(VALU_DEP_2)
+; GFX1164_DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164_DPP-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
; GFX1164_DPP-NEXT: s_or_saveexec_b64 s[0:1], -1
; GFX1164_DPP-NEXT: v_writelane_b32 v3, s2, 48
; GFX1164_DPP-NEXT: s_mov_b64 exec, s[0:1]
+; GFX1164_DPP-NEXT: s_delay_alu instid0(VALU_DEP_2) | instid1(SALU_CYCLE_1)
; GFX1164_DPP-NEXT: v_cmp_eq_u32_e32 vcc, 0, v0
; GFX1164_DPP-NEXT: s_mov_b32 s2, -1
; GFX1164_DPP-NEXT: ; implicit-def: $vgpr0
+; GFX1164_DPP-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1164_DPP-NEXT: s_and_saveexec_b64 s[0:1], vcc
; GFX1164_DPP-NEXT: s_cbranch_execz .LBB21_2
; GFX1164_DPP-NEXT: ; %bb.1:
@@ -13002,7 +13071,7 @@ define amdgpu_kernel void @max_i32_varying(ptr addrspace(1) %out) {
; GFX1132_DPP-NEXT: s_or_saveexec_b32 s0, -1
; GFX1132_DPP-NEXT: v_writelane_b32 v3, s1, 16
; GFX1132_DPP-NEXT: s_mov_b32 exec_lo, s0
-; GFX1132_DPP-NEXT: s_delay_alu instid0(VALU_DEP_2)
+; GFX1132_DPP-NEXT: s_delay_alu instid0(VALU_DEP_2) | instid1(SALU_CYCLE_1)
; GFX1132_DPP-NEXT: v_cmp_eq_u32_e32 vcc_lo, 0, v0
; GFX1132_DPP-NEXT: s_mov_b32 s0, s2
; GFX1132_DPP-NEXT: s_mov_b32 s2, -1
@@ -13060,14 +13129,16 @@ define amdgpu_kernel void @max_i32_varying(ptr addrspace(1) %out) {
; GFX1364-NEXT: v_readlane_b32 s6, v1, 63
; GFX1364-NEXT: v_writelane_b32 v3, s3, 32
; GFX1364-NEXT: s_mov_b64 exec, s[0:1]
-; GFX1364-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_3) | instid1(VALU_DEP_2)
+; GFX1364-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1364-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
; GFX1364-NEXT: s_or_saveexec_b64 s[0:1], -1
; GFX1364-NEXT: v_writelane_b32 v3, s2, 48
; GFX1364-NEXT: s_mov_b64 exec, s[0:1]
+; GFX1364-NEXT: s_delay_alu instid0(VALU_DEP_2) | instid1(SALU_CYCLE_1)
; GFX1364-NEXT: v_cmp_eq_u32_e32 vcc, 0, v0
; GFX1364-NEXT: s_mov_b32 s2, -1
; GFX1364-NEXT: ; implicit-def: $vgpr0
+; GFX1364-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1364-NEXT: s_and_saveexec_b64 s[0:1], vcc
; GFX1364-NEXT: s_cbranch_execz .LBB21_2
; GFX1364-NEXT: ; %bb.1:
@@ -13113,7 +13184,7 @@ define amdgpu_kernel void @max_i32_varying(ptr addrspace(1) %out) {
; GFX1332-NEXT: s_or_saveexec_b32 s0, -1
; GFX1332-NEXT: v_writelane_b32 v3, s1, 16
; GFX1332-NEXT: s_mov_b32 exec_lo, s0
-; GFX1332-NEXT: s_delay_alu instid0(VALU_DEP_2)
+; GFX1332-NEXT: s_delay_alu instid0(VALU_DEP_2) | instid1(SALU_CYCLE_1)
; GFX1332-NEXT: v_cmp_eq_u32_e32 vcc_lo, 0, v0
; GFX1332-NEXT: s_mov_b32 s0, s2
; GFX1332-NEXT: s_mov_b32 s2, -1
@@ -13315,6 +13386,7 @@ define amdgpu_kernel void @max_i64_constant(ptr addrspace(1) %out) {
; GFX1164-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
; GFX1164-NEXT: v_cmp_eq_u32_e32 vcc, 0, v0
; GFX1164-NEXT: ; implicit-def: $vgpr0_vgpr1
+; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1164-NEXT: s_and_saveexec_b64 s[0:1], vcc
; GFX1164-NEXT: s_cbranch_execz .LBB22_2
; GFX1164-NEXT: ; %bb.1:
@@ -13326,12 +13398,12 @@ define amdgpu_kernel void @max_i64_constant(ptr addrspace(1) %out) {
; GFX1164-NEXT: buffer_gl0_inv
; GFX1164-NEXT: .LBB22_2:
; GFX1164-NEXT: s_or_b64 exec, exec, s[0:1]
+; GFX1164-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_4) | instid1(VALU_DEP_1)
; GFX1164-NEXT: v_readfirstlane_b32 s3, v1
; GFX1164-NEXT: v_readfirstlane_b32 s2, v0
; GFX1164-NEXT: v_cndmask_b32_e64 v1, 0, 0x80000000, vcc
; GFX1164-NEXT: v_cndmask_b32_e64 v0, 5, 0, vcc
; GFX1164-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
-; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1164-NEXT: v_cmp_gt_i64_e32 vcc, s[2:3], v[0:1]
; GFX1164-NEXT: s_waitcnt_depctr depctr_va_vcc(0)
; GFX1164-NEXT: v_cndmask_b32_e64 v1, v1, s3, vcc
@@ -13380,6 +13452,7 @@ define amdgpu_kernel void @max_i64_constant(ptr addrspace(1) %out) {
; GFX1364-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
; GFX1364-NEXT: v_cmp_eq_u32_e32 vcc, 0, v0
; GFX1364-NEXT: ; implicit-def: $vgpr0_vgpr1
+; GFX1364-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1364-NEXT: s_and_saveexec_b64 s[0:1], vcc
; GFX1364-NEXT: s_cbranch_execz .LBB22_2
; GFX1364-NEXT: ; %bb.1:
@@ -13741,8 +13814,8 @@ define amdgpu_kernel void @max_i64_varying(ptr addrspace(1) %out) {
; GFX1164_ITERATIVE-NEXT: v_mbcnt_hi_u32_b32 v2, exec_hi, v2
; GFX1164_ITERATIVE-NEXT: v_cmp_eq_u32_e32 vcc, 0, v2
; GFX1164_ITERATIVE-NEXT: ; implicit-def: $vgpr2_vgpr3
+; GFX1164_ITERATIVE-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1164_ITERATIVE-NEXT: s_and_saveexec_b64 s[2:3], vcc
-; GFX1164_ITERATIVE-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164_ITERATIVE-NEXT: s_xor_b64 s[2:3], exec, s[2:3]
; GFX1164_ITERATIVE-NEXT: s_cbranch_execz .LBB23_4
; GFX1164_ITERATIVE-NEXT: ; %bb.3:
@@ -13754,6 +13827,7 @@ define amdgpu_kernel void @max_i64_varying(ptr addrspace(1) %out) {
; GFX1164_ITERATIVE-NEXT: buffer_gl0_inv
; GFX1164_ITERATIVE-NEXT: .LBB23_4:
; GFX1164_ITERATIVE-NEXT: s_or_b64 exec, exec, s[2:3]
+; GFX1164_ITERATIVE-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164_ITERATIVE-NEXT: v_readfirstlane_b32 s3, v3
; GFX1164_ITERATIVE-NEXT: v_readfirstlane_b32 s2, v2
; GFX1164_ITERATIVE-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
@@ -14307,6 +14381,7 @@ define amdgpu_kernel void @max_i64_varying(ptr addrspace(1) %out) {
; GFX1164_DPP-NEXT: v_writelane_b32 v5, s8, 48
; GFX1164_DPP-NEXT: v_writelane_b32 v4, s9, 48
; GFX1164_DPP-NEXT: s_mov_b64 exec, s[6:7]
+; GFX1164_DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(VALU_DEP_1)
; GFX1164_DPP-NEXT: v_cmp_eq_u32_e32 vcc, 0, v7
; GFX1164_DPP-NEXT: s_mov_b32 s2, -1
; GFX1164_DPP-NEXT: ; implicit-def: $vgpr7_vgpr8
@@ -14399,6 +14474,7 @@ define amdgpu_kernel void @max_i64_varying(ptr addrspace(1) %out) {
; GFX1132_DPP-NEXT: v_writelane_b32 v5, s3, 16
; GFX1132_DPP-NEXT: v_writelane_b32 v4, s6, 16
; GFX1132_DPP-NEXT: s_mov_b32 exec_lo, s2
+; GFX1132_DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1132_DPP-NEXT: v_cmp_eq_u32_e32 vcc_lo, 0, v7
; GFX1132_DPP-NEXT: s_mov_b32 s2, -1
; GFX1132_DPP-NEXT: ; implicit-def: $vgpr7_vgpr8
@@ -14519,6 +14595,7 @@ define amdgpu_kernel void @max_i64_varying(ptr addrspace(1) %out) {
; GFX1364-NEXT: v_writelane_b32 v5, s8, 48
; GFX1364-NEXT: v_writelane_b32 v4, s9, 48
; GFX1364-NEXT: s_mov_b64 exec, s[6:7]
+; GFX1364-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX1364-NEXT: v_cmp_eq_u32_e32 vcc, 0, v7
; GFX1364-NEXT: s_mov_b32 s2, -1
; GFX1364-NEXT: ; implicit-def: $vgpr7_vgpr8
@@ -14607,6 +14684,7 @@ define amdgpu_kernel void @max_i64_varying(ptr addrspace(1) %out) {
; GFX1332-NEXT: v_writelane_b32 v5, s3, 16
; GFX1332-NEXT: v_writelane_b32 v4, s6, 16
; GFX1332-NEXT: s_mov_b32 exec_lo, s2
+; GFX1332-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1332-NEXT: v_cmp_eq_u32_e32 vcc_lo, 0, v7
; GFX1332-NEXT: s_mov_b32 s2, -1
; GFX1332-NEXT: ; implicit-def: $vgpr7_vgpr8
@@ -14862,8 +14940,8 @@ define amdgpu_kernel void @min_i32_varying(ptr addrspace(1) %out) {
; GFX1164_ITERATIVE-NEXT: v_mbcnt_hi_u32_b32 v1, exec_hi, v1
; GFX1164_ITERATIVE-NEXT: v_cmp_eq_u32_e32 vcc, 0, v1
; GFX1164_ITERATIVE-NEXT: ; implicit-def: $vgpr1
+; GFX1164_ITERATIVE-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1164_ITERATIVE-NEXT: s_and_saveexec_b64 s[0:1], vcc
-; GFX1164_ITERATIVE-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164_ITERATIVE-NEXT: s_xor_b64 s[0:1], exec, s[0:1]
; GFX1164_ITERATIVE-NEXT: s_cbranch_execz .LBB24_4
; GFX1164_ITERATIVE-NEXT: ; %bb.3:
@@ -15154,14 +15232,16 @@ define amdgpu_kernel void @min_i32_varying(ptr addrspace(1) %out) {
; GFX1164_DPP-NEXT: v_writelane_b32 v3, s3, 32
; GFX1164_DPP-NEXT: s_waitcnt_depctr depctr_sa_sdst(0)
; GFX1164_DPP-NEXT: s_mov_b64 exec, s[0:1]
-; GFX1164_DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_3) | instid1(VALU_DEP_2)
+; GFX1164_DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164_DPP-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
; GFX1164_DPP-NEXT: s_or_saveexec_b64 s[0:1], -1
; GFX1164_DPP-NEXT: v_writelane_b32 v3, s2, 48
; GFX1164_DPP-NEXT: s_mov_b64 exec, s[0:1]
+; GFX1164_DPP-NEXT: s_delay_alu instid0(VALU_DEP_2) | instid1(SALU_CYCLE_1)
; GFX1164_DPP-NEXT: v_cmp_eq_u32_e32 vcc, 0, v0
; GFX1164_DPP-NEXT: s_mov_b32 s2, -1
; GFX1164_DPP-NEXT: ; implicit-def: $vgpr0
+; GFX1164_DPP-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1164_DPP-NEXT: s_and_saveexec_b64 s[0:1], vcc
; GFX1164_DPP-NEXT: s_cbranch_execz .LBB24_2
; GFX1164_DPP-NEXT: ; %bb.1:
@@ -15207,7 +15287,7 @@ define amdgpu_kernel void @min_i32_varying(ptr addrspace(1) %out) {
; GFX1132_DPP-NEXT: s_or_saveexec_b32 s0, -1
; GFX1132_DPP-NEXT: v_writelane_b32 v3, s1, 16
; GFX1132_DPP-NEXT: s_mov_b32 exec_lo, s0
-; GFX1132_DPP-NEXT: s_delay_alu instid0(VALU_DEP_2)
+; GFX1132_DPP-NEXT: s_delay_alu instid0(VALU_DEP_2) | instid1(SALU_CYCLE_1)
; GFX1132_DPP-NEXT: v_cmp_eq_u32_e32 vcc_lo, 0, v0
; GFX1132_DPP-NEXT: s_mov_b32 s0, s2
; GFX1132_DPP-NEXT: s_mov_b32 s2, -1
@@ -15265,14 +15345,16 @@ define amdgpu_kernel void @min_i32_varying(ptr addrspace(1) %out) {
; GFX1364-NEXT: v_readlane_b32 s6, v1, 63
; GFX1364-NEXT: v_writelane_b32 v3, s3, 32
; GFX1364-NEXT: s_mov_b64 exec, s[0:1]
-; GFX1364-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_3) | instid1(VALU_DEP_2)
+; GFX1364-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1364-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
; GFX1364-NEXT: s_or_saveexec_b64 s[0:1], -1
; GFX1364-NEXT: v_writelane_b32 v3, s2, 48
; GFX1364-NEXT: s_mov_b64 exec, s[0:1]
+; GFX1364-NEXT: s_delay_alu instid0(VALU_DEP_2) | instid1(SALU_CYCLE_1)
; GFX1364-NEXT: v_cmp_eq_u32_e32 vcc, 0, v0
; GFX1364-NEXT: s_mov_b32 s2, -1
; GFX1364-NEXT: ; implicit-def: $vgpr0
+; GFX1364-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1364-NEXT: s_and_saveexec_b64 s[0:1], vcc
; GFX1364-NEXT: s_cbranch_execz .LBB24_2
; GFX1364-NEXT: ; %bb.1:
@@ -15318,7 +15400,7 @@ define amdgpu_kernel void @min_i32_varying(ptr addrspace(1) %out) {
; GFX1332-NEXT: s_or_saveexec_b32 s0, -1
; GFX1332-NEXT: v_writelane_b32 v3, s1, 16
; GFX1332-NEXT: s_mov_b32 exec_lo, s0
-; GFX1332-NEXT: s_delay_alu instid0(VALU_DEP_2)
+; GFX1332-NEXT: s_delay_alu instid0(VALU_DEP_2) | instid1(SALU_CYCLE_1)
; GFX1332-NEXT: v_cmp_eq_u32_e32 vcc_lo, 0, v0
; GFX1332-NEXT: s_mov_b32 s0, s2
; GFX1332-NEXT: s_mov_b32 s2, -1
@@ -15520,6 +15602,7 @@ define amdgpu_kernel void @min_i64_constant(ptr addrspace(1) %out) {
; GFX1164-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
; GFX1164-NEXT: v_cmp_eq_u32_e32 vcc, 0, v0
; GFX1164-NEXT: ; implicit-def: $vgpr0_vgpr1
+; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1164-NEXT: s_and_saveexec_b64 s[0:1], vcc
; GFX1164-NEXT: s_cbranch_execz .LBB25_2
; GFX1164-NEXT: ; %bb.1:
@@ -15531,12 +15614,12 @@ define amdgpu_kernel void @min_i64_constant(ptr addrspace(1) %out) {
; GFX1164-NEXT: buffer_gl0_inv
; GFX1164-NEXT: .LBB25_2:
; GFX1164-NEXT: s_or_b64 exec, exec, s[0:1]
+; GFX1164-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_4) | instid1(VALU_DEP_1)
; GFX1164-NEXT: v_readfirstlane_b32 s3, v1
; GFX1164-NEXT: v_readfirstlane_b32 s2, v0
; GFX1164-NEXT: v_cndmask_b32_e64 v1, 0, 0x7fffffff, vcc
; GFX1164-NEXT: v_cndmask_b32_e64 v0, 5, -1, vcc
; GFX1164-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
-; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1164-NEXT: v_cmp_lt_i64_e32 vcc, s[2:3], v[0:1]
; GFX1164-NEXT: s_waitcnt_depctr depctr_va_vcc(0)
; GFX1164-NEXT: v_cndmask_b32_e64 v1, v1, s3, vcc
@@ -15585,6 +15668,7 @@ define amdgpu_kernel void @min_i64_constant(ptr addrspace(1) %out) {
; GFX1364-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
; GFX1364-NEXT: v_cmp_eq_u32_e32 vcc, 0, v0
; GFX1364-NEXT: ; implicit-def: $vgpr0_vgpr1
+; GFX1364-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1364-NEXT: s_and_saveexec_b64 s[0:1], vcc
; GFX1364-NEXT: s_cbranch_execz .LBB25_2
; GFX1364-NEXT: ; %bb.1:
@@ -15946,8 +16030,8 @@ define amdgpu_kernel void @min_i64_varying(ptr addrspace(1) %out) {
; GFX1164_ITERATIVE-NEXT: v_mbcnt_hi_u32_b32 v2, exec_hi, v2
; GFX1164_ITERATIVE-NEXT: v_cmp_eq_u32_e32 vcc, 0, v2
; GFX1164_ITERATIVE-NEXT: ; implicit-def: $vgpr2_vgpr3
+; GFX1164_ITERATIVE-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1164_ITERATIVE-NEXT: s_and_saveexec_b64 s[2:3], vcc
-; GFX1164_ITERATIVE-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164_ITERATIVE-NEXT: s_xor_b64 s[2:3], exec, s[2:3]
; GFX1164_ITERATIVE-NEXT: s_cbranch_execz .LBB26_4
; GFX1164_ITERATIVE-NEXT: ; %bb.3:
@@ -15959,6 +16043,7 @@ define amdgpu_kernel void @min_i64_varying(ptr addrspace(1) %out) {
; GFX1164_ITERATIVE-NEXT: buffer_gl0_inv
; GFX1164_ITERATIVE-NEXT: .LBB26_4:
; GFX1164_ITERATIVE-NEXT: s_or_b64 exec, exec, s[2:3]
+; GFX1164_ITERATIVE-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164_ITERATIVE-NEXT: v_readfirstlane_b32 s3, v3
; GFX1164_ITERATIVE-NEXT: v_readfirstlane_b32 s2, v2
; GFX1164_ITERATIVE-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
@@ -16512,6 +16597,7 @@ define amdgpu_kernel void @min_i64_varying(ptr addrspace(1) %out) {
; GFX1164_DPP-NEXT: v_writelane_b32 v5, s8, 48
; GFX1164_DPP-NEXT: v_writelane_b32 v4, s9, 48
; GFX1164_DPP-NEXT: s_mov_b64 exec, s[6:7]
+; GFX1164_DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(VALU_DEP_1)
; GFX1164_DPP-NEXT: v_cmp_eq_u32_e32 vcc, 0, v7
; GFX1164_DPP-NEXT: s_mov_b32 s2, -1
; GFX1164_DPP-NEXT: ; implicit-def: $vgpr7_vgpr8
@@ -16604,6 +16690,7 @@ define amdgpu_kernel void @min_i64_varying(ptr addrspace(1) %out) {
; GFX1132_DPP-NEXT: v_writelane_b32 v5, s3, 16
; GFX1132_DPP-NEXT: v_writelane_b32 v4, s6, 16
; GFX1132_DPP-NEXT: s_mov_b32 exec_lo, s2
+; GFX1132_DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1132_DPP-NEXT: v_cmp_eq_u32_e32 vcc_lo, 0, v7
; GFX1132_DPP-NEXT: s_mov_b32 s2, -1
; GFX1132_DPP-NEXT: ; implicit-def: $vgpr7_vgpr8
@@ -16724,6 +16811,7 @@ define amdgpu_kernel void @min_i64_varying(ptr addrspace(1) %out) {
; GFX1364-NEXT: v_writelane_b32 v5, s8, 48
; GFX1364-NEXT: v_writelane_b32 v4, s9, 48
; GFX1364-NEXT: s_mov_b64 exec, s[6:7]
+; GFX1364-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX1364-NEXT: v_cmp_eq_u32_e32 vcc, 0, v7
; GFX1364-NEXT: s_mov_b32 s2, -1
; GFX1364-NEXT: ; implicit-def: $vgpr7_vgpr8
@@ -16812,6 +16900,7 @@ define amdgpu_kernel void @min_i64_varying(ptr addrspace(1) %out) {
; GFX1332-NEXT: v_writelane_b32 v5, s3, 16
; GFX1332-NEXT: v_writelane_b32 v4, s6, 16
; GFX1332-NEXT: s_mov_b32 exec_lo, s2
+; GFX1332-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1332-NEXT: v_cmp_eq_u32_e32 vcc_lo, 0, v7
; GFX1332-NEXT: s_mov_b32 s2, -1
; GFX1332-NEXT: ; implicit-def: $vgpr7_vgpr8
@@ -17067,8 +17156,8 @@ define amdgpu_kernel void @umax_i32_varying(ptr addrspace(1) %out) {
; GFX1164_ITERATIVE-NEXT: v_mbcnt_hi_u32_b32 v1, exec_hi, v1
; GFX1164_ITERATIVE-NEXT: v_cmp_eq_u32_e32 vcc, 0, v1
; GFX1164_ITERATIVE-NEXT: ; implicit-def: $vgpr1
+; GFX1164_ITERATIVE-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1164_ITERATIVE-NEXT: s_and_saveexec_b64 s[0:1], vcc
-; GFX1164_ITERATIVE-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164_ITERATIVE-NEXT: s_xor_b64 s[0:1], exec, s[0:1]
; GFX1164_ITERATIVE-NEXT: s_cbranch_execz .LBB27_4
; GFX1164_ITERATIVE-NEXT: ; %bb.3:
@@ -17359,7 +17448,7 @@ define amdgpu_kernel void @umax_i32_varying(ptr addrspace(1) %out) {
; GFX1164_DPP-NEXT: v_writelane_b32 v3, s3, 32
; GFX1164_DPP-NEXT: s_waitcnt_depctr depctr_sa_sdst(0)
; GFX1164_DPP-NEXT: s_mov_b64 exec, s[0:1]
-; GFX1164_DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1164_DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_4) | instid1(SALU_CYCLE_1)
; GFX1164_DPP-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
; GFX1164_DPP-NEXT: v_mov_b32_e32 v4, 0
; GFX1164_DPP-NEXT: s_or_saveexec_b64 s[0:1], -1
@@ -17368,6 +17457,7 @@ define amdgpu_kernel void @umax_i32_varying(ptr addrspace(1) %out) {
; GFX1164_DPP-NEXT: v_cmp_eq_u32_e32 vcc, 0, v0
; GFX1164_DPP-NEXT: s_mov_b32 s2, -1
; GFX1164_DPP-NEXT: ; implicit-def: $vgpr0
+; GFX1164_DPP-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1164_DPP-NEXT: s_and_saveexec_b64 s[0:1], vcc
; GFX1164_DPP-NEXT: s_cbranch_execz .LBB27_2
; GFX1164_DPP-NEXT: ; %bb.1:
@@ -17413,6 +17503,7 @@ define amdgpu_kernel void @umax_i32_varying(ptr addrspace(1) %out) {
; GFX1132_DPP-NEXT: s_or_saveexec_b32 s0, -1
; GFX1132_DPP-NEXT: v_writelane_b32 v3, s1, 16
; GFX1132_DPP-NEXT: s_mov_b32 exec_lo, s0
+; GFX1132_DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1132_DPP-NEXT: v_cmp_eq_u32_e32 vcc_lo, 0, v0
; GFX1132_DPP-NEXT: s_mov_b32 s0, s2
; GFX1132_DPP-NEXT: s_mov_b32 s2, -1
@@ -17469,7 +17560,7 @@ define amdgpu_kernel void @umax_i32_varying(ptr addrspace(1) %out) {
; GFX1364-NEXT: v_readlane_b32 s6, v1, 63
; GFX1364-NEXT: v_writelane_b32 v3, s3, 32
; GFX1364-NEXT: s_mov_b64 exec, s[0:1]
-; GFX1364-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1364-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_4) | instid1(SALU_CYCLE_1)
; GFX1364-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
; GFX1364-NEXT: v_mov_b32_e32 v4, 0
; GFX1364-NEXT: s_or_saveexec_b64 s[0:1], -1
@@ -17478,6 +17569,7 @@ define amdgpu_kernel void @umax_i32_varying(ptr addrspace(1) %out) {
; GFX1364-NEXT: v_cmp_eq_u32_e32 vcc, 0, v0
; GFX1364-NEXT: s_mov_b32 s2, -1
; GFX1364-NEXT: ; implicit-def: $vgpr0
+; GFX1364-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1364-NEXT: s_and_saveexec_b64 s[0:1], vcc
; GFX1364-NEXT: s_cbranch_execz .LBB27_2
; GFX1364-NEXT: ; %bb.1:
@@ -17522,6 +17614,7 @@ define amdgpu_kernel void @umax_i32_varying(ptr addrspace(1) %out) {
; GFX1332-NEXT: s_or_saveexec_b32 s0, -1
; GFX1332-NEXT: v_writelane_b32 v3, s1, 16
; GFX1332-NEXT: s_mov_b32 exec_lo, s0
+; GFX1332-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1332-NEXT: v_cmp_eq_u32_e32 vcc_lo, 0, v0
; GFX1332-NEXT: s_mov_b32 s0, s2
; GFX1332-NEXT: s_mov_b32 s2, -1
@@ -17720,6 +17813,7 @@ define amdgpu_kernel void @umax_i64_constant(ptr addrspace(1) %out) {
; GFX1164-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
; GFX1164-NEXT: v_cmp_eq_u32_e32 vcc, 0, v0
; GFX1164-NEXT: ; implicit-def: $vgpr0_vgpr1
+; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1164-NEXT: s_and_saveexec_b64 s[0:1], vcc
; GFX1164-NEXT: s_cbranch_execz .LBB28_2
; GFX1164-NEXT: ; %bb.1:
@@ -17731,12 +17825,12 @@ define amdgpu_kernel void @umax_i64_constant(ptr addrspace(1) %out) {
; GFX1164-NEXT: buffer_gl0_inv
; GFX1164-NEXT: .LBB28_2:
; GFX1164-NEXT: s_or_b64 exec, exec, s[0:1]
+; GFX1164-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_4) | instid1(VALU_DEP_1)
; GFX1164-NEXT: v_readfirstlane_b32 s3, v1
; GFX1164-NEXT: v_readfirstlane_b32 s2, v0
; GFX1164-NEXT: v_mov_b32_e32 v1, 0
; GFX1164-NEXT: v_cndmask_b32_e64 v0, 5, 0, vcc
; GFX1164-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
-; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1164-NEXT: v_cmp_gt_u64_e32 vcc, s[2:3], v[0:1]
; GFX1164-NEXT: s_waitcnt_depctr depctr_va_vcc(0)
; GFX1164-NEXT: v_cndmask_b32_e64 v0, v0, s2, vcc
@@ -17785,6 +17879,7 @@ define amdgpu_kernel void @umax_i64_constant(ptr addrspace(1) %out) {
; GFX1364-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
; GFX1364-NEXT: v_cmp_eq_u32_e32 vcc, 0, v0
; GFX1364-NEXT: ; implicit-def: $vgpr0_vgpr1
+; GFX1364-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1364-NEXT: s_and_saveexec_b64 s[0:1], vcc
; GFX1364-NEXT: s_cbranch_execz .LBB28_2
; GFX1364-NEXT: ; %bb.1:
@@ -18140,8 +18235,8 @@ define amdgpu_kernel void @umax_i64_varying(ptr addrspace(1) %out) {
; GFX1164_ITERATIVE-NEXT: v_mbcnt_hi_u32_b32 v2, exec_hi, v2
; GFX1164_ITERATIVE-NEXT: v_cmp_eq_u32_e32 vcc, 0, v2
; GFX1164_ITERATIVE-NEXT: ; implicit-def: $vgpr2_vgpr3
+; GFX1164_ITERATIVE-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1164_ITERATIVE-NEXT: s_and_saveexec_b64 s[2:3], vcc
-; GFX1164_ITERATIVE-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164_ITERATIVE-NEXT: s_xor_b64 s[2:3], exec, s[2:3]
; GFX1164_ITERATIVE-NEXT: s_cbranch_execz .LBB29_4
; GFX1164_ITERATIVE-NEXT: ; %bb.3:
@@ -18153,6 +18248,7 @@ define amdgpu_kernel void @umax_i64_varying(ptr addrspace(1) %out) {
; GFX1164_ITERATIVE-NEXT: buffer_gl0_inv
; GFX1164_ITERATIVE-NEXT: .LBB29_4:
; GFX1164_ITERATIVE-NEXT: s_or_b64 exec, exec, s[2:3]
+; GFX1164_ITERATIVE-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164_ITERATIVE-NEXT: v_readfirstlane_b32 s3, v3
; GFX1164_ITERATIVE-NEXT: v_readfirstlane_b32 s2, v2
; GFX1164_ITERATIVE-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
@@ -18707,6 +18803,7 @@ define amdgpu_kernel void @umax_i64_varying(ptr addrspace(1) %out) {
; GFX1164_DPP-NEXT: v_writelane_b32 v5, s8, 48
; GFX1164_DPP-NEXT: v_writelane_b32 v4, s9, 48
; GFX1164_DPP-NEXT: s_mov_b64 exec, s[6:7]
+; GFX1164_DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(VALU_DEP_1)
; GFX1164_DPP-NEXT: v_cmp_eq_u32_e32 vcc, 0, v7
; GFX1164_DPP-NEXT: s_mov_b32 s2, -1
; GFX1164_DPP-NEXT: ; implicit-def: $vgpr7_vgpr8
@@ -18793,6 +18890,7 @@ define amdgpu_kernel void @umax_i64_varying(ptr addrspace(1) %out) {
; GFX1132_DPP-NEXT: v_writelane_b32 v5, s3, 16
; GFX1132_DPP-NEXT: v_writelane_b32 v4, s6, 16
; GFX1132_DPP-NEXT: s_mov_b32 exec_lo, s2
+; GFX1132_DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1132_DPP-NEXT: v_cmp_eq_u32_e32 vcc_lo, 0, v7
; GFX1132_DPP-NEXT: s_mov_b32 s2, -1
; GFX1132_DPP-NEXT: ; implicit-def: $vgpr7_vgpr8
@@ -18913,6 +19011,7 @@ define amdgpu_kernel void @umax_i64_varying(ptr addrspace(1) %out) {
; GFX1364-NEXT: v_writelane_b32 v5, s8, 48
; GFX1364-NEXT: v_writelane_b32 v4, s9, 48
; GFX1364-NEXT: s_mov_b64 exec, s[6:7]
+; GFX1364-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX1364-NEXT: v_cmp_eq_u32_e32 vcc, 0, v7
; GFX1364-NEXT: s_mov_b32 s2, -1
; GFX1364-NEXT: ; implicit-def: $vgpr7_vgpr8
@@ -18999,6 +19098,7 @@ define amdgpu_kernel void @umax_i64_varying(ptr addrspace(1) %out) {
; GFX1332-NEXT: v_writelane_b32 v5, s3, 16
; GFX1332-NEXT: v_writelane_b32 v4, s6, 16
; GFX1332-NEXT: s_mov_b32 exec_lo, s2
+; GFX1332-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1332-NEXT: v_cmp_eq_u32_e32 vcc_lo, 0, v7
; GFX1332-NEXT: s_mov_b32 s2, -1
; GFX1332-NEXT: ; implicit-def: $vgpr7_vgpr8
@@ -19254,8 +19354,8 @@ define amdgpu_kernel void @umin_i32_varying(ptr addrspace(1) %out) {
; GFX1164_ITERATIVE-NEXT: v_mbcnt_hi_u32_b32 v1, exec_hi, v1
; GFX1164_ITERATIVE-NEXT: v_cmp_eq_u32_e32 vcc, 0, v1
; GFX1164_ITERATIVE-NEXT: ; implicit-def: $vgpr1
+; GFX1164_ITERATIVE-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1164_ITERATIVE-NEXT: s_and_saveexec_b64 s[0:1], vcc
-; GFX1164_ITERATIVE-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164_ITERATIVE-NEXT: s_xor_b64 s[0:1], exec, s[0:1]
; GFX1164_ITERATIVE-NEXT: s_cbranch_execz .LBB30_4
; GFX1164_ITERATIVE-NEXT: ; %bb.3:
@@ -19546,14 +19646,16 @@ define amdgpu_kernel void @umin_i32_varying(ptr addrspace(1) %out) {
; GFX1164_DPP-NEXT: v_writelane_b32 v3, s3, 32
; GFX1164_DPP-NEXT: s_waitcnt_depctr depctr_sa_sdst(0)
; GFX1164_DPP-NEXT: s_mov_b64 exec, s[0:1]
-; GFX1164_DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_3) | instid1(VALU_DEP_2)
+; GFX1164_DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164_DPP-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
; GFX1164_DPP-NEXT: s_or_saveexec_b64 s[0:1], -1
; GFX1164_DPP-NEXT: v_writelane_b32 v3, s2, 48
; GFX1164_DPP-NEXT: s_mov_b64 exec, s[0:1]
+; GFX1164_DPP-NEXT: s_delay_alu instid0(VALU_DEP_2) | instid1(SALU_CYCLE_1)
; GFX1164_DPP-NEXT: v_cmp_eq_u32_e32 vcc, 0, v0
; GFX1164_DPP-NEXT: s_mov_b32 s2, -1
; GFX1164_DPP-NEXT: ; implicit-def: $vgpr0
+; GFX1164_DPP-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1164_DPP-NEXT: s_and_saveexec_b64 s[0:1], vcc
; GFX1164_DPP-NEXT: s_cbranch_execz .LBB30_2
; GFX1164_DPP-NEXT: ; %bb.1:
@@ -19599,7 +19701,7 @@ define amdgpu_kernel void @umin_i32_varying(ptr addrspace(1) %out) {
; GFX1132_DPP-NEXT: s_or_saveexec_b32 s0, -1
; GFX1132_DPP-NEXT: v_writelane_b32 v3, s1, 16
; GFX1132_DPP-NEXT: s_mov_b32 exec_lo, s0
-; GFX1132_DPP-NEXT: s_delay_alu instid0(VALU_DEP_2)
+; GFX1132_DPP-NEXT: s_delay_alu instid0(VALU_DEP_2) | instid1(SALU_CYCLE_1)
; GFX1132_DPP-NEXT: v_cmp_eq_u32_e32 vcc_lo, 0, v0
; GFX1132_DPP-NEXT: s_mov_b32 s0, s2
; GFX1132_DPP-NEXT: s_mov_b32 s2, -1
@@ -19657,14 +19759,16 @@ define amdgpu_kernel void @umin_i32_varying(ptr addrspace(1) %out) {
; GFX1364-NEXT: v_readlane_b32 s6, v1, 63
; GFX1364-NEXT: v_writelane_b32 v3, s3, 32
; GFX1364-NEXT: s_mov_b64 exec, s[0:1]
-; GFX1364-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_3) | instid1(VALU_DEP_2)
+; GFX1364-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1364-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
; GFX1364-NEXT: s_or_saveexec_b64 s[0:1], -1
; GFX1364-NEXT: v_writelane_b32 v3, s2, 48
; GFX1364-NEXT: s_mov_b64 exec, s[0:1]
+; GFX1364-NEXT: s_delay_alu instid0(VALU_DEP_2) | instid1(SALU_CYCLE_1)
; GFX1364-NEXT: v_cmp_eq_u32_e32 vcc, 0, v0
; GFX1364-NEXT: s_mov_b32 s2, -1
; GFX1364-NEXT: ; implicit-def: $vgpr0
+; GFX1364-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1364-NEXT: s_and_saveexec_b64 s[0:1], vcc
; GFX1364-NEXT: s_cbranch_execz .LBB30_2
; GFX1364-NEXT: ; %bb.1:
@@ -19709,7 +19813,7 @@ define amdgpu_kernel void @umin_i32_varying(ptr addrspace(1) %out) {
; GFX1332-NEXT: s_or_saveexec_b32 s0, -1
; GFX1332-NEXT: v_writelane_b32 v3, s1, 16
; GFX1332-NEXT: s_mov_b32 exec_lo, s0
-; GFX1332-NEXT: s_delay_alu instid0(VALU_DEP_2)
+; GFX1332-NEXT: s_delay_alu instid0(VALU_DEP_2) | instid1(SALU_CYCLE_1)
; GFX1332-NEXT: v_cmp_eq_u32_e32 vcc_lo, 0, v0
; GFX1332-NEXT: s_mov_b32 s0, s2
; GFX1332-NEXT: s_mov_b32 s2, -1
@@ -19908,6 +20012,7 @@ define amdgpu_kernel void @umin_i64_constant(ptr addrspace(1) %out) {
; GFX1164-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
; GFX1164-NEXT: v_cmp_eq_u32_e32 vcc, 0, v0
; GFX1164-NEXT: ; implicit-def: $vgpr0_vgpr1
+; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1164-NEXT: s_and_saveexec_b64 s[0:1], vcc
; GFX1164-NEXT: s_cbranch_execz .LBB31_2
; GFX1164-NEXT: ; %bb.1:
@@ -19919,12 +20024,12 @@ define amdgpu_kernel void @umin_i64_constant(ptr addrspace(1) %out) {
; GFX1164-NEXT: buffer_gl0_inv
; GFX1164-NEXT: .LBB31_2:
; GFX1164-NEXT: s_or_b64 exec, exec, s[0:1]
+; GFX1164-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_4) | instid1(VALU_DEP_1)
; GFX1164-NEXT: v_readfirstlane_b32 s3, v1
; GFX1164-NEXT: v_readfirstlane_b32 s2, v0
; GFX1164-NEXT: v_cndmask_b32_e64 v1, 0, -1, vcc
; GFX1164-NEXT: v_cndmask_b32_e64 v0, 5, -1, vcc
; GFX1164-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
-; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1164-NEXT: v_cmp_lt_u64_e32 vcc, s[2:3], v[0:1]
; GFX1164-NEXT: s_waitcnt_depctr depctr_va_vcc(0)
; GFX1164-NEXT: v_cndmask_b32_e64 v1, v1, s3, vcc
@@ -19973,6 +20078,7 @@ define amdgpu_kernel void @umin_i64_constant(ptr addrspace(1) %out) {
; GFX1364-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
; GFX1364-NEXT: v_cmp_eq_u32_e32 vcc, 0, v0
; GFX1364-NEXT: ; implicit-def: $vgpr0_vgpr1
+; GFX1364-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1364-NEXT: s_and_saveexec_b64 s[0:1], vcc
; GFX1364-NEXT: s_cbranch_execz .LBB31_2
; GFX1364-NEXT: ; %bb.1:
@@ -20328,8 +20434,8 @@ define amdgpu_kernel void @umin_i64_varying(ptr addrspace(1) %out) {
; GFX1164_ITERATIVE-NEXT: v_mbcnt_hi_u32_b32 v2, exec_hi, v2
; GFX1164_ITERATIVE-NEXT: v_cmp_eq_u32_e32 vcc, 0, v2
; GFX1164_ITERATIVE-NEXT: ; implicit-def: $vgpr2_vgpr3
+; GFX1164_ITERATIVE-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1164_ITERATIVE-NEXT: s_and_saveexec_b64 s[2:3], vcc
-; GFX1164_ITERATIVE-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164_ITERATIVE-NEXT: s_xor_b64 s[2:3], exec, s[2:3]
; GFX1164_ITERATIVE-NEXT: s_cbranch_execz .LBB32_4
; GFX1164_ITERATIVE-NEXT: ; %bb.3:
@@ -20341,6 +20447,7 @@ define amdgpu_kernel void @umin_i64_varying(ptr addrspace(1) %out) {
; GFX1164_ITERATIVE-NEXT: buffer_gl0_inv
; GFX1164_ITERATIVE-NEXT: .LBB32_4:
; GFX1164_ITERATIVE-NEXT: s_or_b64 exec, exec, s[2:3]
+; GFX1164_ITERATIVE-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164_ITERATIVE-NEXT: v_readfirstlane_b32 s3, v3
; GFX1164_ITERATIVE-NEXT: v_readfirstlane_b32 s2, v2
; GFX1164_ITERATIVE-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
@@ -20893,6 +21000,7 @@ define amdgpu_kernel void @umin_i64_varying(ptr addrspace(1) %out) {
; GFX1164_DPP-NEXT: v_writelane_b32 v5, s8, 48
; GFX1164_DPP-NEXT: v_writelane_b32 v4, s9, 48
; GFX1164_DPP-NEXT: s_mov_b64 exec, s[6:7]
+; GFX1164_DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(VALU_DEP_1)
; GFX1164_DPP-NEXT: v_cmp_eq_u32_e32 vcc, 0, v7
; GFX1164_DPP-NEXT: s_mov_b32 s2, -1
; GFX1164_DPP-NEXT: ; implicit-def: $vgpr7_vgpr8
@@ -20979,6 +21087,7 @@ define amdgpu_kernel void @umin_i64_varying(ptr addrspace(1) %out) {
; GFX1132_DPP-NEXT: v_writelane_b32 v5, s3, 16
; GFX1132_DPP-NEXT: v_writelane_b32 v4, s6, 16
; GFX1132_DPP-NEXT: s_mov_b32 exec_lo, s2
+; GFX1132_DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1132_DPP-NEXT: v_cmp_eq_u32_e32 vcc_lo, 0, v7
; GFX1132_DPP-NEXT: s_mov_b32 s2, -1
; GFX1132_DPP-NEXT: ; implicit-def: $vgpr7_vgpr8
@@ -21099,6 +21208,7 @@ define amdgpu_kernel void @umin_i64_varying(ptr addrspace(1) %out) {
; GFX1364-NEXT: v_writelane_b32 v5, s8, 48
; GFX1364-NEXT: v_writelane_b32 v4, s9, 48
; GFX1364-NEXT: s_mov_b64 exec, s[6:7]
+; GFX1364-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX1364-NEXT: v_cmp_eq_u32_e32 vcc, 0, v7
; GFX1364-NEXT: s_mov_b32 s2, -1
; GFX1364-NEXT: ; implicit-def: $vgpr7_vgpr8
@@ -21185,6 +21295,7 @@ define amdgpu_kernel void @umin_i64_varying(ptr addrspace(1) %out) {
; GFX1332-NEXT: v_writelane_b32 v5, s3, 16
; GFX1332-NEXT: v_writelane_b32 v4, s6, 16
; GFX1332-NEXT: s_mov_b32 exec_lo, s2
+; GFX1332-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1332-NEXT: v_cmp_eq_u32_e32 vcc_lo, 0, v7
; GFX1332-NEXT: s_mov_b32 s2, -1
; GFX1332-NEXT: ; implicit-def: $vgpr7_vgpr8
diff --git a/llvm/test/CodeGen/AMDGPU/atomic_optimizations_pixelshader.ll b/llvm/test/CodeGen/AMDGPU/atomic_optimizations_pixelshader.ll
index 4c87e6936e3142..1fab78ab17b705 100644
--- a/llvm/test/CodeGen/AMDGPU/atomic_optimizations_pixelshader.ll
+++ b/llvm/test/CodeGen/AMDGPU/atomic_optimizations_pixelshader.ll
@@ -180,6 +180,7 @@ define amdgpu_ps void @add_i32_constant(ptr addrspace(8) inreg %out, ptr addrspa
; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164-NEXT: v_mbcnt_hi_u32_b32 v0, s13, v0
; GFX1164-NEXT: v_cmpx_eq_u32_e32 0, v0
+; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1164-NEXT: s_cbranch_execz .LBB0_3
; GFX1164-NEXT: ; %bb.2:
; GFX1164-NEXT: s_bcnt1_i32_b64 s12, s[12:13]
@@ -200,7 +201,7 @@ define amdgpu_ps void @add_i32_constant(ptr addrspace(8) inreg %out, ptr addrspa
; GFX1164-NEXT: s_and_b64 s[4:5], s[4:5], s[4:5]
; GFX1164-NEXT: s_and_b64 s[4:5], s[4:5], exec
; GFX1164-NEXT: s_cselect_b32 s4, 1, 0
-; GFX1164-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1164-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1164-NEXT: s_cmp_lg_u32 s4, 1
; GFX1164-NEXT: s_cbranch_scc1 .LBB0_6
; GFX1164-NEXT: ; %bb.5: ; %if
@@ -220,7 +221,7 @@ define amdgpu_ps void @add_i32_constant(ptr addrspace(8) inreg %out, ptr addrspa
; GFX1132-NEXT: s_mov_b32 s9, exec_lo
; GFX1132-NEXT: v_mbcnt_lo_u32_b32 v0, s10, 0
; GFX1132-NEXT: ; implicit-def: $vgpr1
-; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX1132-NEXT: s_cbranch_execz .LBB0_3
; GFX1132-NEXT: ; %bb.2:
@@ -242,7 +243,7 @@ define amdgpu_ps void @add_i32_constant(ptr addrspace(8) inreg %out, ptr addrspa
; GFX1132-NEXT: s_and_b32 s4, s4, s4
; GFX1132-NEXT: s_and_b32 s4, s4, exec_lo
; GFX1132-NEXT: s_cselect_b32 s4, 1, 0
-; GFX1132-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1132-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1132-NEXT: s_cmp_lg_u32 s4, 1
; GFX1132-NEXT: s_cbranch_scc1 .LBB0_6
; GFX1132-NEXT: ; %bb.5: ; %if
@@ -265,6 +266,7 @@ define amdgpu_ps void @add_i32_constant(ptr addrspace(8) inreg %out, ptr addrspa
; GFX1364-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1364-NEXT: v_mbcnt_hi_u32_b32 v0, s13, v0
; GFX1364-NEXT: v_cmpx_eq_u32_e32 0, v0
+; GFX1364-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1364-NEXT: s_cbranch_execz .LBB0_3
; GFX1364-NEXT: ; %bb.2:
; GFX1364-NEXT: s_bcnt1_i32_b64 s12, s[12:13]
@@ -285,7 +287,7 @@ define amdgpu_ps void @add_i32_constant(ptr addrspace(8) inreg %out, ptr addrspa
; GFX1364-NEXT: s_and_b64 s[4:5], s[4:5], s[4:5]
; GFX1364-NEXT: s_and_b64 s[4:5], s[4:5], exec
; GFX1364-NEXT: s_cselect_b32 s4, 1, 0
-; GFX1364-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1364-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1364-NEXT: s_cmp_lg_u32 s4, 1
; GFX1364-NEXT: s_cbranch_scc1 .LBB0_6
; GFX1364-NEXT: ; %bb.5: ; %if
@@ -305,7 +307,7 @@ define amdgpu_ps void @add_i32_constant(ptr addrspace(8) inreg %out, ptr addrspa
; GFX1332-NEXT: s_mov_b32 s9, exec_lo
; GFX1332-NEXT: v_mbcnt_lo_u32_b32 v0, s10, 0
; GFX1332-NEXT: ; implicit-def: $vgpr1
-; GFX1332-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1332-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1332-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX1332-NEXT: s_cbranch_execz .LBB0_3
; GFX1332-NEXT: ; %bb.2:
@@ -327,7 +329,7 @@ define amdgpu_ps void @add_i32_constant(ptr addrspace(8) inreg %out, ptr addrspa
; GFX1332-NEXT: s_and_b32 s4, s4, s4
; GFX1332-NEXT: s_and_b32 s4, s4, exec_lo
; GFX1332-NEXT: s_cselect_b32 s4, 1, 0
-; GFX1332-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1332-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1332-NEXT: s_cmp_lg_u32 s4, 1
; GFX1332-NEXT: s_cbranch_scc1 .LBB0_6
; GFX1332-NEXT: ; %bb.5: ; %if
@@ -631,13 +633,15 @@ define amdgpu_ps void @add_i32_varying(ptr addrspace(8) inreg %out, ptr addrspac
; GFX1164-NEXT: v_writelane_b32 v3, s13, 32
; GFX1164-NEXT: s_waitcnt_depctr depctr_sa_sdst(0)
; GFX1164-NEXT: s_mov_b64 exec, s[10:11]
-; GFX1164-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_3) | instid1(VALU_DEP_2)
+; GFX1164-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
; GFX1164-NEXT: s_or_saveexec_b64 s[10:11], -1
; GFX1164-NEXT: v_writelane_b32 v3, s14, 48
; GFX1164-NEXT: s_mov_b64 exec, s[10:11]
+; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_2) | instid1(SALU_CYCLE_1)
; GFX1164-NEXT: v_cmp_eq_u32_e32 vcc, 0, v0
; GFX1164-NEXT: ; implicit-def: $vgpr0
+; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1164-NEXT: s_and_saveexec_b64 s[10:11], vcc
; GFX1164-NEXT: s_cbranch_execz .LBB1_3
; GFX1164-NEXT: ; %bb.2:
@@ -657,7 +661,7 @@ define amdgpu_ps void @add_i32_varying(ptr addrspace(8) inreg %out, ptr addrspac
; GFX1164-NEXT: s_and_b64 s[4:5], s[4:5], s[4:5]
; GFX1164-NEXT: s_and_b64 s[4:5], s[4:5], exec
; GFX1164-NEXT: s_cselect_b32 s4, 1, 0
-; GFX1164-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1164-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1164-NEXT: s_cmp_lg_u32 s4, 1
; GFX1164-NEXT: s_cbranch_scc1 .LBB1_6
; GFX1164-NEXT: ; %bb.5: ; %if
@@ -691,11 +695,12 @@ define amdgpu_ps void @add_i32_varying(ptr addrspace(8) inreg %out, ptr addrspac
; GFX1132-NEXT: v_mov_b32_dpp v3, v1 row_shr:1 row_mask:0xf bank_mask:0xf
; GFX1132-NEXT: v_readlane_b32 s10, v1, 15
; GFX1132-NEXT: s_mov_b32 exec_lo, s9
-; GFX1132-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_3) | instid1(VALU_DEP_2)
+; GFX1132-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1132-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
; GFX1132-NEXT: s_or_saveexec_b32 s9, -1
; GFX1132-NEXT: v_writelane_b32 v3, s10, 16
; GFX1132-NEXT: s_mov_b32 exec_lo, s9
+; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_2) | instid1(SALU_CYCLE_1)
; GFX1132-NEXT: v_cmp_eq_u32_e32 vcc_lo, 0, v0
; GFX1132-NEXT: ; implicit-def: $vgpr0
; GFX1132-NEXT: s_and_saveexec_b32 s9, vcc_lo
@@ -717,7 +722,7 @@ define amdgpu_ps void @add_i32_varying(ptr addrspace(8) inreg %out, ptr addrspac
; GFX1132-NEXT: s_and_b32 s4, s4, s4
; GFX1132-NEXT: s_and_b32 s4, s4, exec_lo
; GFX1132-NEXT: s_cselect_b32 s4, 1, 0
-; GFX1132-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1132-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1132-NEXT: s_cmp_lg_u32 s4, 1
; GFX1132-NEXT: s_cbranch_scc1 .LBB1_6
; GFX1132-NEXT: ; %bb.5: ; %if
@@ -764,13 +769,15 @@ define amdgpu_ps void @add_i32_varying(ptr addrspace(8) inreg %out, ptr addrspac
; GFX1364-NEXT: v_readlane_b32 s14, v1, 47
; GFX1364-NEXT: v_writelane_b32 v3, s13, 32
; GFX1364-NEXT: s_mov_b64 exec, s[10:11]
-; GFX1364-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_3) | instid1(VALU_DEP_2)
+; GFX1364-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1364-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
; GFX1364-NEXT: s_or_saveexec_b64 s[10:11], -1
; GFX1364-NEXT: v_writelane_b32 v3, s14, 48
; GFX1364-NEXT: s_mov_b64 exec, s[10:11]
+; GFX1364-NEXT: s_delay_alu instid0(VALU_DEP_2) | instid1(SALU_CYCLE_1)
; GFX1364-NEXT: v_cmp_eq_u32_e32 vcc, 0, v0
; GFX1364-NEXT: ; implicit-def: $vgpr0
+; GFX1364-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1364-NEXT: s_and_saveexec_b64 s[10:11], vcc
; GFX1364-NEXT: s_cbranch_execz .LBB1_3
; GFX1364-NEXT: ; %bb.2:
@@ -790,7 +797,7 @@ define amdgpu_ps void @add_i32_varying(ptr addrspace(8) inreg %out, ptr addrspac
; GFX1364-NEXT: s_and_b64 s[4:5], s[4:5], s[4:5]
; GFX1364-NEXT: s_and_b64 s[4:5], s[4:5], exec
; GFX1364-NEXT: s_cselect_b32 s4, 1, 0
-; GFX1364-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1364-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1364-NEXT: s_cmp_lg_u32 s4, 1
; GFX1364-NEXT: s_cbranch_scc1 .LBB1_6
; GFX1364-NEXT: ; %bb.5: ; %if
@@ -823,11 +830,12 @@ define amdgpu_ps void @add_i32_varying(ptr addrspace(8) inreg %out, ptr addrspac
; GFX1332-NEXT: v_mov_b32_dpp v3, v1 row_shr:1 row_mask:0xf bank_mask:0xf
; GFX1332-NEXT: v_readlane_b32 s10, v1, 15
; GFX1332-NEXT: s_mov_b32 exec_lo, s9
-; GFX1332-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_3) | instid1(VALU_DEP_2)
+; GFX1332-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1332-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
; GFX1332-NEXT: s_or_saveexec_b32 s9, -1
; GFX1332-NEXT: v_writelane_b32 v3, s10, 16
; GFX1332-NEXT: s_mov_b32 exec_lo, s9
+; GFX1332-NEXT: s_delay_alu instid0(VALU_DEP_2) | instid1(SALU_CYCLE_1)
; GFX1332-NEXT: v_cmp_eq_u32_e32 vcc_lo, 0, v0
; GFX1332-NEXT: ; implicit-def: $vgpr0
; GFX1332-NEXT: s_and_saveexec_b32 s9, vcc_lo
@@ -849,7 +857,7 @@ define amdgpu_ps void @add_i32_varying(ptr addrspace(8) inreg %out, ptr addrspac
; GFX1332-NEXT: s_and_b32 s4, s4, s4
; GFX1332-NEXT: s_and_b32 s4, s4, exec_lo
; GFX1332-NEXT: s_cselect_b32 s4, 1, 0
-; GFX1332-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1332-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1332-NEXT: s_cmp_lg_u32 s4, 1
; GFX1332-NEXT: s_cbranch_scc1 .LBB1_6
; GFX1332-NEXT: ; %bb.5: ; %if
diff --git a/llvm/test/CodeGen/AMDGPU/atomic_optimizations_raw_buffer.ll b/llvm/test/CodeGen/AMDGPU/atomic_optimizations_raw_buffer.ll
index debb51db298c1e..dbd8f3cc378163 100644
--- a/llvm/test/CodeGen/AMDGPU/atomic_optimizations_raw_buffer.ll
+++ b/llvm/test/CodeGen/AMDGPU/atomic_optimizations_raw_buffer.ll
@@ -165,6 +165,7 @@ define amdgpu_kernel void @add_i32_constant(ptr addrspace(1) %out, ptr addrspace
; GFX11W64-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11W64-NEXT: v_mbcnt_hi_u32_b32 v0, s3, v0
; GFX11W64-NEXT: v_cmpx_eq_u32_e32 0, v0
+; GFX11W64-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11W64-NEXT: s_cbranch_execz .LBB0_2
; GFX11W64-NEXT: ; %bb.1:
; GFX11W64-NEXT: s_load_b128 s[8:11], s[4:5], 0x34
@@ -192,7 +193,7 @@ define amdgpu_kernel void @add_i32_constant(ptr addrspace(1) %out, ptr addrspace
; GFX11W32-NEXT: s_mov_b32 s0, exec_lo
; GFX11W32-NEXT: v_mbcnt_lo_u32_b32 v0, s1, 0
; GFX11W32-NEXT: ; implicit-def: $vgpr1
-; GFX11W32-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX11W32-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11W32-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX11W32-NEXT: s_cbranch_execz .LBB0_2
; GFX11W32-NEXT: ; %bb.1:
@@ -224,6 +225,7 @@ define amdgpu_kernel void @add_i32_constant(ptr addrspace(1) %out, ptr addrspace
; GFX12W64-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX12W64-NEXT: v_mbcnt_hi_u32_b32 v0, s3, v0
; GFX12W64-NEXT: v_cmpx_eq_u32_e32 0, v0
+; GFX12W64-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12W64-NEXT: s_cbranch_execz .LBB0_2
; GFX12W64-NEXT: ; %bb.1:
; GFX12W64-NEXT: s_load_b128 s[8:11], s[4:5], 0x34
@@ -253,7 +255,7 @@ define amdgpu_kernel void @add_i32_constant(ptr addrspace(1) %out, ptr addrspace
; GFX12W32-NEXT: s_mov_b32 s0, exec_lo
; GFX12W32-NEXT: v_mbcnt_lo_u32_b32 v0, s1, 0
; GFX12W32-NEXT: ; implicit-def: $vgpr1
-; GFX12W32-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX12W32-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX12W32-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX12W32-NEXT: s_cbranch_execz .LBB0_2
; GFX12W32-NEXT: ; %bb.1:
@@ -286,6 +288,7 @@ define amdgpu_kernel void @add_i32_constant(ptr addrspace(1) %out, ptr addrspace
; GFX13W64-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX13W64-NEXT: v_mbcnt_hi_u32_b32 v0, s3, v0
; GFX13W64-NEXT: v_cmpx_eq_u32_e32 0, v0
+; GFX13W64-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX13W64-NEXT: s_cbranch_execz .LBB0_2
; GFX13W64-NEXT: ; %bb.1:
; GFX13W64-NEXT: s_load_b128 s[8:11], s[4:5], 0x34 nv
@@ -313,7 +316,7 @@ define amdgpu_kernel void @add_i32_constant(ptr addrspace(1) %out, ptr addrspace
; GFX13W32-NEXT: s_mov_b32 s0, exec_lo
; GFX13W32-NEXT: v_mbcnt_lo_u32_b32 v0, s1, 0
; GFX13W32-NEXT: ; implicit-def: $vgpr1
-; GFX13W32-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX13W32-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX13W32-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX13W32-NEXT: s_cbranch_execz .LBB0_2
; GFX13W32-NEXT: ; %bb.1:
@@ -499,6 +502,7 @@ define amdgpu_kernel void @add_i32_uniform(ptr addrspace(1) %out, ptr addrspace(
; GFX11W64-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11W64-NEXT: v_mbcnt_hi_u32_b32 v0, s3, v0
; GFX11W64-NEXT: v_cmpx_eq_u32_e32 0, v0
+; GFX11W64-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11W64-NEXT: s_cbranch_execz .LBB1_2
; GFX11W64-NEXT: ; %bb.1:
; GFX11W64-NEXT: s_load_b128 s[8:11], s[4:5], 0x34
@@ -527,7 +531,7 @@ define amdgpu_kernel void @add_i32_uniform(ptr addrspace(1) %out, ptr addrspace(
; GFX11W32-NEXT: s_mov_b32 s1, exec_lo
; GFX11W32-NEXT: v_mbcnt_lo_u32_b32 v0, s2, 0
; GFX11W32-NEXT: ; implicit-def: $vgpr1
-; GFX11W32-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX11W32-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11W32-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX11W32-NEXT: s_cbranch_execz .LBB1_2
; GFX11W32-NEXT: ; %bb.1:
@@ -560,6 +564,7 @@ define amdgpu_kernel void @add_i32_uniform(ptr addrspace(1) %out, ptr addrspace(
; GFX12W64-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX12W64-NEXT: v_mbcnt_hi_u32_b32 v0, s3, v0
; GFX12W64-NEXT: v_cmpx_eq_u32_e32 0, v0
+; GFX12W64-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12W64-NEXT: s_cbranch_execz .LBB1_2
; GFX12W64-NEXT: ; %bb.1:
; GFX12W64-NEXT: s_load_b128 s[8:11], s[4:5], 0x34
@@ -590,7 +595,7 @@ define amdgpu_kernel void @add_i32_uniform(ptr addrspace(1) %out, ptr addrspace(
; GFX12W32-NEXT: s_mov_b32 s1, exec_lo
; GFX12W32-NEXT: v_mbcnt_lo_u32_b32 v0, s2, 0
; GFX12W32-NEXT: ; implicit-def: $vgpr1
-; GFX12W32-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX12W32-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX12W32-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX12W32-NEXT: s_cbranch_execz .LBB1_2
; GFX12W32-NEXT: ; %bb.1:
@@ -624,6 +629,7 @@ define amdgpu_kernel void @add_i32_uniform(ptr addrspace(1) %out, ptr addrspace(
; GFX13W64-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX13W64-NEXT: v_mbcnt_hi_u32_b32 v0, s3, v0
; GFX13W64-NEXT: v_cmpx_eq_u32_e32 0, v0
+; GFX13W64-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX13W64-NEXT: s_cbranch_execz .LBB1_2
; GFX13W64-NEXT: ; %bb.1:
; GFX13W64-NEXT: s_load_b128 s[8:11], s[4:5], 0x34 nv
@@ -652,7 +658,7 @@ define amdgpu_kernel void @add_i32_uniform(ptr addrspace(1) %out, ptr addrspace(
; GFX13W32-NEXT: s_mov_b32 s1, exec_lo
; GFX13W32-NEXT: v_mbcnt_lo_u32_b32 v0, s2, 0
; GFX13W32-NEXT: ; implicit-def: $vgpr1
-; GFX13W32-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX13W32-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX13W32-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX13W32-NEXT: s_cbranch_execz .LBB1_2
; GFX13W32-NEXT: ; %bb.1:
@@ -902,8 +908,8 @@ define amdgpu_kernel void @add_i32_varying_vdata(ptr addrspace(1) %out, ptr addr
; GFX11W64-NEXT: v_mbcnt_hi_u32_b32 v1, exec_hi, v1
; GFX11W64-NEXT: v_cmp_eq_u32_e32 vcc, 0, v1
; GFX11W64-NEXT: ; implicit-def: $vgpr1
+; GFX11W64-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11W64-NEXT: s_and_saveexec_b64 s[0:1], vcc
-; GFX11W64-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11W64-NEXT: s_xor_b64 s[0:1], exec, s[0:1]
; GFX11W64-NEXT: s_cbranch_execz .LBB2_4
; GFX11W64-NEXT: ; %bb.3:
@@ -985,8 +991,8 @@ define amdgpu_kernel void @add_i32_varying_vdata(ptr addrspace(1) %out, ptr addr
; GFX12W64-NEXT: v_mbcnt_hi_u32_b32 v1, exec_hi, v1
; GFX12W64-NEXT: v_cmp_eq_u32_e32 vcc, 0, v1
; GFX12W64-NEXT: ; implicit-def: $vgpr1
+; GFX12W64-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12W64-NEXT: s_and_saveexec_b64 s[0:1], vcc
-; GFX12W64-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12W64-NEXT: s_xor_b64 s[0:1], exec, s[0:1]
; GFX12W64-NEXT: s_cbranch_execz .LBB2_4
; GFX12W64-NEXT: ; %bb.3:
@@ -1074,8 +1080,8 @@ define amdgpu_kernel void @add_i32_varying_vdata(ptr addrspace(1) %out, ptr addr
; GFX13W64-NEXT: v_mbcnt_hi_u32_b32 v1, exec_hi, v1
; GFX13W64-NEXT: v_cmp_eq_u32_e32 vcc, 0, v1
; GFX13W64-NEXT: ; implicit-def: $vgpr1
+; GFX13W64-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX13W64-NEXT: s_and_saveexec_b64 s[0:1], vcc
-; GFX13W64-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX13W64-NEXT: s_xor_b64 s[0:1], exec, s[0:1]
; GFX13W64-NEXT: s_cbranch_execz .LBB2_4
; GFX13W64-NEXT: ; %bb.3:
@@ -1419,6 +1425,7 @@ define amdgpu_kernel void @sub_i32_constant(ptr addrspace(1) %out, ptr addrspace
; GFX11W64-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11W64-NEXT: v_mbcnt_hi_u32_b32 v0, s3, v0
; GFX11W64-NEXT: v_cmpx_eq_u32_e32 0, v0
+; GFX11W64-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11W64-NEXT: s_cbranch_execz .LBB4_2
; GFX11W64-NEXT: ; %bb.1:
; GFX11W64-NEXT: s_load_b128 s[8:11], s[4:5], 0x34
@@ -1447,7 +1454,7 @@ define amdgpu_kernel void @sub_i32_constant(ptr addrspace(1) %out, ptr addrspace
; GFX11W32-NEXT: s_mov_b32 s0, exec_lo
; GFX11W32-NEXT: v_mbcnt_lo_u32_b32 v0, s1, 0
; GFX11W32-NEXT: ; implicit-def: $vgpr1
-; GFX11W32-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX11W32-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11W32-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX11W32-NEXT: s_cbranch_execz .LBB4_2
; GFX11W32-NEXT: ; %bb.1:
@@ -1480,6 +1487,7 @@ define amdgpu_kernel void @sub_i32_constant(ptr addrspace(1) %out, ptr addrspace
; GFX12W64-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX12W64-NEXT: v_mbcnt_hi_u32_b32 v0, s3, v0
; GFX12W64-NEXT: v_cmpx_eq_u32_e32 0, v0
+; GFX12W64-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12W64-NEXT: s_cbranch_execz .LBB4_2
; GFX12W64-NEXT: ; %bb.1:
; GFX12W64-NEXT: s_load_b128 s[8:11], s[4:5], 0x34
@@ -1510,7 +1518,7 @@ define amdgpu_kernel void @sub_i32_constant(ptr addrspace(1) %out, ptr addrspace
; GFX12W32-NEXT: s_mov_b32 s0, exec_lo
; GFX12W32-NEXT: v_mbcnt_lo_u32_b32 v0, s1, 0
; GFX12W32-NEXT: ; implicit-def: $vgpr1
-; GFX12W32-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX12W32-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX12W32-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX12W32-NEXT: s_cbranch_execz .LBB4_2
; GFX12W32-NEXT: ; %bb.1:
@@ -1544,6 +1552,7 @@ define amdgpu_kernel void @sub_i32_constant(ptr addrspace(1) %out, ptr addrspace
; GFX13W64-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX13W64-NEXT: v_mbcnt_hi_u32_b32 v0, s3, v0
; GFX13W64-NEXT: v_cmpx_eq_u32_e32 0, v0
+; GFX13W64-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX13W64-NEXT: s_cbranch_execz .LBB4_2
; GFX13W64-NEXT: ; %bb.1:
; GFX13W64-NEXT: s_load_b128 s[8:11], s[4:5], 0x34 nv
@@ -1572,7 +1581,7 @@ define amdgpu_kernel void @sub_i32_constant(ptr addrspace(1) %out, ptr addrspace
; GFX13W32-NEXT: s_mov_b32 s0, exec_lo
; GFX13W32-NEXT: v_mbcnt_lo_u32_b32 v0, s1, 0
; GFX13W32-NEXT: ; implicit-def: $vgpr1
-; GFX13W32-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX13W32-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX13W32-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX13W32-NEXT: s_cbranch_execz .LBB4_2
; GFX13W32-NEXT: ; %bb.1:
@@ -1759,6 +1768,7 @@ define amdgpu_kernel void @sub_i32_uniform(ptr addrspace(1) %out, ptr addrspace(
; GFX11W64-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11W64-NEXT: v_mbcnt_hi_u32_b32 v0, s3, v0
; GFX11W64-NEXT: v_cmpx_eq_u32_e32 0, v0
+; GFX11W64-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11W64-NEXT: s_cbranch_execz .LBB5_2
; GFX11W64-NEXT: ; %bb.1:
; GFX11W64-NEXT: s_load_b128 s[8:11], s[4:5], 0x34
@@ -1788,7 +1798,7 @@ define amdgpu_kernel void @sub_i32_uniform(ptr addrspace(1) %out, ptr addrspace(
; GFX11W32-NEXT: s_mov_b32 s1, exec_lo
; GFX11W32-NEXT: v_mbcnt_lo_u32_b32 v0, s2, 0
; GFX11W32-NEXT: ; implicit-def: $vgpr1
-; GFX11W32-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX11W32-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11W32-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX11W32-NEXT: s_cbranch_execz .LBB5_2
; GFX11W32-NEXT: ; %bb.1:
@@ -1822,6 +1832,7 @@ define amdgpu_kernel void @sub_i32_uniform(ptr addrspace(1) %out, ptr addrspace(
; GFX12W64-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX12W64-NEXT: v_mbcnt_hi_u32_b32 v0, s3, v0
; GFX12W64-NEXT: v_cmpx_eq_u32_e32 0, v0
+; GFX12W64-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12W64-NEXT: s_cbranch_execz .LBB5_2
; GFX12W64-NEXT: ; %bb.1:
; GFX12W64-NEXT: s_load_b128 s[8:11], s[4:5], 0x34
@@ -1853,7 +1864,7 @@ define amdgpu_kernel void @sub_i32_uniform(ptr addrspace(1) %out, ptr addrspace(
; GFX12W32-NEXT: s_mov_b32 s1, exec_lo
; GFX12W32-NEXT: v_mbcnt_lo_u32_b32 v0, s2, 0
; GFX12W32-NEXT: ; implicit-def: $vgpr1
-; GFX12W32-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX12W32-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX12W32-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX12W32-NEXT: s_cbranch_execz .LBB5_2
; GFX12W32-NEXT: ; %bb.1:
@@ -1889,6 +1900,7 @@ define amdgpu_kernel void @sub_i32_uniform(ptr addrspace(1) %out, ptr addrspace(
; GFX13W64-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX13W64-NEXT: v_mbcnt_hi_u32_b32 v0, s3, v0
; GFX13W64-NEXT: v_cmpx_eq_u32_e32 0, v0
+; GFX13W64-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX13W64-NEXT: s_cbranch_execz .LBB5_2
; GFX13W64-NEXT: ; %bb.1:
; GFX13W64-NEXT: s_load_b128 s[8:11], s[4:5], 0x34 nv
@@ -1918,7 +1930,7 @@ define amdgpu_kernel void @sub_i32_uniform(ptr addrspace(1) %out, ptr addrspace(
; GFX13W32-NEXT: s_mov_b32 s1, exec_lo
; GFX13W32-NEXT: v_mbcnt_lo_u32_b32 v0, s2, 0
; GFX13W32-NEXT: ; implicit-def: $vgpr1
-; GFX13W32-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX13W32-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX13W32-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX13W32-NEXT: s_cbranch_execz .LBB5_2
; GFX13W32-NEXT: ; %bb.1:
@@ -2168,8 +2180,8 @@ define amdgpu_kernel void @sub_i32_varying_vdata(ptr addrspace(1) %out, ptr addr
; GFX11W64-NEXT: v_mbcnt_hi_u32_b32 v1, exec_hi, v1
; GFX11W64-NEXT: v_cmp_eq_u32_e32 vcc, 0, v1
; GFX11W64-NEXT: ; implicit-def: $vgpr1
+; GFX11W64-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11W64-NEXT: s_and_saveexec_b64 s[0:1], vcc
-; GFX11W64-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11W64-NEXT: s_xor_b64 s[0:1], exec, s[0:1]
; GFX11W64-NEXT: s_cbranch_execz .LBB6_4
; GFX11W64-NEXT: ; %bb.3:
@@ -2252,8 +2264,8 @@ define amdgpu_kernel void @sub_i32_varying_vdata(ptr addrspace(1) %out, ptr addr
; GFX12W64-NEXT: v_mbcnt_hi_u32_b32 v1, exec_hi, v1
; GFX12W64-NEXT: v_cmp_eq_u32_e32 vcc, 0, v1
; GFX12W64-NEXT: ; implicit-def: $vgpr1
+; GFX12W64-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12W64-NEXT: s_and_saveexec_b64 s[0:1], vcc
-; GFX12W64-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12W64-NEXT: s_xor_b64 s[0:1], exec, s[0:1]
; GFX12W64-NEXT: s_cbranch_execz .LBB6_4
; GFX12W64-NEXT: ; %bb.3:
@@ -2342,8 +2354,8 @@ define amdgpu_kernel void @sub_i32_varying_vdata(ptr addrspace(1) %out, ptr addr
; GFX13W64-NEXT: v_mbcnt_hi_u32_b32 v1, exec_hi, v1
; GFX13W64-NEXT: v_cmp_eq_u32_e32 vcc, 0, v1
; GFX13W64-NEXT: ; implicit-def: $vgpr1
+; GFX13W64-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX13W64-NEXT: s_and_saveexec_b64 s[0:1], vcc
-; GFX13W64-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX13W64-NEXT: s_xor_b64 s[0:1], exec, s[0:1]
; GFX13W64-NEXT: s_cbranch_execz .LBB6_4
; GFX13W64-NEXT: ; %bb.3:
diff --git a/llvm/test/CodeGen/AMDGPU/atomic_optimizations_struct_buffer.ll b/llvm/test/CodeGen/AMDGPU/atomic_optimizations_struct_buffer.ll
index c140252b6003de..3dfc81503f2f02 100644
--- a/llvm/test/CodeGen/AMDGPU/atomic_optimizations_struct_buffer.ll
+++ b/llvm/test/CodeGen/AMDGPU/atomic_optimizations_struct_buffer.ll
@@ -170,6 +170,7 @@ define amdgpu_kernel void @add_i32_constant(ptr addrspace(1) %out, ptr addrspace
; GFX11W64-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11W64-NEXT: v_mbcnt_hi_u32_b32 v0, s3, v0
; GFX11W64-NEXT: v_cmpx_eq_u32_e32 0, v0
+; GFX11W64-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11W64-NEXT: s_cbranch_execz .LBB0_2
; GFX11W64-NEXT: ; %bb.1:
; GFX11W64-NEXT: s_load_b128 s[8:11], s[4:5], 0x34
@@ -198,7 +199,7 @@ define amdgpu_kernel void @add_i32_constant(ptr addrspace(1) %out, ptr addrspace
; GFX11W32-NEXT: s_mov_b32 s0, exec_lo
; GFX11W32-NEXT: v_mbcnt_lo_u32_b32 v0, s1, 0
; GFX11W32-NEXT: ; implicit-def: $vgpr1
-; GFX11W32-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX11W32-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11W32-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX11W32-NEXT: s_cbranch_execz .LBB0_2
; GFX11W32-NEXT: ; %bb.1:
@@ -231,6 +232,7 @@ define amdgpu_kernel void @add_i32_constant(ptr addrspace(1) %out, ptr addrspace
; GFX12W64-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX12W64-NEXT: v_mbcnt_hi_u32_b32 v0, s3, v0
; GFX12W64-NEXT: v_cmpx_eq_u32_e32 0, v0
+; GFX12W64-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12W64-NEXT: s_cbranch_execz .LBB0_2
; GFX12W64-NEXT: ; %bb.1:
; GFX12W64-NEXT: s_load_b128 s[8:11], s[4:5], 0x34
@@ -261,7 +263,7 @@ define amdgpu_kernel void @add_i32_constant(ptr addrspace(1) %out, ptr addrspace
; GFX12W32-NEXT: s_mov_b32 s0, exec_lo
; GFX12W32-NEXT: v_mbcnt_lo_u32_b32 v0, s1, 0
; GFX12W32-NEXT: ; implicit-def: $vgpr1
-; GFX12W32-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX12W32-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX12W32-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX12W32-NEXT: s_cbranch_execz .LBB0_2
; GFX12W32-NEXT: ; %bb.1:
@@ -294,6 +296,7 @@ define amdgpu_kernel void @add_i32_constant(ptr addrspace(1) %out, ptr addrspace
; GFX13W64-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX13W64-NEXT: v_mbcnt_hi_u32_b32 v0, s3, v0
; GFX13W64-NEXT: v_cmpx_eq_u32_e32 0, v0
+; GFX13W64-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX13W64-NEXT: s_cbranch_execz .LBB0_2
; GFX13W64-NEXT: ; %bb.1:
; GFX13W64-NEXT: s_load_b128 s[8:11], s[4:5], 0x34 nv
@@ -322,7 +325,7 @@ define amdgpu_kernel void @add_i32_constant(ptr addrspace(1) %out, ptr addrspace
; GFX13W32-NEXT: s_mov_b32 s0, exec_lo
; GFX13W32-NEXT: v_mbcnt_lo_u32_b32 v0, s1, 0
; GFX13W32-NEXT: ; implicit-def: $vgpr1
-; GFX13W32-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX13W32-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX13W32-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX13W32-NEXT: s_cbranch_execz .LBB0_2
; GFX13W32-NEXT: ; %bb.1:
@@ -513,6 +516,7 @@ define amdgpu_kernel void @add_i32_uniform(ptr addrspace(1) %out, ptr addrspace(
; GFX11W64-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11W64-NEXT: v_mbcnt_hi_u32_b32 v0, s3, v0
; GFX11W64-NEXT: v_cmpx_eq_u32_e32 0, v0
+; GFX11W64-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11W64-NEXT: s_cbranch_execz .LBB1_2
; GFX11W64-NEXT: ; %bb.1:
; GFX11W64-NEXT: s_load_b128 s[8:11], s[4:5], 0x34
@@ -542,7 +546,7 @@ define amdgpu_kernel void @add_i32_uniform(ptr addrspace(1) %out, ptr addrspace(
; GFX11W32-NEXT: s_mov_b32 s1, exec_lo
; GFX11W32-NEXT: v_mbcnt_lo_u32_b32 v0, s2, 0
; GFX11W32-NEXT: ; implicit-def: $vgpr1
-; GFX11W32-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX11W32-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11W32-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX11W32-NEXT: s_cbranch_execz .LBB1_2
; GFX11W32-NEXT: ; %bb.1:
@@ -576,6 +580,7 @@ define amdgpu_kernel void @add_i32_uniform(ptr addrspace(1) %out, ptr addrspace(
; GFX12W64-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX12W64-NEXT: v_mbcnt_hi_u32_b32 v0, s3, v0
; GFX12W64-NEXT: v_cmpx_eq_u32_e32 0, v0
+; GFX12W64-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12W64-NEXT: s_cbranch_execz .LBB1_2
; GFX12W64-NEXT: ; %bb.1:
; GFX12W64-NEXT: s_load_b128 s[8:11], s[4:5], 0x34
@@ -607,7 +612,7 @@ define amdgpu_kernel void @add_i32_uniform(ptr addrspace(1) %out, ptr addrspace(
; GFX12W32-NEXT: s_mov_b32 s1, exec_lo
; GFX12W32-NEXT: v_mbcnt_lo_u32_b32 v0, s2, 0
; GFX12W32-NEXT: ; implicit-def: $vgpr1
-; GFX12W32-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX12W32-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX12W32-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX12W32-NEXT: s_cbranch_execz .LBB1_2
; GFX12W32-NEXT: ; %bb.1:
@@ -641,6 +646,7 @@ define amdgpu_kernel void @add_i32_uniform(ptr addrspace(1) %out, ptr addrspace(
; GFX13W64-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX13W64-NEXT: v_mbcnt_hi_u32_b32 v0, s3, v0
; GFX13W64-NEXT: v_cmpx_eq_u32_e32 0, v0
+; GFX13W64-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX13W64-NEXT: s_cbranch_execz .LBB1_2
; GFX13W64-NEXT: ; %bb.1:
; GFX13W64-NEXT: s_load_b128 s[8:11], s[4:5], 0x34 nv
@@ -670,7 +676,7 @@ define amdgpu_kernel void @add_i32_uniform(ptr addrspace(1) %out, ptr addrspace(
; GFX13W32-NEXT: s_mov_b32 s1, exec_lo
; GFX13W32-NEXT: v_mbcnt_lo_u32_b32 v0, s2, 0
; GFX13W32-NEXT: ; implicit-def: $vgpr1
-; GFX13W32-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX13W32-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX13W32-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX13W32-NEXT: s_cbranch_execz .LBB1_2
; GFX13W32-NEXT: ; %bb.1:
@@ -925,8 +931,8 @@ define amdgpu_kernel void @add_i32_varying_vdata(ptr addrspace(1) %out, ptr addr
; GFX11W64-NEXT: v_mbcnt_hi_u32_b32 v1, exec_hi, v1
; GFX11W64-NEXT: v_cmp_eq_u32_e32 vcc, 0, v1
; GFX11W64-NEXT: ; implicit-def: $vgpr1
+; GFX11W64-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11W64-NEXT: s_and_saveexec_b64 s[0:1], vcc
-; GFX11W64-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11W64-NEXT: s_xor_b64 s[0:1], exec, s[0:1]
; GFX11W64-NEXT: s_cbranch_execz .LBB2_4
; GFX11W64-NEXT: ; %bb.3:
@@ -1009,8 +1015,8 @@ define amdgpu_kernel void @add_i32_varying_vdata(ptr addrspace(1) %out, ptr addr
; GFX12W64-NEXT: v_mbcnt_hi_u32_b32 v1, exec_hi, v1
; GFX12W64-NEXT: v_cmp_eq_u32_e32 vcc, 0, v1
; GFX12W64-NEXT: ; implicit-def: $vgpr1
+; GFX12W64-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12W64-NEXT: s_and_saveexec_b64 s[0:1], vcc
-; GFX12W64-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12W64-NEXT: s_xor_b64 s[0:1], exec, s[0:1]
; GFX12W64-NEXT: s_cbranch_execz .LBB2_4
; GFX12W64-NEXT: ; %bb.3:
@@ -1099,8 +1105,8 @@ define amdgpu_kernel void @add_i32_varying_vdata(ptr addrspace(1) %out, ptr addr
; GFX13W64-NEXT: v_mbcnt_hi_u32_b32 v1, exec_hi, v1
; GFX13W64-NEXT: v_cmp_eq_u32_e32 vcc, 0, v1
; GFX13W64-NEXT: ; implicit-def: $vgpr1
+; GFX13W64-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX13W64-NEXT: s_and_saveexec_b64 s[0:1], vcc
-; GFX13W64-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX13W64-NEXT: s_xor_b64 s[0:1], exec, s[0:1]
; GFX13W64-NEXT: s_cbranch_execz .LBB2_4
; GFX13W64-NEXT: ; %bb.3:
@@ -1605,6 +1611,7 @@ define amdgpu_kernel void @sub_i32_constant(ptr addrspace(1) %out, ptr addrspace
; GFX11W64-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11W64-NEXT: v_mbcnt_hi_u32_b32 v0, s3, v0
; GFX11W64-NEXT: v_cmpx_eq_u32_e32 0, v0
+; GFX11W64-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11W64-NEXT: s_cbranch_execz .LBB5_2
; GFX11W64-NEXT: ; %bb.1:
; GFX11W64-NEXT: s_load_b128 s[8:11], s[4:5], 0x34
@@ -1634,7 +1641,7 @@ define amdgpu_kernel void @sub_i32_constant(ptr addrspace(1) %out, ptr addrspace
; GFX11W32-NEXT: s_mov_b32 s0, exec_lo
; GFX11W32-NEXT: v_mbcnt_lo_u32_b32 v0, s1, 0
; GFX11W32-NEXT: ; implicit-def: $vgpr1
-; GFX11W32-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX11W32-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11W32-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX11W32-NEXT: s_cbranch_execz .LBB5_2
; GFX11W32-NEXT: ; %bb.1:
@@ -1668,6 +1675,7 @@ define amdgpu_kernel void @sub_i32_constant(ptr addrspace(1) %out, ptr addrspace
; GFX12W64-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX12W64-NEXT: v_mbcnt_hi_u32_b32 v0, s3, v0
; GFX12W64-NEXT: v_cmpx_eq_u32_e32 0, v0
+; GFX12W64-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12W64-NEXT: s_cbranch_execz .LBB5_2
; GFX12W64-NEXT: ; %bb.1:
; GFX12W64-NEXT: s_load_b128 s[8:11], s[4:5], 0x34
@@ -1699,7 +1707,7 @@ define amdgpu_kernel void @sub_i32_constant(ptr addrspace(1) %out, ptr addrspace
; GFX12W32-NEXT: s_mov_b32 s0, exec_lo
; GFX12W32-NEXT: v_mbcnt_lo_u32_b32 v0, s1, 0
; GFX12W32-NEXT: ; implicit-def: $vgpr1
-; GFX12W32-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX12W32-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX12W32-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX12W32-NEXT: s_cbranch_execz .LBB5_2
; GFX12W32-NEXT: ; %bb.1:
@@ -1733,6 +1741,7 @@ define amdgpu_kernel void @sub_i32_constant(ptr addrspace(1) %out, ptr addrspace
; GFX13W64-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX13W64-NEXT: v_mbcnt_hi_u32_b32 v0, s3, v0
; GFX13W64-NEXT: v_cmpx_eq_u32_e32 0, v0
+; GFX13W64-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX13W64-NEXT: s_cbranch_execz .LBB5_2
; GFX13W64-NEXT: ; %bb.1:
; GFX13W64-NEXT: s_load_b128 s[8:11], s[4:5], 0x34 nv
@@ -1762,7 +1771,7 @@ define amdgpu_kernel void @sub_i32_constant(ptr addrspace(1) %out, ptr addrspace
; GFX13W32-NEXT: s_mov_b32 s0, exec_lo
; GFX13W32-NEXT: v_mbcnt_lo_u32_b32 v0, s1, 0
; GFX13W32-NEXT: ; implicit-def: $vgpr1
-; GFX13W32-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX13W32-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX13W32-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX13W32-NEXT: s_cbranch_execz .LBB5_2
; GFX13W32-NEXT: ; %bb.1:
@@ -1954,6 +1963,7 @@ define amdgpu_kernel void @sub_i32_uniform(ptr addrspace(1) %out, ptr addrspace(
; GFX11W64-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11W64-NEXT: v_mbcnt_hi_u32_b32 v0, s3, v0
; GFX11W64-NEXT: v_cmpx_eq_u32_e32 0, v0
+; GFX11W64-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11W64-NEXT: s_cbranch_execz .LBB6_2
; GFX11W64-NEXT: ; %bb.1:
; GFX11W64-NEXT: s_load_b128 s[8:11], s[4:5], 0x34
@@ -1984,7 +1994,7 @@ define amdgpu_kernel void @sub_i32_uniform(ptr addrspace(1) %out, ptr addrspace(
; GFX11W32-NEXT: s_mov_b32 s1, exec_lo
; GFX11W32-NEXT: v_mbcnt_lo_u32_b32 v0, s2, 0
; GFX11W32-NEXT: ; implicit-def: $vgpr1
-; GFX11W32-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX11W32-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11W32-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX11W32-NEXT: s_cbranch_execz .LBB6_2
; GFX11W32-NEXT: ; %bb.1:
@@ -2019,6 +2029,7 @@ define amdgpu_kernel void @sub_i32_uniform(ptr addrspace(1) %out, ptr addrspace(
; GFX12W64-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX12W64-NEXT: v_mbcnt_hi_u32_b32 v0, s3, v0
; GFX12W64-NEXT: v_cmpx_eq_u32_e32 0, v0
+; GFX12W64-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12W64-NEXT: s_cbranch_execz .LBB6_2
; GFX12W64-NEXT: ; %bb.1:
; GFX12W64-NEXT: s_load_b128 s[8:11], s[4:5], 0x34
@@ -2051,7 +2062,7 @@ define amdgpu_kernel void @sub_i32_uniform(ptr addrspace(1) %out, ptr addrspace(
; GFX12W32-NEXT: s_mov_b32 s1, exec_lo
; GFX12W32-NEXT: v_mbcnt_lo_u32_b32 v0, s2, 0
; GFX12W32-NEXT: ; implicit-def: $vgpr1
-; GFX12W32-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX12W32-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX12W32-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX12W32-NEXT: s_cbranch_execz .LBB6_2
; GFX12W32-NEXT: ; %bb.1:
@@ -2087,6 +2098,7 @@ define amdgpu_kernel void @sub_i32_uniform(ptr addrspace(1) %out, ptr addrspace(
; GFX13W64-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX13W64-NEXT: v_mbcnt_hi_u32_b32 v0, s3, v0
; GFX13W64-NEXT: v_cmpx_eq_u32_e32 0, v0
+; GFX13W64-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX13W64-NEXT: s_cbranch_execz .LBB6_2
; GFX13W64-NEXT: ; %bb.1:
; GFX13W64-NEXT: s_load_b128 s[8:11], s[4:5], 0x34 nv
@@ -2117,7 +2129,7 @@ define amdgpu_kernel void @sub_i32_uniform(ptr addrspace(1) %out, ptr addrspace(
; GFX13W32-NEXT: s_mov_b32 s1, exec_lo
; GFX13W32-NEXT: v_mbcnt_lo_u32_b32 v0, s2, 0
; GFX13W32-NEXT: ; implicit-def: $vgpr1
-; GFX13W32-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX13W32-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX13W32-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX13W32-NEXT: s_cbranch_execz .LBB6_2
; GFX13W32-NEXT: ; %bb.1:
@@ -2372,8 +2384,8 @@ define amdgpu_kernel void @sub_i32_varying_vdata(ptr addrspace(1) %out, ptr addr
; GFX11W64-NEXT: v_mbcnt_hi_u32_b32 v1, exec_hi, v1
; GFX11W64-NEXT: v_cmp_eq_u32_e32 vcc, 0, v1
; GFX11W64-NEXT: ; implicit-def: $vgpr1
+; GFX11W64-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11W64-NEXT: s_and_saveexec_b64 s[0:1], vcc
-; GFX11W64-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11W64-NEXT: s_xor_b64 s[0:1], exec, s[0:1]
; GFX11W64-NEXT: s_cbranch_execz .LBB7_4
; GFX11W64-NEXT: ; %bb.3:
@@ -2457,8 +2469,8 @@ define amdgpu_kernel void @sub_i32_varying_vdata(ptr addrspace(1) %out, ptr addr
; GFX12W64-NEXT: v_mbcnt_hi_u32_b32 v1, exec_hi, v1
; GFX12W64-NEXT: v_cmp_eq_u32_e32 vcc, 0, v1
; GFX12W64-NEXT: ; implicit-def: $vgpr1
+; GFX12W64-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12W64-NEXT: s_and_saveexec_b64 s[0:1], vcc
-; GFX12W64-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12W64-NEXT: s_xor_b64 s[0:1], exec, s[0:1]
; GFX12W64-NEXT: s_cbranch_execz .LBB7_4
; GFX12W64-NEXT: ; %bb.3:
@@ -2548,8 +2560,8 @@ define amdgpu_kernel void @sub_i32_varying_vdata(ptr addrspace(1) %out, ptr addr
; GFX13W64-NEXT: v_mbcnt_hi_u32_b32 v1, exec_hi, v1
; GFX13W64-NEXT: v_cmp_eq_u32_e32 vcc, 0, v1
; GFX13W64-NEXT: ; implicit-def: $vgpr1
+; GFX13W64-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX13W64-NEXT: s_and_saveexec_b64 s[0:1], vcc
-; GFX13W64-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX13W64-NEXT: s_xor_b64 s[0:1], exec, s[0:1]
; GFX13W64-NEXT: s_cbranch_execz .LBB7_4
; GFX13W64-NEXT: ; %bb.3:
diff --git a/llvm/test/CodeGen/AMDGPU/atomicrmw-expand.ll b/llvm/test/CodeGen/AMDGPU/atomicrmw-expand.ll
index 3dadcddd468760..e6339b843d154c 100644
--- a/llvm/test/CodeGen/AMDGPU/atomicrmw-expand.ll
+++ b/llvm/test/CodeGen/AMDGPU/atomicrmw-expand.ll
@@ -422,11 +422,12 @@ define float @no_unsafe(ptr %addr, float %val) {
; GFX1100-NEXT: buffer_gl0_inv
; GFX1100-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX1100-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX1100-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1100-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1100-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX1100-NEXT: s_cbranch_execnz .LBB3_1
; GFX1100-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX1100-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX1100-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1100-NEXT: v_mov_b32_e32 v0, v3
; GFX1100-NEXT: s_setpc_b64 s[30:31]
;
diff --git a/llvm/test/CodeGen/AMDGPU/atomicrmw_usub_sat.ll b/llvm/test/CodeGen/AMDGPU/atomicrmw_usub_sat.ll
index d1a7aa6d8d29d7..3d64d4cc926ea5 100644
--- a/llvm/test/CodeGen/AMDGPU/atomicrmw_usub_sat.ll
+++ b/llvm/test/CodeGen/AMDGPU/atomicrmw_usub_sat.ll
@@ -162,7 +162,6 @@ define i32 @global_atomic_usub_sat_offset(ptr addrspace(1) %ptr, i32 %data) {
; GFX11-GISEL: ; %bb.0:
; GFX11-GISEL-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, 0x1000, v0
-; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-GISEL-NEXT: s_waitcnt_vscnt null, 0x0
; GFX11-GISEL-NEXT: global_atomic_csub_u32 v0, v[0:1], v2, off glc
@@ -226,7 +225,6 @@ define i32 @global_atomic_usub_sat_offset(ptr addrspace(1) %ptr, i32 %data) {
; GFX11-SDAG: ; %bb.0:
; GFX11-SDAG-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-SDAG-NEXT: v_add_co_u32 v0, vcc_lo, 0x1000, v0
-; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-SDAG-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-SDAG-NEXT: s_waitcnt_vscnt null, 0x0
; GFX11-SDAG-NEXT: global_atomic_csub_u32 v0, v[0:1], v2, off glc
@@ -404,7 +402,6 @@ define void @global_atomic_usub_sat_offset_nortn(ptr addrspace(1) %ptr, i32 %dat
; GFX11-GISEL: ; %bb.0:
; GFX11-GISEL-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, 0x1000, v0
-; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-GISEL-NEXT: s_waitcnt_vscnt null, 0x0
; GFX11-GISEL-NEXT: global_atomic_csub_u32 v0, v[0:1], v2, off glc
@@ -468,7 +465,6 @@ define void @global_atomic_usub_sat_offset_nortn(ptr addrspace(1) %ptr, i32 %dat
; GFX11-SDAG: ; %bb.0:
; GFX11-SDAG-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-SDAG-NEXT: v_add_co_u32 v0, vcc_lo, 0x1000, v0
-; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-SDAG-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-SDAG-NEXT: s_waitcnt_vscnt null, 0x0
; GFX11-SDAG-NEXT: global_atomic_csub_u32 v0, v[0:1], v2, off glc
@@ -853,11 +849,12 @@ define i16 @global_atomic_usub_sat_16(ptr addrspace(1) %ptr, i16 %data) {
; GFX11-GISEL-NEXT: buffer_gl0_inv
; GFX11-GISEL-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-GISEL-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-GISEL-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-GISEL-NEXT: s_cbranch_execnz .LBB6_1
; GFX11-GISEL-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-GISEL-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-GISEL-NEXT: v_mov_b32_e32 v0, v3
; GFX11-GISEL-NEXT: s_setpc_b64 s[30:31]
;
@@ -888,9 +885,11 @@ define i16 @global_atomic_usub_sat_16(ptr addrspace(1) %ptr, i16 %data) {
; GFX12-GISEL-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-GISEL-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-GISEL-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-GISEL-NEXT: s_cbranch_execnz .LBB6_1
; GFX12-GISEL-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-GISEL-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-GISEL-NEXT: v_mov_b32_e32 v0, v3
; GFX12-GISEL-NEXT: s_setpc_b64 s[30:31]
;
@@ -965,11 +964,12 @@ define i16 @global_atomic_usub_sat_16(ptr addrspace(1) %ptr, i16 %data) {
; GFX11-SDAG-NEXT: buffer_gl0_inv
; GFX11-SDAG-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-SDAG-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-SDAG-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-SDAG-NEXT: s_cbranch_execnz .LBB6_1
; GFX11-SDAG-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-SDAG-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-SDAG-NEXT: v_mov_b16_e32 v0.l, v3.l
; GFX11-SDAG-NEXT: s_setpc_b64 s[30:31]
;
@@ -1000,9 +1000,11 @@ define i16 @global_atomic_usub_sat_16(ptr addrspace(1) %ptr, i16 %data) {
; GFX12-SDAG-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-SDAG-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-SDAG-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-SDAG-NEXT: s_cbranch_execnz .LBB6_1
; GFX12-SDAG-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-SDAG-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-SDAG-NEXT: v_mov_b16_e32 v0.l, v3.l
; GFX12-SDAG-NEXT: s_setpc_b64 s[30:31]
%ret = atomicrmw usub_sat ptr addrspace(1) %ptr, i16 %data syncscope("agent") seq_cst, align 4, !amdgpu.no.remote.memory !0
@@ -1082,11 +1084,12 @@ define i16 @global_atomic_usub_sat_offset_16(ptr addrspace(1) %ptr, i16 %data) {
; GFX11-GISEL-NEXT: buffer_gl0_inv
; GFX11-GISEL-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-GISEL-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-GISEL-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-GISEL-NEXT: s_cbranch_execnz .LBB7_1
; GFX11-GISEL-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-GISEL-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-GISEL-NEXT: v_mov_b32_e32 v0, v3
; GFX11-GISEL-NEXT: s_setpc_b64 s[30:31]
;
@@ -1117,9 +1120,11 @@ define i16 @global_atomic_usub_sat_offset_16(ptr addrspace(1) %ptr, i16 %data) {
; GFX12-GISEL-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-GISEL-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-GISEL-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-GISEL-NEXT: s_cbranch_execnz .LBB7_1
; GFX12-GISEL-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-GISEL-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-GISEL-NEXT: v_mov_b32_e32 v0, v3
; GFX12-GISEL-NEXT: s_setpc_b64 s[30:31]
;
@@ -1195,11 +1200,12 @@ define i16 @global_atomic_usub_sat_offset_16(ptr addrspace(1) %ptr, i16 %data) {
; GFX11-SDAG-NEXT: buffer_gl0_inv
; GFX11-SDAG-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-SDAG-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-SDAG-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-SDAG-NEXT: s_cbranch_execnz .LBB7_1
; GFX11-SDAG-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-SDAG-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-SDAG-NEXT: v_mov_b16_e32 v0.l, v3.l
; GFX11-SDAG-NEXT: s_setpc_b64 s[30:31]
;
@@ -1230,9 +1236,11 @@ define i16 @global_atomic_usub_sat_offset_16(ptr addrspace(1) %ptr, i16 %data) {
; GFX12-SDAG-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-SDAG-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-SDAG-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-SDAG-NEXT: s_cbranch_execnz .LBB7_1
; GFX12-SDAG-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-SDAG-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-SDAG-NEXT: v_mov_b16_e32 v0.l, v3.l
; GFX12-SDAG-NEXT: s_setpc_b64 s[30:31]
%gep = getelementptr i16, ptr addrspace(1) %ptr, i64 1024
@@ -1309,7 +1317,7 @@ define void @global_atomic_usub_sat_nortn_16(ptr addrspace(1) %ptr, i16 %data) {
; GFX11-GISEL-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-GISEL-NEXT: v_mov_b32_e32 v4, v3
; GFX11-GISEL-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-GISEL-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-GISEL-NEXT: s_cbranch_execnz .LBB8_1
; GFX11-GISEL-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -1342,6 +1350,7 @@ define void @global_atomic_usub_sat_nortn_16(ptr addrspace(1) %ptr, i16 %data) {
; GFX12-GISEL-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-GISEL-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-GISEL-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-GISEL-NEXT: s_cbranch_execnz .LBB8_1
; GFX12-GISEL-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-GISEL-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -1415,7 +1424,7 @@ define void @global_atomic_usub_sat_nortn_16(ptr addrspace(1) %ptr, i16 %data) {
; GFX11-SDAG-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-SDAG-NEXT: v_mov_b32_e32 v4, v3
; GFX11-SDAG-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-SDAG-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-SDAG-NEXT: s_cbranch_execnz .LBB8_1
; GFX11-SDAG-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -1448,6 +1457,7 @@ define void @global_atomic_usub_sat_nortn_16(ptr addrspace(1) %ptr, i16 %data) {
; GFX12-SDAG-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-SDAG-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-SDAG-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-SDAG-NEXT: s_cbranch_execnz .LBB8_1
; GFX12-SDAG-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-SDAG-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -1527,7 +1537,7 @@ define void @global_atomic_usub_sat_offset_nortn_16(ptr addrspace(1) %ptr, i16 %
; GFX11-GISEL-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-GISEL-NEXT: v_mov_b32_e32 v4, v3
; GFX11-GISEL-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-GISEL-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-GISEL-NEXT: s_cbranch_execnz .LBB9_1
; GFX11-GISEL-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -1560,6 +1570,7 @@ define void @global_atomic_usub_sat_offset_nortn_16(ptr addrspace(1) %ptr, i16 %
; GFX12-GISEL-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-GISEL-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-GISEL-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-GISEL-NEXT: s_cbranch_execnz .LBB9_1
; GFX12-GISEL-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-GISEL-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -1635,7 +1646,7 @@ define void @global_atomic_usub_sat_offset_nortn_16(ptr addrspace(1) %ptr, i16 %
; GFX11-SDAG-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-SDAG-NEXT: v_mov_b32_e32 v4, v3
; GFX11-SDAG-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-SDAG-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-SDAG-NEXT: s_cbranch_execnz .LBB9_1
; GFX11-SDAG-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -1668,6 +1679,7 @@ define void @global_atomic_usub_sat_offset_nortn_16(ptr addrspace(1) %ptr, i16 %
; GFX12-SDAG-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-SDAG-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-SDAG-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-SDAG-NEXT: s_cbranch_execnz .LBB9_1
; GFX12-SDAG-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-SDAG-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -1766,7 +1778,7 @@ define amdgpu_kernel void @global_atomic_usub_sat_sgpr_base_offset_16(ptr addrsp
; GFX11-GISEL-NEXT: buffer_gl0_inv
; GFX11-GISEL-NEXT: v_cmp_eq_u32_e32 vcc_lo, v1, v2
; GFX11-GISEL-NEXT: s_or_b32 s3, vcc_lo, s3
-; GFX11-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-GISEL-NEXT: s_and_not1_b32 exec_lo, exec_lo, s3
; GFX11-GISEL-NEXT: s_cbranch_execnz .LBB10_1
; GFX11-GISEL-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -1802,6 +1814,7 @@ define amdgpu_kernel void @global_atomic_usub_sat_sgpr_base_offset_16(ptr addrsp
; GFX12-GISEL-NEXT: s_or_b32 s3, vcc_lo, s3
; GFX12-GISEL-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-GISEL-NEXT: s_and_not1_b32 exec_lo, exec_lo, s3
+; GFX12-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-GISEL-NEXT: s_cbranch_execnz .LBB10_1
; GFX12-GISEL-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-GISEL-NEXT: s_or_b32 exec_lo, exec_lo, s3
@@ -1902,7 +1915,7 @@ define amdgpu_kernel void @global_atomic_usub_sat_sgpr_base_offset_16(ptr addrsp
; GFX11-SDAG-NEXT: buffer_gl0_inv
; GFX11-SDAG-NEXT: v_cmp_eq_u32_e32 vcc_lo, v1, v2
; GFX11-SDAG-NEXT: s_or_b32 s3, vcc_lo, s3
-; GFX11-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-SDAG-NEXT: s_and_not1_b32 exec_lo, exec_lo, s3
; GFX11-SDAG-NEXT: s_cbranch_execnz .LBB10_1
; GFX11-SDAG-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -1939,6 +1952,7 @@ define amdgpu_kernel void @global_atomic_usub_sat_sgpr_base_offset_16(ptr addrsp
; GFX12-SDAG-NEXT: s_or_b32 s3, vcc_lo, s3
; GFX12-SDAG-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-SDAG-NEXT: s_and_not1_b32 exec_lo, exec_lo, s3
+; GFX12-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-SDAG-NEXT: s_cbranch_execnz .LBB10_1
; GFX12-SDAG-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-SDAG-NEXT: s_or_b32 exec_lo, exec_lo, s3
@@ -2032,7 +2046,7 @@ define amdgpu_kernel void @global_atomic_usub_sat_sgpr_base_offset_nortn_16(ptr
; GFX11-GISEL-NEXT: v_cmp_eq_u32_e32 vcc_lo, v0, v1
; GFX11-GISEL-NEXT: v_mov_b32_e32 v1, v0
; GFX11-GISEL-NEXT: s_or_b32 s3, vcc_lo, s3
-; GFX11-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-GISEL-NEXT: s_and_not1_b32 exec_lo, exec_lo, s3
; GFX11-GISEL-NEXT: s_cbranch_execnz .LBB11_1
; GFX11-GISEL-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -2063,6 +2077,7 @@ define amdgpu_kernel void @global_atomic_usub_sat_sgpr_base_offset_nortn_16(ptr
; GFX12-GISEL-NEXT: s_or_b32 s3, vcc_lo, s3
; GFX12-GISEL-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-GISEL-NEXT: s_and_not1_b32 exec_lo, exec_lo, s3
+; GFX12-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-GISEL-NEXT: s_cbranch_execnz .LBB11_1
; GFX12-GISEL-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-GISEL-NEXT: s_endpgm
@@ -2148,7 +2163,7 @@ define amdgpu_kernel void @global_atomic_usub_sat_sgpr_base_offset_nortn_16(ptr
; GFX11-SDAG-NEXT: v_cmp_eq_u32_e32 vcc_lo, v0, v1
; GFX11-SDAG-NEXT: v_mov_b32_e32 v1, v0
; GFX11-SDAG-NEXT: s_or_b32 s3, vcc_lo, s3
-; GFX11-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-SDAG-NEXT: s_and_not1_b32 exec_lo, exec_lo, s3
; GFX11-SDAG-NEXT: s_cbranch_execnz .LBB11_1
; GFX11-SDAG-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -2180,6 +2195,7 @@ define amdgpu_kernel void @global_atomic_usub_sat_sgpr_base_offset_nortn_16(ptr
; GFX12-SDAG-NEXT: s_or_b32 s3, vcc_lo, s3
; GFX12-SDAG-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-SDAG-NEXT: s_and_not1_b32 exec_lo, exec_lo, s3
+; GFX12-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-SDAG-NEXT: s_cbranch_execnz .LBB11_1
; GFX12-SDAG-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-SDAG-NEXT: s_endpgm
@@ -2273,11 +2289,12 @@ define i8 @global_atomic_usub_sat_8(ptr addrspace(1) %ptr, i8 %data) {
; GFX11-GISEL-NEXT: buffer_gl0_inv
; GFX11-GISEL-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-GISEL-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-GISEL-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-GISEL-NEXT: s_cbranch_execnz .LBB12_1
; GFX11-GISEL-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-GISEL-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-GISEL-NEXT: v_mov_b32_e32 v0, v3
; GFX11-GISEL-NEXT: s_setpc_b64 s[30:31]
;
@@ -2312,9 +2329,11 @@ define i8 @global_atomic_usub_sat_8(ptr addrspace(1) %ptr, i8 %data) {
; GFX12-GISEL-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-GISEL-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-GISEL-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-GISEL-NEXT: s_cbranch_execnz .LBB12_1
; GFX12-GISEL-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-GISEL-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-GISEL-NEXT: v_mov_b32_e32 v0, v3
; GFX12-GISEL-NEXT: s_setpc_b64 s[30:31]
;
@@ -2393,11 +2412,12 @@ define i8 @global_atomic_usub_sat_8(ptr addrspace(1) %ptr, i8 %data) {
; GFX11-SDAG-NEXT: buffer_gl0_inv
; GFX11-SDAG-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-SDAG-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-SDAG-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-SDAG-NEXT: s_cbranch_execnz .LBB12_1
; GFX11-SDAG-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-SDAG-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-SDAG-NEXT: v_mov_b16_e32 v0.l, v3.l
; GFX11-SDAG-NEXT: s_setpc_b64 s[30:31]
;
@@ -2430,9 +2450,11 @@ define i8 @global_atomic_usub_sat_8(ptr addrspace(1) %ptr, i8 %data) {
; GFX12-SDAG-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-SDAG-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-SDAG-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-SDAG-NEXT: s_cbranch_execnz .LBB12_1
; GFX12-SDAG-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-SDAG-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-SDAG-NEXT: v_mov_b16_e32 v0.l, v3.l
; GFX12-SDAG-NEXT: s_setpc_b64 s[30:31]
%ret = atomicrmw usub_sat ptr addrspace(1) %ptr, i8 %data syncscope("agent") seq_cst, align 4, !amdgpu.no.remote.memory !0
@@ -2523,11 +2545,12 @@ define i8 @global_atomic_usub_sat_offset_8(ptr addrspace(1) %ptr, i8 %data) {
; GFX11-GISEL-NEXT: buffer_gl0_inv
; GFX11-GISEL-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-GISEL-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-GISEL-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-GISEL-NEXT: s_cbranch_execnz .LBB13_1
; GFX11-GISEL-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-GISEL-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-GISEL-NEXT: v_mov_b32_e32 v0, v3
; GFX11-GISEL-NEXT: s_setpc_b64 s[30:31]
;
@@ -2562,9 +2585,11 @@ define i8 @global_atomic_usub_sat_offset_8(ptr addrspace(1) %ptr, i8 %data) {
; GFX12-GISEL-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-GISEL-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-GISEL-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-GISEL-NEXT: s_cbranch_execnz .LBB13_1
; GFX12-GISEL-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-GISEL-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-GISEL-NEXT: v_mov_b32_e32 v0, v3
; GFX12-GISEL-NEXT: s_setpc_b64 s[30:31]
;
@@ -2643,11 +2668,12 @@ define i8 @global_atomic_usub_sat_offset_8(ptr addrspace(1) %ptr, i8 %data) {
; GFX11-SDAG-NEXT: buffer_gl0_inv
; GFX11-SDAG-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-SDAG-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-SDAG-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-SDAG-NEXT: s_cbranch_execnz .LBB13_1
; GFX11-SDAG-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-SDAG-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-SDAG-NEXT: v_mov_b16_e32 v0.l, v3.l
; GFX11-SDAG-NEXT: s_setpc_b64 s[30:31]
;
@@ -2680,9 +2706,11 @@ define i8 @global_atomic_usub_sat_offset_8(ptr addrspace(1) %ptr, i8 %data) {
; GFX12-SDAG-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-SDAG-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-SDAG-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-SDAG-NEXT: s_cbranch_execnz .LBB13_1
; GFX12-SDAG-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-SDAG-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-SDAG-NEXT: v_mov_b16_e32 v0.l, v3.l
; GFX12-SDAG-NEXT: s_setpc_b64 s[30:31]
%gep = getelementptr i8, ptr addrspace(1) %ptr, i64 1024
@@ -2771,7 +2799,7 @@ define void @global_atomic_usub_sat_nortn_8(ptr addrspace(1) %ptr, i8 %data) {
; GFX11-GISEL-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-GISEL-NEXT: v_mov_b32_e32 v4, v3
; GFX11-GISEL-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-GISEL-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-GISEL-NEXT: s_cbranch_execnz .LBB14_1
; GFX11-GISEL-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -2808,6 +2836,7 @@ define void @global_atomic_usub_sat_nortn_8(ptr addrspace(1) %ptr, i8 %data) {
; GFX12-GISEL-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-GISEL-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-GISEL-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-GISEL-NEXT: s_cbranch_execnz .LBB14_1
; GFX12-GISEL-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-GISEL-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -2886,7 +2915,7 @@ define void @global_atomic_usub_sat_nortn_8(ptr addrspace(1) %ptr, i8 %data) {
; GFX11-SDAG-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-SDAG-NEXT: v_mov_b32_e32 v4, v3
; GFX11-SDAG-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-SDAG-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-SDAG-NEXT: s_cbranch_execnz .LBB14_1
; GFX11-SDAG-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -2922,6 +2951,7 @@ define void @global_atomic_usub_sat_nortn_8(ptr addrspace(1) %ptr, i8 %data) {
; GFX12-SDAG-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-SDAG-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-SDAG-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-SDAG-NEXT: s_cbranch_execnz .LBB14_1
; GFX12-SDAG-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-SDAG-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -3011,7 +3041,7 @@ define void @global_atomic_usub_sat_offset_nortn_8(ptr addrspace(1) %ptr, i8 %da
; GFX11-GISEL-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-GISEL-NEXT: v_mov_b32_e32 v4, v3
; GFX11-GISEL-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-GISEL-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-GISEL-NEXT: s_cbranch_execnz .LBB15_1
; GFX11-GISEL-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -3048,6 +3078,7 @@ define void @global_atomic_usub_sat_offset_nortn_8(ptr addrspace(1) %ptr, i8 %da
; GFX12-GISEL-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-GISEL-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-GISEL-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-GISEL-NEXT: s_cbranch_execnz .LBB15_1
; GFX12-GISEL-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-GISEL-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -3126,7 +3157,7 @@ define void @global_atomic_usub_sat_offset_nortn_8(ptr addrspace(1) %ptr, i8 %da
; GFX11-SDAG-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-SDAG-NEXT: v_mov_b32_e32 v4, v3
; GFX11-SDAG-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-SDAG-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-SDAG-NEXT: s_cbranch_execnz .LBB15_1
; GFX11-SDAG-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -3162,6 +3193,7 @@ define void @global_atomic_usub_sat_offset_nortn_8(ptr addrspace(1) %ptr, i8 %da
; GFX12-SDAG-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-SDAG-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-SDAG-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-SDAG-NEXT: s_cbranch_execnz .LBB15_1
; GFX12-SDAG-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-SDAG-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -3273,7 +3305,7 @@ define amdgpu_kernel void @global_atomic_usub_sat_sgpr_base_offset_8(ptr addrspa
; GFX11-GISEL-NEXT: buffer_gl0_inv
; GFX11-GISEL-NEXT: v_cmp_eq_u32_e32 vcc_lo, v1, v2
; GFX11-GISEL-NEXT: s_or_b32 s2, vcc_lo, s2
-; GFX11-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-GISEL-NEXT: s_and_not1_b32 exec_lo, exec_lo, s2
; GFX11-GISEL-NEXT: s_cbranch_execnz .LBB16_1
; GFX11-GISEL-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -3314,6 +3346,7 @@ define amdgpu_kernel void @global_atomic_usub_sat_sgpr_base_offset_8(ptr addrspa
; GFX12-GISEL-NEXT: s_or_b32 s3, vcc_lo, s3
; GFX12-GISEL-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-GISEL-NEXT: s_and_not1_b32 exec_lo, exec_lo, s3
+; GFX12-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-GISEL-NEXT: s_cbranch_execnz .LBB16_1
; GFX12-GISEL-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-GISEL-NEXT: s_or_b32 exec_lo, exec_lo, s3
@@ -3418,7 +3451,7 @@ define amdgpu_kernel void @global_atomic_usub_sat_sgpr_base_offset_8(ptr addrspa
; GFX11-SDAG-NEXT: buffer_gl0_inv
; GFX11-SDAG-NEXT: v_cmp_eq_u32_e32 vcc_lo, v1, v2
; GFX11-SDAG-NEXT: s_or_b32 s2, vcc_lo, s2
-; GFX11-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-SDAG-NEXT: s_and_not1_b32 exec_lo, exec_lo, s2
; GFX11-SDAG-NEXT: s_cbranch_execnz .LBB16_1
; GFX11-SDAG-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -3458,6 +3491,7 @@ define amdgpu_kernel void @global_atomic_usub_sat_sgpr_base_offset_8(ptr addrspa
; GFX12-SDAG-NEXT: s_or_b32 s3, vcc_lo, s3
; GFX12-SDAG-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-SDAG-NEXT: s_and_not1_b32 exec_lo, exec_lo, s3
+; GFX12-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-SDAG-NEXT: s_cbranch_execnz .LBB16_1
; GFX12-SDAG-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-SDAG-NEXT: s_or_b32 exec_lo, exec_lo, s3
@@ -3564,7 +3598,7 @@ define amdgpu_kernel void @global_atomic_usub_sat_sgpr_base_offset_nortn_8(ptr a
; GFX11-GISEL-NEXT: v_cmp_eq_u32_e32 vcc_lo, v0, v1
; GFX11-GISEL-NEXT: v_mov_b32_e32 v1, v0
; GFX11-GISEL-NEXT: s_or_b32 s2, vcc_lo, s2
-; GFX11-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-GISEL-NEXT: s_and_not1_b32 exec_lo, exec_lo, s2
; GFX11-GISEL-NEXT: s_cbranch_execnz .LBB17_1
; GFX11-GISEL-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -3600,6 +3634,7 @@ define amdgpu_kernel void @global_atomic_usub_sat_sgpr_base_offset_nortn_8(ptr a
; GFX12-GISEL-NEXT: s_or_b32 s3, vcc_lo, s3
; GFX12-GISEL-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-GISEL-NEXT: s_and_not1_b32 exec_lo, exec_lo, s3
+; GFX12-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-GISEL-NEXT: s_cbranch_execnz .LBB17_1
; GFX12-GISEL-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-GISEL-NEXT: s_endpgm
@@ -3688,7 +3723,7 @@ define amdgpu_kernel void @global_atomic_usub_sat_sgpr_base_offset_nortn_8(ptr a
; GFX11-SDAG-NEXT: v_cmp_eq_u32_e32 vcc_lo, v0, v1
; GFX11-SDAG-NEXT: v_mov_b32_e32 v1, v0
; GFX11-SDAG-NEXT: s_or_b32 s2, vcc_lo, s2
-; GFX11-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-SDAG-NEXT: s_and_not1_b32 exec_lo, exec_lo, s2
; GFX11-SDAG-NEXT: s_cbranch_execnz .LBB17_1
; GFX11-SDAG-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -3722,6 +3757,7 @@ define amdgpu_kernel void @global_atomic_usub_sat_sgpr_base_offset_nortn_8(ptr a
; GFX12-SDAG-NEXT: s_or_b32 s3, vcc_lo, s3
; GFX12-SDAG-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-SDAG-NEXT: s_and_not1_b32 exec_lo, exec_lo, s3
+; GFX12-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-SDAG-NEXT: s_cbranch_execnz .LBB17_1
; GFX12-SDAG-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-SDAG-NEXT: s_endpgm
diff --git a/llvm/test/CodeGen/AMDGPU/atomics-system-scope.ll b/llvm/test/CodeGen/AMDGPU/atomics-system-scope.ll
index ad1746033361f4..a238d1517e78cf 100644
--- a/llvm/test/CodeGen/AMDGPU/atomics-system-scope.ll
+++ b/llvm/test/CodeGen/AMDGPU/atomics-system-scope.ll
@@ -372,9 +372,11 @@ define i16 @global_one_as_atomic_min_i16(ptr addrspace(1) %ptr, i16 %val) {
; FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
; FAKE16-NEXT: s_wait_xcnt 0x0
; FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; FAKE16-NEXT: s_cbranch_execnz .LBB28_1
; FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; FAKE16-NEXT: v_lshrrev_b32_e32 v0, v3, v5
; FAKE16-NEXT: s_set_pc_i64 s[30:31]
;
@@ -411,9 +413,11 @@ define i16 @global_one_as_atomic_min_i16(ptr addrspace(1) %ptr, i16 %val) {
; REAL16-NEXT: s_or_b32 s0, vcc_lo, s0
; REAL16-NEXT: s_wait_xcnt 0x0
; REAL16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; REAL16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; REAL16-NEXT: s_cbranch_execnz .LBB28_1
; REAL16-NEXT: ; %bb.2: ; %atomicrmw.end
; REAL16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; REAL16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; REAL16-NEXT: v_lshrrev_b32_e32 v0, v3, v5
; REAL16-NEXT: s_set_pc_i64 s[30:31]
%result = atomicrmw min ptr addrspace(1) %ptr, i16 %val syncscope("one-as") monotonic
@@ -454,9 +458,11 @@ define i16 @global_one_as_atomic_umin_i16(ptr addrspace(1) %ptr, i16 %val) {
; FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
; FAKE16-NEXT: s_wait_xcnt 0x0
; FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; FAKE16-NEXT: s_cbranch_execnz .LBB29_1
; FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; FAKE16-NEXT: v_lshrrev_b32_e32 v0, v3, v5
; FAKE16-NEXT: s_set_pc_i64 s[30:31]
;
@@ -493,9 +499,11 @@ define i16 @global_one_as_atomic_umin_i16(ptr addrspace(1) %ptr, i16 %val) {
; REAL16-NEXT: s_or_b32 s0, vcc_lo, s0
; REAL16-NEXT: s_wait_xcnt 0x0
; REAL16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; REAL16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; REAL16-NEXT: s_cbranch_execnz .LBB29_1
; REAL16-NEXT: ; %bb.2: ; %atomicrmw.end
; REAL16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; REAL16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; REAL16-NEXT: v_lshrrev_b32_e32 v0, v3, v5
; REAL16-NEXT: s_set_pc_i64 s[30:31]
%result = atomicrmw umin ptr addrspace(1) %ptr, i16 %val syncscope("one-as") monotonic
@@ -536,9 +544,11 @@ define i16 @global_one_as_atomic_max_i16(ptr addrspace(1) %ptr, i16 %val) {
; FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
; FAKE16-NEXT: s_wait_xcnt 0x0
; FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; FAKE16-NEXT: s_cbranch_execnz .LBB30_1
; FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; FAKE16-NEXT: v_lshrrev_b32_e32 v0, v3, v5
; FAKE16-NEXT: s_set_pc_i64 s[30:31]
;
@@ -575,9 +585,11 @@ define i16 @global_one_as_atomic_max_i16(ptr addrspace(1) %ptr, i16 %val) {
; REAL16-NEXT: s_or_b32 s0, vcc_lo, s0
; REAL16-NEXT: s_wait_xcnt 0x0
; REAL16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; REAL16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; REAL16-NEXT: s_cbranch_execnz .LBB30_1
; REAL16-NEXT: ; %bb.2: ; %atomicrmw.end
; REAL16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; REAL16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; REAL16-NEXT: v_lshrrev_b32_e32 v0, v3, v5
; REAL16-NEXT: s_set_pc_i64 s[30:31]
%result = atomicrmw max ptr addrspace(1) %ptr, i16 %val syncscope("one-as") monotonic
@@ -618,9 +630,11 @@ define i16 @global_one_as_atomic_umax_i16(ptr addrspace(1) %ptr, i16 %val) {
; FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
; FAKE16-NEXT: s_wait_xcnt 0x0
; FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; FAKE16-NEXT: s_cbranch_execnz .LBB31_1
; FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; FAKE16-NEXT: v_lshrrev_b32_e32 v0, v3, v5
; FAKE16-NEXT: s_set_pc_i64 s[30:31]
;
@@ -657,9 +671,11 @@ define i16 @global_one_as_atomic_umax_i16(ptr addrspace(1) %ptr, i16 %val) {
; REAL16-NEXT: s_or_b32 s0, vcc_lo, s0
; REAL16-NEXT: s_wait_xcnt 0x0
; REAL16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; REAL16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; REAL16-NEXT: s_cbranch_execnz .LBB31_1
; REAL16-NEXT: ; %bb.2: ; %atomicrmw.end
; REAL16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; REAL16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; REAL16-NEXT: v_lshrrev_b32_e32 v0, v3, v5
; REAL16-NEXT: s_set_pc_i64 s[30:31]
%result = atomicrmw umax ptr addrspace(1) %ptr, i16 %val syncscope("one-as") monotonic
@@ -699,9 +715,10 @@ define double @flat_system_atomic_fadd_f64(ptr %ptr, double %val) {
; GFX1250-NEXT: s_mov_b64 s[0:1], src_shared_base
; GFX1250-NEXT: s_mov_b32 s0, exec_lo
; GFX1250-NEXT: ; implicit-def: $vgpr0_vgpr1
-; GFX1250-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1250-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1250-NEXT: v_cmpx_ne_u32_e32 s1, v5
; GFX1250-NEXT: s_xor_b32 s0, exec_lo, s0
+; GFX1250-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1250-NEXT: s_cbranch_execnz .LBB34_3
; GFX1250-NEXT: ; %bb.1: ; %Flow2
; GFX1250-NEXT: s_and_not1_saveexec_b32 s0, s0
@@ -741,6 +758,7 @@ define double @flat_system_atomic_fadd_f64(ptr %ptr, double %val) {
; GFX1250-NEXT: s_or_b32 exec_lo, exec_lo, s1
; GFX1250-NEXT: ; implicit-def: $vgpr4_vgpr5
; GFX1250-NEXT: ; implicit-def: $vgpr2_vgpr3
+; GFX1250-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1250-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX1250-NEXT: s_cbranch_execz .LBB34_2
; GFX1250-NEXT: .LBB34_8: ; %atomicrmw.shared
@@ -764,9 +782,10 @@ define double @flat_one_as_atomic_fadd_f64(ptr %ptr, double %val) {
; GFX1250-NEXT: s_mov_b64 s[0:1], src_shared_base
; GFX1250-NEXT: s_mov_b32 s0, exec_lo
; GFX1250-NEXT: ; implicit-def: $vgpr0_vgpr1
-; GFX1250-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1250-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1250-NEXT: v_cmpx_ne_u32_e32 s1, v5
; GFX1250-NEXT: s_xor_b32 s0, exec_lo, s0
+; GFX1250-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1250-NEXT: s_cbranch_execnz .LBB35_3
; GFX1250-NEXT: ; %bb.1: ; %Flow2
; GFX1250-NEXT: s_and_not1_saveexec_b32 s0, s0
@@ -806,6 +825,7 @@ define double @flat_one_as_atomic_fadd_f64(ptr %ptr, double %val) {
; GFX1250-NEXT: s_or_b32 exec_lo, exec_lo, s1
; GFX1250-NEXT: ; implicit-def: $vgpr4_vgpr5
; GFX1250-NEXT: ; implicit-def: $vgpr2_vgpr3
+; GFX1250-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1250-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX1250-NEXT: s_cbranch_execz .LBB35_2
; GFX1250-NEXT: .LBB35_8: ; %atomicrmw.shared
@@ -1502,9 +1522,11 @@ define i16 @flat_one_as_atomic_min_i16(ptr %ptr, i16 %val) {
; FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
; FAKE16-NEXT: s_wait_xcnt 0x0
; FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; FAKE16-NEXT: s_cbranch_execnz .LBB60_1
; FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; FAKE16-NEXT: v_lshrrev_b32_e32 v0, v3, v5
; FAKE16-NEXT: s_set_pc_i64 s[30:31]
;
@@ -1541,9 +1563,11 @@ define i16 @flat_one_as_atomic_min_i16(ptr %ptr, i16 %val) {
; REAL16-NEXT: s_or_b32 s0, vcc_lo, s0
; REAL16-NEXT: s_wait_xcnt 0x0
; REAL16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; REAL16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; REAL16-NEXT: s_cbranch_execnz .LBB60_1
; REAL16-NEXT: ; %bb.2: ; %atomicrmw.end
; REAL16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; REAL16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; REAL16-NEXT: v_lshrrev_b32_e32 v0, v3, v5
; REAL16-NEXT: s_set_pc_i64 s[30:31]
%result = atomicrmw min ptr %ptr, i16 %val syncscope("one-as") monotonic
@@ -1584,9 +1608,11 @@ define i16 @flat_one_as_atomic_umin_i16(ptr %ptr, i16 %val) {
; FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
; FAKE16-NEXT: s_wait_xcnt 0x0
; FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; FAKE16-NEXT: s_cbranch_execnz .LBB61_1
; FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; FAKE16-NEXT: v_lshrrev_b32_e32 v0, v3, v5
; FAKE16-NEXT: s_set_pc_i64 s[30:31]
;
@@ -1623,9 +1649,11 @@ define i16 @flat_one_as_atomic_umin_i16(ptr %ptr, i16 %val) {
; REAL16-NEXT: s_or_b32 s0, vcc_lo, s0
; REAL16-NEXT: s_wait_xcnt 0x0
; REAL16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; REAL16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; REAL16-NEXT: s_cbranch_execnz .LBB61_1
; REAL16-NEXT: ; %bb.2: ; %atomicrmw.end
; REAL16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; REAL16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; REAL16-NEXT: v_lshrrev_b32_e32 v0, v3, v5
; REAL16-NEXT: s_set_pc_i64 s[30:31]
%result = atomicrmw umin ptr %ptr, i16 %val syncscope("one-as") monotonic
@@ -1666,9 +1694,11 @@ define i16 @flat_one_as_atomic_max_i16(ptr %ptr, i16 %val) {
; FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
; FAKE16-NEXT: s_wait_xcnt 0x0
; FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; FAKE16-NEXT: s_cbranch_execnz .LBB62_1
; FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; FAKE16-NEXT: v_lshrrev_b32_e32 v0, v3, v5
; FAKE16-NEXT: s_set_pc_i64 s[30:31]
;
@@ -1705,9 +1735,11 @@ define i16 @flat_one_as_atomic_max_i16(ptr %ptr, i16 %val) {
; REAL16-NEXT: s_or_b32 s0, vcc_lo, s0
; REAL16-NEXT: s_wait_xcnt 0x0
; REAL16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; REAL16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; REAL16-NEXT: s_cbranch_execnz .LBB62_1
; REAL16-NEXT: ; %bb.2: ; %atomicrmw.end
; REAL16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; REAL16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; REAL16-NEXT: v_lshrrev_b32_e32 v0, v3, v5
; REAL16-NEXT: s_set_pc_i64 s[30:31]
%result = atomicrmw max ptr %ptr, i16 %val syncscope("one-as") monotonic
@@ -1748,9 +1780,11 @@ define i16 @flat_one_as_atomic_umax_i16(ptr %ptr, i16 %val) {
; FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
; FAKE16-NEXT: s_wait_xcnt 0x0
; FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; FAKE16-NEXT: s_cbranch_execnz .LBB63_1
; FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; FAKE16-NEXT: v_lshrrev_b32_e32 v0, v3, v5
; FAKE16-NEXT: s_set_pc_i64 s[30:31]
;
@@ -1787,9 +1821,11 @@ define i16 @flat_one_as_atomic_umax_i16(ptr %ptr, i16 %val) {
; REAL16-NEXT: s_or_b32 s0, vcc_lo, s0
; REAL16-NEXT: s_wait_xcnt 0x0
; REAL16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; REAL16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; REAL16-NEXT: s_cbranch_execnz .LBB63_1
; REAL16-NEXT: ; %bb.2: ; %atomicrmw.end
; REAL16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; REAL16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; REAL16-NEXT: v_lshrrev_b32_e32 v0, v3, v5
; REAL16-NEXT: s_set_pc_i64 s[30:31]
%result = atomicrmw umax ptr %ptr, i16 %val syncscope("one-as") monotonic
diff --git a/llvm/test/CodeGen/AMDGPU/barrier-signal-wait-latency.ll b/llvm/test/CodeGen/AMDGPU/barrier-signal-wait-latency.ll
index 8bacc9ee8efb3d..4f0eec01c96fc7 100644
--- a/llvm/test/CodeGen/AMDGPU/barrier-signal-wait-latency.ll
+++ b/llvm/test/CodeGen/AMDGPU/barrier-signal-wait-latency.ll
@@ -21,7 +21,7 @@ define amdgpu_kernel void @test_barrier_independent_valu(ptr addrspace(1) %out,
; OPT-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; OPT-NEXT: v_lshlrev_b64_e32 v[0:1], 2, v[0:1]
; OPT-NEXT: v_add_co_u32 v0, vcc_lo, s0, v0
-; OPT-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; OPT-NEXT: s_delay_alu instid0(VALU_DEP_2)
; OPT-NEXT: v_add_co_ci_u32_e64 v1, null, s1, v1, vcc_lo
; OPT-NEXT: s_barrier_wait -1
; OPT-NEXT: global_load_b32 v0, v[0:1], off
@@ -44,7 +44,7 @@ define amdgpu_kernel void @test_barrier_independent_valu(ptr addrspace(1) %out,
; NOOPT-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; NOOPT-NEXT: v_lshlrev_b64_e32 v[0:1], 2, v[0:1]
; NOOPT-NEXT: v_add_co_u32 v0, vcc_lo, s0, v0
-; NOOPT-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; NOOPT-NEXT: s_delay_alu instid0(VALU_DEP_2)
; NOOPT-NEXT: v_add_co_ci_u32_e64 v1, null, s1, v1, vcc_lo
; NOOPT-NEXT: global_load_b32 v0, v[0:1], off
; NOOPT-NEXT: s_wait_loadcnt 0x0
@@ -125,7 +125,7 @@ define amdgpu_kernel void @test_barrier_multiple(ptr addrspace(1) %out, i32 %siz
; OPT-NEXT: v_ashrrev_i32_e32 v1, 31, v0
; OPT-NEXT: v_lshlrev_b64_e32 v[0:1], 2, v[0:1]
; OPT-NEXT: s_barrier_wait -1
-; OPT-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; OPT-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; OPT-NEXT: v_add_co_u32 v0, vcc_lo, s0, v0
; OPT-NEXT: v_add_co_ci_u32_e64 v1, null, s1, v1, vcc_lo
; OPT-NEXT: global_load_b32 v4, v[0:1], off
@@ -161,7 +161,7 @@ define amdgpu_kernel void @test_barrier_multiple(ptr addrspace(1) %out, i32 %siz
; NOOPT-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; NOOPT-NEXT: v_ashrrev_i32_e32 v1, 31, v0
; NOOPT-NEXT: v_lshlrev_b64_e32 v[0:1], 2, v[0:1]
-; NOOPT-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; NOOPT-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; NOOPT-NEXT: v_add_co_u32 v0, vcc_lo, s0, v0
; NOOPT-NEXT: v_add_co_ci_u32_e64 v1, null, s1, v1, vcc_lo
; NOOPT-NEXT: global_load_b32 v2, v[0:1], off
diff --git a/llvm/test/CodeGen/AMDGPU/bf16-conversions.ll b/llvm/test/CodeGen/AMDGPU/bf16-conversions.ll
index 43100e18394266..3f6ca942ec4625 100644
--- a/llvm/test/CodeGen/AMDGPU/bf16-conversions.ll
+++ b/llvm/test/CodeGen/AMDGPU/bf16-conversions.ll
@@ -239,11 +239,10 @@ define amdgpu_ps float @v_test_cvt_v2f64_v2bf16_v(<2 x double> %src) {
; GFX1250-NEXT: v_cmp_gt_f64_e64 s1, |v[2:3]|, |v[4:5]|
; GFX1250-NEXT: v_cmp_nlg_f64_e32 vcc_lo, v[2:3], v[4:5]
; GFX1250-NEXT: v_cmp_nlg_f64_e64 s0, v[0:1], v[6:7]
-; GFX1250-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX1250-NEXT: v_cndmask_b32_e64 v2, -1, 1, s1
; GFX1250-NEXT: v_cmp_gt_f64_e64 s1, |v[0:1]|, |v[6:7]|
+; GFX1250-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_2) | instid1(VALU_DEP_3)
; GFX1250-NEXT: v_dual_add_nc_u32 v1, v8, v2 :: v_dual_bitop2_b32 v10, 1, v8 bitop3:0x40
-; GFX1250-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_3)
; GFX1250-NEXT: v_cndmask_b32_e64 v0, -1, 1, s1
; GFX1250-NEXT: v_and_b32_e32 v11, 1, v9
; GFX1250-NEXT: v_cmp_ne_u32_e64 s1, 0, v10
@@ -251,6 +250,7 @@ define amdgpu_ps float @v_test_cvt_v2f64_v2bf16_v(<2 x double> %src) {
; GFX1250-NEXT: v_add_nc_u32_e32 v0, v9, v0
; GFX1250-NEXT: v_cmp_ne_u32_e64 s2, 0, v11
; GFX1250-NEXT: s_or_b32 vcc_lo, s1, vcc_lo
+; GFX1250-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX1250-NEXT: v_cndmask_b32_e32 v1, v1, v8, vcc_lo
; GFX1250-NEXT: s_or_b32 vcc_lo, s2, s0
; GFX1250-NEXT: v_cndmask_b32_e32 v0, v0, v9, vcc_lo
@@ -521,12 +521,12 @@ define amdgpu_ps void @fptrunc_f64_to_bf16(double %a, ptr %out) {
; GFX1250-NEXT: v_cvt_f64_f32_e32 v[4:5], v6
; GFX1250-NEXT: v_cmp_gt_f64_e64 s0, |v[0:1]|, |v[4:5]|
; GFX1250-NEXT: v_cmp_nlg_f64_e32 vcc_lo, v[0:1], v[4:5]
-; GFX1250-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1250-NEXT: v_cndmask_b32_e64 v0, -1, 1, s0
+; GFX1250-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1250-NEXT: v_dual_add_nc_u32 v0, v6, v0 :: v_dual_bitop2_b32 v7, 1, v6 bitop3:0x40
-; GFX1250-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_2) | instid1(VALU_DEP_1)
; GFX1250-NEXT: v_cmp_eq_u32_e64 s0, 1, v7
; GFX1250-NEXT: s_or_b32 vcc_lo, vcc_lo, s0
+; GFX1250-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1250-NEXT: v_cndmask_b32_e32 v0, v0, v6, vcc_lo
; GFX1250-NEXT: v_cvt_pk_bf16_f32 v0, v0, s0
; GFX1250-NEXT: flat_store_b16 v[2:3], v0
@@ -587,12 +587,12 @@ define amdgpu_ps void @fptrunc_f64_to_bf16_neg(double %a, ptr %out) {
; GFX1250-NEXT: v_cvt_f64_f32_e32 v[4:5], v6
; GFX1250-NEXT: v_cmp_gt_f64_e64 s1, |v[0:1]|, |v[4:5]|
; GFX1250-NEXT: v_cmp_nlg_f64_e64 s0, -v[0:1], v[4:5]
-; GFX1250-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1250-NEXT: v_cndmask_b32_e64 v0, -1, 1, s1
+; GFX1250-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1250-NEXT: v_dual_add_nc_u32 v0, v6, v0 :: v_dual_bitop2_b32 v7, 1, v6 bitop3:0x40
-; GFX1250-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX1250-NEXT: v_cmp_eq_u32_e32 vcc_lo, 1, v7
; GFX1250-NEXT: s_or_b32 vcc_lo, s0, vcc_lo
+; GFX1250-NEXT: s_delay_alu instid0(VALU_DEP_2) | instid1(SALU_CYCLE_1)
; GFX1250-NEXT: v_cndmask_b32_e32 v0, v0, v6, vcc_lo
; GFX1250-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1250-NEXT: v_cvt_pk_bf16_f32 v0, v0, s0
@@ -655,12 +655,12 @@ define amdgpu_ps void @fptrunc_f64_to_bf16_abs(double %a, ptr %out) {
; GFX1250-NEXT: v_cvt_f64_f32_e32 v[4:5], v6
; GFX1250-NEXT: v_cmp_gt_f64_e64 s1, |v[0:1]|, |v[4:5]|
; GFX1250-NEXT: v_cmp_nlg_f64_e64 s0, |v[0:1]|, v[4:5]
-; GFX1250-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1250-NEXT: v_cndmask_b32_e64 v0, -1, 1, s1
+; GFX1250-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1250-NEXT: v_dual_add_nc_u32 v0, v6, v0 :: v_dual_bitop2_b32 v7, 1, v6 bitop3:0x40
-; GFX1250-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX1250-NEXT: v_cmp_eq_u32_e32 vcc_lo, 1, v7
; GFX1250-NEXT: s_or_b32 vcc_lo, s0, vcc_lo
+; GFX1250-NEXT: s_delay_alu instid0(VALU_DEP_2) | instid1(SALU_CYCLE_1)
; GFX1250-NEXT: v_cndmask_b32_e32 v0, v0, v6, vcc_lo
; GFX1250-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1250-NEXT: v_cvt_pk_bf16_f32 v0, v0, s0
diff --git a/llvm/test/CodeGen/AMDGPU/bf16.ll b/llvm/test/CodeGen/AMDGPU/bf16.ll
index d3e89cd1e469ba..1eeffdcc28a93d 100644
--- a/llvm/test/CodeGen/AMDGPU/bf16.ll
+++ b/llvm/test/CodeGen/AMDGPU/bf16.ll
@@ -1732,12 +1732,12 @@ define void @test_load_store_f64_to_bf16(ptr addrspace(1) %in, ptr addrspace(1)
; GFX11-NEXT: v_and_b32_e32 v7, 1, v6
; GFX11-NEXT: v_cmp_gt_f64_e64 s0, |v[0:1]|, |v[4:5]|
; GFX11-NEXT: v_cmp_nlg_f64_e32 vcc_lo, v[0:1], v[4:5]
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_4)
; GFX11-NEXT: v_cndmask_b32_e64 v4, -1, 1, s0
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-NEXT: v_cmp_eq_u32_e64 s0, 1, v7
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_nc_u32_e32 v4, v6, v4
; GFX11-NEXT: s_or_b32 vcc_lo, vcc_lo, s0
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: v_cndmask_b32_e32 v4, v4, v6, vcc_lo
; GFX11-NEXT: v_cmp_u_f64_e32 vcc_lo, v[0:1], v[0:1]
; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
@@ -1761,12 +1761,12 @@ define void @test_load_store_f64_to_bf16(ptr addrspace(1) %in, ptr addrspace(1)
; GFX1250-NEXT: v_cmp_gt_f64_e64 s0, |v[0:1]|, |v[4:5]|
; GFX1250-NEXT: v_cmp_nlg_f64_e32 vcc_lo, v[0:1], v[4:5]
; GFX1250-NEXT: s_wait_xcnt 0x0
-; GFX1250-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1250-NEXT: v_cndmask_b32_e64 v0, -1, 1, s0
+; GFX1250-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1250-NEXT: v_dual_add_nc_u32 v0, v6, v0 :: v_dual_bitop2_b32 v7, 1, v6 bitop3:0x40
-; GFX1250-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_2) | instid1(VALU_DEP_1)
; GFX1250-NEXT: v_cmp_eq_u32_e64 s0, 1, v7
; GFX1250-NEXT: s_or_b32 vcc_lo, vcc_lo, s0
+; GFX1250-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1250-NEXT: v_cndmask_b32_e32 v0, v0, v6, vcc_lo
; GFX1250-NEXT: v_cvt_pk_bf16_f32 v0, v0, s0
; GFX1250-NEXT: global_store_b16 v[2:3], v0, off
@@ -28287,23 +28287,23 @@ define bfloat @v_sqrt_bf16(bfloat %a) #0 {
; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-NEXT: v_fma_f32 v5, -v3, v1, v0
; GFX11-NEXT: v_cmp_ge_f32_e64 s0, 0, v4
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_3)
; GFX11-NEXT: v_cndmask_b32_e64 v1, v1, v2, s0
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-NEXT: v_cmp_lt_f32_e64 s0, 0, v5
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_cndmask_b32_e64 v1, v1, v3, s0
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_mul_f32_e32 v2, 0x37800000, v1
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX11-NEXT: v_cndmask_b32_e32 v1, v1, v2, vcc_lo
; GFX11-NEXT: v_cmp_class_f32_e64 vcc_lo, v0, 0x260
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_cndmask_b32_e32 v0, v1, v0, vcc_lo
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_2) | instid1(VALU_DEP_3)
; GFX11-NEXT: v_bfe_u32 v1, v0, 16, 1
; GFX11-NEXT: v_or_b32_e32 v2, 0x400000, v0
; GFX11-NEXT: v_cmp_u_f32_e32 vcc_lo, v0, v0
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add3_u32 v1, v1, v0, 0x7fff
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_cndmask_b32_e32 v0, v1, v2, vcc_lo
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_lshrrev_b32_e32 v0, 16, v0
; GFX11-NEXT: s_setpc_b64 s[30:31]
;
@@ -28603,48 +28603,47 @@ define bfloat @v_rsq_bf16(bfloat %x) #0 {
; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-NEXT: v_fma_f32 v5, -v3, v1, v0
; GFX11-NEXT: v_cmp_ge_f32_e64 s0, 0, v4
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_3)
; GFX11-NEXT: v_cndmask_b32_e64 v1, v1, v2, s0
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-NEXT: v_cmp_lt_f32_e64 s0, 0, v5
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_cndmask_b32_e64 v1, v1, v3, s0
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_mul_f32_e32 v2, 0x37800000, v1
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX11-NEXT: v_cndmask_b32_e32 v1, v1, v2, vcc_lo
; GFX11-NEXT: v_cmp_class_f32_e64 vcc_lo, v0, 0x260
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_cndmask_b32_e32 v0, v1, v0, vcc_lo
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_2) | instid1(VALU_DEP_3)
; GFX11-NEXT: v_bfe_u32 v1, v0, 16, 1
; GFX11-NEXT: v_or_b32_e32 v2, 0x400000, v0
; GFX11-NEXT: v_cmp_u_f32_e32 vcc_lo, v0, v0
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add3_u32 v1, v1, v0, 0x7fff
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_cndmask_b32_e32 v0, v1, v2, vcc_lo
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_and_b32_e32 v0, 0xffff0000, v0
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX11-NEXT: v_div_scale_f32 v1, null, v0, v0, 1.0
; GFX11-NEXT: v_div_scale_f32 v4, vcc_lo, 1.0, v0, 1.0
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_2) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_rcp_f32_e32 v2, v1
; GFX11-NEXT: s_waitcnt_depctr depctr_va_vdst(0)
; GFX11-NEXT: v_fma_f32 v3, -v1, v2, 1.0
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_fmac_f32_e32 v2, v3, v2
-; GFX11-NEXT: v_mul_f32_e32 v3, v4, v2
; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX11-NEXT: v_mul_f32_e32 v3, v4, v2
; GFX11-NEXT: v_fma_f32 v5, -v1, v3, v4
-; GFX11-NEXT: v_fmac_f32_e32 v3, v5, v2
; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX11-NEXT: v_fmac_f32_e32 v3, v5, v2
; GFX11-NEXT: v_fma_f32 v1, -v1, v3, v4
-; GFX11-NEXT: v_div_fmas_f32 v1, v1, v2, v3
; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX11-NEXT: v_div_fmas_f32 v1, v1, v2, v3
; GFX11-NEXT: v_div_fixup_f32 v0, v1, v0, 1.0
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_2) | instid1(VALU_DEP_3)
; GFX11-NEXT: v_bfe_u32 v1, v0, 16, 1
; GFX11-NEXT: v_or_b32_e32 v2, 0x400000, v0
; GFX11-NEXT: v_cmp_u_f32_e32 vcc_lo, v0, v0
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add3_u32 v1, v1, v0, 0x7fff
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_cndmask_b32_e32 v0, v1, v2, vcc_lo
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_lshrrev_b32_e32 v0, 16, v0
; GFX11-NEXT: s_setpc_b64 s[30:31]
;
@@ -28945,48 +28944,47 @@ define bfloat @v_neg_rsq_bf16(bfloat %x) #0 {
; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-NEXT: v_fma_f32 v5, -v3, v1, v0
; GFX11-NEXT: v_cmp_ge_f32_e64 s0, 0, v4
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_3)
; GFX11-NEXT: v_cndmask_b32_e64 v1, v1, v2, s0
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-NEXT: v_cmp_lt_f32_e64 s0, 0, v5
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_cndmask_b32_e64 v1, v1, v3, s0
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_mul_f32_e32 v2, 0x37800000, v1
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX11-NEXT: v_cndmask_b32_e32 v1, v1, v2, vcc_lo
; GFX11-NEXT: v_cmp_class_f32_e64 vcc_lo, v0, 0x260
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_cndmask_b32_e32 v0, v1, v0, vcc_lo
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_2) | instid1(VALU_DEP_3)
; GFX11-NEXT: v_bfe_u32 v1, v0, 16, 1
; GFX11-NEXT: v_or_b32_e32 v2, 0x400000, v0
; GFX11-NEXT: v_cmp_u_f32_e32 vcc_lo, v0, v0
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add3_u32 v1, v1, v0, 0x7fff
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_cndmask_b32_e32 v0, v1, v2, vcc_lo
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_and_b32_e32 v0, 0xffff0000, v0
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX11-NEXT: v_div_scale_f32 v1, null, v0, v0, -1.0
; GFX11-NEXT: v_div_scale_f32 v4, vcc_lo, -1.0, v0, -1.0
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_2) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_rcp_f32_e32 v2, v1
; GFX11-NEXT: s_waitcnt_depctr depctr_va_vdst(0)
; GFX11-NEXT: v_fma_f32 v3, -v1, v2, 1.0
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_fmac_f32_e32 v2, v3, v2
-; GFX11-NEXT: v_mul_f32_e32 v3, v4, v2
; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX11-NEXT: v_mul_f32_e32 v3, v4, v2
; GFX11-NEXT: v_fma_f32 v5, -v1, v3, v4
-; GFX11-NEXT: v_fmac_f32_e32 v3, v5, v2
; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX11-NEXT: v_fmac_f32_e32 v3, v5, v2
; GFX11-NEXT: v_fma_f32 v1, -v1, v3, v4
-; GFX11-NEXT: v_div_fmas_f32 v1, v1, v2, v3
; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX11-NEXT: v_div_fmas_f32 v1, v1, v2, v3
; GFX11-NEXT: v_div_fixup_f32 v0, v1, v0, -1.0
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_2) | instid1(VALU_DEP_3)
; GFX11-NEXT: v_bfe_u32 v1, v0, 16, 1
; GFX11-NEXT: v_or_b32_e32 v2, 0x400000, v0
; GFX11-NEXT: v_cmp_u_f32_e32 vcc_lo, v0, v0
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add3_u32 v1, v1, v0, 0x7fff
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_cndmask_b32_e32 v0, v1, v2, vcc_lo
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_lshrrev_b32_e32 v0, 16, v0
; GFX11-NEXT: s_setpc_b64 s[30:31]
;
@@ -30995,19 +30993,19 @@ define bfloat @v_round_bf16(bfloat %a) #0 {
; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_trunc_f32_e32 v1, v0
; GFX11-NEXT: v_sub_f32_e32 v2, v0, v1
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_cmp_ge_f32_e64 s0, |v2|, 0.5
; GFX11-NEXT: v_cndmask_b32_e64 v2, 0, 1.0, s0
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_bfi_b32 v0, 0x7fffffff, v2, v0
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_f32_e32 v0, v1, v0
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_2) | instid1(VALU_DEP_3)
; GFX11-NEXT: v_bfe_u32 v1, v0, 16, 1
; GFX11-NEXT: v_or_b32_e32 v2, 0x400000, v0
; GFX11-NEXT: v_cmp_u_f32_e32 vcc_lo, v0, v0
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add3_u32 v1, v1, v0, 0x7fff
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_cndmask_b32_e32 v0, v1, v2, vcc_lo
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_lshrrev_b32_e32 v0, 16, v0
; GFX11-NEXT: s_setpc_b64 s[30:31]
;
@@ -31019,13 +31017,12 @@ define bfloat @v_round_bf16(bfloat %a) #0 {
; GFX1250-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1250-NEXT: v_trunc_f32_e32 v2, v1
; GFX1250-NEXT: v_fma_mix_f32_bf16 v0, -v2, 1.0, v0 op_sel:[0,1,0] op_sel_hi:[0,1,1]
-; GFX1250-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX1250-NEXT: v_cmp_ge_f32_e64 s0, |v0|, 0.5
; GFX1250-NEXT: v_cndmask_b32_e64 v0, 0, 1.0, s0
-; GFX1250-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1250-NEXT: v_bfi_b32 v0, 0x7fffffff, v0, v1
+; GFX1250-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1250-NEXT: v_add_f32_e32 v0, v2, v0
-; GFX1250-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1250-NEXT: v_cvt_pk_bf16_f32 v0, v0, s0
; GFX1250-NEXT: s_set_pc_i64 s[30:31]
%op = call bfloat @llvm.round.bf16(bfloat %a)
@@ -33290,7 +33287,7 @@ define i64 @v_fptosi_bf16_to_i64(bfloat %x) #0 {
; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_xor_b32_e32 v0, v0, v3
; GFX11-NEXT: v_sub_co_u32 v0, vcc_lo, v0, v3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX11-NEXT: v_sub_co_ci_u32_e64 v1, null, v1, v3, vcc_lo
; GFX11-NEXT: s_setpc_b64 s[30:31]
;
@@ -33534,10 +33531,9 @@ define <2 x i64> @v_fptosi_v2bf16_to_v2i64(<2 x bfloat> %x) #0 {
; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
; GFX11-NEXT: v_xor_b32_e32 v0, v0, v1
; GFX11-NEXT: v_xor_b32_e32 v4, v4, v6
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_3)
; GFX11-NEXT: v_sub_co_u32 v0, vcc_lo, v0, v1
; GFX11-NEXT: v_sub_co_ci_u32_e64 v1, null, v2, v1, vcc_lo
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_sub_co_u32 v2, vcc_lo, v4, v6
; GFX11-NEXT: v_sub_co_ci_u32_e64 v3, null, v3, v6, vcc_lo
; GFX11-NEXT: s_setpc_b64 s[30:31]
@@ -33885,12 +33881,10 @@ define <3 x i64> @v_fptosi_v3bf16_to_v3i64(<3 x bfloat> %x) #0 {
; GFX11-NEXT: v_xor_b32_e32 v10, v1, v8
; GFX11-NEXT: v_xor_b32_e32 v6, v6, v8
; GFX11-NEXT: v_sub_co_u32 v0, vcc_lo, v2, v5
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_sub_co_ci_u32_e64 v1, null, v3, v5, vcc_lo
; GFX11-NEXT: v_sub_co_u32 v2, vcc_lo, v9, v7
; GFX11-NEXT: v_sub_co_ci_u32_e64 v3, null, v4, v7, vcc_lo
; GFX11-NEXT: v_sub_co_u32 v4, vcc_lo, v10, v8
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_sub_co_ci_u32_e64 v5, null, v6, v8, vcc_lo
; GFX11-NEXT: s_setpc_b64 s[30:31]
;
@@ -34332,10 +34326,10 @@ define <4 x i64> @v_fptosi_v4bf16_to_v4i64(<4 x bfloat> %x) #0 {
; GFX11-NEXT: v_xor_b32_e32 v6, v11, v13
; GFX11-NEXT: v_xor_b32_e32 v7, v9, v13
; GFX11-NEXT: v_sub_co_u32 v4, vcc_lo, v4, v10
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_4)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
; GFX11-NEXT: v_sub_co_ci_u32_e64 v5, null, v5, v10, vcc_lo
; GFX11-NEXT: v_sub_co_u32 v6, vcc_lo, v6, v13
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_4)
; GFX11-NEXT: v_sub_co_ci_u32_e64 v7, null, v7, v13, vcc_lo
; GFX11-NEXT: s_setpc_b64 s[30:31]
;
@@ -40776,7 +40770,7 @@ define <2 x bfloat> @v_vselect_v2bf16(<2 x i1> %cond, <2 x bfloat> %a, <2 x bflo
; GFX11TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
; GFX11TRUE16-NEXT: v_cmp_eq_u16_e32 vcc_lo, 1, v0.h
; GFX11TRUE16-NEXT: v_cmp_eq_u16_e64 s0, 1, v0.l
-; GFX11TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX11TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX11TRUE16-NEXT: v_cndmask_b16 v0.h, v4.l, v1.l, vcc_lo
; GFX11TRUE16-NEXT: v_cndmask_b16 v0.l, v3.l, v2.l, s0
; GFX11TRUE16-NEXT: s_setpc_b64 s[30:31]
@@ -40806,7 +40800,7 @@ define <2 x bfloat> @v_vselect_v2bf16(<2 x i1> %cond, <2 x bfloat> %a, <2 x bflo
; GFX1250TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_3)
; GFX1250TRUE16-NEXT: v_cmp_eq_u16_e32 vcc_lo, 1, v0.h
; GFX1250TRUE16-NEXT: v_cmp_eq_u16_e64 s0, 1, v0.l
-; GFX1250TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX1250TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX1250TRUE16-NEXT: v_cndmask_b16 v0.h, v4.l, v1.l, vcc_lo
; GFX1250TRUE16-NEXT: v_cndmask_b16 v0.l, v3.l, v2.l, s0
; GFX1250TRUE16-NEXT: s_set_pc_i64 s[30:31]
@@ -42537,10 +42531,9 @@ define <4 x bfloat> @v_vselect_v4bf16(<4 x i1> %cond, <4 x bfloat> %a, <4 x bflo
; GFX11TRUE16-NEXT: v_cmp_eq_u16_e64 s1, 1, v0.l
; GFX11TRUE16-NEXT: v_cmp_eq_u16_e64 s2, 1, v0.h
; GFX11TRUE16-NEXT: v_cndmask_b16 v0.h, v8.l, v3.l, vcc_lo
-; GFX11TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
+; GFX11TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_4)
; GFX11TRUE16-NEXT: v_cndmask_b16 v1.h, v2.l, v1.l, s0
; GFX11TRUE16-NEXT: v_cndmask_b16 v0.l, v6.l, v4.l, s1
-; GFX11TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_4)
; GFX11TRUE16-NEXT: v_cndmask_b16 v1.l, v7.l, v5.l, s2
; GFX11TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -42584,10 +42577,9 @@ define <4 x bfloat> @v_vselect_v4bf16(<4 x i1> %cond, <4 x bfloat> %a, <4 x bflo
; GFX1250TRUE16-NEXT: v_cmp_eq_u16_e64 s1, 1, v0.l
; GFX1250TRUE16-NEXT: v_cmp_eq_u16_e64 s2, 1, v0.h
; GFX1250TRUE16-NEXT: v_cndmask_b16 v0.h, v8.l, v3.l, vcc_lo
-; GFX1250TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
+; GFX1250TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_4)
; GFX1250TRUE16-NEXT: v_cndmask_b16 v1.h, v2.l, v1.l, s0
; GFX1250TRUE16-NEXT: v_cndmask_b16 v0.l, v6.l, v4.l, s1
-; GFX1250TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_4)
; GFX1250TRUE16-NEXT: v_cndmask_b16 v1.l, v7.l, v5.l, s2
; GFX1250TRUE16-NEXT: s_set_pc_i64 s[30:31]
;
diff --git a/llvm/test/CodeGen/AMDGPU/branch-relaxation-gfx1250.ll b/llvm/test/CodeGen/AMDGPU/branch-relaxation-gfx1250.ll
index 63bcd96eb76abf..6718f02decdb8c 100644
--- a/llvm/test/CodeGen/AMDGPU/branch-relaxation-gfx1250.ll
+++ b/llvm/test/CodeGen/AMDGPU/branch-relaxation-gfx1250.ll
@@ -30,6 +30,7 @@ define amdgpu_kernel void @uniform_conditional_max_short_forward_branch(ptr addr
; GCN-NEXT: s_load_b32 s0, s[4:5], 0x2c nv
; GCN-NEXT: s_wait_kmcnt 0x0
; GCN-NEXT: s_cmp_eq_u32 s0, 0
+; GCN-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GCN-NEXT: s_cbranch_scc0 .LBB0_1
; GCN-NEXT: ; %bb.3: ; %bb
; GCN-NEXT: s_get_pc_i64 s[2:3]
@@ -60,6 +61,7 @@ define amdgpu_kernel void @uniform_conditional_max_short_forward_branch(ptr addr
; GCN-ADD-PC64-NEXT: s_load_b32 s0, s[4:5], 0x2c nv
; GCN-ADD-PC64-NEXT: s_wait_kmcnt 0x0
; GCN-ADD-PC64-NEXT: s_cmp_eq_u32 s0, 0
+; GCN-ADD-PC64-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GCN-ADD-PC64-NEXT: s_cbranch_scc0 .LBB0_1
; GCN-ADD-PC64-NEXT: ; %bb.3: ; %bb
; GCN-ADD-PC64-NEXT: s_add_pc_i64 .LBB0_2-.Lpost_addpc0
@@ -127,6 +129,7 @@ define amdgpu_kernel void @uniform_conditional_min_long_forward_branch(ptr addrs
; GCN-NEXT: s_load_b32 s0, s[4:5], 0x2c nv
; GCN-NEXT: s_wait_kmcnt 0x0
; GCN-NEXT: s_cmp_eq_u32 s0, 0
+; GCN-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GCN-NEXT: s_cbranch_scc0 .LBB1_1
; GCN-NEXT: ; %bb.3: ; %bb0
; GCN-NEXT: s_get_pc_i64 s[2:3]
@@ -158,6 +161,7 @@ define amdgpu_kernel void @uniform_conditional_min_long_forward_branch(ptr addrs
; GCN-ADD-PC64-NEXT: s_load_b32 s0, s[4:5], 0x2c nv
; GCN-ADD-PC64-NEXT: s_wait_kmcnt 0x0
; GCN-ADD-PC64-NEXT: s_cmp_eq_u32 s0, 0
+; GCN-ADD-PC64-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GCN-ADD-PC64-NEXT: s_cbranch_scc0 .LBB1_1
; GCN-ADD-PC64-NEXT: ; %bb.3: ; %bb0
; GCN-ADD-PC64-NEXT: s_add_pc_i64 .LBB1_2-.Lpost_addpc1
@@ -228,6 +232,7 @@ define amdgpu_kernel void @uniform_conditional_min_long_forward_vcnd_branch(ptr
; GCN-NEXT: s_load_b32 s0, s[4:5], 0x2c nv
; GCN-NEXT: s_wait_kmcnt 0x0
; GCN-NEXT: s_cmp_eq_f32 s0, 0
+; GCN-NEXT: s_delay_alu instid0(SALU_CYCLE_3)
; GCN-NEXT: s_cbranch_scc0 .LBB2_1
; GCN-NEXT: ; %bb.3: ; %bb0
; GCN-NEXT: s_get_pc_i64 s[2:3]
@@ -260,6 +265,7 @@ define amdgpu_kernel void @uniform_conditional_min_long_forward_vcnd_branch(ptr
; GCN-ADD-PC64-NEXT: s_load_b32 s0, s[4:5], 0x2c nv
; GCN-ADD-PC64-NEXT: s_wait_kmcnt 0x0
; GCN-ADD-PC64-NEXT: s_cmp_eq_f32 s0, 0
+; GCN-ADD-PC64-NEXT: s_delay_alu instid0(SALU_CYCLE_3)
; GCN-ADD-PC64-NEXT: s_cbranch_scc0 .LBB2_1
; GCN-ADD-PC64-NEXT: ; %bb.3: ; %bb0
; GCN-ADD-PC64-NEXT: s_add_pc_i64 .LBB2_2-.Lpost_addpc2
@@ -335,7 +341,7 @@ define amdgpu_kernel void @min_long_forward_vbranch(ptr addrspace(1) %arg) #0 {
; GCN-NEXT: global_load_b32 v2, v0, s[0:1] scale_offset scope:SCOPE_SYS
; GCN-NEXT: s_wait_loadcnt 0x0
; GCN-NEXT: v_lshlrev_b32_e32 v0, 2, v0
-; GCN-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GCN-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_2) | instid1(VALU_DEP_1)
; GCN-NEXT: v_add_nc_u64_e32 v[0:1], s[0:1], v[0:1]
; GCN-NEXT: s_mov_b32 s0, exec_lo
; GCN-NEXT: v_cmpx_ne_u32_e32 0, v2
@@ -373,7 +379,7 @@ define amdgpu_kernel void @min_long_forward_vbranch(ptr addrspace(1) %arg) #0 {
; GCN-ADD-PC64-NEXT: global_load_b32 v2, v0, s[0:1] scale_offset scope:SCOPE_SYS
; GCN-ADD-PC64-NEXT: s_wait_loadcnt 0x0
; GCN-ADD-PC64-NEXT: v_lshlrev_b32_e32 v0, 2, v0
-; GCN-ADD-PC64-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GCN-ADD-PC64-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_2) | instid1(VALU_DEP_1)
; GCN-ADD-PC64-NEXT: v_add_nc_u64_e32 v[0:1], s[0:1], v[0:1]
; GCN-ADD-PC64-NEXT: s_mov_b32 s0, exec_lo
; GCN-ADD-PC64-NEXT: v_cmpx_ne_u32_e32 0, v2
@@ -456,7 +462,7 @@ define amdgpu_kernel void @long_backward_sbranch(ptr addrspace(1) %arg) #0 {
; GCN-NEXT: s_mov_b32 s0, 0
; GCN-NEXT: .LBB4_1: ; %bb2
; GCN-NEXT: ; =>This Inner Loop Header: Depth=1
-; GCN-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GCN-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
; GCN-NEXT: s_add_co_i32 s0, s0, 1
; GCN-NEXT: ;;#ASMSTART
; GCN-NEXT: v_nop_e64
@@ -483,7 +489,7 @@ define amdgpu_kernel void @long_backward_sbranch(ptr addrspace(1) %arg) #0 {
; GCN-ADD-PC64-NEXT: s_mov_b32 s0, 0
; GCN-ADD-PC64-NEXT: .LBB4_1: ; %bb2
; GCN-ADD-PC64-NEXT: ; =>This Inner Loop Header: Depth=1
-; GCN-ADD-PC64-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GCN-ADD-PC64-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
; GCN-ADD-PC64-NEXT: s_add_co_i32 s0, s0, 1
; GCN-ADD-PC64-NEXT: ;;#ASMSTART
; GCN-ADD-PC64-NEXT: v_nop_e64
@@ -566,7 +572,7 @@ define amdgpu_kernel void @uniform_unconditional_min_long_forward_branch(ptr add
; GCN-NEXT: .LBB5_2: ; %Flow
; GCN-NEXT: s_and_b32 s0, s0, exec_lo
; GCN-NEXT: s_cselect_b32 s0, 1, 0
-; GCN-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GCN-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GCN-NEXT: s_cmp_lg_u32 s0, 1
; GCN-NEXT: s_cbranch_scc1 .LBB5_4
; GCN-NEXT: ; %bb.3: ; %bb2
@@ -606,7 +612,7 @@ define amdgpu_kernel void @uniform_unconditional_min_long_forward_branch(ptr add
; GCN-ADD-PC64-NEXT: .LBB5_2: ; %Flow
; GCN-ADD-PC64-NEXT: s_and_b32 s0, s0, exec_lo
; GCN-ADD-PC64-NEXT: s_cselect_b32 s0, 1, 0
-; GCN-ADD-PC64-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GCN-ADD-PC64-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GCN-ADD-PC64-NEXT: s_cmp_lg_u32 s0, 1
; GCN-ADD-PC64-NEXT: s_cbranch_scc1 .LBB5_4
; GCN-ADD-PC64-NEXT: ; %bb.3: ; %bb2
@@ -775,20 +781,23 @@ define amdgpu_kernel void @expand_requires_expand(i32 %cond0) #0 {
; GCN-NEXT: s_load_b32 s0, s[4:5], 0x24 nv
; GCN-NEXT: s_wait_kmcnt 0x0
; GCN-NEXT: s_cmp_lt_i32 s0, 0
+; GCN-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GCN-NEXT: s_cselect_b32 s0, -1, 0
-; GCN-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GCN-NEXT: s_and_b32 vcc_lo, exec_lo, s0
+; GCN-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GCN-NEXT: s_cbranch_vccnz .LBB7_2
; GCN-NEXT: ; %bb.1: ; %bb1
; GCN-NEXT: s_load_b32 s0, s[0:1], 0x0
; GCN-NEXT: s_wait_kmcnt 0x0
; GCN-NEXT: s_cmp_lg_u32 s0, 3
+; GCN-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GCN-NEXT: s_cselect_b32 s0, -1, 0
; GCN-NEXT: .LBB7_2: ; %Flow
; GCN-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GCN-NEXT: s_and_b32 s0, s0, exec_lo
; GCN-NEXT: s_cselect_b32 s0, 1, 0
; GCN-NEXT: s_cmp_lg_u32 s0, 1
+; GCN-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GCN-NEXT: s_cbranch_scc0 .LBB7_3
; GCN-NEXT: ; %bb.5: ; %Flow
; GCN-NEXT: s_get_pc_i64 s[0:1]
@@ -821,20 +830,23 @@ define amdgpu_kernel void @expand_requires_expand(i32 %cond0) #0 {
; GCN-ADD-PC64-NEXT: s_load_b32 s0, s[4:5], 0x24 nv
; GCN-ADD-PC64-NEXT: s_wait_kmcnt 0x0
; GCN-ADD-PC64-NEXT: s_cmp_lt_i32 s0, 0
+; GCN-ADD-PC64-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GCN-ADD-PC64-NEXT: s_cselect_b32 s0, -1, 0
-; GCN-ADD-PC64-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GCN-ADD-PC64-NEXT: s_and_b32 vcc_lo, exec_lo, s0
+; GCN-ADD-PC64-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GCN-ADD-PC64-NEXT: s_cbranch_vccnz .LBB7_2
; GCN-ADD-PC64-NEXT: ; %bb.1: ; %bb1
; GCN-ADD-PC64-NEXT: s_load_b32 s0, s[0:1], 0x0
; GCN-ADD-PC64-NEXT: s_wait_kmcnt 0x0
; GCN-ADD-PC64-NEXT: s_cmp_lg_u32 s0, 3
+; GCN-ADD-PC64-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GCN-ADD-PC64-NEXT: s_cselect_b32 s0, -1, 0
; GCN-ADD-PC64-NEXT: .LBB7_2: ; %Flow
; GCN-ADD-PC64-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GCN-ADD-PC64-NEXT: s_and_b32 s0, s0, exec_lo
; GCN-ADD-PC64-NEXT: s_cselect_b32 s0, 1, 0
; GCN-ADD-PC64-NEXT: s_cmp_lg_u32 s0, 1
+; GCN-ADD-PC64-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GCN-ADD-PC64-NEXT: s_cbranch_scc0 .LBB7_3
; GCN-ADD-PC64-NEXT: ; %bb.5: ; %Flow
; GCN-ADD-PC64-NEXT: s_add_pc_i64 .LBB7_4-.Lpost_addpc7
@@ -930,7 +942,7 @@ define amdgpu_kernel void @uniform_inside_divergent(ptr addrspace(1) %out, i32 %
; GCN-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GCN-NEXT: v_and_b32_e32 v0, 0x3ff, v0
; GCN-NEXT: s_mov_b32 s3, exec_lo
-; GCN-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GCN-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GCN-NEXT: v_cmpx_gt_u32_e32 16, v0
; GCN-NEXT: s_cbranch_execnz .LBB8_1
; GCN-NEXT: ; %bb.4: ; %entry
@@ -963,7 +975,7 @@ define amdgpu_kernel void @uniform_inside_divergent(ptr addrspace(1) %out, i32 %
; GCN-ADD-PC64-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GCN-ADD-PC64-NEXT: v_and_b32_e32 v0, 0x3ff, v0
; GCN-ADD-PC64-NEXT: s_mov_b32 s3, exec_lo
-; GCN-ADD-PC64-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GCN-ADD-PC64-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GCN-ADD-PC64-NEXT: v_cmpx_gt_u32_e32 16, v0
; GCN-ADD-PC64-NEXT: s_cbranch_execnz .LBB8_1
; GCN-ADD-PC64-NEXT: ; %bb.4: ; %entry
@@ -1045,6 +1057,7 @@ define amdgpu_kernel void @analyze_mask_branch() #0 {
; GCN-NEXT: v_mov_b32_e64 v0, 0
; GCN-NEXT: ;;#ASMEND
; GCN-NEXT: v_cmpx_nlt_f32_e32 0, v0
+; GCN-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GCN-NEXT: s_xor_b32 s0, exec_lo, s0
; GCN-NEXT: s_cbranch_execz .LBB9_2
; GCN-NEXT: ; %bb.1: ; %ret
@@ -1097,6 +1110,7 @@ define amdgpu_kernel void @analyze_mask_branch() #0 {
; GCN-ADD-PC64-NEXT: v_mov_b32_e64 v0, 0
; GCN-ADD-PC64-NEXT: ;;#ASMEND
; GCN-ADD-PC64-NEXT: v_cmpx_nlt_f32_e32 0, v0
+; GCN-ADD-PC64-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GCN-ADD-PC64-NEXT: s_xor_b32 s0, exec_lo, s0
; GCN-ADD-PC64-NEXT: s_cbranch_execz .LBB9_2
; GCN-ADD-PC64-NEXT: ; %bb.1: ; %ret
@@ -1210,6 +1224,7 @@ define amdgpu_kernel void @long_branch_hang(ptr addrspace(1) nocapture %arg, i32
; GCN-NEXT: s_mov_b32 s8, -1
; GCN-NEXT: s_wait_kmcnt 0x0
; GCN-NEXT: s_cmp_eq_u32 s0, 0
+; GCN-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_4) | instid1(SALU_CYCLE_1)
; GCN-NEXT: s_cselect_b32 s6, -1, 0
; GCN-NEXT: s_cmp_lg_u32 s0, 0
; GCN-NEXT: s_mov_b32 s0, 0
@@ -1234,11 +1249,12 @@ define amdgpu_kernel void @long_branch_hang(ptr addrspace(1) nocapture %arg, i32
; GCN-NEXT: .LBB10_2: ; %Flow
; GCN-NEXT: s_and_b32 s7, s8, exec_lo
; GCN-NEXT: s_cselect_b32 s7, 1, 0
-; GCN-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GCN-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GCN-NEXT: s_cmp_lg_u32 s7, 1
; GCN-NEXT: s_cbranch_scc1 .LBB10_4
; GCN-NEXT: ; %bb.3: ; %bb9
; GCN-NEXT: s_cmp_lt_i32 s3, 11
+; GCN-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GCN-NEXT: s_cselect_b32 s0, -1, 0
; GCN-NEXT: s_cmp_ge_i32 s2, s3
; GCN-NEXT: s_cselect_b32 s7, -1, 0
@@ -1250,6 +1266,7 @@ define amdgpu_kernel void @long_branch_hang(ptr addrspace(1) nocapture %arg, i32
; GCN-NEXT: s_cselect_b32 s0, 1, 0
; GCN-NEXT: s_cmp_lg_u32 s0, 1
; GCN-NEXT: ; implicit-def: $sgpr0
+; GCN-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GCN-NEXT: s_cbranch_scc0 .LBB10_5
; GCN-NEXT: ; %bb.9: ; %Flow5
; GCN-NEXT: s_get_pc_i64 s[2:3]
@@ -1259,6 +1276,7 @@ define amdgpu_kernel void @long_branch_hang(ptr addrspace(1) nocapture %arg, i32
; GCN-NEXT: s_set_pc_i64 s[2:3]
; GCN-NEXT: .LBB10_5: ; %bb14
; GCN-NEXT: s_cmp_lt_i32 s1, 9
+; GCN-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GCN-NEXT: s_cselect_b32 s0, -1, 0
; GCN-NEXT: s_cmp_lt_i32 s2, s3
; GCN-NEXT: s_cselect_b32 s1, -1, 0
@@ -1290,6 +1308,7 @@ define amdgpu_kernel void @long_branch_hang(ptr addrspace(1) nocapture %arg, i32
; GCN-ADD-PC64-NEXT: s_mov_b32 s8, -1
; GCN-ADD-PC64-NEXT: s_wait_kmcnt 0x0
; GCN-ADD-PC64-NEXT: s_cmp_eq_u32 s0, 0
+; GCN-ADD-PC64-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_4) | instid1(SALU_CYCLE_1)
; GCN-ADD-PC64-NEXT: s_cselect_b32 s6, -1, 0
; GCN-ADD-PC64-NEXT: s_cmp_lg_u32 s0, 0
; GCN-ADD-PC64-NEXT: s_mov_b32 s0, 0
@@ -1311,11 +1330,12 @@ define amdgpu_kernel void @long_branch_hang(ptr addrspace(1) nocapture %arg, i32
; GCN-ADD-PC64-NEXT: .LBB10_2: ; %Flow
; GCN-ADD-PC64-NEXT: s_and_b32 s7, s8, exec_lo
; GCN-ADD-PC64-NEXT: s_cselect_b32 s7, 1, 0
-; GCN-ADD-PC64-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GCN-ADD-PC64-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GCN-ADD-PC64-NEXT: s_cmp_lg_u32 s7, 1
; GCN-ADD-PC64-NEXT: s_cbranch_scc1 .LBB10_4
; GCN-ADD-PC64-NEXT: ; %bb.3: ; %bb9
; GCN-ADD-PC64-NEXT: s_cmp_lt_i32 s3, 11
+; GCN-ADD-PC64-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GCN-ADD-PC64-NEXT: s_cselect_b32 s0, -1, 0
; GCN-ADD-PC64-NEXT: s_cmp_ge_i32 s2, s3
; GCN-ADD-PC64-NEXT: s_cselect_b32 s7, -1, 0
@@ -1327,12 +1347,14 @@ define amdgpu_kernel void @long_branch_hang(ptr addrspace(1) nocapture %arg, i32
; GCN-ADD-PC64-NEXT: s_cselect_b32 s0, 1, 0
; GCN-ADD-PC64-NEXT: s_cmp_lg_u32 s0, 1
; GCN-ADD-PC64-NEXT: ; implicit-def: $sgpr0
+; GCN-ADD-PC64-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GCN-ADD-PC64-NEXT: s_cbranch_scc0 .LBB10_5
; GCN-ADD-PC64-NEXT: ; %bb.9: ; %Flow5
; GCN-ADD-PC64-NEXT: s_add_pc_i64 .LBB10_6-.Lpost_addpc12
; GCN-ADD-PC64-NEXT: .Lpost_addpc12:
; GCN-ADD-PC64-NEXT: .LBB10_5: ; %bb14
; GCN-ADD-PC64-NEXT: s_cmp_lt_i32 s1, 9
+; GCN-ADD-PC64-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GCN-ADD-PC64-NEXT: s_cselect_b32 s0, -1, 0
; GCN-ADD-PC64-NEXT: s_cmp_lt_i32 s2, s3
; GCN-ADD-PC64-NEXT: s_cselect_b32 s1, -1, 0
diff --git a/llvm/test/CodeGen/AMDGPU/branch-relaxation-inst-size-gfx10.ll b/llvm/test/CodeGen/AMDGPU/branch-relaxation-inst-size-gfx10.ll
index 13912aababb7e8..68968e2576a4ad 100644
--- a/llvm/test/CodeGen/AMDGPU/branch-relaxation-inst-size-gfx10.ll
+++ b/llvm/test/CodeGen/AMDGPU/branch-relaxation-inst-size-gfx10.ll
@@ -1,6 +1,6 @@
; RUN: llc -mtriple=amdgpu10.10 -amdgpu-s-branch-bits=4 < %s | FileCheck -enable-var-scope -check-prefixes=GCN,GFX10 %s
; RUN: llc -mtriple=amdgpu9.00 -amdgpu-s-branch-bits=4 < %s | FileCheck -enable-var-scope -check-prefixes=GCN,GFX9 %s
-; RUN: llc -mtriple=amdgpu11.00 -amdgpu-s-branch-bits=4 < %s | FileCheck -enable-var-scope -check-prefixes=GCN,GFX10 %s
+; RUN: llc -mtriple=amdgpu11.00 -amdgpu-s-branch-bits=4 < %s | FileCheck -enable-var-scope -check-prefixes=GCN,GFX11 %s
; Make sure the code size estimate for inline asm is 12-bytes per
; instruction, rather than 8 in previous generations.
@@ -15,6 +15,14 @@
; GFX10: s_add_u32
; GFX10: s_addc_u32
; GFX10: s_setpc_b64
+
+; GFX11: s_cmp_eq_u32
+; GFX11-NEXT: s_delay_alu
+; GFX11-NEXT: s_cbranch_scc0
+; GFX11: s_getpc_b64
+; GFX11: s_add_u32
+; GFX11: s_addc_u32
+; GFX11: s_setpc_b64
define amdgpu_kernel void @long_forward_branch_gfx10only(ptr addrspace(1) %arg, i32 %cnd) #0 {
bb0:
%cmp = icmp eq i32 %cnd, 0
diff --git a/llvm/test/CodeGen/AMDGPU/branch-relaxation-inst-size-gfx11.ll b/llvm/test/CodeGen/AMDGPU/branch-relaxation-inst-size-gfx11.ll
index 6707863663f8c3..24c5a77ee5b2c3 100644
--- a/llvm/test/CodeGen/AMDGPU/branch-relaxation-inst-size-gfx11.ll
+++ b/llvm/test/CodeGen/AMDGPU/branch-relaxation-inst-size-gfx11.ll
@@ -9,6 +9,7 @@ define amdgpu_kernel void @long_forward_branch_gfx11plus(ptr addrspace(1) %in, p
; GFX11-NEXT: s_load_b32 s0, s[4:5], 0x34
; GFX11-NEXT: s_waitcnt lgkmcnt(0)
; GFX11-NEXT: s_cmp_eq_u32 s0, 0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc0 .LBB0_1
; GFX11-NEXT: ; %bb.3: ; %bb0
; GFX11-NEXT: s_getpc_b64 s[6:7]
diff --git a/llvm/test/CodeGen/AMDGPU/branch-relaxation.ll b/llvm/test/CodeGen/AMDGPU/branch-relaxation.ll
index b40636776d84fe..c7065355cd622b 100644
--- a/llvm/test/CodeGen/AMDGPU/branch-relaxation.ll
+++ b/llvm/test/CodeGen/AMDGPU/branch-relaxation.ll
@@ -50,6 +50,7 @@ define amdgpu_kernel void @uniform_conditional_max_short_forward_branch(ptr addr
; GFX11-NEXT: s_load_b32 s0, s[4:5], 0x2c
; GFX11-NEXT: s_waitcnt lgkmcnt(0)
; GFX11-NEXT: s_cmp_eq_u32 s0, 0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc0 .LBB0_1
; GFX11-NEXT: ; %bb.3: ; %bb
; GFX11-NEXT: s_getpc_b64 s[2:3]
@@ -80,6 +81,7 @@ define amdgpu_kernel void @uniform_conditional_max_short_forward_branch(ptr addr
; GFX12-NEXT: s_load_b32 s0, s[4:5], 0x2c
; GFX12-NEXT: s_wait_kmcnt 0x0
; GFX12-NEXT: s_cmp_eq_u32 s0, 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_scc0 .LBB0_1
; GFX12-NEXT: ; %bb.3: ; %bb
; GFX12-NEXT: s_getpc_b64 s[2:3]
@@ -156,6 +158,7 @@ define amdgpu_kernel void @uniform_conditional_min_long_forward_branch(ptr addrs
; GFX11-NEXT: s_load_b32 s0, s[4:5], 0x2c
; GFX11-NEXT: s_waitcnt lgkmcnt(0)
; GFX11-NEXT: s_cmp_eq_u32 s0, 0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc0 .LBB1_1
; GFX11-NEXT: ; %bb.3: ; %bb0
; GFX11-NEXT: s_getpc_b64 s[2:3]
@@ -186,6 +189,7 @@ define amdgpu_kernel void @uniform_conditional_min_long_forward_branch(ptr addrs
; GFX12-NEXT: s_load_b32 s0, s[4:5], 0x2c
; GFX12-NEXT: s_wait_kmcnt 0x0
; GFX12-NEXT: s_cmp_eq_u32 s0, 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_scc0 .LBB1_1
; GFX12-NEXT: ; %bb.3: ; %bb0
; GFX12-NEXT: s_getpc_b64 s[2:3]
@@ -265,6 +269,7 @@ define amdgpu_kernel void @uniform_conditional_min_long_forward_vcnd_branch(ptr
; GFX11-NEXT: s_waitcnt lgkmcnt(0)
; GFX11-NEXT: v_cmp_eq_f32_e64 s[2:3], s0, 0
; GFX11-NEXT: s_and_b64 vcc, exec, s[2:3]
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_vccz .LBB2_1
; GFX11-NEXT: ; %bb.3: ; %bb0
; GFX11-NEXT: s_getpc_b64 s[2:3]
@@ -296,6 +301,7 @@ define amdgpu_kernel void @uniform_conditional_min_long_forward_vcnd_branch(ptr
; GFX12-NEXT: s_load_b32 s0, s[4:5], 0x2c
; GFX12-NEXT: s_wait_kmcnt 0x0
; GFX12-NEXT: s_cmp_eq_f32 s0, 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_3)
; GFX12-NEXT: s_cbranch_scc0 .LBB2_1
; GFX12-NEXT: ; %bb.3: ; %bb0
; GFX12-NEXT: s_getpc_b64 s[2:3]
@@ -380,7 +386,7 @@ define amdgpu_kernel void @min_long_forward_vbranch(ptr addrspace(1) %arg) #0 {
; GFX11: ; %bb.0: ; %bb
; GFX11-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
; GFX11-NEXT: v_and_b32_e32 v0, 0x3ff, v0
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_4) | instid1(VALU_DEP_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_lshlrev_b32_e32 v0, 2, v0
; GFX11-NEXT: s_waitcnt lgkmcnt(0)
; GFX11-NEXT: global_load_b32 v2, v0, s[0:1] glc dlc
@@ -389,6 +395,7 @@ define amdgpu_kernel void @min_long_forward_vbranch(ptr addrspace(1) %arg) #0 {
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, s1, 0, s[2:3]
; GFX11-NEXT: s_mov_b64 s[0:1], exec
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v2
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_cbranch_execnz .LBB3_1
; GFX11-NEXT: ; %bb.3: ; %bb
; GFX11-NEXT: s_getpc_b64 s[2:3]
@@ -426,6 +433,7 @@ define amdgpu_kernel void @min_long_forward_vbranch(ptr addrspace(1) %arg) #0 {
; GFX12-NEXT: v_add_co_ci_u32_e64 v1, null, s1, 0, s0
; GFX12-NEXT: s_mov_b32 s0, exec_lo
; GFX12-NEXT: v_cmpx_ne_u32_e32 0, v2
+; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-NEXT: s_cbranch_execnz .LBB3_1
; GFX12-NEXT: ; %bb.3: ; %bb
; GFX12-NEXT: s_getpc_b64 s[2:3]
@@ -500,7 +508,7 @@ define amdgpu_kernel void @long_backward_sbranch(ptr addrspace(1) %arg) #0 {
; GFX11-NEXT: .p2align 6
; GFX11-NEXT: .LBB4_1: ; %bb2
; GFX11-NEXT: ; =>This Inner Loop Header: Depth=1
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_add_i32 s0, s0, 1
; GFX11-NEXT: ;;#ASMSTART
; GFX11-NEXT: v_nop_e64
@@ -526,7 +534,7 @@ define amdgpu_kernel void @long_backward_sbranch(ptr addrspace(1) %arg) #0 {
; GFX12-NEXT: s_mov_b32 s0, 0
; GFX12-NEXT: .LBB4_1: ; %bb2
; GFX12-NEXT: ; =>This Inner Loop Header: Depth=1
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_add_co_i32 s0, s0, 1
; GFX12-NEXT: ;;#ASMSTART
; GFX12-NEXT: v_nop_e64
@@ -637,7 +645,7 @@ define amdgpu_kernel void @uniform_unconditional_min_long_forward_branch(ptr add
; GFX11-NEXT: .LBB5_2: ; %Flow
; GFX11-NEXT: s_and_b64 s[0:1], s[0:1], exec
; GFX11-NEXT: s_cselect_b32 s0, 1, 0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_cmp_lg_u32 s0, 1
; GFX11-NEXT: s_cbranch_scc1 .LBB5_4
; GFX11-NEXT: ; %bb.3: ; %bb2
@@ -679,7 +687,7 @@ define amdgpu_kernel void @uniform_unconditional_min_long_forward_branch(ptr add
; GFX12-NEXT: .LBB5_2: ; %Flow
; GFX12-NEXT: s_and_b32 s0, s0, exec_lo
; GFX12-NEXT: s_cselect_b32 s0, 1, 0
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_lg_u32 s0, 1
; GFX12-NEXT: s_cbranch_scc1 .LBB5_4
; GFX12-NEXT: ; %bb.3: ; %bb2
@@ -849,20 +857,23 @@ define amdgpu_kernel void @expand_requires_expand(i32 %cond0) #0 {
; GFX11-NEXT: s_load_b32 s0, s[4:5], 0x24
; GFX11-NEXT: s_waitcnt lgkmcnt(0)
; GFX11-NEXT: s_cmp_lt_i32 s0, 0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_cselect_b64 s[0:1], -1, 0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_and_b64 vcc, exec, s[0:1]
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_vccnz .LBB7_2
; GFX11-NEXT: ; %bb.1: ; %bb1
; GFX11-NEXT: s_load_b32 s0, s[0:1], 0x0
; GFX11-NEXT: s_waitcnt lgkmcnt(0)
; GFX11-NEXT: s_cmp_lg_u32 s0, 3
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cselect_b64 s[0:1], -1, 0
; GFX11-NEXT: .LBB7_2: ; %Flow
; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_b64 s[0:1], s[0:1], exec
; GFX11-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc0 .LBB7_3
; GFX11-NEXT: ; %bb.5: ; %Flow
; GFX11-NEXT: s_getpc_b64 s[0:1]
@@ -893,20 +904,23 @@ define amdgpu_kernel void @expand_requires_expand(i32 %cond0) #0 {
; GFX12-NEXT: s_load_b32 s0, s[4:5], 0x24
; GFX12-NEXT: s_wait_kmcnt 0x0
; GFX12-NEXT: s_cmp_lt_i32 s0, 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s0, -1, 0
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_and_b32 vcc_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_vccnz .LBB7_2
; GFX12-NEXT: ; %bb.1: ; %bb1
; GFX12-NEXT: s_load_b32 s0, s[0:1], 0x0
; GFX12-NEXT: s_wait_kmcnt 0x0
; GFX12-NEXT: s_cmp_lg_u32 s0, 3
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s0, -1, 0
; GFX12-NEXT: .LBB7_2: ; %Flow
; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_and_b32 s0, s0, exec_lo
; GFX12-NEXT: s_cselect_b32 s0, 1, 0
; GFX12-NEXT: s_cmp_lg_u32 s0, 1
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_scc0 .LBB7_3
; GFX12-NEXT: ; %bb.5: ; %Flow
; GFX12-NEXT: s_getpc_b64 s[0:1]
@@ -996,7 +1010,7 @@ define amdgpu_kernel void @uniform_inside_divergent(ptr addrspace(1) %out, i32 %
; GFX11: ; %bb.0: ; %entry
; GFX11-NEXT: v_and_b32_e32 v0, 0x3ff, v0
; GFX11-NEXT: s_mov_b64 s[0:1], exec
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_cmpx_gt_u32_e32 16, v0
; GFX11-NEXT: s_cbranch_execz .LBB8_3
; GFX11-NEXT: ; %bb.1: ; %if
@@ -1020,7 +1034,7 @@ define amdgpu_kernel void @uniform_inside_divergent(ptr addrspace(1) %out, i32 %
; GFX12: ; %bb.0: ; %entry
; GFX12-NEXT: v_and_b32_e32 v0, 0x3ff, v0
; GFX12-NEXT: s_mov_b32 s3, exec_lo
-; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX12-NEXT: v_cmpx_gt_u32_e32 16, v0
; GFX12-NEXT: s_cbranch_execz .LBB8_3
; GFX12-NEXT: ; %bb.1: ; %if
@@ -1118,6 +1132,7 @@ define amdgpu_kernel void @analyze_mask_branch() #0 {
; GFX11-NEXT: v_mov_b32_e64 v0, 0
; GFX11-NEXT: ;;#ASMEND
; GFX11-NEXT: v_cmpx_nlt_f32_e32 0, v0
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b64 s[0:1], exec, s[0:1]
; GFX11-NEXT: s_cbranch_execz .LBB9_2
; GFX11-NEXT: ; %bb.1: ; %ret
@@ -1170,6 +1185,7 @@ define amdgpu_kernel void @analyze_mask_branch() #0 {
; GFX12-NEXT: v_mov_b32_e64 v0, 0
; GFX12-NEXT: ;;#ASMEND
; GFX12-NEXT: v_cmpx_nlt_f32_e32 0, v0
+; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX12-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX12-NEXT: s_cbranch_execz .LBB9_2
; GFX12-NEXT: ; %bb.1: ; %ret
@@ -1318,10 +1334,12 @@ define amdgpu_kernel void @long_branch_hang(ptr addrspace(1) nocapture %arg, i32
; GFX11-NEXT: s_load_b128 s[0:3], s[4:5], 0x2c
; GFX11-NEXT: s_waitcnt lgkmcnt(0)
; GFX11-NEXT: s_cmp_eq_u32 s0, 0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_cselect_b64 s[6:7], -1, 0
; GFX11-NEXT: s_cmp_lg_u32 s0, 0
; GFX11-NEXT: s_cselect_b64 s[8:9], -1, 0
; GFX11-NEXT: s_cmp_lt_i32 s3, 6
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB10_1
; GFX11-NEXT: ; %bb.8: ; %bb
; GFX11-NEXT: s_getpc_b64 s[8:9]
@@ -1346,11 +1364,12 @@ define amdgpu_kernel void @long_branch_hang(ptr addrspace(1) nocapture %arg, i32
; GFX11-NEXT: .LBB10_3: ; %Flow
; GFX11-NEXT: s_and_b64 s[10:11], s[10:11], exec
; GFX11-NEXT: s_cselect_b32 s0, 1, 0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_cmp_lg_u32 s0, 1
; GFX11-NEXT: s_cbranch_scc1 .LBB10_5
; GFX11-NEXT: ; %bb.4: ; %bb9
; GFX11-NEXT: s_cmp_lt_i32 s3, 11
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_cselect_b64 s[8:9], -1, 0
; GFX11-NEXT: s_cmp_ge_i32 s2, s3
; GFX11-NEXT: s_cselect_b64 s[10:11], -1, 0
@@ -1362,9 +1381,11 @@ define amdgpu_kernel void @long_branch_hang(ptr addrspace(1) nocapture %arg, i32
; GFX11-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s0, 1
; GFX11-NEXT: ; implicit-def: $sgpr0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB10_7
; GFX11-NEXT: ; %bb.6: ; %bb14
; GFX11-NEXT: s_cmp_lt_i32 s1, 9
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_cselect_b64 s[0:1], -1, 0
; GFX11-NEXT: s_cmp_lt_i32 s2, s3
; GFX11-NEXT: s_cselect_b64 s[2:3], -1, 0
@@ -1394,6 +1415,7 @@ define amdgpu_kernel void @long_branch_hang(ptr addrspace(1) nocapture %arg, i32
; GFX12-NEXT: s_mov_b32 s7, -1
; GFX12-NEXT: s_wait_kmcnt 0x0
; GFX12-NEXT: s_cmp_eq_u32 s0, 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_4) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s6, -1, 0
; GFX12-NEXT: s_cmp_lg_u32 s0, 0
; GFX12-NEXT: s_mov_b32 s0, 0
@@ -1420,11 +1442,12 @@ define amdgpu_kernel void @long_branch_hang(ptr addrspace(1) nocapture %arg, i32
; GFX12-NEXT: .LBB10_2: ; %Flow
; GFX12-NEXT: s_and_b32 s7, s7, exec_lo
; GFX12-NEXT: s_cselect_b32 s7, 1, 0
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_lg_u32 s7, 1
; GFX12-NEXT: s_cbranch_scc1 .LBB10_4
; GFX12-NEXT: ; %bb.3: ; %bb9
; GFX12-NEXT: s_cmp_lt_i32 s3, 11
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s0, -1, 0
; GFX12-NEXT: s_cmp_ge_i32 s2, s3
; GFX12-NEXT: s_cselect_b32 s7, -1, 0
@@ -1436,9 +1459,11 @@ define amdgpu_kernel void @long_branch_hang(ptr addrspace(1) nocapture %arg, i32
; GFX12-NEXT: s_cselect_b32 s0, 1, 0
; GFX12-NEXT: s_cmp_lg_u32 s0, 1
; GFX12-NEXT: ; implicit-def: $sgpr0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_scc1 .LBB10_6
; GFX12-NEXT: ; %bb.5: ; %bb14
; GFX12-NEXT: s_cmp_lt_i32 s1, 9
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s0, -1, 0
; GFX12-NEXT: s_cmp_lt_i32 s2, s3
; GFX12-NEXT: s_cselect_b32 s1, -1, 0
diff --git a/llvm/test/CodeGen/AMDGPU/buffer-fat-pointer-atomicrmw-fadd.ll b/llvm/test/CodeGen/AMDGPU/buffer-fat-pointer-atomicrmw-fadd.ll
index 4719d6edbaaa50..73c98e5af79a78 100644
--- a/llvm/test/CodeGen/AMDGPU/buffer-fat-pointer-atomicrmw-fadd.ll
+++ b/llvm/test/CodeGen/AMDGPU/buffer-fat-pointer-atomicrmw-fadd.ll
@@ -1213,7 +1213,7 @@ define float @buffer_fat_ptr_agent_atomic_fadd_ret_f32__offset(ptr addrspace(7)
; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v0, v5
; GFX11-NEXT: s_or_b32 s4, vcc_lo, s4
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
; GFX11-NEXT: s_cbranch_execnz .LBB5_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -1431,7 +1431,7 @@ define float @buffer_fat_ptr_agent_atomic_fadd_ret_f32__offset__amdgpu_no_remote
; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v0, v5
; GFX11-NEXT: s_or_b32 s4, vcc_lo, s4
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
; GFX11-NEXT: s_cbranch_execnz .LBB6_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -1649,7 +1649,7 @@ define float @buffer_fat_ptr_agent_atomic_fadd_ret_f32__offset__amdgpu_no_remote
; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v0, v5
; GFX11-NEXT: s_or_b32 s4, vcc_lo, s4
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
; GFX11-NEXT: s_cbranch_execnz .LBB7_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -1853,6 +1853,7 @@ define double @buffer_fat_ptr_agent_atomic_fadd_ret_f64__offset__amdgpu_no_fine_
; GFX12-NEXT: s_or_b32 s4, vcc_lo, s4
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB8_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s4
@@ -1892,7 +1893,7 @@ define double @buffer_fat_ptr_agent_atomic_fadd_ret_f64__offset__amdgpu_no_fine_
; GFX11-NEXT: v_cmp_eq_u64_e32 vcc_lo, v[0:1], v[8:9]
; GFX11-NEXT: v_dual_mov_b32 v9, v1 :: v_dual_mov_b32 v8, v0
; GFX11-NEXT: s_or_b32 s4, vcc_lo, s4
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
; GFX11-NEXT: s_cbranch_execnz .LBB8_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -2095,6 +2096,7 @@ define void @buffer_fat_ptr_agent_atomic_fadd_noret_f64__offset__amdgpu_no_fine_
; GFX12-NEXT: s_or_b32 s4, vcc_lo, s4
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB9_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s4
@@ -2133,7 +2135,7 @@ define void @buffer_fat_ptr_agent_atomic_fadd_noret_f64__offset__amdgpu_no_fine_
; GFX11-NEXT: v_cmp_eq_u64_e32 vcc_lo, v[7:8], v[4:5]
; GFX11-NEXT: v_dual_mov_b32 v4, v7 :: v_dual_mov_b32 v5, v8
; GFX11-NEXT: s_or_b32 s4, vcc_lo, s4
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
; GFX11-NEXT: s_cbranch_execnz .LBB9_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -2363,6 +2365,7 @@ define double @buffer_fat_ptr_agent_atomic_fadd_ret_f64__offset__waterfall__amdg
; GFX12-NEXT: s_or_b32 s4, vcc_lo, s4
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB10_3
; GFX12-NEXT: ; %bb.6: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s4
@@ -2455,7 +2458,7 @@ define double @buffer_fat_ptr_agent_atomic_fadd_ret_f64__offset__waterfall__amdg
; GFX11-NEXT: buffer_gl1_inv
; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: s_or_b32 s4, vcc_lo, s4
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
; GFX11-NEXT: s_cbranch_execnz .LBB10_3
; GFX11-NEXT: ; %bb.6: ; %atomicrmw.end
@@ -2840,6 +2843,7 @@ define double @buffer_fat_ptr_agent_atomic_fadd_ret_f64__offset__amdgpu_no_remot
; GFX12-NEXT: s_or_b32 s4, vcc_lo, s4
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB11_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s4
@@ -2879,7 +2883,7 @@ define double @buffer_fat_ptr_agent_atomic_fadd_ret_f64__offset__amdgpu_no_remot
; GFX11-NEXT: v_cmp_eq_u64_e32 vcc_lo, v[0:1], v[8:9]
; GFX11-NEXT: v_dual_mov_b32 v9, v1 :: v_dual_mov_b32 v8, v0
; GFX11-NEXT: s_or_b32 s4, vcc_lo, s4
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
; GFX11-NEXT: s_cbranch_execnz .LBB11_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -3101,6 +3105,7 @@ define double @buffer_fat_ptr_agent_atomic_fadd_ret_f64__offset__amdgpu_no_fine_
; GFX12-NEXT: s_or_b32 s4, vcc_lo, s4
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB12_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s4
@@ -3140,7 +3145,7 @@ define double @buffer_fat_ptr_agent_atomic_fadd_ret_f64__offset__amdgpu_no_fine_
; GFX11-NEXT: v_cmp_eq_u64_e32 vcc_lo, v[0:1], v[8:9]
; GFX11-NEXT: v_dual_mov_b32 v9, v1 :: v_dual_mov_b32 v8, v0
; GFX11-NEXT: s_or_b32 s4, vcc_lo, s4
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
; GFX11-NEXT: s_cbranch_execnz .LBB12_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -3362,9 +3367,11 @@ define half @buffer_fat_ptr_agent_atomic_fadd_ret_f16__offset__amdgpu_no_fine_gr
; GFX12-TRUE16-NEXT: s_or_b32 s5, vcc_lo, s5
; GFX12-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s5
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_cbranch_execnz .LBB13_1
; GFX12-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s5
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: v_lshrrev_b32_e32 v0, s4, v3
; GFX12-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -3410,9 +3417,11 @@ define half @buffer_fat_ptr_agent_atomic_fadd_ret_f16__offset__amdgpu_no_fine_gr
; GFX12-FAKE16-NEXT: s_or_b32 s5, vcc_lo, s5
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s5
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: s_cbranch_execnz .LBB13_1
; GFX12-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s5
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: v_lshrrev_b32_e32 v0, s4, v3
; GFX12-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -3485,11 +3494,12 @@ define half @buffer_fat_ptr_agent_atomic_fadd_ret_f16__offset__amdgpu_no_fine_gr
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v2
; GFX11-TRUE16-NEXT: v_mov_b32_e32 v2, v3
; GFX11-TRUE16-NEXT: s_or_b32 s5, vcc_lo, s5
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s5
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB13_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s5
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v0, s4, v3
; GFX11-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -3527,11 +3537,12 @@ define half @buffer_fat_ptr_agent_atomic_fadd_ret_f16__offset__amdgpu_no_fine_gr
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v2
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v2, v3
; GFX11-FAKE16-NEXT: s_or_b32 s5, vcc_lo, s5
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s5
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB13_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s5
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v0, s4, v3
; GFX11-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -3798,6 +3809,7 @@ define void @buffer_fat_ptr_agent_atomic_fadd_noret_f16__offset__amdgpu_no_fine_
; GFX12-TRUE16-NEXT: s_or_b32 s5, vcc_lo, s5
; GFX12-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s5
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_cbranch_execnz .LBB14_1
; GFX12-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s5
@@ -3845,6 +3857,7 @@ define void @buffer_fat_ptr_agent_atomic_fadd_noret_f16__offset__amdgpu_no_fine_
; GFX12-FAKE16-NEXT: s_or_b32 s5, vcc_lo, s5
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s5
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: s_cbranch_execnz .LBB14_1
; GFX12-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s5
@@ -3918,7 +3931,7 @@ define void @buffer_fat_ptr_agent_atomic_fadd_noret_f16__offset__amdgpu_no_fine_
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v4, v2
; GFX11-TRUE16-NEXT: v_mov_b32_e32 v2, v4
; GFX11-TRUE16-NEXT: s_or_b32 s5, vcc_lo, s5
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s5
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB14_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -3959,7 +3972,7 @@ define void @buffer_fat_ptr_agent_atomic_fadd_noret_f16__offset__amdgpu_no_fine_
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v4, v2
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v2, v4
; GFX11-FAKE16-NEXT: s_or_b32 s5, vcc_lo, s5
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s5
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB14_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -4254,9 +4267,11 @@ define half @buffer_fat_ptr_agent_atomic_fadd_ret_f16__offset__waterfall__amdgpu
; GFX12-TRUE16-NEXT: s_or_b32 s4, vcc_lo, s4
; GFX12-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_cbranch_execnz .LBB15_3
; GFX12-TRUE16-NEXT: ; %bb.6: ; %atomicrmw.end
; GFX12-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s4
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: v_lshrrev_b32_e32 v0, v4, v8
; GFX12-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -4333,9 +4348,11 @@ define half @buffer_fat_ptr_agent_atomic_fadd_ret_f16__offset__waterfall__amdgpu
; GFX12-FAKE16-NEXT: s_or_b32 s4, vcc_lo, s4
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: s_cbranch_execnz .LBB15_3
; GFX12-FAKE16-NEXT: ; %bb.6: ; %atomicrmw.end
; GFX12-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s4
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: v_lshrrev_b32_e32 v0, v4, v8
; GFX12-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -4472,11 +4489,12 @@ define half @buffer_fat_ptr_agent_atomic_fadd_ret_f16__offset__waterfall__amdgpu
; GFX11-TRUE16-NEXT: buffer_gl1_inv
; GFX11-TRUE16-NEXT: buffer_gl0_inv
; GFX11-TRUE16-NEXT: s_or_b32 s4, vcc_lo, s4
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB15_3
; GFX11-TRUE16-NEXT: ; %bb.6: ; %atomicrmw.end
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s4
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v0, v4, v8
; GFX11-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -4545,11 +4563,12 @@ define half @buffer_fat_ptr_agent_atomic_fadd_ret_f16__offset__waterfall__amdgpu
; GFX11-FAKE16-NEXT: buffer_gl1_inv
; GFX11-FAKE16-NEXT: buffer_gl0_inv
; GFX11-FAKE16-NEXT: s_or_b32 s4, vcc_lo, s4
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB15_3
; GFX11-FAKE16-NEXT: ; %bb.6: ; %atomicrmw.end
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s4
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v0, v4, v8
; GFX11-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -5018,9 +5037,11 @@ define bfloat @buffer_fat_ptr_agent_atomic_fadd_ret_bf16__offset__amdgpu_no_fine
; GFX12-NEXT: s_or_b32 s5, vcc_lo, s5
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s5
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB16_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s5
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: v_lshrrev_b32_e32 v0, s4, v2
; GFX12-NEXT: s_setpc_b64 s[30:31]
;
@@ -5111,11 +5132,12 @@ define bfloat @buffer_fat_ptr_agent_atomic_fadd_ret_bf16__offset__amdgpu_no_fine
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v1
; GFX11-NEXT: v_mov_b32_e32 v1, v2
; GFX11-NEXT: s_or_b32 s5, vcc_lo, s5
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s5
; GFX11-NEXT: s_cbranch_execnz .LBB16_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s5
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: v_lshrrev_b32_e32 v0, s4, v2
; GFX11-NEXT: s_setpc_b64 s[30:31]
;
@@ -5424,6 +5446,7 @@ define void @buffer_fat_ptr_agent_atomic_fadd_noret_bf16__offset__amdgpu_no_fine
; GFX12-NEXT: s_or_b32 s5, vcc_lo, s5
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s5
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB17_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s5
@@ -5515,7 +5538,7 @@ define void @buffer_fat_ptr_agent_atomic_fadd_noret_bf16__offset__amdgpu_no_fine
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v4, v1
; GFX11-NEXT: v_mov_b32_e32 v1, v4
; GFX11-NEXT: s_or_b32 s5, vcc_lo, s5
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s5
; GFX11-NEXT: s_cbranch_execnz .LBB17_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -5802,6 +5825,7 @@ define bfloat @buffer_fat_ptr_agent_atomic_fadd_ret_bf16__offset__waterfall__amd
; GFX12-NEXT: s_cbranch_execnz .LBB18_1
; GFX12-NEXT: ; %bb.2:
; GFX12-NEXT: s_mov_b32 exec_lo, s4
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: v_lshlrev_b32_e32 v10, 16, v5
; GFX12-NEXT: s_mov_b32 s4, 0
; GFX12-NEXT: .LBB18_3: ; %atomicrmw.start
@@ -5852,9 +5876,11 @@ define bfloat @buffer_fat_ptr_agent_atomic_fadd_ret_bf16__offset__waterfall__amd
; GFX12-NEXT: s_or_b32 s4, vcc_lo, s4
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB18_3
; GFX12-NEXT: ; %bb.6: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s4
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: v_lshrrev_b32_e32 v0, v7, v4
; GFX12-NEXT: s_setpc_b64 s[30:31]
;
@@ -5960,6 +5986,7 @@ define bfloat @buffer_fat_ptr_agent_atomic_fadd_ret_bf16__offset__waterfall__amd
; GFX11-NEXT: s_cbranch_execnz .LBB18_1
; GFX11-NEXT: ; %bb.2:
; GFX11-NEXT: s_mov_b32 exec_lo, s5
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: v_lshlrev_b32_e32 v10, 16, v5
; GFX11-NEXT: s_set_inst_prefetch_distance 0x1
; GFX11-NEXT: .p2align 6
@@ -6008,12 +6035,13 @@ define bfloat @buffer_fat_ptr_agent_atomic_fadd_ret_bf16__offset__waterfall__amd
; GFX11-NEXT: buffer_gl1_inv
; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: s_or_b32 s4, vcc_lo, s4
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
; GFX11-NEXT: s_cbranch_execnz .LBB18_3
; GFX11-NEXT: ; %bb.6: ; %atomicrmw.end
; GFX11-NEXT: s_set_inst_prefetch_distance 0x2
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s4
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: v_lshrrev_b32_e32 v0, v7, v4
; GFX11-NEXT: s_setpc_b64 s[30:31]
;
@@ -6512,7 +6540,7 @@ define <2 x half> @buffer_fat_ptr_agent_atomic_fadd_ret_v2f16__offset__amdgpu_no
; GFX11-TRUE16-NEXT: buffer_gl0_inv
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v0, v5
; GFX11-TRUE16-NEXT: s_or_b32 s4, vcc_lo, s4
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB19_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -6547,7 +6575,7 @@ define <2 x half> @buffer_fat_ptr_agent_atomic_fadd_ret_v2f16__offset__amdgpu_no
; GFX11-FAKE16-NEXT: buffer_gl0_inv
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v0, v5
; GFX11-FAKE16-NEXT: s_or_b32 s4, vcc_lo, s4
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB19_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -6803,7 +6831,7 @@ define void @buffer_fat_ptr_agent_atomic_fadd_noret_v2f16__offset__amdgpu_no_fin
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v4, v2
; GFX11-TRUE16-NEXT: v_mov_b32_e32 v2, v4
; GFX11-TRUE16-NEXT: s_or_b32 s4, vcc_lo, s4
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB20_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -6838,7 +6866,7 @@ define void @buffer_fat_ptr_agent_atomic_fadd_noret_v2f16__offset__amdgpu_no_fin
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v4, v2
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v2, v4
; GFX11-FAKE16-NEXT: s_or_b32 s4, vcc_lo, s4
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB20_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -7145,11 +7173,12 @@ define <2 x half> @buffer_fat_ptr_agent_atomic_fadd_ret_v2f16__offset__waterfall
; GFX11-TRUE16-NEXT: buffer_gl1_inv
; GFX11-TRUE16-NEXT: buffer_gl0_inv
; GFX11-TRUE16-NEXT: s_or_b32 s4, vcc_lo, s4
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB21_5
; GFX11-TRUE16-NEXT: ; %bb.8: ; %atomicrmw.end
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s4
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_mov_b32_e32 v0, v6
; GFX11-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -7221,11 +7250,12 @@ define <2 x half> @buffer_fat_ptr_agent_atomic_fadd_ret_v2f16__offset__waterfall
; GFX11-FAKE16-NEXT: buffer_gl1_inv
; GFX11-FAKE16-NEXT: buffer_gl0_inv
; GFX11-FAKE16-NEXT: s_or_b32 s4, vcc_lo, s4
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB21_5
; GFX11-FAKE16-NEXT: ; %bb.8: ; %atomicrmw.end
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s4
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v0, v6
; GFX11-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -7711,7 +7741,7 @@ define <2 x half> @buffer_fat_ptr_agent_atomic_fadd_ret_v2f16__offset(ptr addrsp
; GFX11-TRUE16-NEXT: buffer_gl0_inv
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v0, v5
; GFX11-TRUE16-NEXT: s_or_b32 s4, vcc_lo, s4
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB22_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -7746,7 +7776,7 @@ define <2 x half> @buffer_fat_ptr_agent_atomic_fadd_ret_v2f16__offset(ptr addrsp
; GFX11-FAKE16-NEXT: buffer_gl0_inv
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v0, v5
; GFX11-FAKE16-NEXT: s_or_b32 s4, vcc_lo, s4
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB22_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -8023,7 +8053,7 @@ define void @buffer_fat_ptr_agent_atomic_fadd_noret_v2f16__offset(ptr addrspace(
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v4, v2
; GFX11-TRUE16-NEXT: v_mov_b32_e32 v2, v4
; GFX11-TRUE16-NEXT: s_or_b32 s4, vcc_lo, s4
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB23_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -8058,7 +8088,7 @@ define void @buffer_fat_ptr_agent_atomic_fadd_noret_v2f16__offset(ptr addrspace(
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v4, v2
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v2, v4
; GFX11-FAKE16-NEXT: s_or_b32 s4, vcc_lo, s4
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB23_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -8325,7 +8355,7 @@ define <2 x half> @buffer_fat_ptr_agent_atomic_fadd_ret_v2f16__offset__amdgpu_no
; GFX11-TRUE16-NEXT: buffer_gl0_inv
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v0, v5
; GFX11-TRUE16-NEXT: s_or_b32 s4, vcc_lo, s4
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB24_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -8360,7 +8390,7 @@ define <2 x half> @buffer_fat_ptr_agent_atomic_fadd_ret_v2f16__offset__amdgpu_no
; GFX11-FAKE16-NEXT: buffer_gl0_inv
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v0, v5
; GFX11-FAKE16-NEXT: s_or_b32 s4, vcc_lo, s4
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB24_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -8637,7 +8667,7 @@ define void @buffer_fat_ptr_agent_atomic_fadd_noret_v2f16__offset__amdgpu_no_rem
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v4, v2
; GFX11-TRUE16-NEXT: v_mov_b32_e32 v2, v4
; GFX11-TRUE16-NEXT: s_or_b32 s4, vcc_lo, s4
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB25_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -8672,7 +8702,7 @@ define void @buffer_fat_ptr_agent_atomic_fadd_noret_v2f16__offset__amdgpu_no_rem
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v4, v2
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v2, v4
; GFX11-FAKE16-NEXT: s_or_b32 s4, vcc_lo, s4
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB25_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -9006,7 +9036,7 @@ define <2 x bfloat> @buffer_fat_ptr_agent_atomic_fadd_ret_v2bf16__offset__amdgpu
; GFX11-TRUE16-NEXT: buffer_gl0_inv
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v0, v6
; GFX11-TRUE16-NEXT: s_or_b32 s4, vcc_lo, s4
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB26_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -9063,7 +9093,7 @@ define <2 x bfloat> @buffer_fat_ptr_agent_atomic_fadd_ret_v2bf16__offset__amdgpu
; GFX11-FAKE16-NEXT: buffer_gl0_inv
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v0, v6
; GFX11-FAKE16-NEXT: s_or_b32 s5, vcc_lo, s5
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s5
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB26_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -9480,7 +9510,7 @@ define void @buffer_fat_ptr_agent_atomic_fadd_noret_v2bf16__offset__amdgpu_no_fi
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v5, v1
; GFX11-TRUE16-NEXT: v_mov_b32_e32 v1, v5
; GFX11-TRUE16-NEXT: s_or_b32 s4, vcc_lo, s4
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB27_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -9522,7 +9552,7 @@ define void @buffer_fat_ptr_agent_atomic_fadd_noret_v2bf16__offset__amdgpu_no_fi
; GFX11-FAKE16-NEXT: v_add3_u32 v7, v7, v5, 0x7fff
; GFX11-FAKE16-NEXT: v_add3_u32 v6, v6, v0, 0x7fff
; GFX11-FAKE16-NEXT: v_cmp_u_f32_e64 s4, v0, v0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_cndmask_b32_e32 v5, v7, v9, vcc_lo
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v0, v6, v8, s4
; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
@@ -9535,7 +9565,7 @@ define void @buffer_fat_ptr_agent_atomic_fadd_noret_v2bf16__offset__amdgpu_no_fi
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v5, v1
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v1, v5
; GFX11-FAKE16-NEXT: s_or_b32 s5, vcc_lo, s5
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s5
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB27_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -10054,12 +10084,13 @@ define <2 x bfloat> @buffer_fat_ptr_agent_atomic_fadd_ret_v2bf16__offset__waterf
; GFX11-TRUE16-NEXT: buffer_gl1_inv
; GFX11-TRUE16-NEXT: buffer_gl0_inv
; GFX11-TRUE16-NEXT: s_or_b32 s4, vcc_lo, s4
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB28_5
; GFX11-TRUE16-NEXT: ; %bb.8: ; %atomicrmw.end
; GFX11-TRUE16-NEXT: s_set_inst_prefetch_distance 0x2
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s4
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_mov_b32_e32 v0, v5
; GFX11-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -10150,12 +10181,13 @@ define <2 x bfloat> @buffer_fat_ptr_agent_atomic_fadd_ret_v2bf16__offset__waterf
; GFX11-FAKE16-NEXT: buffer_gl1_inv
; GFX11-FAKE16-NEXT: buffer_gl0_inv
; GFX11-FAKE16-NEXT: s_or_b32 s4, vcc_lo, s4
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB28_5
; GFX11-FAKE16-NEXT: ; %bb.8: ; %atomicrmw.end
; GFX11-FAKE16-NEXT: s_set_inst_prefetch_distance 0x2
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s4
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v0, v5
; GFX11-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -10831,7 +10863,7 @@ define <2 x bfloat> @buffer_fat_ptr_agent_atomic_fadd_ret_v2bf16__offset(ptr add
; GFX11-TRUE16-NEXT: buffer_gl0_inv
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v0, v6
; GFX11-TRUE16-NEXT: s_or_b32 s4, vcc_lo, s4
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB29_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -10888,7 +10920,7 @@ define <2 x bfloat> @buffer_fat_ptr_agent_atomic_fadd_ret_v2bf16__offset(ptr add
; GFX11-FAKE16-NEXT: buffer_gl0_inv
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v0, v6
; GFX11-FAKE16-NEXT: s_or_b32 s5, vcc_lo, s5
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s5
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB29_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -11305,7 +11337,7 @@ define void @buffer_fat_ptr_agent_atomic_fadd_noret_v2bf16__offset(ptr addrspace
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v5, v1
; GFX11-TRUE16-NEXT: v_mov_b32_e32 v1, v5
; GFX11-TRUE16-NEXT: s_or_b32 s4, vcc_lo, s4
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB30_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -11347,7 +11379,7 @@ define void @buffer_fat_ptr_agent_atomic_fadd_noret_v2bf16__offset(ptr addrspace
; GFX11-FAKE16-NEXT: v_add3_u32 v7, v7, v5, 0x7fff
; GFX11-FAKE16-NEXT: v_add3_u32 v6, v6, v0, 0x7fff
; GFX11-FAKE16-NEXT: v_cmp_u_f32_e64 s4, v0, v0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_cndmask_b32_e32 v5, v7, v9, vcc_lo
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v0, v6, v8, s4
; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
@@ -11360,7 +11392,7 @@ define void @buffer_fat_ptr_agent_atomic_fadd_noret_v2bf16__offset(ptr addrspace
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v5, v1
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v1, v5
; GFX11-FAKE16-NEXT: s_or_b32 s5, vcc_lo, s5
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s5
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB30_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -11773,7 +11805,7 @@ define <2 x bfloat> @buffer_fat_ptr_agent_atomic_fadd_ret_v2bf16__offset__amdgpu
; GFX11-TRUE16-NEXT: buffer_gl0_inv
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v0, v6
; GFX11-TRUE16-NEXT: s_or_b32 s4, vcc_lo, s4
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB31_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -11830,7 +11862,7 @@ define <2 x bfloat> @buffer_fat_ptr_agent_atomic_fadd_ret_v2bf16__offset__amdgpu
; GFX11-FAKE16-NEXT: buffer_gl0_inv
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v0, v6
; GFX11-FAKE16-NEXT: s_or_b32 s5, vcc_lo, s5
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s5
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB31_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -12247,7 +12279,7 @@ define void @buffer_fat_ptr_agent_atomic_fadd_noret_v2bf16__offset__amdgpu_no_re
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v5, v1
; GFX11-TRUE16-NEXT: v_mov_b32_e32 v1, v5
; GFX11-TRUE16-NEXT: s_or_b32 s4, vcc_lo, s4
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB32_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -12289,7 +12321,7 @@ define void @buffer_fat_ptr_agent_atomic_fadd_noret_v2bf16__offset__amdgpu_no_re
; GFX11-FAKE16-NEXT: v_add3_u32 v7, v7, v5, 0x7fff
; GFX11-FAKE16-NEXT: v_add3_u32 v6, v6, v0, 0x7fff
; GFX11-FAKE16-NEXT: v_cmp_u_f32_e64 s4, v0, v0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_cndmask_b32_e32 v5, v7, v9, vcc_lo
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v0, v6, v8, s4
; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
@@ -12302,7 +12334,7 @@ define void @buffer_fat_ptr_agent_atomic_fadd_noret_v2bf16__offset__amdgpu_no_re
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v5, v1
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v1, v5
; GFX11-FAKE16-NEXT: s_or_b32 s5, vcc_lo, s5
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s5
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB32_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -12711,7 +12743,7 @@ define void @buffer_fat_ptr_agent_atomic_fadd_noret_v2bf16__offset__amdgpu_no_fi
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v5, v1
; GFX11-TRUE16-NEXT: v_mov_b32_e32 v1, v5
; GFX11-TRUE16-NEXT: s_or_b32 s4, vcc_lo, s4
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB33_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -12753,7 +12785,7 @@ define void @buffer_fat_ptr_agent_atomic_fadd_noret_v2bf16__offset__amdgpu_no_fi
; GFX11-FAKE16-NEXT: v_add3_u32 v7, v7, v5, 0x7fff
; GFX11-FAKE16-NEXT: v_add3_u32 v6, v6, v0, 0x7fff
; GFX11-FAKE16-NEXT: v_cmp_u_f32_e64 s4, v0, v0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_cndmask_b32_e32 v5, v7, v9, vcc_lo
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v0, v6, v8, s4
; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
@@ -12766,7 +12798,7 @@ define void @buffer_fat_ptr_agent_atomic_fadd_noret_v2bf16__offset__amdgpu_no_fi
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v5, v1
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v1, v5
; GFX11-FAKE16-NEXT: s_or_b32 s5, vcc_lo, s5
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s5
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB33_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
diff --git a/llvm/test/CodeGen/AMDGPU/buffer-fat-pointer-atomicrmw-fmax.ll b/llvm/test/CodeGen/AMDGPU/buffer-fat-pointer-atomicrmw-fmax.ll
index fbfbf7249f2fd5..519222e24e09cd 100644
--- a/llvm/test/CodeGen/AMDGPU/buffer-fat-pointer-atomicrmw-fmax.ll
+++ b/llvm/test/CodeGen/AMDGPU/buffer-fat-pointer-atomicrmw-fmax.ll
@@ -805,7 +805,7 @@ define float @buffer_fat_ptr_agent_atomic_fmax_ret_f32__offset__amdgpu_no_remote
; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v0, v5
; GFX11-NEXT: s_or_b32 s4, vcc_lo, s4
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
; GFX11-NEXT: s_cbranch_execnz .LBB3_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -1195,6 +1195,7 @@ define double @buffer_fat_ptr_agent_atomic_fmax_ret_f64__offset__amdgpu_no_fine_
; GFX12-NEXT: s_or_b32 s4, vcc_lo, s4
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB5_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s4
@@ -1236,7 +1237,7 @@ define double @buffer_fat_ptr_agent_atomic_fmax_ret_f64__offset__amdgpu_no_fine_
; GFX11-NEXT: v_cmp_eq_u64_e32 vcc_lo, v[0:1], v[4:5]
; GFX11-NEXT: v_dual_mov_b32 v5, v1 :: v_dual_mov_b32 v4, v0
; GFX11-NEXT: s_or_b32 s4, vcc_lo, s4
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
; GFX11-NEXT: s_cbranch_execnz .LBB5_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -1376,6 +1377,7 @@ define void @buffer_fat_ptr_agent_atomic_fmax_noret_f64__offset__amdgpu_no_fine_
; GFX12-NEXT: s_or_b32 s4, vcc_lo, s4
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB6_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s4
@@ -1416,7 +1418,7 @@ define void @buffer_fat_ptr_agent_atomic_fmax_noret_f64__offset__amdgpu_no_fine_
; GFX11-NEXT: v_cmp_eq_u64_e32 vcc_lo, v[7:8], v[2:3]
; GFX11-NEXT: v_dual_mov_b32 v2, v7 :: v_dual_mov_b32 v3, v8
; GFX11-NEXT: s_or_b32 s4, vcc_lo, s4
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
; GFX11-NEXT: s_cbranch_execnz .LBB6_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -1555,6 +1557,7 @@ define double @buffer_fat_ptr_agent_atomic_fmax_ret_f64__offset__waterfall__amdg
; GFX12-NEXT: s_cbranch_execnz .LBB7_1
; GFX12-NEXT: ; %bb.2:
; GFX12-NEXT: s_mov_b32 exec_lo, s4
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: v_max_num_f64_e32 v[5:6], v[5:6], v[5:6]
; GFX12-NEXT: s_mov_b32 s4, 0
; GFX12-NEXT: .LBB7_3: ; %atomicrmw.start
@@ -1593,6 +1596,7 @@ define double @buffer_fat_ptr_agent_atomic_fmax_ret_f64__offset__waterfall__amdg
; GFX12-NEXT: s_or_b32 s4, vcc_lo, s4
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB7_3
; GFX12-NEXT: ; %bb.6: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s4
@@ -1652,6 +1656,7 @@ define double @buffer_fat_ptr_agent_atomic_fmax_ret_f64__offset__waterfall__amdg
; GFX11-NEXT: s_cbranch_execnz .LBB7_1
; GFX11-NEXT: ; %bb.2:
; GFX11-NEXT: s_mov_b32 exec_lo, s5
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: v_max_f64 v[5:6], v[5:6], v[5:6]
; GFX11-NEXT: .p2align 6
; GFX11-NEXT: .LBB7_3: ; %atomicrmw.start
@@ -1687,7 +1692,7 @@ define double @buffer_fat_ptr_agent_atomic_fmax_ret_f64__offset__waterfall__amdg
; GFX11-NEXT: buffer_gl1_inv
; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: s_or_b32 s4, vcc_lo, s4
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
; GFX11-NEXT: s_cbranch_execnz .LBB7_3
; GFX11-NEXT: ; %bb.6: ; %atomicrmw.end
@@ -1971,6 +1976,7 @@ define double @buffer_fat_ptr_agent_atomic_fmax_ret_f64__offset__amdgpu_no_remot
; GFX12-NEXT: s_or_b32 s4, vcc_lo, s4
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB8_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s4
@@ -2012,7 +2018,7 @@ define double @buffer_fat_ptr_agent_atomic_fmax_ret_f64__offset__amdgpu_no_remot
; GFX11-NEXT: v_cmp_eq_u64_e32 vcc_lo, v[0:1], v[4:5]
; GFX11-NEXT: v_dual_mov_b32 v5, v1 :: v_dual_mov_b32 v4, v0
; GFX11-NEXT: s_or_b32 s4, vcc_lo, s4
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
; GFX11-NEXT: s_cbranch_execnz .LBB8_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -2235,6 +2241,7 @@ define double @buffer_fat_ptr_agent_atomic_fmax_ret_f64__offset__amdgpu_no_fine_
; GFX12-NEXT: s_or_b32 s4, vcc_lo, s4
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB9_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s4
@@ -2276,7 +2283,7 @@ define double @buffer_fat_ptr_agent_atomic_fmax_ret_f64__offset__amdgpu_no_fine_
; GFX11-NEXT: v_cmp_eq_u64_e32 vcc_lo, v[0:1], v[4:5]
; GFX11-NEXT: v_dual_mov_b32 v5, v1 :: v_dual_mov_b32 v4, v0
; GFX11-NEXT: s_or_b32 s4, vcc_lo, s4
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
; GFX11-NEXT: s_cbranch_execnz .LBB9_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -2435,9 +2442,11 @@ define half @buffer_fat_ptr_agent_atomic_fmax_ret_f16__offset__amdgpu_no_fine_gr
; GFX12-TRUE16-NEXT: s_or_b32 s5, vcc_lo, s5
; GFX12-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s5
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_cbranch_execnz .LBB10_1
; GFX12-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s5
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: v_lshrrev_b32_e32 v0, s4, v3
; GFX12-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -2485,9 +2494,11 @@ define half @buffer_fat_ptr_agent_atomic_fmax_ret_f16__offset__amdgpu_no_fine_gr
; GFX12-FAKE16-NEXT: s_or_b32 s5, vcc_lo, s5
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s5
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: s_cbranch_execnz .LBB10_1
; GFX12-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s5
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: v_lshrrev_b32_e32 v0, s4, v2
; GFX12-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -2564,11 +2575,12 @@ define half @buffer_fat_ptr_agent_atomic_fmax_ret_f16__offset__amdgpu_no_fine_gr
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v2
; GFX11-TRUE16-NEXT: v_mov_b32_e32 v2, v3
; GFX11-TRUE16-NEXT: s_or_b32 s5, vcc_lo, s5
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s5
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB10_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s5
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v0, s4, v3
; GFX11-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -2609,11 +2621,12 @@ define half @buffer_fat_ptr_agent_atomic_fmax_ret_f16__offset__amdgpu_no_fine_gr
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v1
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v1, v2
; GFX11-FAKE16-NEXT: s_or_b32 s5, vcc_lo, s5
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s5
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB10_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s5
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v0, s4, v2
; GFX11-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -2890,6 +2903,7 @@ define void @buffer_fat_ptr_agent_atomic_fmax_noret_f16__offset__amdgpu_no_fine_
; GFX12-TRUE16-NEXT: s_or_b32 s5, vcc_lo, s5
; GFX12-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s5
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_cbranch_execnz .LBB11_1
; GFX12-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s5
@@ -2939,6 +2953,7 @@ define void @buffer_fat_ptr_agent_atomic_fmax_noret_f16__offset__amdgpu_no_fine_
; GFX12-FAKE16-NEXT: s_or_b32 s5, vcc_lo, s5
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s5
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: s_cbranch_execnz .LBB11_1
; GFX12-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s5
@@ -3016,7 +3031,7 @@ define void @buffer_fat_ptr_agent_atomic_fmax_noret_f16__offset__amdgpu_no_fine_
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v4, v2
; GFX11-TRUE16-NEXT: v_mov_b32_e32 v2, v4
; GFX11-TRUE16-NEXT: s_or_b32 s5, vcc_lo, s5
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s5
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB11_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -3060,7 +3075,7 @@ define void @buffer_fat_ptr_agent_atomic_fmax_noret_f16__offset__amdgpu_no_fine_
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v4, v1
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v1, v4
; GFX11-FAKE16-NEXT: s_or_b32 s5, vcc_lo, s5
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s5
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB11_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -3323,6 +3338,7 @@ define half @buffer_fat_ptr_agent_atomic_fmax_ret_f16__offset__waterfall__amdgpu
; GFX12-TRUE16-NEXT: s_cbranch_execnz .LBB12_1
; GFX12-TRUE16-NEXT: ; %bb.2:
; GFX12-TRUE16-NEXT: s_mov_b32 exec_lo, s4
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: v_max_num_f16_e32 v4.l, v5.l, v5.l
; GFX12-TRUE16-NEXT: s_mov_b32 s4, 0
; GFX12-TRUE16-NEXT: .LBB12_3: ; %atomicrmw.start
@@ -3366,9 +3382,11 @@ define half @buffer_fat_ptr_agent_atomic_fmax_ret_f16__offset__waterfall__amdgpu
; GFX12-TRUE16-NEXT: s_or_b32 s4, vcc_lo, s4
; GFX12-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_cbranch_execnz .LBB12_3
; GFX12-TRUE16-NEXT: ; %bb.6: ; %atomicrmw.end
; GFX12-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s4
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: v_lshrrev_b32_e32 v0, v9, v7
; GFX12-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -3405,6 +3423,7 @@ define half @buffer_fat_ptr_agent_atomic_fmax_ret_f16__offset__waterfall__amdgpu
; GFX12-FAKE16-NEXT: s_cbranch_execnz .LBB12_1
; GFX12-FAKE16-NEXT: ; %bb.2:
; GFX12-FAKE16-NEXT: s_mov_b32 exec_lo, s4
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: v_max_num_f16_e32 v10, v5, v5
; GFX12-FAKE16-NEXT: s_mov_b32 s4, 0
; GFX12-FAKE16-NEXT: .LBB12_3: ; %atomicrmw.start
@@ -3448,9 +3467,11 @@ define half @buffer_fat_ptr_agent_atomic_fmax_ret_f16__offset__waterfall__amdgpu
; GFX12-FAKE16-NEXT: s_or_b32 s4, vcc_lo, s4
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: s_cbranch_execnz .LBB12_3
; GFX12-FAKE16-NEXT: ; %bb.6: ; %atomicrmw.end
; GFX12-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s4
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: v_lshrrev_b32_e32 v0, v7, v4
; GFX12-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -3551,6 +3572,7 @@ define half @buffer_fat_ptr_agent_atomic_fmax_ret_f16__offset__waterfall__amdgpu
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB12_1
; GFX11-TRUE16-NEXT: ; %bb.2:
; GFX11-TRUE16-NEXT: s_mov_b32 exec_lo, s5
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_max_f16_e32 v4.l, v5.l, v5.l
; GFX11-TRUE16-NEXT: .p2align 6
; GFX11-TRUE16-NEXT: .LBB12_3: ; %atomicrmw.start
@@ -3591,11 +3613,12 @@ define half @buffer_fat_ptr_agent_atomic_fmax_ret_f16__offset__waterfall__amdgpu
; GFX11-TRUE16-NEXT: buffer_gl1_inv
; GFX11-TRUE16-NEXT: buffer_gl0_inv
; GFX11-TRUE16-NEXT: s_or_b32 s4, vcc_lo, s4
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB12_3
; GFX11-TRUE16-NEXT: ; %bb.6: ; %atomicrmw.end
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s4
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v0, v9, v7
; GFX11-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -3626,6 +3649,7 @@ define half @buffer_fat_ptr_agent_atomic_fmax_ret_f16__offset__waterfall__amdgpu
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB12_1
; GFX11-FAKE16-NEXT: ; %bb.2:
; GFX11-FAKE16-NEXT: s_mov_b32 exec_lo, s5
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_max_f16_e32 v10, v5, v5
; GFX11-FAKE16-NEXT: .p2align 6
; GFX11-FAKE16-NEXT: .LBB12_3: ; %atomicrmw.start
@@ -3666,11 +3690,12 @@ define half @buffer_fat_ptr_agent_atomic_fmax_ret_f16__offset__waterfall__amdgpu
; GFX11-FAKE16-NEXT: buffer_gl1_inv
; GFX11-FAKE16-NEXT: buffer_gl0_inv
; GFX11-FAKE16-NEXT: s_or_b32 s4, vcc_lo, s4
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB12_3
; GFX11-FAKE16-NEXT: ; %bb.6: ; %atomicrmw.end
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s4
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v0, v7, v4
; GFX11-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -4147,9 +4172,11 @@ define bfloat @buffer_fat_ptr_agent_atomic_fmax_ret_bf16__offset__amdgpu_no_fine
; GFX12-NEXT: s_or_b32 s5, vcc_lo, s5
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s5
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB13_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s5
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: v_lshrrev_b32_e32 v0, s4, v2
; GFX12-NEXT: s_setpc_b64 s[30:31]
;
@@ -4240,11 +4267,12 @@ define bfloat @buffer_fat_ptr_agent_atomic_fmax_ret_bf16__offset__amdgpu_no_fine
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v1
; GFX11-NEXT: v_mov_b32_e32 v1, v2
; GFX11-NEXT: s_or_b32 s5, vcc_lo, s5
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s5
; GFX11-NEXT: s_cbranch_execnz .LBB13_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s5
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: v_lshrrev_b32_e32 v0, s4, v2
; GFX11-NEXT: s_setpc_b64 s[30:31]
;
@@ -4555,6 +4583,7 @@ define void @buffer_fat_ptr_agent_atomic_fmax_noret_bf16__offset__amdgpu_no_fine
; GFX12-NEXT: s_or_b32 s5, vcc_lo, s5
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s5
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB14_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s5
@@ -4646,7 +4675,7 @@ define void @buffer_fat_ptr_agent_atomic_fmax_noret_bf16__offset__amdgpu_no_fine
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v4, v1
; GFX11-NEXT: v_mov_b32_e32 v1, v4
; GFX11-NEXT: s_or_b32 s5, vcc_lo, s5
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s5
; GFX11-NEXT: s_cbranch_execnz .LBB14_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -4935,6 +4964,7 @@ define bfloat @buffer_fat_ptr_agent_atomic_fmax_ret_bf16__offset__waterfall__amd
; GFX12-NEXT: s_cbranch_execnz .LBB15_1
; GFX12-NEXT: ; %bb.2:
; GFX12-NEXT: s_mov_b32 exec_lo, s4
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: v_lshlrev_b32_e32 v10, 16, v5
; GFX12-NEXT: s_mov_b32 s4, 0
; GFX12-NEXT: .LBB15_3: ; %atomicrmw.start
@@ -4985,9 +5015,11 @@ define bfloat @buffer_fat_ptr_agent_atomic_fmax_ret_bf16__offset__waterfall__amd
; GFX12-NEXT: s_or_b32 s4, vcc_lo, s4
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB15_3
; GFX12-NEXT: ; %bb.6: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s4
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: v_lshrrev_b32_e32 v0, v7, v4
; GFX12-NEXT: s_setpc_b64 s[30:31]
;
@@ -5093,6 +5125,7 @@ define bfloat @buffer_fat_ptr_agent_atomic_fmax_ret_bf16__offset__waterfall__amd
; GFX11-NEXT: s_cbranch_execnz .LBB15_1
; GFX11-NEXT: ; %bb.2:
; GFX11-NEXT: s_mov_b32 exec_lo, s5
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: v_lshlrev_b32_e32 v10, 16, v5
; GFX11-NEXT: s_set_inst_prefetch_distance 0x1
; GFX11-NEXT: .p2align 6
@@ -5141,12 +5174,13 @@ define bfloat @buffer_fat_ptr_agent_atomic_fmax_ret_bf16__offset__waterfall__amd
; GFX11-NEXT: buffer_gl1_inv
; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: s_or_b32 s4, vcc_lo, s4
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
; GFX11-NEXT: s_cbranch_execnz .LBB15_3
; GFX11-NEXT: ; %bb.6: ; %atomicrmw.end
; GFX11-NEXT: s_set_inst_prefetch_distance 0x2
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s4
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: v_lshrrev_b32_e32 v0, v7, v4
; GFX11-NEXT: s_setpc_b64 s[30:31]
;
@@ -5630,6 +5664,7 @@ define <2 x half> @buffer_fat_ptr_agent_atomic_fmax_ret_v2f16__offset__amdgpu_no
; GFX12-TRUE16-NEXT: s_or_b32 s4, vcc_lo, s4
; GFX12-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_cbranch_execnz .LBB16_1
; GFX12-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s4
@@ -5671,6 +5706,7 @@ define <2 x half> @buffer_fat_ptr_agent_atomic_fmax_ret_v2f16__offset__amdgpu_no
; GFX12-FAKE16-NEXT: s_or_b32 s4, vcc_lo, s4
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: s_cbranch_execnz .LBB16_1
; GFX12-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s4
@@ -5737,7 +5773,7 @@ define <2 x half> @buffer_fat_ptr_agent_atomic_fmax_ret_v2f16__offset__amdgpu_no
; GFX11-TRUE16-NEXT: buffer_gl0_inv
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v0, v5
; GFX11-TRUE16-NEXT: s_or_b32 s4, vcc_lo, s4
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB16_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -5774,7 +5810,7 @@ define <2 x half> @buffer_fat_ptr_agent_atomic_fmax_ret_v2f16__offset__amdgpu_no
; GFX11-FAKE16-NEXT: buffer_gl0_inv
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v0, v5
; GFX11-FAKE16-NEXT: s_or_b32 s4, vcc_lo, s4
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB16_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -6040,6 +6076,7 @@ define void @buffer_fat_ptr_agent_atomic_fmax_noret_v2f16__offset__amdgpu_no_fin
; GFX12-TRUE16-NEXT: s_or_b32 s4, vcc_lo, s4
; GFX12-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_cbranch_execnz .LBB17_1
; GFX12-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s4
@@ -6081,6 +6118,7 @@ define void @buffer_fat_ptr_agent_atomic_fmax_noret_v2f16__offset__amdgpu_no_fin
; GFX12-FAKE16-NEXT: s_or_b32 s4, vcc_lo, s4
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: s_cbranch_execnz .LBB17_1
; GFX12-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s4
@@ -6147,7 +6185,7 @@ define void @buffer_fat_ptr_agent_atomic_fmax_noret_v2f16__offset__amdgpu_no_fin
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v4, v1
; GFX11-TRUE16-NEXT: v_mov_b32_e32 v1, v4
; GFX11-TRUE16-NEXT: s_or_b32 s4, vcc_lo, s4
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB17_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -6184,7 +6222,7 @@ define void @buffer_fat_ptr_agent_atomic_fmax_noret_v2f16__offset__amdgpu_no_fin
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v4, v1
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v1, v4
; GFX11-FAKE16-NEXT: s_or_b32 s4, vcc_lo, s4
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB17_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -6495,9 +6533,11 @@ define <2 x half> @buffer_fat_ptr_agent_atomic_fmax_ret_v2f16__offset__waterfall
; GFX12-TRUE16-NEXT: s_or_b32 s4, vcc_lo, s4
; GFX12-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_cbranch_execnz .LBB18_5
; GFX12-TRUE16-NEXT: ; %bb.8: ; %atomicrmw.end
; GFX12-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s4
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: v_mov_b32_e32 v0, v5
; GFX12-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -6585,9 +6625,11 @@ define <2 x half> @buffer_fat_ptr_agent_atomic_fmax_ret_v2f16__offset__waterfall
; GFX12-FAKE16-NEXT: s_or_b32 s4, vcc_lo, s4
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: s_cbranch_execnz .LBB18_5
; GFX12-FAKE16-NEXT: ; %bb.8: ; %atomicrmw.end
; GFX12-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s4
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: v_mov_b32_e32 v0, v5
; GFX12-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -6740,11 +6782,12 @@ define <2 x half> @buffer_fat_ptr_agent_atomic_fmax_ret_v2f16__offset__waterfall
; GFX11-TRUE16-NEXT: buffer_gl1_inv
; GFX11-TRUE16-NEXT: buffer_gl0_inv
; GFX11-TRUE16-NEXT: s_or_b32 s4, vcc_lo, s4
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB18_5
; GFX11-TRUE16-NEXT: ; %bb.8: ; %atomicrmw.end
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s4
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_mov_b32_e32 v0, v5
; GFX11-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -6819,11 +6862,12 @@ define <2 x half> @buffer_fat_ptr_agent_atomic_fmax_ret_v2f16__offset__waterfall
; GFX11-FAKE16-NEXT: buffer_gl1_inv
; GFX11-FAKE16-NEXT: buffer_gl0_inv
; GFX11-FAKE16-NEXT: s_or_b32 s4, vcc_lo, s4
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB18_5
; GFX11-FAKE16-NEXT: ; %bb.8: ; %atomicrmw.end
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s4
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v0, v5
; GFX11-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -7376,6 +7420,7 @@ define <2 x bfloat> @buffer_fat_ptr_agent_atomic_fmax_ret_v2bf16__offset__amdgpu
; GFX12-TRUE16-NEXT: s_or_b32 s4, vcc_lo, s4
; GFX12-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_cbranch_execnz .LBB19_1
; GFX12-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s4
@@ -7436,6 +7481,7 @@ define <2 x bfloat> @buffer_fat_ptr_agent_atomic_fmax_ret_v2bf16__offset__amdgpu
; GFX12-FAKE16-NEXT: s_or_b32 s5, vcc_lo, s5
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s5
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: s_cbranch_execnz .LBB19_1
; GFX12-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s5
@@ -7540,7 +7586,7 @@ define <2 x bfloat> @buffer_fat_ptr_agent_atomic_fmax_ret_v2bf16__offset__amdgpu
; GFX11-TRUE16-NEXT: buffer_gl0_inv
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v0, v6
; GFX11-TRUE16-NEXT: s_or_b32 s4, vcc_lo, s4
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB19_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -7597,7 +7643,7 @@ define <2 x bfloat> @buffer_fat_ptr_agent_atomic_fmax_ret_v2bf16__offset__amdgpu
; GFX11-FAKE16-NEXT: buffer_gl0_inv
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v0, v6
; GFX11-FAKE16-NEXT: s_or_b32 s5, vcc_lo, s5
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s5
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB19_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -7957,6 +8003,7 @@ define void @buffer_fat_ptr_agent_atomic_fmax_noret_v2bf16__offset__amdgpu_no_fi
; GFX12-TRUE16-NEXT: s_or_b32 s4, vcc_lo, s4
; GFX12-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_cbranch_execnz .LBB20_1
; GFX12-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s4
@@ -8000,12 +8047,12 @@ define void @buffer_fat_ptr_agent_atomic_fmax_noret_v2bf16__offset__amdgpu_no_fi
; GFX12-FAKE16-NEXT: v_add3_u32 v6, v6, v0, 0x7fff
; GFX12-FAKE16-NEXT: v_cmp_u_f32_e64 s4, v0, v0
; GFX12-FAKE16-NEXT: s_wait_alu depctr_va_vcc(0)
-; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_1) | instid1(VALU_DEP_2)
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_2) | instid1(VALU_DEP_1)
; GFX12-FAKE16-NEXT: v_cndmask_b32_e32 v5, v7, v9, vcc_lo
; GFX12-FAKE16-NEXT: s_wait_alu depctr_va_sdst(0)
; GFX12-FAKE16-NEXT: v_cndmask_b32_e64 v0, v6, v8, s4
-; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX12-FAKE16-NEXT: v_perm_b32 v0, v5, v0, 0x7060302
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-FAKE16-NEXT: v_dual_mov_b32 v6, v1 :: v_dual_mov_b32 v5, v0
; GFX12-FAKE16-NEXT: buffer_atomic_cmpswap_b32 v[5:6], v4, s[0:3], null offen offset:1024 th:TH_ATOMIC_RETURN
; GFX12-FAKE16-NEXT: s_wait_loadcnt 0x0
@@ -8015,6 +8062,7 @@ define void @buffer_fat_ptr_agent_atomic_fmax_noret_v2bf16__offset__amdgpu_no_fi
; GFX12-FAKE16-NEXT: s_or_b32 s5, vcc_lo, s5
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s5
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: s_cbranch_execnz .LBB20_1
; GFX12-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s5
@@ -8115,7 +8163,7 @@ define void @buffer_fat_ptr_agent_atomic_fmax_noret_v2bf16__offset__amdgpu_no_fi
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v5, v1
; GFX11-TRUE16-NEXT: v_mov_b32_e32 v1, v5
; GFX11-TRUE16-NEXT: s_or_b32 s4, vcc_lo, s4
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB20_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -8157,7 +8205,7 @@ define void @buffer_fat_ptr_agent_atomic_fmax_noret_v2bf16__offset__amdgpu_no_fi
; GFX11-FAKE16-NEXT: v_add3_u32 v7, v7, v5, 0x7fff
; GFX11-FAKE16-NEXT: v_add3_u32 v6, v6, v0, 0x7fff
; GFX11-FAKE16-NEXT: v_cmp_u_f32_e64 s4, v0, v0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_cndmask_b32_e32 v5, v7, v9, vcc_lo
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v0, v6, v8, s4
; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
@@ -8170,7 +8218,7 @@ define void @buffer_fat_ptr_agent_atomic_fmax_noret_v2bf16__offset__amdgpu_no_fi
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v5, v1
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v1, v5
; GFX11-FAKE16-NEXT: s_or_b32 s5, vcc_lo, s5
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s5
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB20_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -8574,9 +8622,11 @@ define <2 x bfloat> @buffer_fat_ptr_agent_atomic_fmax_ret_v2bf16__offset__waterf
; GFX12-TRUE16-NEXT: s_or_b32 s4, vcc_lo, s4
; GFX12-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_cbranch_execnz .LBB21_5
; GFX12-TRUE16-NEXT: ; %bb.8: ; %atomicrmw.end
; GFX12-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s4
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: v_mov_b32_e32 v0, v5
; GFX12-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -8681,9 +8731,11 @@ define <2 x bfloat> @buffer_fat_ptr_agent_atomic_fmax_ret_v2bf16__offset__waterf
; GFX12-FAKE16-NEXT: s_or_b32 s4, vcc_lo, s4
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: s_cbranch_execnz .LBB21_5
; GFX12-FAKE16-NEXT: ; %bb.8: ; %atomicrmw.end
; GFX12-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s4
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: v_mov_b32_e32 v0, v5
; GFX12-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -8872,12 +8924,13 @@ define <2 x bfloat> @buffer_fat_ptr_agent_atomic_fmax_ret_v2bf16__offset__waterf
; GFX11-TRUE16-NEXT: buffer_gl1_inv
; GFX11-TRUE16-NEXT: buffer_gl0_inv
; GFX11-TRUE16-NEXT: s_or_b32 s4, vcc_lo, s4
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB21_5
; GFX11-TRUE16-NEXT: ; %bb.8: ; %atomicrmw.end
; GFX11-TRUE16-NEXT: s_set_inst_prefetch_distance 0x2
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s4
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_mov_b32_e32 v0, v5
; GFX11-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -8968,12 +9021,13 @@ define <2 x bfloat> @buffer_fat_ptr_agent_atomic_fmax_ret_v2bf16__offset__waterf
; GFX11-FAKE16-NEXT: buffer_gl1_inv
; GFX11-FAKE16-NEXT: buffer_gl0_inv
; GFX11-FAKE16-NEXT: s_or_b32 s4, vcc_lo, s4
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB21_5
; GFX11-FAKE16-NEXT: ; %bb.8: ; %atomicrmw.end
; GFX11-FAKE16-NEXT: s_set_inst_prefetch_distance 0x2
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s4
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v0, v5
; GFX11-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
diff --git a/llvm/test/CodeGen/AMDGPU/buffer-fat-pointer-atomicrmw-fmin.ll b/llvm/test/CodeGen/AMDGPU/buffer-fat-pointer-atomicrmw-fmin.ll
index b7efd8ea18d8cb..cd98a2b4f254ea 100644
--- a/llvm/test/CodeGen/AMDGPU/buffer-fat-pointer-atomicrmw-fmin.ll
+++ b/llvm/test/CodeGen/AMDGPU/buffer-fat-pointer-atomicrmw-fmin.ll
@@ -805,7 +805,7 @@ define float @buffer_fat_ptr_agent_atomic_fmin_ret_f32__offset__amdgpu_no_remote
; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v0, v5
; GFX11-NEXT: s_or_b32 s4, vcc_lo, s4
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
; GFX11-NEXT: s_cbranch_execnz .LBB3_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -1195,6 +1195,7 @@ define double @buffer_fat_ptr_agent_atomic_fmin_ret_f64__offset__amdgpu_no_fine_
; GFX12-NEXT: s_or_b32 s4, vcc_lo, s4
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB5_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s4
@@ -1236,7 +1237,7 @@ define double @buffer_fat_ptr_agent_atomic_fmin_ret_f64__offset__amdgpu_no_fine_
; GFX11-NEXT: v_cmp_eq_u64_e32 vcc_lo, v[0:1], v[4:5]
; GFX11-NEXT: v_dual_mov_b32 v5, v1 :: v_dual_mov_b32 v4, v0
; GFX11-NEXT: s_or_b32 s4, vcc_lo, s4
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
; GFX11-NEXT: s_cbranch_execnz .LBB5_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -1376,6 +1377,7 @@ define void @buffer_fat_ptr_agent_atomic_fmin_noret_f64__offset__amdgpu_no_fine_
; GFX12-NEXT: s_or_b32 s4, vcc_lo, s4
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB6_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s4
@@ -1416,7 +1418,7 @@ define void @buffer_fat_ptr_agent_atomic_fmin_noret_f64__offset__amdgpu_no_fine_
; GFX11-NEXT: v_cmp_eq_u64_e32 vcc_lo, v[7:8], v[2:3]
; GFX11-NEXT: v_dual_mov_b32 v2, v7 :: v_dual_mov_b32 v3, v8
; GFX11-NEXT: s_or_b32 s4, vcc_lo, s4
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
; GFX11-NEXT: s_cbranch_execnz .LBB6_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -1555,6 +1557,7 @@ define double @buffer_fat_ptr_agent_atomic_fmin_ret_f64__offset__waterfall__amdg
; GFX12-NEXT: s_cbranch_execnz .LBB7_1
; GFX12-NEXT: ; %bb.2:
; GFX12-NEXT: s_mov_b32 exec_lo, s4
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: v_max_num_f64_e32 v[5:6], v[5:6], v[5:6]
; GFX12-NEXT: s_mov_b32 s4, 0
; GFX12-NEXT: .LBB7_3: ; %atomicrmw.start
@@ -1593,6 +1596,7 @@ define double @buffer_fat_ptr_agent_atomic_fmin_ret_f64__offset__waterfall__amdg
; GFX12-NEXT: s_or_b32 s4, vcc_lo, s4
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB7_3
; GFX12-NEXT: ; %bb.6: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s4
@@ -1652,6 +1656,7 @@ define double @buffer_fat_ptr_agent_atomic_fmin_ret_f64__offset__waterfall__amdg
; GFX11-NEXT: s_cbranch_execnz .LBB7_1
; GFX11-NEXT: ; %bb.2:
; GFX11-NEXT: s_mov_b32 exec_lo, s5
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: v_max_f64 v[5:6], v[5:6], v[5:6]
; GFX11-NEXT: .p2align 6
; GFX11-NEXT: .LBB7_3: ; %atomicrmw.start
@@ -1687,7 +1692,7 @@ define double @buffer_fat_ptr_agent_atomic_fmin_ret_f64__offset__waterfall__amdg
; GFX11-NEXT: buffer_gl1_inv
; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: s_or_b32 s4, vcc_lo, s4
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
; GFX11-NEXT: s_cbranch_execnz .LBB7_3
; GFX11-NEXT: ; %bb.6: ; %atomicrmw.end
@@ -1971,6 +1976,7 @@ define double @buffer_fat_ptr_agent_atomic_fmin_ret_f64__offset__amdgpu_no_remot
; GFX12-NEXT: s_or_b32 s4, vcc_lo, s4
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB8_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s4
@@ -2012,7 +2018,7 @@ define double @buffer_fat_ptr_agent_atomic_fmin_ret_f64__offset__amdgpu_no_remot
; GFX11-NEXT: v_cmp_eq_u64_e32 vcc_lo, v[0:1], v[4:5]
; GFX11-NEXT: v_dual_mov_b32 v5, v1 :: v_dual_mov_b32 v4, v0
; GFX11-NEXT: s_or_b32 s4, vcc_lo, s4
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
; GFX11-NEXT: s_cbranch_execnz .LBB8_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -2235,6 +2241,7 @@ define double @buffer_fat_ptr_agent_atomic_fmin_ret_f64__offset__amdgpu_no_fine_
; GFX12-NEXT: s_or_b32 s4, vcc_lo, s4
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB9_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s4
@@ -2276,7 +2283,7 @@ define double @buffer_fat_ptr_agent_atomic_fmin_ret_f64__offset__amdgpu_no_fine_
; GFX11-NEXT: v_cmp_eq_u64_e32 vcc_lo, v[0:1], v[4:5]
; GFX11-NEXT: v_dual_mov_b32 v5, v1 :: v_dual_mov_b32 v4, v0
; GFX11-NEXT: s_or_b32 s4, vcc_lo, s4
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
; GFX11-NEXT: s_cbranch_execnz .LBB9_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -2435,9 +2442,11 @@ define half @buffer_fat_ptr_agent_atomic_fmin_ret_f16__offset__amdgpu_no_fine_gr
; GFX12-TRUE16-NEXT: s_or_b32 s5, vcc_lo, s5
; GFX12-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s5
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_cbranch_execnz .LBB10_1
; GFX12-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s5
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: v_lshrrev_b32_e32 v0, s4, v3
; GFX12-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -2485,9 +2494,11 @@ define half @buffer_fat_ptr_agent_atomic_fmin_ret_f16__offset__amdgpu_no_fine_gr
; GFX12-FAKE16-NEXT: s_or_b32 s5, vcc_lo, s5
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s5
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: s_cbranch_execnz .LBB10_1
; GFX12-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s5
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: v_lshrrev_b32_e32 v0, s4, v2
; GFX12-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -2564,11 +2575,12 @@ define half @buffer_fat_ptr_agent_atomic_fmin_ret_f16__offset__amdgpu_no_fine_gr
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v2
; GFX11-TRUE16-NEXT: v_mov_b32_e32 v2, v3
; GFX11-TRUE16-NEXT: s_or_b32 s5, vcc_lo, s5
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s5
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB10_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s5
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v0, s4, v3
; GFX11-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -2609,11 +2621,12 @@ define half @buffer_fat_ptr_agent_atomic_fmin_ret_f16__offset__amdgpu_no_fine_gr
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v1
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v1, v2
; GFX11-FAKE16-NEXT: s_or_b32 s5, vcc_lo, s5
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s5
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB10_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s5
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v0, s4, v2
; GFX11-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -2890,6 +2903,7 @@ define void @buffer_fat_ptr_agent_atomic_fmin_noret_f16__offset__amdgpu_no_fine_
; GFX12-TRUE16-NEXT: s_or_b32 s5, vcc_lo, s5
; GFX12-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s5
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_cbranch_execnz .LBB11_1
; GFX12-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s5
@@ -2939,6 +2953,7 @@ define void @buffer_fat_ptr_agent_atomic_fmin_noret_f16__offset__amdgpu_no_fine_
; GFX12-FAKE16-NEXT: s_or_b32 s5, vcc_lo, s5
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s5
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: s_cbranch_execnz .LBB11_1
; GFX12-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s5
@@ -3016,7 +3031,7 @@ define void @buffer_fat_ptr_agent_atomic_fmin_noret_f16__offset__amdgpu_no_fine_
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v4, v2
; GFX11-TRUE16-NEXT: v_mov_b32_e32 v2, v4
; GFX11-TRUE16-NEXT: s_or_b32 s5, vcc_lo, s5
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s5
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB11_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -3060,7 +3075,7 @@ define void @buffer_fat_ptr_agent_atomic_fmin_noret_f16__offset__amdgpu_no_fine_
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v4, v1
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v1, v4
; GFX11-FAKE16-NEXT: s_or_b32 s5, vcc_lo, s5
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s5
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB11_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -3323,6 +3338,7 @@ define half @buffer_fat_ptr_agent_atomic_fmin_ret_f16__offset__waterfall__amdgpu
; GFX12-TRUE16-NEXT: s_cbranch_execnz .LBB12_1
; GFX12-TRUE16-NEXT: ; %bb.2:
; GFX12-TRUE16-NEXT: s_mov_b32 exec_lo, s4
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: v_max_num_f16_e32 v4.l, v5.l, v5.l
; GFX12-TRUE16-NEXT: s_mov_b32 s4, 0
; GFX12-TRUE16-NEXT: .LBB12_3: ; %atomicrmw.start
@@ -3366,9 +3382,11 @@ define half @buffer_fat_ptr_agent_atomic_fmin_ret_f16__offset__waterfall__amdgpu
; GFX12-TRUE16-NEXT: s_or_b32 s4, vcc_lo, s4
; GFX12-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_cbranch_execnz .LBB12_3
; GFX12-TRUE16-NEXT: ; %bb.6: ; %atomicrmw.end
; GFX12-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s4
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: v_lshrrev_b32_e32 v0, v9, v7
; GFX12-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -3405,6 +3423,7 @@ define half @buffer_fat_ptr_agent_atomic_fmin_ret_f16__offset__waterfall__amdgpu
; GFX12-FAKE16-NEXT: s_cbranch_execnz .LBB12_1
; GFX12-FAKE16-NEXT: ; %bb.2:
; GFX12-FAKE16-NEXT: s_mov_b32 exec_lo, s4
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: v_max_num_f16_e32 v10, v5, v5
; GFX12-FAKE16-NEXT: s_mov_b32 s4, 0
; GFX12-FAKE16-NEXT: .LBB12_3: ; %atomicrmw.start
@@ -3448,9 +3467,11 @@ define half @buffer_fat_ptr_agent_atomic_fmin_ret_f16__offset__waterfall__amdgpu
; GFX12-FAKE16-NEXT: s_or_b32 s4, vcc_lo, s4
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: s_cbranch_execnz .LBB12_3
; GFX12-FAKE16-NEXT: ; %bb.6: ; %atomicrmw.end
; GFX12-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s4
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: v_lshrrev_b32_e32 v0, v7, v4
; GFX12-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -3551,6 +3572,7 @@ define half @buffer_fat_ptr_agent_atomic_fmin_ret_f16__offset__waterfall__amdgpu
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB12_1
; GFX11-TRUE16-NEXT: ; %bb.2:
; GFX11-TRUE16-NEXT: s_mov_b32 exec_lo, s5
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_max_f16_e32 v4.l, v5.l, v5.l
; GFX11-TRUE16-NEXT: .p2align 6
; GFX11-TRUE16-NEXT: .LBB12_3: ; %atomicrmw.start
@@ -3591,11 +3613,12 @@ define half @buffer_fat_ptr_agent_atomic_fmin_ret_f16__offset__waterfall__amdgpu
; GFX11-TRUE16-NEXT: buffer_gl1_inv
; GFX11-TRUE16-NEXT: buffer_gl0_inv
; GFX11-TRUE16-NEXT: s_or_b32 s4, vcc_lo, s4
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB12_3
; GFX11-TRUE16-NEXT: ; %bb.6: ; %atomicrmw.end
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s4
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v0, v9, v7
; GFX11-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -3626,6 +3649,7 @@ define half @buffer_fat_ptr_agent_atomic_fmin_ret_f16__offset__waterfall__amdgpu
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB12_1
; GFX11-FAKE16-NEXT: ; %bb.2:
; GFX11-FAKE16-NEXT: s_mov_b32 exec_lo, s5
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_max_f16_e32 v10, v5, v5
; GFX11-FAKE16-NEXT: .p2align 6
; GFX11-FAKE16-NEXT: .LBB12_3: ; %atomicrmw.start
@@ -3666,11 +3690,12 @@ define half @buffer_fat_ptr_agent_atomic_fmin_ret_f16__offset__waterfall__amdgpu
; GFX11-FAKE16-NEXT: buffer_gl1_inv
; GFX11-FAKE16-NEXT: buffer_gl0_inv
; GFX11-FAKE16-NEXT: s_or_b32 s4, vcc_lo, s4
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB12_3
; GFX11-FAKE16-NEXT: ; %bb.6: ; %atomicrmw.end
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s4
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v0, v7, v4
; GFX11-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -4147,9 +4172,11 @@ define bfloat @buffer_fat_ptr_agent_atomic_fmin_ret_bf16__offset__amdgpu_no_fine
; GFX12-NEXT: s_or_b32 s5, vcc_lo, s5
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s5
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB13_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s5
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: v_lshrrev_b32_e32 v0, s4, v2
; GFX12-NEXT: s_setpc_b64 s[30:31]
;
@@ -4240,11 +4267,12 @@ define bfloat @buffer_fat_ptr_agent_atomic_fmin_ret_bf16__offset__amdgpu_no_fine
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v1
; GFX11-NEXT: v_mov_b32_e32 v1, v2
; GFX11-NEXT: s_or_b32 s5, vcc_lo, s5
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s5
; GFX11-NEXT: s_cbranch_execnz .LBB13_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s5
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: v_lshrrev_b32_e32 v0, s4, v2
; GFX11-NEXT: s_setpc_b64 s[30:31]
;
@@ -4555,6 +4583,7 @@ define void @buffer_fat_ptr_agent_atomic_fmin_noret_bf16__offset__amdgpu_no_fine
; GFX12-NEXT: s_or_b32 s5, vcc_lo, s5
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s5
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB14_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s5
@@ -4646,7 +4675,7 @@ define void @buffer_fat_ptr_agent_atomic_fmin_noret_bf16__offset__amdgpu_no_fine
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v4, v1
; GFX11-NEXT: v_mov_b32_e32 v1, v4
; GFX11-NEXT: s_or_b32 s5, vcc_lo, s5
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s5
; GFX11-NEXT: s_cbranch_execnz .LBB14_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -4935,6 +4964,7 @@ define bfloat @buffer_fat_ptr_agent_atomic_fmin_ret_bf16__offset__waterfall__amd
; GFX12-NEXT: s_cbranch_execnz .LBB15_1
; GFX12-NEXT: ; %bb.2:
; GFX12-NEXT: s_mov_b32 exec_lo, s4
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: v_lshlrev_b32_e32 v10, 16, v5
; GFX12-NEXT: s_mov_b32 s4, 0
; GFX12-NEXT: .LBB15_3: ; %atomicrmw.start
@@ -4985,9 +5015,11 @@ define bfloat @buffer_fat_ptr_agent_atomic_fmin_ret_bf16__offset__waterfall__amd
; GFX12-NEXT: s_or_b32 s4, vcc_lo, s4
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB15_3
; GFX12-NEXT: ; %bb.6: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s4
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: v_lshrrev_b32_e32 v0, v7, v4
; GFX12-NEXT: s_setpc_b64 s[30:31]
;
@@ -5093,6 +5125,7 @@ define bfloat @buffer_fat_ptr_agent_atomic_fmin_ret_bf16__offset__waterfall__amd
; GFX11-NEXT: s_cbranch_execnz .LBB15_1
; GFX11-NEXT: ; %bb.2:
; GFX11-NEXT: s_mov_b32 exec_lo, s5
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: v_lshlrev_b32_e32 v10, 16, v5
; GFX11-NEXT: s_set_inst_prefetch_distance 0x1
; GFX11-NEXT: .p2align 6
@@ -5141,12 +5174,13 @@ define bfloat @buffer_fat_ptr_agent_atomic_fmin_ret_bf16__offset__waterfall__amd
; GFX11-NEXT: buffer_gl1_inv
; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: s_or_b32 s4, vcc_lo, s4
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
; GFX11-NEXT: s_cbranch_execnz .LBB15_3
; GFX11-NEXT: ; %bb.6: ; %atomicrmw.end
; GFX11-NEXT: s_set_inst_prefetch_distance 0x2
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s4
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: v_lshrrev_b32_e32 v0, v7, v4
; GFX11-NEXT: s_setpc_b64 s[30:31]
;
@@ -5630,6 +5664,7 @@ define <2 x half> @buffer_fat_ptr_agent_atomic_fmin_ret_v2f16__offset__amdgpu_no
; GFX12-TRUE16-NEXT: s_or_b32 s4, vcc_lo, s4
; GFX12-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_cbranch_execnz .LBB16_1
; GFX12-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s4
@@ -5671,6 +5706,7 @@ define <2 x half> @buffer_fat_ptr_agent_atomic_fmin_ret_v2f16__offset__amdgpu_no
; GFX12-FAKE16-NEXT: s_or_b32 s4, vcc_lo, s4
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: s_cbranch_execnz .LBB16_1
; GFX12-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s4
@@ -5737,7 +5773,7 @@ define <2 x half> @buffer_fat_ptr_agent_atomic_fmin_ret_v2f16__offset__amdgpu_no
; GFX11-TRUE16-NEXT: buffer_gl0_inv
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v0, v5
; GFX11-TRUE16-NEXT: s_or_b32 s4, vcc_lo, s4
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB16_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -5774,7 +5810,7 @@ define <2 x half> @buffer_fat_ptr_agent_atomic_fmin_ret_v2f16__offset__amdgpu_no
; GFX11-FAKE16-NEXT: buffer_gl0_inv
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v0, v5
; GFX11-FAKE16-NEXT: s_or_b32 s4, vcc_lo, s4
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB16_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -6040,6 +6076,7 @@ define void @buffer_fat_ptr_agent_atomic_fmin_noret_v2f16__offset__amdgpu_no_fin
; GFX12-TRUE16-NEXT: s_or_b32 s4, vcc_lo, s4
; GFX12-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_cbranch_execnz .LBB17_1
; GFX12-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s4
@@ -6081,6 +6118,7 @@ define void @buffer_fat_ptr_agent_atomic_fmin_noret_v2f16__offset__amdgpu_no_fin
; GFX12-FAKE16-NEXT: s_or_b32 s4, vcc_lo, s4
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: s_cbranch_execnz .LBB17_1
; GFX12-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s4
@@ -6147,7 +6185,7 @@ define void @buffer_fat_ptr_agent_atomic_fmin_noret_v2f16__offset__amdgpu_no_fin
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v4, v1
; GFX11-TRUE16-NEXT: v_mov_b32_e32 v1, v4
; GFX11-TRUE16-NEXT: s_or_b32 s4, vcc_lo, s4
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB17_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -6184,7 +6222,7 @@ define void @buffer_fat_ptr_agent_atomic_fmin_noret_v2f16__offset__amdgpu_no_fin
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v4, v1
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v1, v4
; GFX11-FAKE16-NEXT: s_or_b32 s4, vcc_lo, s4
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB17_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -6495,9 +6533,11 @@ define <2 x half> @buffer_fat_ptr_agent_atomic_fmin_ret_v2f16__offset__waterfall
; GFX12-TRUE16-NEXT: s_or_b32 s4, vcc_lo, s4
; GFX12-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_cbranch_execnz .LBB18_5
; GFX12-TRUE16-NEXT: ; %bb.8: ; %atomicrmw.end
; GFX12-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s4
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: v_mov_b32_e32 v0, v5
; GFX12-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -6585,9 +6625,11 @@ define <2 x half> @buffer_fat_ptr_agent_atomic_fmin_ret_v2f16__offset__waterfall
; GFX12-FAKE16-NEXT: s_or_b32 s4, vcc_lo, s4
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: s_cbranch_execnz .LBB18_5
; GFX12-FAKE16-NEXT: ; %bb.8: ; %atomicrmw.end
; GFX12-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s4
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: v_mov_b32_e32 v0, v5
; GFX12-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -6740,11 +6782,12 @@ define <2 x half> @buffer_fat_ptr_agent_atomic_fmin_ret_v2f16__offset__waterfall
; GFX11-TRUE16-NEXT: buffer_gl1_inv
; GFX11-TRUE16-NEXT: buffer_gl0_inv
; GFX11-TRUE16-NEXT: s_or_b32 s4, vcc_lo, s4
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB18_5
; GFX11-TRUE16-NEXT: ; %bb.8: ; %atomicrmw.end
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s4
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_mov_b32_e32 v0, v5
; GFX11-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -6819,11 +6862,12 @@ define <2 x half> @buffer_fat_ptr_agent_atomic_fmin_ret_v2f16__offset__waterfall
; GFX11-FAKE16-NEXT: buffer_gl1_inv
; GFX11-FAKE16-NEXT: buffer_gl0_inv
; GFX11-FAKE16-NEXT: s_or_b32 s4, vcc_lo, s4
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB18_5
; GFX11-FAKE16-NEXT: ; %bb.8: ; %atomicrmw.end
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s4
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v0, v5
; GFX11-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -7376,6 +7420,7 @@ define <2 x bfloat> @buffer_fat_ptr_agent_atomic_fmin_ret_v2bf16__offset__amdgpu
; GFX12-TRUE16-NEXT: s_or_b32 s4, vcc_lo, s4
; GFX12-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_cbranch_execnz .LBB19_1
; GFX12-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s4
@@ -7436,6 +7481,7 @@ define <2 x bfloat> @buffer_fat_ptr_agent_atomic_fmin_ret_v2bf16__offset__amdgpu
; GFX12-FAKE16-NEXT: s_or_b32 s5, vcc_lo, s5
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s5
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: s_cbranch_execnz .LBB19_1
; GFX12-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s5
@@ -7540,7 +7586,7 @@ define <2 x bfloat> @buffer_fat_ptr_agent_atomic_fmin_ret_v2bf16__offset__amdgpu
; GFX11-TRUE16-NEXT: buffer_gl0_inv
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v0, v6
; GFX11-TRUE16-NEXT: s_or_b32 s4, vcc_lo, s4
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB19_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -7597,7 +7643,7 @@ define <2 x bfloat> @buffer_fat_ptr_agent_atomic_fmin_ret_v2bf16__offset__amdgpu
; GFX11-FAKE16-NEXT: buffer_gl0_inv
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v0, v6
; GFX11-FAKE16-NEXT: s_or_b32 s5, vcc_lo, s5
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s5
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB19_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -7957,6 +8003,7 @@ define void @buffer_fat_ptr_agent_atomic_fmin_noret_v2bf16__offset__amdgpu_no_fi
; GFX12-TRUE16-NEXT: s_or_b32 s4, vcc_lo, s4
; GFX12-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_cbranch_execnz .LBB20_1
; GFX12-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s4
@@ -8000,12 +8047,12 @@ define void @buffer_fat_ptr_agent_atomic_fmin_noret_v2bf16__offset__amdgpu_no_fi
; GFX12-FAKE16-NEXT: v_add3_u32 v6, v6, v0, 0x7fff
; GFX12-FAKE16-NEXT: v_cmp_u_f32_e64 s4, v0, v0
; GFX12-FAKE16-NEXT: s_wait_alu depctr_va_vcc(0)
-; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_1) | instid1(VALU_DEP_2)
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_2) | instid1(VALU_DEP_1)
; GFX12-FAKE16-NEXT: v_cndmask_b32_e32 v5, v7, v9, vcc_lo
; GFX12-FAKE16-NEXT: s_wait_alu depctr_va_sdst(0)
; GFX12-FAKE16-NEXT: v_cndmask_b32_e64 v0, v6, v8, s4
-; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX12-FAKE16-NEXT: v_perm_b32 v0, v5, v0, 0x7060302
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-FAKE16-NEXT: v_dual_mov_b32 v6, v1 :: v_dual_mov_b32 v5, v0
; GFX12-FAKE16-NEXT: buffer_atomic_cmpswap_b32 v[5:6], v4, s[0:3], null offen offset:1024 th:TH_ATOMIC_RETURN
; GFX12-FAKE16-NEXT: s_wait_loadcnt 0x0
@@ -8015,6 +8062,7 @@ define void @buffer_fat_ptr_agent_atomic_fmin_noret_v2bf16__offset__amdgpu_no_fi
; GFX12-FAKE16-NEXT: s_or_b32 s5, vcc_lo, s5
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s5
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: s_cbranch_execnz .LBB20_1
; GFX12-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s5
@@ -8115,7 +8163,7 @@ define void @buffer_fat_ptr_agent_atomic_fmin_noret_v2bf16__offset__amdgpu_no_fi
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v5, v1
; GFX11-TRUE16-NEXT: v_mov_b32_e32 v1, v5
; GFX11-TRUE16-NEXT: s_or_b32 s4, vcc_lo, s4
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB20_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -8157,7 +8205,7 @@ define void @buffer_fat_ptr_agent_atomic_fmin_noret_v2bf16__offset__amdgpu_no_fi
; GFX11-FAKE16-NEXT: v_add3_u32 v7, v7, v5, 0x7fff
; GFX11-FAKE16-NEXT: v_add3_u32 v6, v6, v0, 0x7fff
; GFX11-FAKE16-NEXT: v_cmp_u_f32_e64 s4, v0, v0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_cndmask_b32_e32 v5, v7, v9, vcc_lo
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v0, v6, v8, s4
; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
@@ -8170,7 +8218,7 @@ define void @buffer_fat_ptr_agent_atomic_fmin_noret_v2bf16__offset__amdgpu_no_fi
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v5, v1
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v1, v5
; GFX11-FAKE16-NEXT: s_or_b32 s5, vcc_lo, s5
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s5
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB20_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -8574,9 +8622,11 @@ define <2 x bfloat> @buffer_fat_ptr_agent_atomic_fmin_ret_v2bf16__offset__waterf
; GFX12-TRUE16-NEXT: s_or_b32 s4, vcc_lo, s4
; GFX12-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_cbranch_execnz .LBB21_5
; GFX12-TRUE16-NEXT: ; %bb.8: ; %atomicrmw.end
; GFX12-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s4
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: v_mov_b32_e32 v0, v5
; GFX12-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -8681,9 +8731,11 @@ define <2 x bfloat> @buffer_fat_ptr_agent_atomic_fmin_ret_v2bf16__offset__waterf
; GFX12-FAKE16-NEXT: s_or_b32 s4, vcc_lo, s4
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: s_cbranch_execnz .LBB21_5
; GFX12-FAKE16-NEXT: ; %bb.8: ; %atomicrmw.end
; GFX12-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s4
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: v_mov_b32_e32 v0, v5
; GFX12-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -8872,12 +8924,13 @@ define <2 x bfloat> @buffer_fat_ptr_agent_atomic_fmin_ret_v2bf16__offset__waterf
; GFX11-TRUE16-NEXT: buffer_gl1_inv
; GFX11-TRUE16-NEXT: buffer_gl0_inv
; GFX11-TRUE16-NEXT: s_or_b32 s4, vcc_lo, s4
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB21_5
; GFX11-TRUE16-NEXT: ; %bb.8: ; %atomicrmw.end
; GFX11-TRUE16-NEXT: s_set_inst_prefetch_distance 0x2
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s4
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_mov_b32_e32 v0, v5
; GFX11-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -8968,12 +9021,13 @@ define <2 x bfloat> @buffer_fat_ptr_agent_atomic_fmin_ret_v2bf16__offset__waterf
; GFX11-FAKE16-NEXT: buffer_gl1_inv
; GFX11-FAKE16-NEXT: buffer_gl0_inv
; GFX11-FAKE16-NEXT: s_or_b32 s4, vcc_lo, s4
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB21_5
; GFX11-FAKE16-NEXT: ; %bb.8: ; %atomicrmw.end
; GFX11-FAKE16-NEXT: s_set_inst_prefetch_distance 0x2
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s4
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v0, v5
; GFX11-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
diff --git a/llvm/test/CodeGen/AMDGPU/calling-conventions.ll b/llvm/test/CodeGen/AMDGPU/calling-conventions.ll
index 8a886acdbc82a7..c4ad984ed3ae31 100644
--- a/llvm/test/CodeGen/AMDGPU/calling-conventions.ll
+++ b/llvm/test/CodeGen/AMDGPU/calling-conventions.ll
@@ -92,6 +92,7 @@ define amdgpu_ps half @ps_ret_cc_f16(half %arg0) {
; GFX1250-TRUE16-NEXT: v_nop
; GFX1250-TRUE16-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-TRUE16-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 2, 2), 0 ; msbs: dst=0 src0=0 src1=0 src2=0
+; GFX1250-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1250-TRUE16-NEXT: v_add_f16_e32 v0.l, 1.0, v0.l
; GFX1250-TRUE16-NEXT: ; return to shader part epilog
;
@@ -102,6 +103,7 @@ define amdgpu_ps half @ps_ret_cc_f16(half %arg0) {
; GFX1250-FAKE16-NEXT: v_nop
; GFX1250-FAKE16-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-FAKE16-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 2, 2), 0 ; msbs: dst=0 src0=0 src1=0 src2=0
+; GFX1250-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1250-FAKE16-NEXT: v_add_f16_e32 v0, 1.0, v0
; GFX1250-FAKE16-NEXT: ; return to shader part epilog
%add = fadd half %arg0, 1.0
@@ -138,8 +140,8 @@ define amdgpu_ps half @ps_ret_cc_inreg_f16(half inreg %arg0) {
; GFX1250-NEXT: v_nop
; GFX1250-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 2, 2), 0 ; msbs: dst=0 src0=0 src1=0 src2=0
+; GFX1250-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_3)
; GFX1250-NEXT: s_add_f16 s0, s0, 1.0
-; GFX1250-NEXT: s_delay_alu instid0(SALU_CYCLE_3)
; GFX1250-NEXT: v_mov_b32_e32 v0, s0
; GFX1250-NEXT: ; return to shader part epilog
%add = fadd half %arg0, 1.0
@@ -465,6 +467,7 @@ define amdgpu_cs half @cs_mesa(half %arg0) {
; GFX1250-TRUE16-NEXT: v_nop
; GFX1250-TRUE16-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-TRUE16-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 2, 2), 0 ; msbs: dst=0 src0=0 src1=0 src2=0
+; GFX1250-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1250-TRUE16-NEXT: v_add_f16_e32 v0.l, 1.0, v0.l
; GFX1250-TRUE16-NEXT: ; return to shader part epilog
;
@@ -475,6 +478,7 @@ define amdgpu_cs half @cs_mesa(half %arg0) {
; GFX1250-FAKE16-NEXT: v_nop
; GFX1250-FAKE16-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-FAKE16-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 2, 2), 0 ; msbs: dst=0 src0=0 src1=0 src2=0
+; GFX1250-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1250-FAKE16-NEXT: v_add_f16_e32 v0, 1.0, v0
; GFX1250-FAKE16-NEXT: ; return to shader part epilog
%add = fadd half %arg0, 1.0
@@ -512,6 +516,7 @@ define amdgpu_ps half @ps_mesa_f16(half %arg0) {
; GFX1250-TRUE16-NEXT: v_nop
; GFX1250-TRUE16-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-TRUE16-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 2, 2), 0 ; msbs: dst=0 src0=0 src1=0 src2=0
+; GFX1250-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1250-TRUE16-NEXT: v_add_f16_e32 v0.l, 1.0, v0.l
; GFX1250-TRUE16-NEXT: ; return to shader part epilog
;
@@ -522,6 +527,7 @@ define amdgpu_ps half @ps_mesa_f16(half %arg0) {
; GFX1250-FAKE16-NEXT: v_nop
; GFX1250-FAKE16-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-FAKE16-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 2, 2), 0 ; msbs: dst=0 src0=0 src1=0 src2=0
+; GFX1250-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1250-FAKE16-NEXT: v_add_f16_e32 v0, 1.0, v0
; GFX1250-FAKE16-NEXT: ; return to shader part epilog
%add = fadd half %arg0, 1.0
@@ -559,6 +565,7 @@ define amdgpu_vs half @vs_mesa(half %arg0) {
; GFX1250-TRUE16-NEXT: v_nop
; GFX1250-TRUE16-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-TRUE16-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 2, 2), 0 ; msbs: dst=0 src0=0 src1=0 src2=0
+; GFX1250-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1250-TRUE16-NEXT: v_add_f16_e32 v0.l, 1.0, v0.l
; GFX1250-TRUE16-NEXT: ; return to shader part epilog
;
@@ -569,6 +576,7 @@ define amdgpu_vs half @vs_mesa(half %arg0) {
; GFX1250-FAKE16-NEXT: v_nop
; GFX1250-FAKE16-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-FAKE16-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 2, 2), 0 ; msbs: dst=0 src0=0 src1=0 src2=0
+; GFX1250-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1250-FAKE16-NEXT: v_add_f16_e32 v0, 1.0, v0
; GFX1250-FAKE16-NEXT: ; return to shader part epilog
%add = fadd half %arg0, 1.0
@@ -606,6 +614,7 @@ define amdgpu_gs half @gs_mesa(half %arg0) {
; GFX1250-TRUE16-NEXT: v_nop
; GFX1250-TRUE16-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-TRUE16-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 2, 2), 0 ; msbs: dst=0 src0=0 src1=0 src2=0
+; GFX1250-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1250-TRUE16-NEXT: v_add_f16_e32 v0.l, 1.0, v0.l
; GFX1250-TRUE16-NEXT: ; return to shader part epilog
;
@@ -616,6 +625,7 @@ define amdgpu_gs half @gs_mesa(half %arg0) {
; GFX1250-FAKE16-NEXT: v_nop
; GFX1250-FAKE16-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-FAKE16-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 2, 2), 0 ; msbs: dst=0 src0=0 src1=0 src2=0
+; GFX1250-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1250-FAKE16-NEXT: v_add_f16_e32 v0, 1.0, v0
; GFX1250-FAKE16-NEXT: ; return to shader part epilog
%add = fadd half %arg0, 1.0
@@ -653,6 +663,7 @@ define amdgpu_hs half @hs_mesa(half %arg0) {
; GFX1250-TRUE16-NEXT: v_nop
; GFX1250-TRUE16-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-TRUE16-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 2, 2), 0 ; msbs: dst=0 src0=0 src1=0 src2=0
+; GFX1250-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1250-TRUE16-NEXT: v_add_f16_e32 v0.l, 1.0, v0.l
; GFX1250-TRUE16-NEXT: ; return to shader part epilog
;
@@ -663,6 +674,7 @@ define amdgpu_hs half @hs_mesa(half %arg0) {
; GFX1250-FAKE16-NEXT: v_nop
; GFX1250-FAKE16-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-FAKE16-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 2, 2), 0 ; msbs: dst=0 src0=0 src1=0 src2=0
+; GFX1250-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1250-FAKE16-NEXT: v_add_f16_e32 v0, 1.0, v0
; GFX1250-FAKE16-NEXT: ; return to shader part epilog
%add = fadd half %arg0, 1.0
@@ -705,6 +717,7 @@ define amdgpu_ps <2 x half> @ps_mesa_v2f16(<2 x half> %arg0) {
; GFX1250-NEXT: v_nop
; GFX1250-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 2, 2), 0 ; msbs: dst=0 src0=0 src1=0 src2=0
+; GFX1250-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1250-NEXT: v_pk_add_f16 v0, v0, 1.0 op_sel_hi:[1,0]
; GFX1250-NEXT: ; return to shader part epilog
%add = fadd <2 x half> %arg0, <half 1.0, half 1.0>
@@ -747,6 +760,7 @@ define amdgpu_ps <2 x half> @ps_mesa_inreg_v2f16(<2 x half> inreg %arg0) {
; GFX1250-NEXT: v_nop
; GFX1250-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 2, 2), 0 ; msbs: dst=0 src0=0 src1=0 src2=0
+; GFX1250-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1250-NEXT: v_pk_add_f16 v0, s0, 1.0 op_sel_hi:[1,0]
; GFX1250-NEXT: ; return to shader part epilog
%add = fadd <2 x half> %arg0, <half 1.0, half 1.0>
@@ -888,6 +902,7 @@ define amdgpu_ps <4 x half> @ps_mesa_v4f16(<4 x half> %arg0) {
; GFX1250-NEXT: v_nop
; GFX1250-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 2, 2), 0 ; msbs: dst=0 src0=0 src1=0 src2=0
+; GFX1250-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1250-NEXT: v_pk_add_f16 v0, v0, 1.0 op_sel_hi:[1,0]
; GFX1250-NEXT: v_pk_add_f16 v1, v1, 1.0 op_sel_hi:[1,0]
; GFX1250-NEXT: ; return to shader part epilog
@@ -946,6 +961,7 @@ define amdgpu_ps <4 x half> @ps_mesa_inreg_v4f16(<4 x half> inreg %arg0) {
; GFX1250-NEXT: v_nop
; GFX1250-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 2, 2), 0 ; msbs: dst=0 src0=0 src1=0 src2=0
+; GFX1250-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1250-NEXT: v_pk_add_f16 v0, s0, 1.0 op_sel_hi:[1,0]
; GFX1250-NEXT: v_pk_add_f16 v1, s1, 1.0 op_sel_hi:[1,0]
; GFX1250-NEXT: ; return to shader part epilog
diff --git a/llvm/test/CodeGen/AMDGPU/carryout-selection.ll b/llvm/test/CodeGen/AMDGPU/carryout-selection.ll
index d3fa414cfda693..55e92204fa2f5d 100644
--- a/llvm/test/CodeGen/AMDGPU/carryout-selection.ll
+++ b/llvm/test/CodeGen/AMDGPU/carryout-selection.ll
@@ -391,7 +391,7 @@ define amdgpu_kernel void @vadd64rr(ptr addrspace(1) %out, i64 %a) {
; GFX11-NEXT: v_and_b32_e32 v0, 0x3ff, v0
; GFX11-NEXT: v_mov_b32_e32 v2, 0
; GFX11-NEXT: s_waitcnt lgkmcnt(0)
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX11-NEXT: v_add_co_u32 v0, s2, s2, v0
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, s3, 0, s2
; GFX11-NEXT: global_store_b64 v2, v[0:1], s[0:1]
@@ -418,7 +418,7 @@ define amdgpu_kernel void @vadd64rr(ptr addrspace(1) %out, i64 %a) {
; GFX13-NEXT: v_and_b32_e32 v0, 0x3ff, v0
; GFX13-NEXT: v_mov_b32_e32 v2, 0
; GFX13-NEXT: s_wait_kmcnt 0x0
-; GFX13-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX13-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX13-NEXT: v_add_co_u32 v0, s2, s2, v0
; GFX13-NEXT: v_add_co_ci_u32_e64 v1, null, s3, 0, s2
; GFX13-NEXT: global_store_b64 v2, v[0:1], s[0:1]
@@ -525,7 +525,7 @@ define amdgpu_kernel void @vadd64ri(ptr addrspace(1) %out) {
; GFX11-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
; GFX11-NEXT: v_and_b32_e32 v0, 0x3ff, v0
; GFX11-NEXT: v_mov_b32_e32 v2, 0
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX11-NEXT: v_add_co_u32 v0, s2, 0x56789876, v0
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, 0x1234, 0, s2
; GFX11-NEXT: s_waitcnt lgkmcnt(0)
@@ -552,7 +552,7 @@ define amdgpu_kernel void @vadd64ri(ptr addrspace(1) %out) {
; GFX13-NEXT: s_load_b64 s[0:1], s[4:5], 0x24 nv
; GFX13-NEXT: v_and_b32_e32 v0, 0x3ff, v0
; GFX13-NEXT: v_mov_b32_e32 v2, 0
-; GFX13-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX13-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX13-NEXT: v_add_co_u32 v0, s2, 0x56789876, v0
; GFX13-NEXT: v_add_co_ci_u32_e64 v1, null, 0x1234, 0, s2
; GFX13-NEXT: s_wait_kmcnt 0x0
@@ -1238,10 +1238,9 @@ define amdgpu_kernel void @vuaddo64(ptr addrspace(1) %out, ptr addrspace(1) %car
; GFX11-NEXT: v_and_b32_e32 v0, 0x3ff, v0
; GFX11-NEXT: v_mov_b32_e32 v2, 0
; GFX11-NEXT: s_waitcnt lgkmcnt(0)
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX11-NEXT: v_add_co_u32 v0, s4, s6, v0
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, s4, s7, 0, s4
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_cndmask_b32_e64 v3, 0, 1, s4
; GFX11-NEXT: s_clause 0x1
; GFX11-NEXT: global_store_b64 v2, v[0:1], s[0:1]
@@ -1260,10 +1259,9 @@ define amdgpu_kernel void @vuaddo64(ptr addrspace(1) %out, ptr addrspace(1) %car
; GFX1250-NEXT: v_and_b32_e32 v0, 0x3ff, v0
; GFX1250-NEXT: v_mov_b32_e32 v2, 0
; GFX1250-NEXT: s_wait_kmcnt 0x0
-; GFX1250-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX1250-NEXT: v_add_co_u32 v0, s4, s6, v0
; GFX1250-NEXT: v_add_co_ci_u32_e64 v1, s4, s7, 0, s4
-; GFX1250-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1250-NEXT: v_cndmask_b32_e64 v3, 0, 1, s4
; GFX1250-NEXT: s_clause 0x1
; GFX1250-NEXT: global_store_b64 v2, v[0:1], s[0:1]
@@ -1278,10 +1276,9 @@ define amdgpu_kernel void @vuaddo64(ptr addrspace(1) %out, ptr addrspace(1) %car
; GFX13-NEXT: v_and_b32_e32 v0, 0x3ff, v0
; GFX13-NEXT: v_mov_b32_e32 v2, 0
; GFX13-NEXT: s_wait_kmcnt 0x0
-; GFX13-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX13-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX13-NEXT: v_add_co_u32 v0, s4, s6, v0
; GFX13-NEXT: v_add_co_ci_u32_e64 v1, s4, s7, 0, s4
-; GFX13-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX13-NEXT: v_cndmask_b32_e64 v3, 0, 1, s4
; GFX13-NEXT: s_clause 0x1
; GFX13-NEXT: global_store_b64 v2, v[0:1], s[0:1]
@@ -1714,7 +1711,7 @@ define amdgpu_kernel void @vsub64rr(ptr addrspace(1) %out, i64 %a) {
; GFX11-NEXT: v_and_b32_e32 v0, 0x3ff, v0
; GFX11-NEXT: v_mov_b32_e32 v2, 0
; GFX11-NEXT: s_waitcnt lgkmcnt(0)
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX11-NEXT: v_sub_co_u32 v0, s2, s2, v0
; GFX11-NEXT: v_sub_co_ci_u32_e64 v1, null, s3, 0, s2
; GFX11-NEXT: global_store_b64 v2, v[0:1], s[0:1]
@@ -1741,7 +1738,7 @@ define amdgpu_kernel void @vsub64rr(ptr addrspace(1) %out, i64 %a) {
; GFX13-NEXT: v_and_b32_e32 v0, 0x3ff, v0
; GFX13-NEXT: v_mov_b32_e32 v2, 0
; GFX13-NEXT: s_wait_kmcnt 0x0
-; GFX13-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX13-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX13-NEXT: v_sub_co_u32 v0, s2, s2, v0
; GFX13-NEXT: v_sub_co_ci_u32_e64 v1, null, s3, 0, s2
; GFX13-NEXT: global_store_b64 v2, v[0:1], s[0:1]
@@ -1848,7 +1845,7 @@ define amdgpu_kernel void @vsub64ri(ptr addrspace(1) %out) {
; GFX11-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
; GFX11-NEXT: v_and_b32_e32 v0, 0x3ff, v0
; GFX11-NEXT: v_mov_b32_e32 v2, 0
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX11-NEXT: v_sub_co_u32 v0, s2, 0x56789876, v0
; GFX11-NEXT: v_sub_co_ci_u32_e64 v1, null, 0x1234, 0, s2
; GFX11-NEXT: s_waitcnt lgkmcnt(0)
@@ -1875,7 +1872,7 @@ define amdgpu_kernel void @vsub64ri(ptr addrspace(1) %out) {
; GFX13-NEXT: s_load_b64 s[0:1], s[4:5], 0x24 nv
; GFX13-NEXT: v_and_b32_e32 v0, 0x3ff, v0
; GFX13-NEXT: v_mov_b32_e32 v2, 0
-; GFX13-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX13-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX13-NEXT: v_sub_co_u32 v0, s2, 0x56789876, v0
; GFX13-NEXT: v_sub_co_ci_u32_e64 v1, null, 0x1234, 0, s2
; GFX13-NEXT: s_wait_kmcnt 0x0
@@ -2562,10 +2559,9 @@ define amdgpu_kernel void @vusubo64(ptr addrspace(1) %out, ptr addrspace(1) %car
; GFX11-NEXT: v_and_b32_e32 v0, 0x3ff, v0
; GFX11-NEXT: v_mov_b32_e32 v2, 0
; GFX11-NEXT: s_waitcnt lgkmcnt(0)
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX11-NEXT: v_sub_co_u32 v0, s4, s6, v0
; GFX11-NEXT: v_sub_co_ci_u32_e64 v1, s4, s7, 0, s4
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_cndmask_b32_e64 v3, 0, 1, s4
; GFX11-NEXT: s_clause 0x1
; GFX11-NEXT: global_store_b64 v2, v[0:1], s[0:1]
@@ -2584,10 +2580,9 @@ define amdgpu_kernel void @vusubo64(ptr addrspace(1) %out, ptr addrspace(1) %car
; GFX1250-NEXT: v_and_b32_e32 v0, 0x3ff, v0
; GFX1250-NEXT: v_mov_b32_e32 v2, 0
; GFX1250-NEXT: s_wait_kmcnt 0x0
-; GFX1250-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX1250-NEXT: v_sub_co_u32 v0, s4, s6, v0
; GFX1250-NEXT: v_sub_co_ci_u32_e64 v1, s4, s7, 0, s4
-; GFX1250-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1250-NEXT: v_cndmask_b32_e64 v3, 0, 1, s4
; GFX1250-NEXT: s_clause 0x1
; GFX1250-NEXT: global_store_b64 v2, v[0:1], s[0:1]
@@ -2602,10 +2597,9 @@ define amdgpu_kernel void @vusubo64(ptr addrspace(1) %out, ptr addrspace(1) %car
; GFX13-NEXT: v_and_b32_e32 v0, 0x3ff, v0
; GFX13-NEXT: v_mov_b32_e32 v2, 0
; GFX13-NEXT: s_wait_kmcnt 0x0
-; GFX13-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX13-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX13-NEXT: v_sub_co_u32 v0, s4, s6, v0
; GFX13-NEXT: v_sub_co_ci_u32_e64 v1, s4, s7, 0, s4
-; GFX13-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX13-NEXT: v_cndmask_b32_e64 v3, 0, 1, s4
; GFX13-NEXT: s_clause 0x1
; GFX13-NEXT: global_store_b64 v2, v[0:1], s[0:1]
@@ -3608,7 +3602,7 @@ define amdgpu_kernel void @sudiv64(ptr addrspace(1) %out, i64 %x, i64 %y) {
; GFX11-NEXT: s_mov_b32 s8, 0
; GFX11-NEXT: s_waitcnt lgkmcnt(0)
; GFX11-NEXT: s_or_b64 s[6:7], s[2:3], s[4:5]
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_cmp_lg_u32 s7, 0
; GFX11-NEXT: s_cbranch_scc0 .LBB16_2
; GFX11-NEXT: ; %bb.1:
@@ -3702,10 +3696,11 @@ define amdgpu_kernel void @sudiv64(ptr addrspace(1) %out, i64 %x, i64 %y) {
; GFX11-NEXT: s_subb_u32 s11, s11, s5
; GFX11-NEXT: s_sub_u32 s13, s10, s4
; GFX11-NEXT: s_subb_u32 s11, s11, 0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_cmp_ge_u32 s11, s5
; GFX11-NEXT: s_cselect_b32 s14, -1, 0
; GFX11-NEXT: s_cmp_ge_u32 s13, s4
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_cselect_b32 s13, -1, 0
; GFX11-NEXT: s_cmp_eq_u32 s11, s5
; GFX11-NEXT: s_cselect_b32 s11, s13, s14
@@ -3714,18 +3709,20 @@ define amdgpu_kernel void @sudiv64(ptr addrspace(1) %out, i64 %x, i64 %y) {
; GFX11-NEXT: s_add_u32 s15, s6, 2
; GFX11-NEXT: s_addc_u32 s16, s7, 0
; GFX11-NEXT: s_cmp_lg_u32 s11, 0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_cselect_b32 s11, s15, s13
; GFX11-NEXT: s_cselect_b32 s13, s16, s14
; GFX11-NEXT: s_cmp_lg_u32 s12, 0
; GFX11-NEXT: s_subb_u32 s3, s3, s9
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_cmp_ge_u32 s3, s5
; GFX11-NEXT: s_cselect_b32 s9, -1, 0
; GFX11-NEXT: s_cmp_ge_u32 s10, s4
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_cselect_b32 s10, -1, 0
; GFX11-NEXT: s_cmp_eq_u32 s3, s5
; GFX11-NEXT: s_cselect_b32 s3, s10, s9
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_cmp_lg_u32 s3, 0
; GFX11-NEXT: s_cselect_b32 s7, s13, s7
; GFX11-NEXT: s_cselect_b32 s6, s11, s6
@@ -3738,6 +3735,7 @@ define amdgpu_kernel void @sudiv64(ptr addrspace(1) %out, i64 %x, i64 %y) {
; GFX11-NEXT: s_and_b32 s3, s8, exec_lo
; GFX11-NEXT: s_cselect_b32 s3, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s3, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB16_5
; GFX11-NEXT: ; %bb.4:
; GFX11-NEXT: v_cvt_f32_u32_e32 v0, s4
@@ -3761,6 +3759,7 @@ define amdgpu_kernel void @sudiv64(ptr addrspace(1) %out, i64 %x, i64 %y) {
; GFX11-NEXT: s_add_i32 s5, s3, 1
; GFX11-NEXT: s_sub_i32 s6, s2, s4
; GFX11-NEXT: s_cmp_ge_u32 s2, s4
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_3) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_cselect_b32 s3, s5, s3
; GFX11-NEXT: s_cselect_b32 s2, s6, s2
; GFX11-NEXT: s_add_i32 s5, s3, 1
@@ -3854,7 +3853,7 @@ define amdgpu_kernel void @sudiv64(ptr addrspace(1) %out, i64 %x, i64 %y) {
; GFX1250-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1250-NEXT: s_or_b32 s10, s10, s8
; GFX1250-NEXT: s_mul_u64 s[8:9], s[6:7], s[10:11]
-; GFX1250-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1250-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_3) | instid1(SALU_CYCLE_1)
; GFX1250-NEXT: s_sub_co_u32 s4, s2, s8
; GFX1250-NEXT: s_cselect_b32 s8, -1, 0
; GFX1250-NEXT: s_sub_co_i32 s12, s3, s9
@@ -3862,28 +3861,31 @@ define amdgpu_kernel void @sudiv64(ptr addrspace(1) %out, i64 %x, i64 %y) {
; GFX1250-NEXT: s_sub_co_ci_u32 s12, s12, s7
; GFX1250-NEXT: s_sub_co_u32 s13, s4, s6
; GFX1250-NEXT: s_sub_co_ci_u32 s12, s12, 0
-; GFX1250-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1250-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1250-NEXT: s_cmp_ge_u32 s12, s7
; GFX1250-NEXT: s_cselect_b32 s14, -1, 0
; GFX1250-NEXT: s_cmp_ge_u32 s13, s6
+; GFX1250-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1250-NEXT: s_cselect_b32 s15, -1, 0
; GFX1250-NEXT: s_cmp_eq_u32 s12, s7
; GFX1250-NEXT: s_add_nc_u64 s[12:13], s[10:11], 1
; GFX1250-NEXT: s_cselect_b32 s16, s15, s14
; GFX1250-NEXT: s_add_nc_u64 s[14:15], s[10:11], 2
; GFX1250-NEXT: s_cmp_lg_u32 s16, 0
+; GFX1250-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
; GFX1250-NEXT: s_cselect_b32 s12, s14, s12
; GFX1250-NEXT: s_cselect_b32 s13, s15, s13
; GFX1250-NEXT: s_cmp_lg_u32 s8, 0
; GFX1250-NEXT: s_sub_co_ci_u32 s3, s3, s9
-; GFX1250-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1250-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1250-NEXT: s_cmp_ge_u32 s3, s7
; GFX1250-NEXT: s_cselect_b32 s8, -1, 0
; GFX1250-NEXT: s_cmp_ge_u32 s4, s6
+; GFX1250-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX1250-NEXT: s_cselect_b32 s4, -1, 0
; GFX1250-NEXT: s_cmp_eq_u32 s3, s7
; GFX1250-NEXT: s_cselect_b32 s3, s4, s8
-; GFX1250-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1250-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1250-NEXT: s_cmp_lg_u32 s3, 0
; GFX1250-NEXT: s_cselect_b32 s9, s13, s11
; GFX1250-NEXT: s_cselect_b32 s8, s12, s10
@@ -3896,6 +3898,7 @@ define amdgpu_kernel void @sudiv64(ptr addrspace(1) %out, i64 %x, i64 %y) {
; GFX1250-NEXT: s_and_b32 s3, s5, exec_lo
; GFX1250-NEXT: s_cselect_b32 s3, 1, 0
; GFX1250-NEXT: s_cmp_lg_u32 s3, 1
+; GFX1250-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1250-NEXT: s_cbranch_scc1 .LBB16_5
; GFX1250-NEXT: ; %bb.4:
; GFX1250-NEXT: v_cvt_f32_u32_e32 v0, s6
@@ -3915,7 +3918,7 @@ define amdgpu_kernel void @sudiv64(ptr addrspace(1) %out, i64 %x, i64 %y) {
; GFX1250-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1250-NEXT: s_mul_hi_u32 s3, s2, s3
; GFX1250-NEXT: s_mul_i32 s4, s3, s6
-; GFX1250-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1250-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_3) | instid1(SALU_CYCLE_1)
; GFX1250-NEXT: s_sub_co_i32 s2, s2, s4
; GFX1250-NEXT: s_add_co_i32 s4, s3, 1
; GFX1250-NEXT: s_sub_co_i32 s5, s2, s6
@@ -3924,6 +3927,7 @@ define amdgpu_kernel void @sudiv64(ptr addrspace(1) %out, i64 %x, i64 %y) {
; GFX1250-NEXT: s_cselect_b32 s2, s5, s2
; GFX1250-NEXT: s_add_co_i32 s4, s3, 1
; GFX1250-NEXT: s_cmp_ge_u32 s2, s6
+; GFX1250-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1250-NEXT: s_cselect_b32 s8, s4, s3
; GFX1250-NEXT: .LBB16_5: ; %.split
; GFX1250-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
@@ -4010,7 +4014,7 @@ define amdgpu_kernel void @sudiv64(ptr addrspace(1) %out, i64 %x, i64 %y) {
; GFX13-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX13-NEXT: s_or_b32 s10, s10, s8
; GFX13-NEXT: s_mul_u64 s[8:9], s[4:5], s[10:11]
-; GFX13-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX13-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_3) | instid1(SALU_CYCLE_1)
; GFX13-NEXT: s_sub_co_u32 s6, s2, s8
; GFX13-NEXT: s_cselect_b32 s8, -1, 0
; GFX13-NEXT: s_sub_co_i32 s12, s3, s9
@@ -4018,28 +4022,31 @@ define amdgpu_kernel void @sudiv64(ptr addrspace(1) %out, i64 %x, i64 %y) {
; GFX13-NEXT: s_sub_co_ci_u32 s12, s12, s5
; GFX13-NEXT: s_sub_co_u32 s13, s6, s4
; GFX13-NEXT: s_sub_co_ci_u32 s12, s12, 0
-; GFX13-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX13-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX13-NEXT: s_cmp_ge_u32 s12, s5
; GFX13-NEXT: s_cselect_b32 s14, -1, 0
; GFX13-NEXT: s_cmp_ge_u32 s13, s4
+; GFX13-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX13-NEXT: s_cselect_b32 s15, -1, 0
; GFX13-NEXT: s_cmp_eq_u32 s12, s5
; GFX13-NEXT: s_add_nc_u64 s[12:13], s[10:11], 1
; GFX13-NEXT: s_cselect_b32 s16, s15, s14
; GFX13-NEXT: s_add_nc_u64 s[14:15], s[10:11], 2
; GFX13-NEXT: s_cmp_lg_u32 s16, 0
+; GFX13-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
; GFX13-NEXT: s_cselect_b32 s12, s14, s12
; GFX13-NEXT: s_cselect_b32 s13, s15, s13
; GFX13-NEXT: s_cmp_lg_u32 s8, 0
; GFX13-NEXT: s_sub_co_ci_u32 s3, s3, s9
-; GFX13-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX13-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX13-NEXT: s_cmp_ge_u32 s3, s5
; GFX13-NEXT: s_cselect_b32 s8, -1, 0
; GFX13-NEXT: s_cmp_ge_u32 s6, s4
+; GFX13-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX13-NEXT: s_cselect_b32 s6, -1, 0
; GFX13-NEXT: s_cmp_eq_u32 s3, s5
; GFX13-NEXT: s_cselect_b32 s3, s6, s8
-; GFX13-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX13-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX13-NEXT: s_cmp_lg_u32 s3, 0
; GFX13-NEXT: s_cselect_b32 s9, s13, s11
; GFX13-NEXT: s_cselect_b32 s8, s12, s10
@@ -4052,6 +4059,7 @@ define amdgpu_kernel void @sudiv64(ptr addrspace(1) %out, i64 %x, i64 %y) {
; GFX13-NEXT: s_and_b32 s3, s7, exec_lo
; GFX13-NEXT: s_cselect_b32 s3, 1, 0
; GFX13-NEXT: s_cmp_lg_u32 s3, 1
+; GFX13-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX13-NEXT: s_cbranch_scc1 .LBB16_5
; GFX13-NEXT: ; %bb.4:
; GFX13-NEXT: v_cvt_f32_u32_e32 v0, s4
@@ -4070,7 +4078,7 @@ define amdgpu_kernel void @sudiv64(ptr addrspace(1) %out, i64 %x, i64 %y) {
; GFX13-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX13-NEXT: s_mul_hi_u32 s3, s2, s3
; GFX13-NEXT: s_mul_i32 s5, s3, s4
-; GFX13-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX13-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_3) | instid1(SALU_CYCLE_1)
; GFX13-NEXT: s_sub_co_i32 s2, s2, s5
; GFX13-NEXT: s_add_co_i32 s5, s3, 1
; GFX13-NEXT: s_sub_co_i32 s6, s2, s4
@@ -4079,6 +4087,7 @@ define amdgpu_kernel void @sudiv64(ptr addrspace(1) %out, i64 %x, i64 %y) {
; GFX13-NEXT: s_cselect_b32 s2, s6, s2
; GFX13-NEXT: s_add_co_i32 s5, s3, 1
; GFX13-NEXT: s_cmp_ge_u32 s2, s4
+; GFX13-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX13-NEXT: s_cselect_b32 s8, s5, s3
; GFX13-NEXT: .LBB16_5: ; %.split
; GFX13-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
diff --git a/llvm/test/CodeGen/AMDGPU/clmul.ll b/llvm/test/CodeGen/AMDGPU/clmul.ll
index 3b04923ea0794a..8d48745940900d 100644
--- a/llvm/test/CodeGen/AMDGPU/clmul.ll
+++ b/llvm/test/CodeGen/AMDGPU/clmul.ll
@@ -1501,174 +1501,203 @@ define amdgpu_kernel void @test_clmulr_i32(ptr addrspace(1) %out, ptr addrspace(
; GFX11-NEXT: s_lshl_b64 s[14:15], s[2:3], 2
; GFX11-NEXT: s_xor_b64 s[8:9], s[8:9], s[12:13]
; GFX11-NEXT: s_cmp_eq_u64 s[10:11], 0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cselect_b32 s13, 0, s15
; GFX11-NEXT: s_cselect_b32 s12, 0, s14
; GFX11-NEXT: s_and_b32 s10, s4, 8
; GFX11-NEXT: s_lshl_b64 s[14:15], s[2:3], 3
; GFX11-NEXT: s_xor_b64 s[8:9], s[8:9], s[12:13]
; GFX11-NEXT: s_cmp_eq_u64 s[10:11], 0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cselect_b32 s13, 0, s15
; GFX11-NEXT: s_cselect_b32 s12, 0, s14
; GFX11-NEXT: s_and_b32 s10, s4, 16
; GFX11-NEXT: s_lshl_b64 s[14:15], s[2:3], 4
; GFX11-NEXT: s_xor_b64 s[8:9], s[8:9], s[12:13]
; GFX11-NEXT: s_cmp_eq_u64 s[10:11], 0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cselect_b32 s13, 0, s15
; GFX11-NEXT: s_cselect_b32 s12, 0, s14
; GFX11-NEXT: s_and_b32 s10, s4, 32
; GFX11-NEXT: s_lshl_b64 s[14:15], s[2:3], 5
; GFX11-NEXT: s_xor_b64 s[8:9], s[8:9], s[12:13]
; GFX11-NEXT: s_cmp_eq_u64 s[10:11], 0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cselect_b32 s13, 0, s15
; GFX11-NEXT: s_cselect_b32 s12, 0, s14
; GFX11-NEXT: s_and_b32 s10, s4, 64
; GFX11-NEXT: s_lshl_b64 s[14:15], s[2:3], 6
; GFX11-NEXT: s_xor_b64 s[8:9], s[8:9], s[12:13]
; GFX11-NEXT: s_cmp_eq_u64 s[10:11], 0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cselect_b32 s13, 0, s15
; GFX11-NEXT: s_cselect_b32 s12, 0, s14
; GFX11-NEXT: s_and_b32 s10, s4, 0x80
; GFX11-NEXT: s_lshl_b64 s[14:15], s[2:3], 7
; GFX11-NEXT: s_xor_b64 s[8:9], s[8:9], s[12:13]
; GFX11-NEXT: s_cmp_eq_u64 s[10:11], 0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cselect_b32 s13, 0, s15
; GFX11-NEXT: s_cselect_b32 s12, 0, s14
; GFX11-NEXT: s_and_b32 s10, s4, 0x100
; GFX11-NEXT: s_lshl_b64 s[14:15], s[2:3], 8
; GFX11-NEXT: s_xor_b64 s[8:9], s[8:9], s[12:13]
; GFX11-NEXT: s_cmp_eq_u64 s[10:11], 0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cselect_b32 s13, 0, s15
; GFX11-NEXT: s_cselect_b32 s12, 0, s14
; GFX11-NEXT: s_and_b32 s10, s4, 0x200
; GFX11-NEXT: s_lshl_b64 s[14:15], s[2:3], 9
; GFX11-NEXT: s_xor_b64 s[8:9], s[8:9], s[12:13]
; GFX11-NEXT: s_cmp_eq_u64 s[10:11], 0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cselect_b32 s13, 0, s15
; GFX11-NEXT: s_cselect_b32 s12, 0, s14
; GFX11-NEXT: s_and_b32 s10, s4, 0x400
; GFX11-NEXT: s_lshl_b64 s[14:15], s[2:3], 10
; GFX11-NEXT: s_xor_b64 s[8:9], s[8:9], s[12:13]
; GFX11-NEXT: s_cmp_eq_u64 s[10:11], 0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cselect_b32 s13, 0, s15
; GFX11-NEXT: s_cselect_b32 s12, 0, s14
; GFX11-NEXT: s_and_b32 s10, s4, 0x800
; GFX11-NEXT: s_lshl_b64 s[14:15], s[2:3], 11
; GFX11-NEXT: s_xor_b64 s[8:9], s[8:9], s[12:13]
; GFX11-NEXT: s_cmp_eq_u64 s[10:11], 0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cselect_b32 s13, 0, s15
; GFX11-NEXT: s_cselect_b32 s12, 0, s14
; GFX11-NEXT: s_and_b32 s10, s4, 0x1000
; GFX11-NEXT: s_lshl_b64 s[14:15], s[2:3], 12
; GFX11-NEXT: s_xor_b64 s[8:9], s[8:9], s[12:13]
; GFX11-NEXT: s_cmp_eq_u64 s[10:11], 0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cselect_b32 s13, 0, s15
; GFX11-NEXT: s_cselect_b32 s12, 0, s14
; GFX11-NEXT: s_and_b32 s10, s4, 0x2000
; GFX11-NEXT: s_lshl_b64 s[14:15], s[2:3], 13
; GFX11-NEXT: s_xor_b64 s[8:9], s[8:9], s[12:13]
; GFX11-NEXT: s_cmp_eq_u64 s[10:11], 0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cselect_b32 s13, 0, s15
; GFX11-NEXT: s_cselect_b32 s12, 0, s14
; GFX11-NEXT: s_and_b32 s10, s4, 0x4000
; GFX11-NEXT: s_lshl_b64 s[14:15], s[2:3], 14
; GFX11-NEXT: s_xor_b64 s[8:9], s[8:9], s[12:13]
; GFX11-NEXT: s_cmp_eq_u64 s[10:11], 0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cselect_b32 s13, 0, s15
; GFX11-NEXT: s_cselect_b32 s12, 0, s14
; GFX11-NEXT: s_and_b32 s10, s4, 0x8000
; GFX11-NEXT: s_lshl_b64 s[14:15], s[2:3], 15
; GFX11-NEXT: s_xor_b64 s[8:9], s[8:9], s[12:13]
; GFX11-NEXT: s_cmp_eq_u64 s[10:11], 0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cselect_b32 s13, 0, s15
; GFX11-NEXT: s_cselect_b32 s12, 0, s14
; GFX11-NEXT: s_and_b32 s10, s4, 0x10000
; GFX11-NEXT: s_lshl_b64 s[14:15], s[2:3], 16
; GFX11-NEXT: s_xor_b64 s[8:9], s[8:9], s[12:13]
; GFX11-NEXT: s_cmp_eq_u64 s[10:11], 0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cselect_b32 s13, 0, s15
; GFX11-NEXT: s_cselect_b32 s12, 0, s14
; GFX11-NEXT: s_and_b32 s10, s4, 0x20000
; GFX11-NEXT: s_lshl_b64 s[14:15], s[2:3], 17
; GFX11-NEXT: s_xor_b64 s[8:9], s[8:9], s[12:13]
; GFX11-NEXT: s_cmp_eq_u64 s[10:11], 0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cselect_b32 s13, 0, s15
; GFX11-NEXT: s_cselect_b32 s12, 0, s14
; GFX11-NEXT: s_and_b32 s10, s4, 0x40000
; GFX11-NEXT: s_lshl_b64 s[14:15], s[2:3], 18
; GFX11-NEXT: s_xor_b64 s[8:9], s[8:9], s[12:13]
; GFX11-NEXT: s_cmp_eq_u64 s[10:11], 0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cselect_b32 s13, 0, s15
; GFX11-NEXT: s_cselect_b32 s12, 0, s14
; GFX11-NEXT: s_and_b32 s10, s4, 0x80000
; GFX11-NEXT: s_lshl_b64 s[14:15], s[2:3], 19
; GFX11-NEXT: s_xor_b64 s[8:9], s[8:9], s[12:13]
; GFX11-NEXT: s_cmp_eq_u64 s[10:11], 0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cselect_b32 s13, 0, s15
; GFX11-NEXT: s_cselect_b32 s12, 0, s14
; GFX11-NEXT: s_and_b32 s10, s4, 0x100000
; GFX11-NEXT: s_lshl_b64 s[14:15], s[2:3], 20
; GFX11-NEXT: s_xor_b64 s[8:9], s[8:9], s[12:13]
; GFX11-NEXT: s_cmp_eq_u64 s[10:11], 0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cselect_b32 s13, 0, s15
; GFX11-NEXT: s_cselect_b32 s12, 0, s14
; GFX11-NEXT: s_and_b32 s10, s4, 0x200000
; GFX11-NEXT: s_lshl_b64 s[14:15], s[2:3], 21
; GFX11-NEXT: s_xor_b64 s[8:9], s[8:9], s[12:13]
; GFX11-NEXT: s_cmp_eq_u64 s[10:11], 0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cselect_b32 s13, 0, s15
; GFX11-NEXT: s_cselect_b32 s12, 0, s14
; GFX11-NEXT: s_and_b32 s10, s4, 0x400000
; GFX11-NEXT: s_lshl_b64 s[14:15], s[2:3], 22
; GFX11-NEXT: s_xor_b64 s[8:9], s[8:9], s[12:13]
; GFX11-NEXT: s_cmp_eq_u64 s[10:11], 0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cselect_b32 s13, 0, s15
; GFX11-NEXT: s_cselect_b32 s12, 0, s14
; GFX11-NEXT: s_and_b32 s10, s4, 0x800000
; GFX11-NEXT: s_lshl_b64 s[14:15], s[2:3], 23
; GFX11-NEXT: s_xor_b64 s[8:9], s[8:9], s[12:13]
; GFX11-NEXT: s_cmp_eq_u64 s[10:11], 0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cselect_b32 s13, 0, s15
; GFX11-NEXT: s_cselect_b32 s12, 0, s14
; GFX11-NEXT: s_and_b32 s10, s4, 0x1000000
; GFX11-NEXT: s_lshl_b64 s[14:15], s[2:3], 24
; GFX11-NEXT: s_xor_b64 s[8:9], s[8:9], s[12:13]
; GFX11-NEXT: s_cmp_eq_u64 s[10:11], 0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cselect_b32 s13, 0, s15
; GFX11-NEXT: s_cselect_b32 s12, 0, s14
; GFX11-NEXT: s_and_b32 s10, s4, 0x2000000
; GFX11-NEXT: s_lshl_b64 s[14:15], s[2:3], 25
; GFX11-NEXT: s_xor_b64 s[8:9], s[8:9], s[12:13]
; GFX11-NEXT: s_cmp_eq_u64 s[10:11], 0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cselect_b32 s13, 0, s15
; GFX11-NEXT: s_cselect_b32 s12, 0, s14
; GFX11-NEXT: s_and_b32 s10, s4, 0x4000000
; GFX11-NEXT: s_lshl_b64 s[14:15], s[2:3], 26
; GFX11-NEXT: s_xor_b64 s[8:9], s[8:9], s[12:13]
; GFX11-NEXT: s_cmp_eq_u64 s[10:11], 0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cselect_b32 s13, 0, s15
; GFX11-NEXT: s_cselect_b32 s12, 0, s14
; GFX11-NEXT: s_and_b32 s10, s4, 0x8000000
; GFX11-NEXT: s_lshl_b64 s[14:15], s[2:3], 27
; GFX11-NEXT: s_xor_b64 s[8:9], s[8:9], s[12:13]
; GFX11-NEXT: s_cmp_eq_u64 s[10:11], 0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cselect_b32 s13, 0, s15
; GFX11-NEXT: s_cselect_b32 s12, 0, s14
; GFX11-NEXT: s_and_b32 s10, s4, 0x10000000
; GFX11-NEXT: s_lshl_b64 s[14:15], s[2:3], 28
; GFX11-NEXT: s_xor_b64 s[8:9], s[8:9], s[12:13]
; GFX11-NEXT: s_cmp_eq_u64 s[10:11], 0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cselect_b32 s13, 0, s15
; GFX11-NEXT: s_cselect_b32 s12, 0, s14
; GFX11-NEXT: s_and_b32 s10, s4, 0x20000000
; GFX11-NEXT: s_lshl_b64 s[14:15], s[2:3], 29
; GFX11-NEXT: s_xor_b64 s[8:9], s[8:9], s[12:13]
; GFX11-NEXT: s_cmp_eq_u64 s[10:11], 0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cselect_b32 s13, 0, s15
; GFX11-NEXT: s_cselect_b32 s12, 0, s14
; GFX11-NEXT: s_and_b32 s10, s4, 2.0
; GFX11-NEXT: s_lshl_b64 s[14:15], s[2:3], 30
; GFX11-NEXT: s_xor_b64 s[8:9], s[8:9], s[12:13]
; GFX11-NEXT: s_cmp_eq_u64 s[10:11], 0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cselect_b32 s11, 0, s15
; GFX11-NEXT: s_cselect_b32 s10, 0, s14
; GFX11-NEXT: s_lshl_b64 s[2:3], s[2:3], 31
@@ -2979,174 +3008,203 @@ define amdgpu_kernel void @test_clmulh_i32(ptr addrspace(1) %out, ptr addrspace(
; GFX11-NEXT: s_lshl_b64 s[14:15], s[2:3], 2
; GFX11-NEXT: s_xor_b64 s[8:9], s[8:9], s[12:13]
; GFX11-NEXT: s_cmp_eq_u64 s[10:11], 0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cselect_b32 s13, 0, s15
; GFX11-NEXT: s_cselect_b32 s12, 0, s14
; GFX11-NEXT: s_and_b32 s10, s4, 8
; GFX11-NEXT: s_lshl_b64 s[14:15], s[2:3], 3
; GFX11-NEXT: s_xor_b64 s[8:9], s[8:9], s[12:13]
; GFX11-NEXT: s_cmp_eq_u64 s[10:11], 0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cselect_b32 s13, 0, s15
; GFX11-NEXT: s_cselect_b32 s12, 0, s14
; GFX11-NEXT: s_and_b32 s10, s4, 16
; GFX11-NEXT: s_lshl_b64 s[14:15], s[2:3], 4
; GFX11-NEXT: s_xor_b64 s[8:9], s[8:9], s[12:13]
; GFX11-NEXT: s_cmp_eq_u64 s[10:11], 0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cselect_b32 s13, 0, s15
; GFX11-NEXT: s_cselect_b32 s12, 0, s14
; GFX11-NEXT: s_and_b32 s10, s4, 32
; GFX11-NEXT: s_lshl_b64 s[14:15], s[2:3], 5
; GFX11-NEXT: s_xor_b64 s[8:9], s[8:9], s[12:13]
; GFX11-NEXT: s_cmp_eq_u64 s[10:11], 0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cselect_b32 s13, 0, s15
; GFX11-NEXT: s_cselect_b32 s12, 0, s14
; GFX11-NEXT: s_and_b32 s10, s4, 64
; GFX11-NEXT: s_lshl_b64 s[14:15], s[2:3], 6
; GFX11-NEXT: s_xor_b64 s[8:9], s[8:9], s[12:13]
; GFX11-NEXT: s_cmp_eq_u64 s[10:11], 0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cselect_b32 s13, 0, s15
; GFX11-NEXT: s_cselect_b32 s12, 0, s14
; GFX11-NEXT: s_and_b32 s10, s4, 0x80
; GFX11-NEXT: s_lshl_b64 s[14:15], s[2:3], 7
; GFX11-NEXT: s_xor_b64 s[8:9], s[8:9], s[12:13]
; GFX11-NEXT: s_cmp_eq_u64 s[10:11], 0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cselect_b32 s13, 0, s15
; GFX11-NEXT: s_cselect_b32 s12, 0, s14
; GFX11-NEXT: s_and_b32 s10, s4, 0x100
; GFX11-NEXT: s_lshl_b64 s[14:15], s[2:3], 8
; GFX11-NEXT: s_xor_b64 s[8:9], s[8:9], s[12:13]
; GFX11-NEXT: s_cmp_eq_u64 s[10:11], 0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cselect_b32 s13, 0, s15
; GFX11-NEXT: s_cselect_b32 s12, 0, s14
; GFX11-NEXT: s_and_b32 s10, s4, 0x200
; GFX11-NEXT: s_lshl_b64 s[14:15], s[2:3], 9
; GFX11-NEXT: s_xor_b64 s[8:9], s[8:9], s[12:13]
; GFX11-NEXT: s_cmp_eq_u64 s[10:11], 0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cselect_b32 s13, 0, s15
; GFX11-NEXT: s_cselect_b32 s12, 0, s14
; GFX11-NEXT: s_and_b32 s10, s4, 0x400
; GFX11-NEXT: s_lshl_b64 s[14:15], s[2:3], 10
; GFX11-NEXT: s_xor_b64 s[8:9], s[8:9], s[12:13]
; GFX11-NEXT: s_cmp_eq_u64 s[10:11], 0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cselect_b32 s13, 0, s15
; GFX11-NEXT: s_cselect_b32 s12, 0, s14
; GFX11-NEXT: s_and_b32 s10, s4, 0x800
; GFX11-NEXT: s_lshl_b64 s[14:15], s[2:3], 11
; GFX11-NEXT: s_xor_b64 s[8:9], s[8:9], s[12:13]
; GFX11-NEXT: s_cmp_eq_u64 s[10:11], 0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cselect_b32 s13, 0, s15
; GFX11-NEXT: s_cselect_b32 s12, 0, s14
; GFX11-NEXT: s_and_b32 s10, s4, 0x1000
; GFX11-NEXT: s_lshl_b64 s[14:15], s[2:3], 12
; GFX11-NEXT: s_xor_b64 s[8:9], s[8:9], s[12:13]
; GFX11-NEXT: s_cmp_eq_u64 s[10:11], 0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cselect_b32 s13, 0, s15
; GFX11-NEXT: s_cselect_b32 s12, 0, s14
; GFX11-NEXT: s_and_b32 s10, s4, 0x2000
; GFX11-NEXT: s_lshl_b64 s[14:15], s[2:3], 13
; GFX11-NEXT: s_xor_b64 s[8:9], s[8:9], s[12:13]
; GFX11-NEXT: s_cmp_eq_u64 s[10:11], 0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cselect_b32 s13, 0, s15
; GFX11-NEXT: s_cselect_b32 s12, 0, s14
; GFX11-NEXT: s_and_b32 s10, s4, 0x4000
; GFX11-NEXT: s_lshl_b64 s[14:15], s[2:3], 14
; GFX11-NEXT: s_xor_b64 s[8:9], s[8:9], s[12:13]
; GFX11-NEXT: s_cmp_eq_u64 s[10:11], 0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cselect_b32 s13, 0, s15
; GFX11-NEXT: s_cselect_b32 s12, 0, s14
; GFX11-NEXT: s_and_b32 s10, s4, 0x8000
; GFX11-NEXT: s_lshl_b64 s[14:15], s[2:3], 15
; GFX11-NEXT: s_xor_b64 s[8:9], s[8:9], s[12:13]
; GFX11-NEXT: s_cmp_eq_u64 s[10:11], 0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cselect_b32 s13, 0, s15
; GFX11-NEXT: s_cselect_b32 s12, 0, s14
; GFX11-NEXT: s_and_b32 s10, s4, 0x10000
; GFX11-NEXT: s_lshl_b64 s[14:15], s[2:3], 16
; GFX11-NEXT: s_xor_b64 s[8:9], s[8:9], s[12:13]
; GFX11-NEXT: s_cmp_eq_u64 s[10:11], 0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cselect_b32 s13, 0, s15
; GFX11-NEXT: s_cselect_b32 s12, 0, s14
; GFX11-NEXT: s_and_b32 s10, s4, 0x20000
; GFX11-NEXT: s_lshl_b64 s[14:15], s[2:3], 17
; GFX11-NEXT: s_xor_b64 s[8:9], s[8:9], s[12:13]
; GFX11-NEXT: s_cmp_eq_u64 s[10:11], 0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cselect_b32 s13, 0, s15
; GFX11-NEXT: s_cselect_b32 s12, 0, s14
; GFX11-NEXT: s_and_b32 s10, s4, 0x40000
; GFX11-NEXT: s_lshl_b64 s[14:15], s[2:3], 18
; GFX11-NEXT: s_xor_b64 s[8:9], s[8:9], s[12:13]
; GFX11-NEXT: s_cmp_eq_u64 s[10:11], 0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cselect_b32 s13, 0, s15
; GFX11-NEXT: s_cselect_b32 s12, 0, s14
; GFX11-NEXT: s_and_b32 s10, s4, 0x80000
; GFX11-NEXT: s_lshl_b64 s[14:15], s[2:3], 19
; GFX11-NEXT: s_xor_b64 s[8:9], s[8:9], s[12:13]
; GFX11-NEXT: s_cmp_eq_u64 s[10:11], 0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cselect_b32 s13, 0, s15
; GFX11-NEXT: s_cselect_b32 s12, 0, s14
; GFX11-NEXT: s_and_b32 s10, s4, 0x100000
; GFX11-NEXT: s_lshl_b64 s[14:15], s[2:3], 20
; GFX11-NEXT: s_xor_b64 s[8:9], s[8:9], s[12:13]
; GFX11-NEXT: s_cmp_eq_u64 s[10:11], 0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cselect_b32 s13, 0, s15
; GFX11-NEXT: s_cselect_b32 s12, 0, s14
; GFX11-NEXT: s_and_b32 s10, s4, 0x200000
; GFX11-NEXT: s_lshl_b64 s[14:15], s[2:3], 21
; GFX11-NEXT: s_xor_b64 s[8:9], s[8:9], s[12:13]
; GFX11-NEXT: s_cmp_eq_u64 s[10:11], 0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cselect_b32 s13, 0, s15
; GFX11-NEXT: s_cselect_b32 s12, 0, s14
; GFX11-NEXT: s_and_b32 s10, s4, 0x400000
; GFX11-NEXT: s_lshl_b64 s[14:15], s[2:3], 22
; GFX11-NEXT: s_xor_b64 s[8:9], s[8:9], s[12:13]
; GFX11-NEXT: s_cmp_eq_u64 s[10:11], 0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cselect_b32 s13, 0, s15
; GFX11-NEXT: s_cselect_b32 s12, 0, s14
; GFX11-NEXT: s_and_b32 s10, s4, 0x800000
; GFX11-NEXT: s_lshl_b64 s[14:15], s[2:3], 23
; GFX11-NEXT: s_xor_b64 s[8:9], s[8:9], s[12:13]
; GFX11-NEXT: s_cmp_eq_u64 s[10:11], 0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cselect_b32 s13, 0, s15
; GFX11-NEXT: s_cselect_b32 s12, 0, s14
; GFX11-NEXT: s_and_b32 s10, s4, 0x1000000
; GFX11-NEXT: s_lshl_b64 s[14:15], s[2:3], 24
; GFX11-NEXT: s_xor_b64 s[8:9], s[8:9], s[12:13]
; GFX11-NEXT: s_cmp_eq_u64 s[10:11], 0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cselect_b32 s13, 0, s15
; GFX11-NEXT: s_cselect_b32 s12, 0, s14
; GFX11-NEXT: s_and_b32 s10, s4, 0x2000000
; GFX11-NEXT: s_lshl_b64 s[14:15], s[2:3], 25
; GFX11-NEXT: s_xor_b64 s[8:9], s[8:9], s[12:13]
; GFX11-NEXT: s_cmp_eq_u64 s[10:11], 0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cselect_b32 s13, 0, s15
; GFX11-NEXT: s_cselect_b32 s12, 0, s14
; GFX11-NEXT: s_and_b32 s10, s4, 0x4000000
; GFX11-NEXT: s_lshl_b64 s[14:15], s[2:3], 26
; GFX11-NEXT: s_xor_b64 s[8:9], s[8:9], s[12:13]
; GFX11-NEXT: s_cmp_eq_u64 s[10:11], 0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cselect_b32 s13, 0, s15
; GFX11-NEXT: s_cselect_b32 s12, 0, s14
; GFX11-NEXT: s_and_b32 s10, s4, 0x8000000
; GFX11-NEXT: s_lshl_b64 s[14:15], s[2:3], 27
; GFX11-NEXT: s_xor_b64 s[8:9], s[8:9], s[12:13]
; GFX11-NEXT: s_cmp_eq_u64 s[10:11], 0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cselect_b32 s13, 0, s15
; GFX11-NEXT: s_cselect_b32 s12, 0, s14
; GFX11-NEXT: s_and_b32 s10, s4, 0x10000000
; GFX11-NEXT: s_lshl_b64 s[14:15], s[2:3], 28
; GFX11-NEXT: s_xor_b64 s[8:9], s[8:9], s[12:13]
; GFX11-NEXT: s_cmp_eq_u64 s[10:11], 0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cselect_b32 s13, 0, s15
; GFX11-NEXT: s_cselect_b32 s12, 0, s14
; GFX11-NEXT: s_and_b32 s10, s4, 0x20000000
; GFX11-NEXT: s_lshl_b64 s[14:15], s[2:3], 29
; GFX11-NEXT: s_xor_b64 s[8:9], s[8:9], s[12:13]
; GFX11-NEXT: s_cmp_eq_u64 s[10:11], 0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cselect_b32 s13, 0, s15
; GFX11-NEXT: s_cselect_b32 s12, 0, s14
; GFX11-NEXT: s_and_b32 s10, s4, 2.0
; GFX11-NEXT: s_lshl_b64 s[14:15], s[2:3], 30
; GFX11-NEXT: s_xor_b64 s[8:9], s[8:9], s[12:13]
; GFX11-NEXT: s_cmp_eq_u64 s[10:11], 0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cselect_b32 s11, 0, s15
; GFX11-NEXT: s_cselect_b32 s10, 0, s14
; GFX11-NEXT: s_lshl_b64 s[2:3], s[2:3], 31
diff --git a/llvm/test/CodeGen/AMDGPU/code-size-estimate.ll b/llvm/test/CodeGen/AMDGPU/code-size-estimate.ll
index 5dd20fd4b55bf6..bc4354771846db 100644
--- a/llvm/test/CodeGen/AMDGPU/code-size-estimate.ll
+++ b/llvm/test/CodeGen/AMDGPU/code-size-estimate.ll
@@ -722,7 +722,6 @@ define i64 @v_add_u64_vop2_literal_32(i64 %x) {
; GFX11: ; %bb.0:
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0) ; encoding: [0x00,0x00,0x89,0xbf]
; GFX11-NEXT: v_add_co_u32 v0, vcc_lo, 0x7b, v0 ; encoding: [0x00,0x6a,0x00,0xd7,0xff,0x00,0x02,0x02,0x7b,0x00,0x00,0x00]
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) ; encoding: [0x01,0x00,0x87,0xbf]
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo ; encoding: [0x01,0x7c,0x20,0xd5,0x80,0x02,0xaa,0x01]
; GFX11-NEXT: s_setpc_b64 s[30:31] ; encoding: [0x1e,0x48,0x80,0xbe]
;
@@ -750,8 +749,8 @@ define i64 @v_add_u64_vop2_literal_32(i64 %x) {
; GFX9: codeLenInByte = 20
; GFX10: codeLenInByte = 28
-; GFX1100: codeLenInByte = 32
-; GFX1150: codeLenInByte = 32
+; GFX1100: codeLenInByte = 28
+; GFX1150: codeLenInByte = 28
; GFX1250: codeLenInByte = 20
define i64 @v_add_u64_vop2_literal_64(i64 %x) {
@@ -773,7 +772,6 @@ define i64 @v_add_u64_vop2_literal_64(i64 %x) {
; GFX11: ; %bb.0:
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0) ; encoding: [0x00,0x00,0x89,0xbf]
; GFX11-NEXT: v_add_co_u32 v0, vcc_lo, 0x12345678, v0 ; encoding: [0x00,0x6a,0x00,0xd7,0xff,0x00,0x02,0x02,0x78,0x56,0x34,0x12]
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) ; encoding: [0x01,0x00,0x87,0xbf]
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, 1, v1, vcc_lo ; encoding: [0x01,0x7c,0x20,0xd5,0x81,0x02,0xaa,0x01]
; GFX11-NEXT: s_setpc_b64 s[30:31] ; encoding: [0x1e,0x48,0x80,0xbe]
;
@@ -801,8 +799,8 @@ define i64 @v_add_u64_vop2_literal_64(i64 %x) {
; GFX9: codeLenInByte = 20
; GFX10: codeLenInByte = 28
-; GFX1100: codeLenInByte = 32
-; GFX1150: codeLenInByte = 32
+; GFX1100: codeLenInByte = 28
+; GFX1150: codeLenInByte = 28
; GFX1250: codeLenInByte = 24
;; NOTE: These prefixes are unused and the list is autogenerated. Do not add tests below this line:
; NOT-GFX12: {{.*}}
diff --git a/llvm/test/CodeGen/AMDGPU/combine-add-zext-xor.ll b/llvm/test/CodeGen/AMDGPU/combine-add-zext-xor.ll
index 1053d0304b0817..84fc6eb52ff6ee 100644
--- a/llvm/test/CodeGen/AMDGPU/combine-add-zext-xor.ll
+++ b/llvm/test/CodeGen/AMDGPU/combine-add-zext-xor.ll
@@ -63,7 +63,7 @@ define i32 @combine_add_zext_xor(i32 inreg %cond) {
; GFX1100-NEXT: ; =>This Inner Loop Header: Depth=1
; GFX1100-NEXT: s_and_b32 s2, s0, exec_lo
; GFX1100-NEXT: s_cselect_b32 s2, 1, 0
-; GFX1100-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1100-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1100-NEXT: s_cmp_lg_u32 s2, 1
; GFX1100-NEXT: ; implicit-def: $sgpr2
; GFX1100-NEXT: s_cbranch_scc1 .LBB0_1
@@ -74,6 +74,7 @@ define i32 @combine_add_zext_xor(i32 inreg %cond) {
; GFX1100-NEXT: s_waitcnt vmcnt(0)
; GFX1100-NEXT: v_readfirstlane_b32 s2, v0
; GFX1100-NEXT: s_cmp_eq_u32 s2, 0
+; GFX1100-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1100-NEXT: s_cselect_b32 s2, -1, 0
; GFX1100-NEXT: s_branch .LBB0_1
; GFX1100-NEXT: .LBB0_4: ; %.exit
@@ -165,7 +166,7 @@ define i32 @combine_sub_zext_xor(i32 inreg %cond) {
; GFX1100-NEXT: ; =>This Inner Loop Header: Depth=1
; GFX1100-NEXT: s_and_b32 s2, s0, exec_lo
; GFX1100-NEXT: s_cselect_b32 s2, 1, 0
-; GFX1100-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1100-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1100-NEXT: s_cmp_lg_u32 s2, 1
; GFX1100-NEXT: ; implicit-def: $sgpr2
; GFX1100-NEXT: s_cbranch_scc1 .LBB1_1
@@ -176,6 +177,7 @@ define i32 @combine_sub_zext_xor(i32 inreg %cond) {
; GFX1100-NEXT: s_waitcnt vmcnt(0)
; GFX1100-NEXT: v_readfirstlane_b32 s2, v0
; GFX1100-NEXT: s_cmp_eq_u32 s2, 0
+; GFX1100-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1100-NEXT: s_cselect_b32 s2, -1, 0
; GFX1100-NEXT: s_branch .LBB1_1
; GFX1100-NEXT: .LBB1_4: ; %.exit
@@ -254,6 +256,7 @@ define i32 @combine_add_zext_or(i32 inreg %cond) {
; GFX1100-NEXT: .LBB2_1: ; %bb9
; GFX1100-NEXT: ; in Loop: Header=BB2_2 Depth=1
; GFX1100-NEXT: s_cmpk_gt_i32 s0, 0xfbe6
+; GFX1100-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
; GFX1100-NEXT: s_cselect_b32 s3, -1, 0
; GFX1100-NEXT: s_add_i32 s0, s0, 1
; GFX1100-NEXT: s_and_b32 vcc_lo, exec_lo, s3
@@ -262,7 +265,7 @@ define i32 @combine_add_zext_or(i32 inreg %cond) {
; GFX1100-NEXT: ; =>This Inner Loop Header: Depth=1
; GFX1100-NEXT: s_and_b32 s2, s1, exec_lo
; GFX1100-NEXT: s_cselect_b32 s2, 1, 0
-; GFX1100-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1100-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1100-NEXT: s_cmp_lg_u32 s2, 1
; GFX1100-NEXT: ; implicit-def: $sgpr2
; GFX1100-NEXT: s_cbranch_scc1 .LBB2_1
@@ -354,6 +357,7 @@ define i32 @combine_sub_zext_or(i32 inreg %cond) {
; GFX1100-NEXT: .LBB3_1: ; %bb9
; GFX1100-NEXT: ; in Loop: Header=BB3_2 Depth=1
; GFX1100-NEXT: s_cmpk_gt_i32 s0, 0xfbe6
+; GFX1100-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
; GFX1100-NEXT: s_cselect_b32 s3, -1, 0
; GFX1100-NEXT: s_add_i32 s0, s0, -1
; GFX1100-NEXT: s_and_b32 vcc_lo, exec_lo, s3
@@ -362,7 +366,7 @@ define i32 @combine_sub_zext_or(i32 inreg %cond) {
; GFX1100-NEXT: ; =>This Inner Loop Header: Depth=1
; GFX1100-NEXT: s_and_b32 s2, s1, exec_lo
; GFX1100-NEXT: s_cselect_b32 s2, 1, 0
-; GFX1100-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1100-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1100-NEXT: s_cmp_lg_u32 s2, 1
; GFX1100-NEXT: ; implicit-def: $sgpr2
; GFX1100-NEXT: s_cbranch_scc1 .LBB3_1
@@ -457,9 +461,10 @@ define i32 @combine_add_zext_and(i32 inreg %cond) {
; GFX1100-NEXT: .LBB4_1: ; %bb9
; GFX1100-NEXT: ; in Loop: Header=BB4_2 Depth=1
; GFX1100-NEXT: s_cmpk_gt_i32 s0, 0xfbe6
-; GFX1100-NEXT: s_cselect_b32 s3, -1, 0
; GFX1100-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
+; GFX1100-NEXT: s_cselect_b32 s3, -1, 0
; GFX1100-NEXT: s_and_b32 s2, s2, s3
+; GFX1100-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1100-NEXT: s_and_b32 s2, s2, exec_lo
; GFX1100-NEXT: s_cselect_b32 s2, 1, 0
; GFX1100-NEXT: s_and_b32 vcc_lo, exec_lo, s3
@@ -469,7 +474,7 @@ define i32 @combine_add_zext_and(i32 inreg %cond) {
; GFX1100-NEXT: ; =>This Inner Loop Header: Depth=1
; GFX1100-NEXT: s_and_b32 s2, s1, exec_lo
; GFX1100-NEXT: s_cselect_b32 s2, 1, 0
-; GFX1100-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1100-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1100-NEXT: s_cmp_lg_u32 s2, 1
; GFX1100-NEXT: ; implicit-def: $sgpr2
; GFX1100-NEXT: s_cbranch_scc1 .LBB4_1
@@ -480,6 +485,7 @@ define i32 @combine_add_zext_and(i32 inreg %cond) {
; GFX1100-NEXT: s_waitcnt vmcnt(0)
; GFX1100-NEXT: v_readfirstlane_b32 s2, v0
; GFX1100-NEXT: s_cmp_eq_u32 s2, 0
+; GFX1100-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1100-NEXT: s_cselect_b32 s2, -1, 0
; GFX1100-NEXT: s_branch .LBB4_1
; GFX1100-NEXT: .LBB4_4: ; %.exit
@@ -562,9 +568,10 @@ define i32 @combine_sub_zext_and(i32 inreg %cond) {
; GFX1100-NEXT: .LBB5_1: ; %bb9
; GFX1100-NEXT: ; in Loop: Header=BB5_2 Depth=1
; GFX1100-NEXT: s_cmpk_gt_i32 s0, 0xfbe6
-; GFX1100-NEXT: s_cselect_b32 s3, -1, 0
; GFX1100-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
+; GFX1100-NEXT: s_cselect_b32 s3, -1, 0
; GFX1100-NEXT: s_and_b32 s2, s2, s3
+; GFX1100-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1100-NEXT: s_and_b32 s2, s2, exec_lo
; GFX1100-NEXT: s_cselect_b32 s2, 1, 0
; GFX1100-NEXT: s_and_b32 vcc_lo, exec_lo, s3
@@ -574,7 +581,7 @@ define i32 @combine_sub_zext_and(i32 inreg %cond) {
; GFX1100-NEXT: ; =>This Inner Loop Header: Depth=1
; GFX1100-NEXT: s_and_b32 s2, s1, exec_lo
; GFX1100-NEXT: s_cselect_b32 s2, 1, 0
-; GFX1100-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1100-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1100-NEXT: s_cmp_lg_u32 s2, 1
; GFX1100-NEXT: ; implicit-def: $sgpr2
; GFX1100-NEXT: s_cbranch_scc1 .LBB5_1
@@ -585,6 +592,7 @@ define i32 @combine_sub_zext_and(i32 inreg %cond) {
; GFX1100-NEXT: s_waitcnt vmcnt(0)
; GFX1100-NEXT: v_readfirstlane_b32 s2, v0
; GFX1100-NEXT: s_cmp_eq_u32 s2, 0
+; GFX1100-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1100-NEXT: s_cselect_b32 s2, -1, 0
; GFX1100-NEXT: s_branch .LBB5_1
; GFX1100-NEXT: .LBB5_4: ; %.exit
diff --git a/llvm/test/CodeGen/AMDGPU/commute-compares-scalar-float.ll b/llvm/test/CodeGen/AMDGPU/commute-compares-scalar-float.ll
index 6c654c8be097ec..134aa77ff12545 100644
--- a/llvm/test/CodeGen/AMDGPU/commute-compares-scalar-float.ll
+++ b/llvm/test/CodeGen/AMDGPU/commute-compares-scalar-float.ll
@@ -6,8 +6,8 @@ define amdgpu_vs void @fcmp_f32_olt_to_ogt(ptr addrspace(1) inreg %out, float in
; SDAG-LABEL: fcmp_f32_olt_to_ogt:
; SDAG: ; %bb.0: ; %entry
; SDAG-NEXT: s_cmp_gt_f32 s2, 2.0
+; SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_3) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; SDAG-NEXT: s_cselect_b32 s2, -1, 0
-; SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; SDAG-NEXT: v_dual_mov_b32 v0, 0 :: v_dual_mov_b32 v1, s2
; SDAG-NEXT: global_store_b32 v0, v1, s[0:1]
; SDAG-NEXT: s_endpgm
@@ -16,8 +16,8 @@ define amdgpu_vs void @fcmp_f32_olt_to_ogt(ptr addrspace(1) inreg %out, float in
; GISEL: ; %bb.0: ; %entry
; GISEL-NEXT: s_cmp_gt_f32 s2, 2.0
; GISEL-NEXT: v_mov_b32_e32 v1, 0
+; GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_2) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GISEL-NEXT: s_cselect_b32 s2, -1, 0
-; GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GISEL-NEXT: v_mov_b32_e32 v0, s2
; GISEL-NEXT: global_store_b32 v1, v0, s[0:1]
; GISEL-NEXT: s_endpgm
@@ -32,8 +32,8 @@ define amdgpu_vs void @fcmp_f32_ogt_to_olt(ptr addrspace(1) inreg %out, float in
; SDAG-LABEL: fcmp_f32_ogt_to_olt:
; SDAG: ; %bb.0: ; %entry
; SDAG-NEXT: s_cmp_lt_f32 s2, 2.0
+; SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_3) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; SDAG-NEXT: s_cselect_b32 s2, -1, 0
-; SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; SDAG-NEXT: v_dual_mov_b32 v0, 0 :: v_dual_mov_b32 v1, s2
; SDAG-NEXT: global_store_b32 v0, v1, s[0:1]
; SDAG-NEXT: s_endpgm
@@ -42,8 +42,8 @@ define amdgpu_vs void @fcmp_f32_ogt_to_olt(ptr addrspace(1) inreg %out, float in
; GISEL: ; %bb.0: ; %entry
; GISEL-NEXT: s_cmp_lt_f32 s2, 2.0
; GISEL-NEXT: v_mov_b32_e32 v1, 0
+; GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_2) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GISEL-NEXT: s_cselect_b32 s2, -1, 0
-; GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GISEL-NEXT: v_mov_b32_e32 v0, s2
; GISEL-NEXT: global_store_b32 v1, v0, s[0:1]
; GISEL-NEXT: s_endpgm
@@ -58,8 +58,8 @@ define amdgpu_vs void @fcmp_f32_ole_to_oge(ptr addrspace(1) inreg %out, float in
; SDAG-LABEL: fcmp_f32_ole_to_oge:
; SDAG: ; %bb.0: ; %entry
; SDAG-NEXT: s_cmp_ge_f32 s2, 2.0
+; SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_3) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; SDAG-NEXT: s_cselect_b32 s2, -1, 0
-; SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; SDAG-NEXT: v_dual_mov_b32 v0, 0 :: v_dual_mov_b32 v1, s2
; SDAG-NEXT: global_store_b32 v0, v1, s[0:1]
; SDAG-NEXT: s_endpgm
@@ -68,8 +68,8 @@ define amdgpu_vs void @fcmp_f32_ole_to_oge(ptr addrspace(1) inreg %out, float in
; GISEL: ; %bb.0: ; %entry
; GISEL-NEXT: s_cmp_ge_f32 s2, 2.0
; GISEL-NEXT: v_mov_b32_e32 v1, 0
+; GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_2) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GISEL-NEXT: s_cselect_b32 s2, -1, 0
-; GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GISEL-NEXT: v_mov_b32_e32 v0, s2
; GISEL-NEXT: global_store_b32 v1, v0, s[0:1]
; GISEL-NEXT: s_endpgm
@@ -84,8 +84,8 @@ define amdgpu_vs void @fcmp_f32_oge_to_ole(ptr addrspace(1) inreg %out, float in
; SDAG-LABEL: fcmp_f32_oge_to_ole:
; SDAG: ; %bb.0: ; %entry
; SDAG-NEXT: s_cmp_le_f32 s2, 2.0
+; SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_3) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; SDAG-NEXT: s_cselect_b32 s2, -1, 0
-; SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; SDAG-NEXT: v_dual_mov_b32 v0, 0 :: v_dual_mov_b32 v1, s2
; SDAG-NEXT: global_store_b32 v0, v1, s[0:1]
; SDAG-NEXT: s_endpgm
@@ -94,8 +94,8 @@ define amdgpu_vs void @fcmp_f32_oge_to_ole(ptr addrspace(1) inreg %out, float in
; GISEL: ; %bb.0: ; %entry
; GISEL-NEXT: s_cmp_le_f32 s2, 2.0
; GISEL-NEXT: v_mov_b32_e32 v1, 0
+; GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_2) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GISEL-NEXT: s_cselect_b32 s2, -1, 0
-; GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GISEL-NEXT: v_mov_b32_e32 v0, s2
; GISEL-NEXT: global_store_b32 v1, v0, s[0:1]
; GISEL-NEXT: s_endpgm
@@ -110,8 +110,8 @@ define amdgpu_vs void @fcmp_f32_ult_to_ugt(ptr addrspace(1) inreg %out, float in
; SDAG-LABEL: fcmp_f32_ult_to_ugt:
; SDAG: ; %bb.0: ; %entry
; SDAG-NEXT: s_cmp_nle_f32 s2, 2.0
+; SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_3) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; SDAG-NEXT: s_cselect_b32 s2, -1, 0
-; SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; SDAG-NEXT: v_dual_mov_b32 v0, 0 :: v_dual_mov_b32 v1, s2
; SDAG-NEXT: global_store_b32 v0, v1, s[0:1]
; SDAG-NEXT: s_endpgm
@@ -120,8 +120,8 @@ define amdgpu_vs void @fcmp_f32_ult_to_ugt(ptr addrspace(1) inreg %out, float in
; GISEL: ; %bb.0: ; %entry
; GISEL-NEXT: s_cmp_nle_f32 s2, 2.0
; GISEL-NEXT: v_mov_b32_e32 v1, 0
+; GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_2) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GISEL-NEXT: s_cselect_b32 s2, -1, 0
-; GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GISEL-NEXT: v_mov_b32_e32 v0, s2
; GISEL-NEXT: global_store_b32 v1, v0, s[0:1]
; GISEL-NEXT: s_endpgm
@@ -136,8 +136,8 @@ define amdgpu_vs void @fcmp_f32_ugt_to_ult(ptr addrspace(1) inreg %out, float in
; SDAG-LABEL: fcmp_f32_ugt_to_ult:
; SDAG: ; %bb.0: ; %entry
; SDAG-NEXT: s_cmp_nge_f32 s2, 2.0
+; SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_3) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; SDAG-NEXT: s_cselect_b32 s2, -1, 0
-; SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; SDAG-NEXT: v_dual_mov_b32 v0, 0 :: v_dual_mov_b32 v1, s2
; SDAG-NEXT: global_store_b32 v0, v1, s[0:1]
; SDAG-NEXT: s_endpgm
@@ -146,8 +146,8 @@ define amdgpu_vs void @fcmp_f32_ugt_to_ult(ptr addrspace(1) inreg %out, float in
; GISEL: ; %bb.0: ; %entry
; GISEL-NEXT: s_cmp_nge_f32 s2, 2.0
; GISEL-NEXT: v_mov_b32_e32 v1, 0
+; GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_2) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GISEL-NEXT: s_cselect_b32 s2, -1, 0
-; GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GISEL-NEXT: v_mov_b32_e32 v0, s2
; GISEL-NEXT: global_store_b32 v1, v0, s[0:1]
; GISEL-NEXT: s_endpgm
@@ -162,8 +162,8 @@ define amdgpu_vs void @fcmp_f32_ule_to_uge(ptr addrspace(1) inreg %out, float in
; SDAG-LABEL: fcmp_f32_ule_to_uge:
; SDAG: ; %bb.0: ; %entry
; SDAG-NEXT: s_cmp_nlt_f32 s2, 2.0
+; SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_3) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; SDAG-NEXT: s_cselect_b32 s2, -1, 0
-; SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; SDAG-NEXT: v_dual_mov_b32 v0, 0 :: v_dual_mov_b32 v1, s2
; SDAG-NEXT: global_store_b32 v0, v1, s[0:1]
; SDAG-NEXT: s_endpgm
@@ -172,8 +172,8 @@ define amdgpu_vs void @fcmp_f32_ule_to_uge(ptr addrspace(1) inreg %out, float in
; GISEL: ; %bb.0: ; %entry
; GISEL-NEXT: s_cmp_nlt_f32 s2, 2.0
; GISEL-NEXT: v_mov_b32_e32 v1, 0
+; GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_2) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GISEL-NEXT: s_cselect_b32 s2, -1, 0
-; GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GISEL-NEXT: v_mov_b32_e32 v0, s2
; GISEL-NEXT: global_store_b32 v1, v0, s[0:1]
; GISEL-NEXT: s_endpgm
@@ -188,8 +188,8 @@ define amdgpu_vs void @fcmp_f32_uge_to_ule(ptr addrspace(1) inreg %out, float in
; SDAG-LABEL: fcmp_f32_uge_to_ule:
; SDAG: ; %bb.0: ; %entry
; SDAG-NEXT: s_cmp_ngt_f32 s2, 2.0
+; SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_3) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; SDAG-NEXT: s_cselect_b32 s2, -1, 0
-; SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; SDAG-NEXT: v_dual_mov_b32 v0, 0 :: v_dual_mov_b32 v1, s2
; SDAG-NEXT: global_store_b32 v0, v1, s[0:1]
; SDAG-NEXT: s_endpgm
@@ -198,8 +198,8 @@ define amdgpu_vs void @fcmp_f32_uge_to_ule(ptr addrspace(1) inreg %out, float in
; GISEL: ; %bb.0: ; %entry
; GISEL-NEXT: s_cmp_ngt_f32 s2, 2.0
; GISEL-NEXT: v_mov_b32_e32 v1, 0
+; GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_2) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GISEL-NEXT: s_cselect_b32 s2, -1, 0
-; GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GISEL-NEXT: v_mov_b32_e32 v0, s2
; GISEL-NEXT: global_store_b32 v1, v0, s[0:1]
; GISEL-NEXT: s_endpgm
@@ -214,8 +214,8 @@ define amdgpu_vs void @fcmp_f16_olt_to_ogt(ptr addrspace(1) inreg %out, half inr
; SDAG-LABEL: fcmp_f16_olt_to_ogt:
; SDAG: ; %bb.0: ; %entry
; SDAG-NEXT: s_cmp_gt_f16 s2, 0x4000
+; SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_3) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; SDAG-NEXT: s_cselect_b32 s2, -1, 0
-; SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; SDAG-NEXT: v_dual_mov_b32 v0, 0 :: v_dual_mov_b32 v1, s2
; SDAG-NEXT: global_store_b32 v0, v1, s[0:1]
; SDAG-NEXT: s_endpgm
@@ -224,8 +224,8 @@ define amdgpu_vs void @fcmp_f16_olt_to_ogt(ptr addrspace(1) inreg %out, half inr
; GISEL: ; %bb.0: ; %entry
; GISEL-NEXT: s_cmp_gt_f16 s2, 0x4000
; GISEL-NEXT: v_mov_b32_e32 v1, 0
+; GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_2) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GISEL-NEXT: s_cselect_b32 s2, -1, 0
-; GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GISEL-NEXT: v_mov_b32_e32 v0, s2
; GISEL-NEXT: global_store_b32 v1, v0, s[0:1]
; GISEL-NEXT: s_endpgm
@@ -240,8 +240,8 @@ define amdgpu_vs void @fcmp_f16_ogt_to_olt(ptr addrspace(1) inreg %out, half inr
; SDAG-LABEL: fcmp_f16_ogt_to_olt:
; SDAG: ; %bb.0: ; %entry
; SDAG-NEXT: s_cmp_lt_f16 s2, 0x4000
+; SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_3) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; SDAG-NEXT: s_cselect_b32 s2, -1, 0
-; SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; SDAG-NEXT: v_dual_mov_b32 v0, 0 :: v_dual_mov_b32 v1, s2
; SDAG-NEXT: global_store_b32 v0, v1, s[0:1]
; SDAG-NEXT: s_endpgm
@@ -250,8 +250,8 @@ define amdgpu_vs void @fcmp_f16_ogt_to_olt(ptr addrspace(1) inreg %out, half inr
; GISEL: ; %bb.0: ; %entry
; GISEL-NEXT: s_cmp_lt_f16 s2, 0x4000
; GISEL-NEXT: v_mov_b32_e32 v1, 0
+; GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_2) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GISEL-NEXT: s_cselect_b32 s2, -1, 0
-; GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GISEL-NEXT: v_mov_b32_e32 v0, s2
; GISEL-NEXT: global_store_b32 v1, v0, s[0:1]
; GISEL-NEXT: s_endpgm
@@ -266,8 +266,8 @@ define amdgpu_vs void @fcmp_f16_ole_to_oge(ptr addrspace(1) inreg %out, half inr
; SDAG-LABEL: fcmp_f16_ole_to_oge:
; SDAG: ; %bb.0: ; %entry
; SDAG-NEXT: s_cmp_ge_f16 s2, 0x4000
+; SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_3) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; SDAG-NEXT: s_cselect_b32 s2, -1, 0
-; SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; SDAG-NEXT: v_dual_mov_b32 v0, 0 :: v_dual_mov_b32 v1, s2
; SDAG-NEXT: global_store_b32 v0, v1, s[0:1]
; SDAG-NEXT: s_endpgm
@@ -276,8 +276,8 @@ define amdgpu_vs void @fcmp_f16_ole_to_oge(ptr addrspace(1) inreg %out, half inr
; GISEL: ; %bb.0: ; %entry
; GISEL-NEXT: s_cmp_ge_f16 s2, 0x4000
; GISEL-NEXT: v_mov_b32_e32 v1, 0
+; GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_2) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GISEL-NEXT: s_cselect_b32 s2, -1, 0
-; GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GISEL-NEXT: v_mov_b32_e32 v0, s2
; GISEL-NEXT: global_store_b32 v1, v0, s[0:1]
; GISEL-NEXT: s_endpgm
@@ -292,8 +292,8 @@ define amdgpu_vs void @fcmp_f16_oge_to_ole(ptr addrspace(1) inreg %out, half inr
; SDAG-LABEL: fcmp_f16_oge_to_ole:
; SDAG: ; %bb.0: ; %entry
; SDAG-NEXT: s_cmp_le_f16 s2, 0x4000
+; SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_3) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; SDAG-NEXT: s_cselect_b32 s2, -1, 0
-; SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; SDAG-NEXT: v_dual_mov_b32 v0, 0 :: v_dual_mov_b32 v1, s2
; SDAG-NEXT: global_store_b32 v0, v1, s[0:1]
; SDAG-NEXT: s_endpgm
@@ -302,8 +302,8 @@ define amdgpu_vs void @fcmp_f16_oge_to_ole(ptr addrspace(1) inreg %out, half inr
; GISEL: ; %bb.0: ; %entry
; GISEL-NEXT: s_cmp_le_f16 s2, 0x4000
; GISEL-NEXT: v_mov_b32_e32 v1, 0
+; GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_2) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GISEL-NEXT: s_cselect_b32 s2, -1, 0
-; GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GISEL-NEXT: v_mov_b32_e32 v0, s2
; GISEL-NEXT: global_store_b32 v1, v0, s[0:1]
; GISEL-NEXT: s_endpgm
@@ -318,8 +318,8 @@ define amdgpu_vs void @fcmp_f16_ult_to_ugt(ptr addrspace(1) inreg %out, half inr
; SDAG-LABEL: fcmp_f16_ult_to_ugt:
; SDAG: ; %bb.0: ; %entry
; SDAG-NEXT: s_cmp_nle_f16 s2, 0x4000
+; SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_3) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; SDAG-NEXT: s_cselect_b32 s2, -1, 0
-; SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; SDAG-NEXT: v_dual_mov_b32 v0, 0 :: v_dual_mov_b32 v1, s2
; SDAG-NEXT: global_store_b32 v0, v1, s[0:1]
; SDAG-NEXT: s_endpgm
@@ -328,8 +328,8 @@ define amdgpu_vs void @fcmp_f16_ult_to_ugt(ptr addrspace(1) inreg %out, half inr
; GISEL: ; %bb.0: ; %entry
; GISEL-NEXT: s_cmp_nle_f16 s2, 0x4000
; GISEL-NEXT: v_mov_b32_e32 v1, 0
+; GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_2) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GISEL-NEXT: s_cselect_b32 s2, -1, 0
-; GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GISEL-NEXT: v_mov_b32_e32 v0, s2
; GISEL-NEXT: global_store_b32 v1, v0, s[0:1]
; GISEL-NEXT: s_endpgm
@@ -344,8 +344,8 @@ define amdgpu_vs void @fcmp_f16_ugt_to_ult(ptr addrspace(1) inreg %out, half inr
; SDAG-LABEL: fcmp_f16_ugt_to_ult:
; SDAG: ; %bb.0: ; %entry
; SDAG-NEXT: s_cmp_nge_f16 s2, 0x4000
+; SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_3) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; SDAG-NEXT: s_cselect_b32 s2, -1, 0
-; SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; SDAG-NEXT: v_dual_mov_b32 v0, 0 :: v_dual_mov_b32 v1, s2
; SDAG-NEXT: global_store_b32 v0, v1, s[0:1]
; SDAG-NEXT: s_endpgm
@@ -354,8 +354,8 @@ define amdgpu_vs void @fcmp_f16_ugt_to_ult(ptr addrspace(1) inreg %out, half inr
; GISEL: ; %bb.0: ; %entry
; GISEL-NEXT: s_cmp_nge_f16 s2, 0x4000
; GISEL-NEXT: v_mov_b32_e32 v1, 0
+; GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_2) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GISEL-NEXT: s_cselect_b32 s2, -1, 0
-; GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GISEL-NEXT: v_mov_b32_e32 v0, s2
; GISEL-NEXT: global_store_b32 v1, v0, s[0:1]
; GISEL-NEXT: s_endpgm
@@ -370,8 +370,8 @@ define amdgpu_vs void @fcmp_ule_to_uge(ptr addrspace(1) inreg %out, half inreg %
; SDAG-LABEL: fcmp_ule_to_uge:
; SDAG: ; %bb.0: ; %entry
; SDAG-NEXT: s_cmp_nlt_f16 s2, 0x4000
+; SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_3) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; SDAG-NEXT: s_cselect_b32 s2, -1, 0
-; SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; SDAG-NEXT: v_dual_mov_b32 v0, 0 :: v_dual_mov_b32 v1, s2
; SDAG-NEXT: global_store_b32 v0, v1, s[0:1]
; SDAG-NEXT: s_endpgm
@@ -380,8 +380,8 @@ define amdgpu_vs void @fcmp_ule_to_uge(ptr addrspace(1) inreg %out, half inreg %
; GISEL: ; %bb.0: ; %entry
; GISEL-NEXT: s_cmp_nlt_f16 s2, 0x4000
; GISEL-NEXT: v_mov_b32_e32 v1, 0
+; GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_2) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GISEL-NEXT: s_cselect_b32 s2, -1, 0
-; GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GISEL-NEXT: v_mov_b32_e32 v0, s2
; GISEL-NEXT: global_store_b32 v1, v0, s[0:1]
; GISEL-NEXT: s_endpgm
@@ -396,8 +396,8 @@ define amdgpu_vs void @fcmp_uge_to_ule(ptr addrspace(1) inreg %out, half inreg %
; SDAG-LABEL: fcmp_uge_to_ule:
; SDAG: ; %bb.0: ; %entry
; SDAG-NEXT: s_cmp_ngt_f16 s2, 0x4000
+; SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_3) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; SDAG-NEXT: s_cselect_b32 s2, -1, 0
-; SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; SDAG-NEXT: v_dual_mov_b32 v0, 0 :: v_dual_mov_b32 v1, s2
; SDAG-NEXT: global_store_b32 v0, v1, s[0:1]
; SDAG-NEXT: s_endpgm
@@ -406,8 +406,8 @@ define amdgpu_vs void @fcmp_uge_to_ule(ptr addrspace(1) inreg %out, half inreg %
; GISEL: ; %bb.0: ; %entry
; GISEL-NEXT: s_cmp_ngt_f16 s2, 0x4000
; GISEL-NEXT: v_mov_b32_e32 v1, 0
+; GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_2) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GISEL-NEXT: s_cselect_b32 s2, -1, 0
-; GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GISEL-NEXT: v_mov_b32_e32 v0, s2
; GISEL-NEXT: global_store_b32 v1, v0, s[0:1]
; GISEL-NEXT: s_endpgm
diff --git a/llvm/test/CodeGen/AMDGPU/ctpop64.ll b/llvm/test/CodeGen/AMDGPU/ctpop64.ll
index edf9ef109a42d9..a61b4e6285be64 100644
--- a/llvm/test/CodeGen/AMDGPU/ctpop64.ll
+++ b/llvm/test/CodeGen/AMDGPU/ctpop64.ll
@@ -535,7 +535,7 @@ define amdgpu_kernel void @ctpop_i64_in_br(ptr addrspace(1) %out, ptr addrspace(
; GFX12-NEXT: ; implicit-def: $sgpr2_sgpr3
; GFX12-NEXT: .LBB7_3: ; %Flow
; GFX12-NEXT: s_xor_b32 s6, s6, 1
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_lg_u32 s6, 0
; GFX12-NEXT: s_cbranch_scc1 .LBB7_5
; GFX12-NEXT: ; %bb.4: ; %if
diff --git a/llvm/test/CodeGen/AMDGPU/d16-write-vgpr32.ll b/llvm/test/CodeGen/AMDGPU/d16-write-vgpr32.ll
index 7fe521102be992..6eb28b83d9b72b 100644
--- a/llvm/test/CodeGen/AMDGPU/d16-write-vgpr32.ll
+++ b/llvm/test/CodeGen/AMDGPU/d16-write-vgpr32.ll
@@ -61,7 +61,7 @@ define amdgpu_kernel void @d16_load_same_order_group(ptr %in1, ptr %in2) {
; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_lshlrev_b64 v[0:1], 1, v[0:1]
; GFX11-NEXT: v_add_co_u32 v4, vcc_lo, s2, v0
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX11-NEXT: v_add_co_ci_u32_e64 v5, null, s3, v1, vcc_lo
; GFX11-NEXT: flat_load_d16_b16 v0, v[2:3]
; GFX11-NEXT: flat_load_d16_hi_b16 v0, v[4:5]
@@ -82,7 +82,7 @@ define amdgpu_kernel void @d16_load_same_order_group(ptr %in1, ptr %in2) {
; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX12-NEXT: v_lshlrev_b64_e32 v[0:1], 1, v[0:1]
; GFX12-NEXT: v_add_co_u32 v4, vcc_lo, s2, v0
-; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX12-NEXT: v_add_co_ci_u32_e64 v5, null, s3, v1, vcc_lo
; GFX12-NEXT: flat_load_d16_b16 v0, v[2:3]
; GFX12-NEXT: flat_load_d16_hi_b16 v0, v[4:5]
diff --git a/llvm/test/CodeGen/AMDGPU/dagcombine-fmul-sel.ll b/llvm/test/CodeGen/AMDGPU/dagcombine-fmul-sel.ll
index d776749dcdbce0..3caea634378764 100644
--- a/llvm/test/CodeGen/AMDGPU/dagcombine-fmul-sel.ll
+++ b/llvm/test/CodeGen/AMDGPU/dagcombine-fmul-sel.ll
@@ -1838,10 +1838,9 @@ define <2 x half> @fmul_select_v2f16_test3(<2 x half> %x, <2 x i32> %bool.arg1,
; GFX11-SDAG-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v4
; GFX11-SDAG-TRUE16-NEXT: v_mov_b16_e32 v2.l, 0x3c00
; GFX11-SDAG-TRUE16-NEXT: v_cmp_eq_u32_e64 s0, v1, v3
-; GFX11-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX11-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-SDAG-TRUE16-NEXT: v_cndmask_b16 v1.h, v2.l, 0x4000, vcc_lo
; GFX11-SDAG-TRUE16-NEXT: v_cndmask_b16 v1.l, v2.l, 0x4000, s0
-; GFX11-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-SDAG-TRUE16-NEXT: v_pk_mul_f16 v0, v0, v1
; GFX11-SDAG-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -2005,10 +2004,9 @@ define <2 x half> @fmul_select_v2f16_test4(<2 x half> %x, <2 x i32> %bool.arg1,
; GFX11-SDAG-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v4
; GFX11-SDAG-TRUE16-NEXT: v_mov_b16_e32 v2.l, 0x3c00
; GFX11-SDAG-TRUE16-NEXT: v_cmp_eq_u32_e64 s0, v1, v3
-; GFX11-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX11-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-SDAG-TRUE16-NEXT: v_cndmask_b16 v1.h, v2.l, 0x3800, vcc_lo
; GFX11-SDAG-TRUE16-NEXT: v_cndmask_b16 v1.l, v2.l, 0x3800, s0
-; GFX11-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-SDAG-TRUE16-NEXT: v_pk_mul_f16 v0, v0, v1
; GFX11-SDAG-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -3276,18 +3274,17 @@ define <2 x bfloat> @fmul_select_v2bf16_test3(<2 x bfloat> %x, <2 x i32> %bool.a
; GFX11-SDAG-TRUE16-NEXT: v_mov_b16_e32 v5.l, 0x3f80
; GFX11-SDAG-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v4
; GFX11-SDAG-TRUE16-NEXT: v_cmp_eq_u32_e64 s0, v1, v3
-; GFX11-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX11-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX11-SDAG-TRUE16-NEXT: v_cndmask_b16 v1.l, v5.l, 0x4000, vcc_lo
; GFX11-SDAG-TRUE16-NEXT: v_cndmask_b16 v2.l, v5.l, 0x4000, s0
-; GFX11-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-SDAG-TRUE16-NEXT: v_lshlrev_b32_e32 v1, 16, v1
+; GFX11-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_2) | instid1(VALU_DEP_1)
; GFX11-SDAG-TRUE16-NEXT: v_lshlrev_b32_e32 v2, 16, v2
; GFX11-SDAG-TRUE16-NEXT: v_and_b32_e32 v3, 0xffff0000, v0
; GFX11-SDAG-TRUE16-NEXT: v_lshlrev_b32_e32 v0, 16, v0
-; GFX11-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-SDAG-TRUE16-NEXT: v_dual_mul_f32 v0, v0, v2 :: v_dual_mul_f32 v1, v3, v1
+; GFX11-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-SDAG-TRUE16-NEXT: v_bfe_u32 v3, v0, 16, 1
-; GFX11-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX11-SDAG-TRUE16-NEXT: v_bfe_u32 v2, v1, 16, 1
; GFX11-SDAG-TRUE16-NEXT: v_or_b32_e32 v4, 0x400000, v1
; GFX11-SDAG-TRUE16-NEXT: v_cmp_u_f32_e32 vcc_lo, v1, v1
@@ -3578,18 +3575,17 @@ define <2 x bfloat> @fmul_select_v2bf16_test4(<2 x bfloat> %x, <2 x i32> %bool.a
; GFX11-SDAG-TRUE16-NEXT: v_mov_b16_e32 v5.l, 0x3f80
; GFX11-SDAG-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v4
; GFX11-SDAG-TRUE16-NEXT: v_cmp_eq_u32_e64 s0, v1, v3
-; GFX11-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX11-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX11-SDAG-TRUE16-NEXT: v_cndmask_b16 v1.l, v5.l, 0x3f00, vcc_lo
; GFX11-SDAG-TRUE16-NEXT: v_cndmask_b16 v2.l, v5.l, 0x3f00, s0
-; GFX11-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-SDAG-TRUE16-NEXT: v_lshlrev_b32_e32 v1, 16, v1
+; GFX11-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_2) | instid1(VALU_DEP_1)
; GFX11-SDAG-TRUE16-NEXT: v_lshlrev_b32_e32 v2, 16, v2
; GFX11-SDAG-TRUE16-NEXT: v_and_b32_e32 v3, 0xffff0000, v0
; GFX11-SDAG-TRUE16-NEXT: v_lshlrev_b32_e32 v0, 16, v0
-; GFX11-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-SDAG-TRUE16-NEXT: v_dual_mul_f32 v0, v0, v2 :: v_dual_mul_f32 v1, v3, v1
+; GFX11-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-SDAG-TRUE16-NEXT: v_bfe_u32 v3, v0, 16, 1
-; GFX11-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX11-SDAG-TRUE16-NEXT: v_bfe_u32 v2, v1, 16, 1
; GFX11-SDAG-TRUE16-NEXT: v_or_b32_e32 v4, 0x400000, v1
; GFX11-SDAG-TRUE16-NEXT: v_cmp_u_f32_e32 vcc_lo, v1, v1
diff --git a/llvm/test/CodeGen/AMDGPU/ds-atomic-barrier-convergent.ll b/llvm/test/CodeGen/AMDGPU/ds-atomic-barrier-convergent.ll
index c42d8b4c9cc143..1f2f1ac28c4130 100644
--- a/llvm/test/CodeGen/AMDGPU/ds-atomic-barrier-convergent.ll
+++ b/llvm/test/CodeGen/AMDGPU/ds-atomic-barrier-convergent.ll
@@ -15,9 +15,10 @@ define void @taildup_ds_atomic_barrier_arrive(ptr addrspace(1) %a, ptr addrspace
; GCN-NEXT: v_dual_mov_b32 v7, v4 :: v_dual_mov_b32 v6, v3
; GCN-NEXT: v_and_b32_e32 v3, 1, v5
; GCN-NEXT: s_mov_b32 s0, exec_lo
-; GCN-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GCN-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GCN-NEXT: v_cmpx_ne_u32_e32 1, v3
; GCN-NEXT: s_xor_b32 s0, exec_lo, s0
+; GCN-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GCN-NEXT: s_cbranch_execz .LBB0_2
; GCN-NEXT: ; %bb.1: ; %bb2
; GCN-NEXT: v_mov_b32_e32 v3, 1
@@ -64,9 +65,10 @@ define void @taildup_ds_atomic_async_barrier_arrive(ptr addrspace(1) %a, ptr add
; GCN-NEXT: s_wait_kmcnt 0x0
; GCN-NEXT: v_and_b32_e32 v3, 1, v5
; GCN-NEXT: s_mov_b32 s0, exec_lo
-; GCN-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GCN-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GCN-NEXT: v_cmpx_ne_u32_e32 1, v3
; GCN-NEXT: s_xor_b32 s0, exec_lo, s0
+; GCN-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GCN-NEXT: s_cbranch_execz .LBB1_2
; GCN-NEXT: ; %bb.1: ; %bb2
; GCN-NEXT: v_mov_b32_e32 v3, 1
diff --git a/llvm/test/CodeGen/AMDGPU/ds_read2-gfx1250.ll b/llvm/test/CodeGen/AMDGPU/ds_read2-gfx1250.ll
index 33634dd0824d32..d79c95457114ed 100644
--- a/llvm/test/CodeGen/AMDGPU/ds_read2-gfx1250.ll
+++ b/llvm/test/CodeGen/AMDGPU/ds_read2-gfx1250.ll
@@ -546,7 +546,7 @@ define amdgpu_kernel void @simple_read2_f64(ptr addrspace(1) %out) #0 {
; GFX1250-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-NEXT: v_lshlrev_b32_e32 v0, 3, v0
; GFX1250-NEXT: s_load_b64 s[0:1], s[4:5], 0x0 nv
-; GFX1250-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1250-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_3) | instid1(SALU_CYCLE_1)
; GFX1250-NEXT: v_and_b32_e32 v4, 0x1ff8, v0
; GFX1250-NEXT: ds_load_2addr_b64 v[0:3], v4 offset1:8
; GFX1250-NEXT: s_wait_dscnt 0x0
@@ -576,7 +576,7 @@ define amdgpu_kernel void @simple_read2_f64_max_offset(ptr addrspace(1) %out) #0
; GFX1250-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-NEXT: v_lshlrev_b32_e32 v0, 3, v0
; GFX1250-NEXT: s_load_b64 s[0:1], s[4:5], 0x0 nv
-; GFX1250-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1250-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_3) | instid1(SALU_CYCLE_1)
; GFX1250-NEXT: v_and_b32_e32 v4, 0x1ff8, v0
; GFX1250-NEXT: ds_load_2addr_b64 v[0:3], v4 offset1:255
; GFX1250-NEXT: s_wait_dscnt 0x0
@@ -606,7 +606,7 @@ define amdgpu_kernel void @simple_read2_f64_too_far(ptr addrspace(1) %out) #0 {
; GFX1250-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-NEXT: v_lshlrev_b32_e32 v0, 3, v0
; GFX1250-NEXT: s_load_b64 s[0:1], s[4:5], 0x0 nv
-; GFX1250-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1250-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_4) | instid1(SALU_CYCLE_1)
; GFX1250-NEXT: v_and_b32_e32 v4, 0x1ff8, v0
; GFX1250-NEXT: ds_load_b64 v[0:1], v4
; GFX1250-NEXT: ds_load_b64 v[2:3], v4 offset:2056
@@ -646,6 +646,7 @@ define amdgpu_kernel void @misaligned_read2_f64(ptr addrspace(1) %out, ptr addrs
; GFX1250-UNALIGNED-NEXT: ds_load_2addr_b32 v[2:3], v2 offset0:14 offset1:15
; GFX1250-UNALIGNED-NEXT: s_wait_dscnt 0x0
; GFX1250-UNALIGNED-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 2, 2), 0 ; msbs: dst=0 src0=0 src1=0 src2=0
+; GFX1250-UNALIGNED-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1250-UNALIGNED-NEXT: v_add_f64_e32 v[0:1], v[0:1], v[2:3]
; GFX1250-UNALIGNED-NEXT: global_store_b64 v4, v[0:1], s[0:1]
; GFX1250-UNALIGNED-NEXT: s_endpgm
@@ -666,6 +667,7 @@ define amdgpu_kernel void @misaligned_read2_f64(ptr addrspace(1) %out, ptr addrs
; GFX1250S-UNALIGNED-NEXT: ds_load_b64 v[2:3], v2 offset:56
; GFX1250S-UNALIGNED-NEXT: s_wait_dscnt 0x0
; GFX1250S-UNALIGNED-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 2, 2), 0 ; msbs: dst=0 src0=0 src1=0 src2=0
+; GFX1250S-UNALIGNED-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1250S-UNALIGNED-NEXT: v_add_f64_e32 v[0:1], v[0:1], v[2:3]
; GFX1250S-UNALIGNED-NEXT: global_store_b64 v4, v[0:1], s[0:1]
; GFX1250S-UNALIGNED-NEXT: s_endpgm
@@ -791,14 +793,15 @@ define amdgpu_kernel void @sgemm_inner_loop_read2_sequence(ptr addrspace(1) %C,
; GFX1250-UNALIGNED-NEXT: s_add_co_i32 s0, s0, 1
; GFX1250-UNALIGNED-NEXT: s_getreg_b32 s2, hwreg(HW_REG_IB_STS2, 6, 4)
; GFX1250-UNALIGNED-NEXT: s_mul_i32 s0, ttmp9, s0
-; GFX1250-UNALIGNED-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
+; GFX1250-UNALIGNED-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX1250-UNALIGNED-NEXT: s_add_co_i32 s1, s1, s0
; GFX1250-UNALIGNED-NEXT: s_cmp_eq_u32 s2, 0
; GFX1250-UNALIGNED-NEXT: s_cselect_b32 s0, ttmp9, s1
+; GFX1250-UNALIGNED-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1250-UNALIGNED-NEXT: s_lshl_b32 s0, s0, 2
-; GFX1250-UNALIGNED-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX1250-UNALIGNED-NEXT: s_add_co_i32 s1, s0, 0xc20
; GFX1250-UNALIGNED-NEXT: s_addk_co_i32 s0, 0xc60
+; GFX1250-UNALIGNED-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1250-UNALIGNED-NEXT: v_dual_mov_b32 v1, s1 :: v_dual_mov_b32 v4, s0
; GFX1250-UNALIGNED-NEXT: s_load_b64 s[0:1], s[4:5], 0x0 nv
; GFX1250-UNALIGNED-NEXT: ds_load_2addr_b32 v[2:3], v1 offset1:1
@@ -843,10 +846,11 @@ define amdgpu_kernel void @sgemm_inner_loop_read2_sequence(ptr addrspace(1) %C,
; GFX1250S-UNALIGNED-NEXT: v_lshrrev_b32_e32 v0, 8, v0
; GFX1250S-UNALIGNED-NEXT: s_add_co_i32 s1, s1, s0
; GFX1250S-UNALIGNED-NEXT: s_cmp_eq_u32 s2, 0
+; GFX1250S-UNALIGNED-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1250S-UNALIGNED-NEXT: s_cselect_b32 s0, ttmp9, s1
-; GFX1250S-UNALIGNED-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX1250S-UNALIGNED-NEXT: v_and_b32_e32 v8, 0xffc, v0
; GFX1250S-UNALIGNED-NEXT: s_lshl_b32 s0, s0, 2
+; GFX1250S-UNALIGNED-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1250S-UNALIGNED-NEXT: v_mov_b32_e32 v1, s0
; GFX1250S-UNALIGNED-NEXT: ds_load_b64 v[2:3], v1 offset:3104
; GFX1250S-UNALIGNED-NEXT: ds_load_b64 v[4:5], v1 offset:3168
diff --git a/llvm/test/CodeGen/AMDGPU/ds_write2.ll b/llvm/test/CodeGen/AMDGPU/ds_write2.ll
index 3a8b59bb3f9002..46d667dd8f5d81 100644
--- a/llvm/test/CodeGen/AMDGPU/ds_write2.ll
+++ b/llvm/test/CodeGen/AMDGPU/ds_write2.ll
@@ -1281,13 +1281,14 @@ define amdgpu_kernel void @write2_sgemm_sequence(ptr addrspace(1) %C, i32 %lda,
; GFX1250-UNALIGNED-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1250-UNALIGNED-NEXT: s_add_co_i32 s1, s1, 1
; GFX1250-UNALIGNED-NEXT: s_mul_i32 s1, ttmp9, s1
-; GFX1250-UNALIGNED-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
+; GFX1250-UNALIGNED-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX1250-UNALIGNED-NEXT: s_add_co_i32 s2, s2, s1
; GFX1250-UNALIGNED-NEXT: s_cmp_eq_u32 s3, 0
; GFX1250-UNALIGNED-NEXT: s_cselect_b32 s1, ttmp9, s2
-; GFX1250-UNALIGNED-NEXT: s_lshl_b32 s1, s1, 2
; GFX1250-UNALIGNED-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
+; GFX1250-UNALIGNED-NEXT: s_lshl_b32 s1, s1, 2
; GFX1250-UNALIGNED-NEXT: s_add_co_i32 s2, s1, 0xc20
+; GFX1250-UNALIGNED-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1250-UNALIGNED-NEXT: v_dual_mov_b32 v1, s2 :: v_dual_lshrrev_b32 v0, 8, v0
; GFX1250-UNALIGNED-NEXT: s_addk_co_i32 s1, 0xc60
; GFX1250-UNALIGNED-NEXT: s_wait_kmcnt 0x0
@@ -1317,17 +1318,17 @@ define amdgpu_kernel void @write2_sgemm_sequence(ptr addrspace(1) %C, i32 %lda,
; GFX1250S-UNALIGNED-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1250S-UNALIGNED-NEXT: s_add_co_i32 s1, s1, 1
; GFX1250S-UNALIGNED-NEXT: s_mul_i32 s1, ttmp9, s1
-; GFX1250S-UNALIGNED-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
+; GFX1250S-UNALIGNED-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX1250S-UNALIGNED-NEXT: s_add_co_i32 s2, s2, s1
; GFX1250S-UNALIGNED-NEXT: s_cmp_eq_u32 s3, 0
; GFX1250S-UNALIGNED-NEXT: s_cselect_b32 s1, ttmp9, s2
+; GFX1250S-UNALIGNED-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1250S-UNALIGNED-NEXT: s_lshl_b32 s2, s1, 2
-; GFX1250S-UNALIGNED-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1250S-UNALIGNED-NEXT: v_dual_mov_b32 v3, s2 :: v_dual_lshrrev_b32 v2, 8, v0
+; GFX1250S-UNALIGNED-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
; GFX1250S-UNALIGNED-NEXT: v_and_b32_e32 v2, 0xffc, v2
; GFX1250S-UNALIGNED-NEXT: s_wait_kmcnt 0x0
; GFX1250S-UNALIGNED-NEXT: s_mov_b32 s1, s0
-; GFX1250S-UNALIGNED-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1250S-UNALIGNED-NEXT: v_mov_b64_e32 v[0:1], s[0:1]
; GFX1250S-UNALIGNED-NEXT: ds_store_b64 v3, v[0:1] offset:3104
; GFX1250S-UNALIGNED-NEXT: ds_store_b64 v3, v[0:1] offset:3168
diff --git a/llvm/test/CodeGen/AMDGPU/dynamic-vgpr-reserve-stack-for-cwsr.ll b/llvm/test/CodeGen/AMDGPU/dynamic-vgpr-reserve-stack-for-cwsr.ll
index 06e71249c4732f..09bf17bc63e459 100644
--- a/llvm/test/CodeGen/AMDGPU/dynamic-vgpr-reserve-stack-for-cwsr.ll
+++ b/llvm/test/CodeGen/AMDGPU/dynamic-vgpr-reserve-stack-for-cwsr.ll
@@ -8,7 +8,7 @@ define amdgpu_cs void @amdgpu_cs() #0 {
; CHECK-LABEL: amdgpu_cs:
; CHECK: ; %bb.0:
; CHECK-NEXT: s_getreg_b32 s33, hwreg(HW_REG_WAVE_HW_ID2, 8, 2)
-; CHECK-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; CHECK-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; CHECK-NEXT: s_cmp_lg_u32 0, s33
; CHECK-NEXT: s_cmovk_i32 s33, 0x1c0
; CHECK-NEXT: s_alloc_vgpr 0
@@ -20,7 +20,7 @@ define amdgpu_kernel void @kernel() #0 {
; CHECK-LABEL: kernel:
; CHECK: ; %bb.0:
; CHECK-NEXT: s_getreg_b32 s33, hwreg(HW_REG_WAVE_HW_ID2, 8, 2)
-; CHECK-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; CHECK-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; CHECK-NEXT: s_cmp_lg_u32 0, s33
; CHECK-NEXT: s_cmovk_i32 s33, 0x1c0
; CHECK-NEXT: s_alloc_vgpr 0
@@ -34,6 +34,7 @@ define amdgpu_cs void @with_local() #0 {
; CHECK-TRUE16-NEXT: s_getreg_b32 s33, hwreg(HW_REG_WAVE_HW_ID2, 8, 2)
; CHECK-TRUE16-NEXT: v_mov_b16_e32 v0.l, 13
; CHECK-TRUE16-NEXT: s_cmp_lg_u32 0, s33
+; CHECK-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; CHECK-TRUE16-NEXT: s_cmovk_i32 s33, 0x1c0
; CHECK-TRUE16-NEXT: scratch_store_b8 off, v0, s33 scope:SCOPE_SYS
; CHECK-TRUE16-NEXT: s_wait_storecnt 0x0
@@ -45,6 +46,7 @@ define amdgpu_cs void @with_local() #0 {
; CHECK-FAKE16-NEXT: s_getreg_b32 s33, hwreg(HW_REG_WAVE_HW_ID2, 8, 2)
; CHECK-FAKE16-NEXT: v_mov_b32_e32 v0, 13
; CHECK-FAKE16-NEXT: s_cmp_lg_u32 0, s33
+; CHECK-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; CHECK-FAKE16-NEXT: s_cmovk_i32 s33, 0x1c0
; CHECK-FAKE16-NEXT: scratch_store_b8 off, v0, s33 scope:SCOPE_SYS
; CHECK-FAKE16-NEXT: s_wait_storecnt 0x0
@@ -141,7 +143,7 @@ define amdgpu_cs void @with_spills() #0 {
; CHECK-LABEL: with_spills:
; CHECK: ; %bb.0:
; CHECK-NEXT: s_getreg_b32 s33, hwreg(HW_REG_WAVE_HW_ID2, 8, 2)
-; CHECK-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; CHECK-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; CHECK-NEXT: s_cmp_lg_u32 0, s33
; CHECK-NEXT: s_cmovk_i32 s33, 0x1c0
; CHECK-NEXT: s_alloc_vgpr 0
@@ -190,6 +192,7 @@ define amdgpu_cs void @frame_pointer_none() #1 {
; CHECK-TRUE16-NEXT: s_getreg_b32 s33, hwreg(HW_REG_WAVE_HW_ID2, 8, 2)
; CHECK-TRUE16-NEXT: v_mov_b16_e32 v0.l, 13
; CHECK-TRUE16-NEXT: s_cmp_lg_u32 0, s33
+; CHECK-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; CHECK-TRUE16-NEXT: s_cmovk_i32 s33, 0x1c0
; CHECK-TRUE16-NEXT: scratch_store_b8 off, v0, s33 scope:SCOPE_SYS
; CHECK-TRUE16-NEXT: s_wait_storecnt 0x0
@@ -201,6 +204,7 @@ define amdgpu_cs void @frame_pointer_none() #1 {
; CHECK-FAKE16-NEXT: s_getreg_b32 s33, hwreg(HW_REG_WAVE_HW_ID2, 8, 2)
; CHECK-FAKE16-NEXT: v_mov_b32_e32 v0, 13
; CHECK-FAKE16-NEXT: s_cmp_lg_u32 0, s33
+; CHECK-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; CHECK-FAKE16-NEXT: s_cmovk_i32 s33, 0x1c0
; CHECK-FAKE16-NEXT: scratch_store_b8 off, v0, s33 scope:SCOPE_SYS
; CHECK-FAKE16-NEXT: s_wait_storecnt 0x0
@@ -217,6 +221,7 @@ define amdgpu_cs void @frame_pointer_all() #2 {
; CHECK-TRUE16-NEXT: s_getreg_b32 s33, hwreg(HW_REG_WAVE_HW_ID2, 8, 2)
; CHECK-TRUE16-NEXT: v_mov_b16_e32 v0.l, 13
; CHECK-TRUE16-NEXT: s_cmp_lg_u32 0, s33
+; CHECK-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; CHECK-TRUE16-NEXT: s_cmovk_i32 s33, 0x1c0
; CHECK-TRUE16-NEXT: scratch_store_b8 off, v0, s33 scope:SCOPE_SYS
; CHECK-TRUE16-NEXT: s_wait_storecnt 0x0
@@ -228,6 +233,7 @@ define amdgpu_cs void @frame_pointer_all() #2 {
; CHECK-FAKE16-NEXT: s_getreg_b32 s33, hwreg(HW_REG_WAVE_HW_ID2, 8, 2)
; CHECK-FAKE16-NEXT: v_mov_b32_e32 v0, 13
; CHECK-FAKE16-NEXT: s_cmp_lg_u32 0, s33
+; CHECK-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; CHECK-FAKE16-NEXT: s_cmovk_i32 s33, 0x1c0
; CHECK-FAKE16-NEXT: scratch_store_b8 off, v0, s33 scope:SCOPE_SYS
; CHECK-FAKE16-NEXT: s_wait_storecnt 0x0
diff --git a/llvm/test/CodeGen/AMDGPU/dynamic_stackalloc.ll b/llvm/test/CodeGen/AMDGPU/dynamic_stackalloc.ll
index 028ec031941d38..fa22f36398b5c3 100644
--- a/llvm/test/CodeGen/AMDGPU/dynamic_stackalloc.ll
+++ b/llvm/test/CodeGen/AMDGPU/dynamic_stackalloc.ll
@@ -350,7 +350,7 @@ define amdgpu_kernel void @test_dynamic_stackalloc_kernel_divergent() #0 {
; GFX11-GISEL-NEXT: ds_swizzle_b32 v2, v1 offset:swizzle(BROADCAST,32,15)
; GFX11-GISEL-NEXT: s_waitcnt lgkmcnt(0)
; GFX11-GISEL-NEXT: v_max_u32_e32 v1, v1, v2
-; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX11-GISEL-NEXT: v_readlane_b32 s2, v1, 31
; GFX11-GISEL-NEXT: s_mov_b32 exec_lo, s1
; GFX11-GISEL-NEXT: v_mov_b32_e32 v0, 0x7b
@@ -485,13 +485,13 @@ define amdgpu_kernel void @test_dynamic_stackalloc_kernel_divergent_over_aligned
; GFX11-GISEL-NEXT: ds_swizzle_b32 v2, v1 offset:swizzle(BROADCAST,32,15)
; GFX11-GISEL-NEXT: s_waitcnt lgkmcnt(0)
; GFX11-GISEL-NEXT: v_max_u32_e32 v1, v1, v2
-; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_3) | instid1(SALU_CYCLE_1)
+; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX11-GISEL-NEXT: v_readlane_b32 s1, v1, 31
; GFX11-GISEL-NEXT: s_mov_b32 exec_lo, s0
; GFX11-GISEL-NEXT: v_mov_b32_e32 v0, 0x1bc
; GFX11-GISEL-NEXT: s_add_u32 s0, s32, 0x7f
+; GFX11-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-GISEL-NEXT: s_and_b32 s0, s0, 0xffffff80
-; GFX11-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-GISEL-NEXT: s_add_u32 s32, s0, s1
; GFX11-GISEL-NEXT: scratch_store_b32 off, v0, s0 dlc
; GFX11-GISEL-NEXT: s_waitcnt_vscnt null, 0x0
@@ -619,7 +619,7 @@ define amdgpu_kernel void @test_dynamic_stackalloc_kernel_divergent_under_aligne
; GFX11-GISEL-NEXT: ds_swizzle_b32 v2, v1 offset:swizzle(BROADCAST,32,15)
; GFX11-GISEL-NEXT: s_waitcnt lgkmcnt(0)
; GFX11-GISEL-NEXT: v_max_u32_e32 v1, v1, v2
-; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX11-GISEL-NEXT: v_readlane_b32 s2, v1, 31
; GFX11-GISEL-NEXT: s_mov_b32 exec_lo, s1
; GFX11-GISEL-NEXT: v_mov_b32_e32 v0, 0x29a
@@ -762,6 +762,7 @@ define amdgpu_kernel void @test_dynamic_stackalloc_kernel_multiple_allocas(i32 %
; GFX11-SDAG-NEXT: s_movk_i32 s32, 0x80
; GFX11-SDAG-NEXT: s_waitcnt lgkmcnt(0)
; GFX11-SDAG-NEXT: s_cmp_lg_u32 s0, 0
+; GFX11-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-SDAG-NEXT: s_cbranch_scc1 .LBB6_2
; GFX11-SDAG-NEXT: ; %bb.1: ; %bb.0
; GFX11-SDAG-NEXT: v_and_b32_e32 v0, 0x3ff, v0
@@ -818,6 +819,7 @@ define amdgpu_kernel void @test_dynamic_stackalloc_kernel_multiple_allocas(i32 %
; GFX11-GISEL-NEXT: s_movk_i32 s32, 0x80
; GFX11-GISEL-NEXT: s_waitcnt lgkmcnt(0)
; GFX11-GISEL-NEXT: s_cmp_lg_u32 s0, 0
+; GFX11-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-GISEL-NEXT: s_cbranch_scc1 .LBB6_2
; GFX11-GISEL-NEXT: ; %bb.1: ; %bb.0
; GFX11-GISEL-NEXT: v_and_b32_e32 v0, 0x3ff, v0
@@ -1004,6 +1006,7 @@ define amdgpu_kernel void @test_dynamic_stackalloc_kernel_control_flow(i32 %n, i
; GFX11-SDAG-NEXT: s_mov_b32 s32, 64
; GFX11-SDAG-NEXT: s_waitcnt lgkmcnt(0)
; GFX11-SDAG-NEXT: s_cmp_lg_u32 s0, 0
+; GFX11-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-SDAG-NEXT: s_cbranch_scc0 .LBB7_2
; GFX11-SDAG-NEXT: ; %bb.1: ; %bb.1
; GFX11-SDAG-NEXT: v_and_b32_e32 v0, 0x3ff, v0
@@ -1040,6 +1043,7 @@ define amdgpu_kernel void @test_dynamic_stackalloc_kernel_control_flow(i32 %n, i
; GFX11-SDAG-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-SDAG-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-SDAG-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-SDAG-NEXT: s_cbranch_scc1 .LBB7_5
; GFX11-SDAG-NEXT: ; %bb.4: ; %bb.0
; GFX11-SDAG-NEXT: v_mov_b32_e32 v0, 2
@@ -1080,7 +1084,7 @@ define amdgpu_kernel void @test_dynamic_stackalloc_kernel_control_flow(i32 %n, i
; GFX11-GISEL-NEXT: ds_swizzle_b32 v2, v1 offset:swizzle(BROADCAST,32,15)
; GFX11-GISEL-NEXT: s_waitcnt lgkmcnt(0)
; GFX11-GISEL-NEXT: v_max_u32_e32 v1, v1, v2
-; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX11-GISEL-NEXT: v_readlane_b32 s3, v1, 31
; GFX11-GISEL-NEXT: s_mov_b32 exec_lo, s0
; GFX11-GISEL-NEXT: v_mov_b32_e32 v0, 1
@@ -1090,7 +1094,7 @@ define amdgpu_kernel void @test_dynamic_stackalloc_kernel_control_flow(i32 %n, i
; GFX11-GISEL-NEXT: s_waitcnt_vscnt null, 0x0
; GFX11-GISEL-NEXT: .LBB7_2: ; %Flow
; GFX11-GISEL-NEXT: s_xor_b32 s0, s0, 1
-; GFX11-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-GISEL-NEXT: s_cmp_lg_u32 s0, 0
; GFX11-GISEL-NEXT: s_cbranch_scc1 .LBB7_4
; GFX11-GISEL-NEXT: ; %bb.3: ; %bb.0
@@ -1216,9 +1220,9 @@ define void @test_dynamic_stackalloc_device_uniform(i32 %n) #0 {
; GFX11-SDAG-NEXT: scratch_store_b32 off, v1, s33
; GFX11-SDAG-NEXT: scratch_store_b32 off, v2, s33 offset:4
; GFX11-SDAG-NEXT: s_mov_b32 exec_lo, s0
+; GFX11-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-SDAG-NEXT: v_lshl_add_u32 v0, v0, 2, 15
; GFX11-SDAG-NEXT: s_add_i32 s32, s32, 16
-; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-SDAG-NEXT: v_and_b32_e32 v0, -16, v0
; GFX11-SDAG-NEXT: s_or_saveexec_b32 s0, -1
; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
@@ -1261,10 +1265,11 @@ define void @test_dynamic_stackalloc_device_uniform(i32 %n) #0 {
; GFX11-GISEL-NEXT: scratch_store_b32 off, v1, s33
; GFX11-GISEL-NEXT: scratch_store_b32 off, v2, s33 offset:4
; GFX11-GISEL-NEXT: s_mov_b32 exec_lo, s0
+; GFX11-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX11-GISEL-NEXT: v_lshl_add_u32 v0, v0, 2, 15
; GFX11-GISEL-NEXT: s_add_i32 s32, s32, 16
-; GFX11-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-GISEL-NEXT: s_mov_b32 s0, s32
+; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-GISEL-NEXT: v_and_b32_e32 v0, -16, v0
; GFX11-GISEL-NEXT: s_or_saveexec_b32 s1, -1
; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
@@ -1278,7 +1283,7 @@ define void @test_dynamic_stackalloc_device_uniform(i32 %n) #0 {
; GFX11-GISEL-NEXT: ds_swizzle_b32 v2, v1 offset:swizzle(BROADCAST,32,15)
; GFX11-GISEL-NEXT: s_waitcnt lgkmcnt(0)
; GFX11-GISEL-NEXT: v_max_u32_e32 v1, v1, v2
-; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX11-GISEL-NEXT: v_readlane_b32 s2, v1, 31
; GFX11-GISEL-NEXT: s_mov_b32 exec_lo, s1
; GFX11-GISEL-NEXT: v_mov_b32_e32 v0, 0x7b
@@ -1406,11 +1411,11 @@ define void @test_dynamic_stackalloc_device_uniform_over_aligned(i32 %n) #0 {
; GFX11-SDAG-NEXT: scratch_store_b32 off, v1, s33
; GFX11-SDAG-NEXT: scratch_store_b32 off, v2, s33 offset:4
; GFX11-SDAG-NEXT: s_mov_b32 exec_lo, s0
+; GFX11-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_3) | instid1(VALU_DEP_1)
; GFX11-SDAG-NEXT: v_lshl_add_u32 v0, v0, 2, 15
; GFX11-SDAG-NEXT: s_mov_b32 s3, s34
; GFX11-SDAG-NEXT: s_mov_b32 s34, s32
; GFX11-SDAG-NEXT: s_addk_i32 s32, 0x100
-; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-SDAG-NEXT: v_and_b32_e32 v0, -16, v0
; GFX11-SDAG-NEXT: s_or_saveexec_b32 s0, -1
; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
@@ -1457,11 +1462,11 @@ define void @test_dynamic_stackalloc_device_uniform_over_aligned(i32 %n) #0 {
; GFX11-GISEL-NEXT: scratch_store_b32 off, v1, s33
; GFX11-GISEL-NEXT: scratch_store_b32 off, v2, s33 offset:4
; GFX11-GISEL-NEXT: s_mov_b32 exec_lo, s0
+; GFX11-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_3) | instid1(VALU_DEP_1)
; GFX11-GISEL-NEXT: v_lshl_add_u32 v0, v0, 2, 15
; GFX11-GISEL-NEXT: s_mov_b32 s3, s34
; GFX11-GISEL-NEXT: s_mov_b32 s34, s32
; GFX11-GISEL-NEXT: s_addk_i32 s32, 0x100
-; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-GISEL-NEXT: v_and_b32_e32 v0, -16, v0
; GFX11-GISEL-NEXT: s_or_saveexec_b32 s0, -1
; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
@@ -1475,13 +1480,13 @@ define void @test_dynamic_stackalloc_device_uniform_over_aligned(i32 %n) #0 {
; GFX11-GISEL-NEXT: ds_swizzle_b32 v2, v1 offset:swizzle(BROADCAST,32,15)
; GFX11-GISEL-NEXT: s_waitcnt lgkmcnt(0)
; GFX11-GISEL-NEXT: v_max_u32_e32 v1, v1, v2
-; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_3) | instid1(SALU_CYCLE_1)
+; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX11-GISEL-NEXT: v_readlane_b32 s1, v1, 31
; GFX11-GISEL-NEXT: s_mov_b32 exec_lo, s0
; GFX11-GISEL-NEXT: v_mov_b32_e32 v0, 10
; GFX11-GISEL-NEXT: s_add_u32 s0, s32, 0x7f
+; GFX11-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-GISEL-NEXT: s_and_b32 s0, s0, 0xffffff80
-; GFX11-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-GISEL-NEXT: s_add_u32 s32, s0, s1
; GFX11-GISEL-NEXT: scratch_store_b32 off, v0, s0 dlc
; GFX11-GISEL-NEXT: s_waitcnt_vscnt null, 0x0
@@ -1595,9 +1600,9 @@ define void @test_dynamic_stackalloc_device_uniform_under_aligned(i32 %n) #0 {
; GFX11-SDAG-NEXT: scratch_store_b32 off, v1, s33
; GFX11-SDAG-NEXT: scratch_store_b32 off, v2, s33 offset:4
; GFX11-SDAG-NEXT: s_mov_b32 exec_lo, s0
+; GFX11-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-SDAG-NEXT: v_lshl_add_u32 v0, v0, 2, 15
; GFX11-SDAG-NEXT: s_add_i32 s32, s32, 16
-; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-SDAG-NEXT: v_and_b32_e32 v0, -16, v0
; GFX11-SDAG-NEXT: s_or_saveexec_b32 s0, -1
; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
@@ -1640,10 +1645,11 @@ define void @test_dynamic_stackalloc_device_uniform_under_aligned(i32 %n) #0 {
; GFX11-GISEL-NEXT: scratch_store_b32 off, v1, s33
; GFX11-GISEL-NEXT: scratch_store_b32 off, v2, s33 offset:4
; GFX11-GISEL-NEXT: s_mov_b32 exec_lo, s0
+; GFX11-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX11-GISEL-NEXT: v_lshl_add_u32 v0, v0, 2, 15
; GFX11-GISEL-NEXT: s_add_i32 s32, s32, 16
-; GFX11-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-GISEL-NEXT: s_mov_b32 s0, s32
+; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-GISEL-NEXT: v_and_b32_e32 v0, -16, v0
; GFX11-GISEL-NEXT: s_or_saveexec_b32 s1, -1
; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
@@ -1657,7 +1663,7 @@ define void @test_dynamic_stackalloc_device_uniform_under_aligned(i32 %n) #0 {
; GFX11-GISEL-NEXT: ds_swizzle_b32 v2, v1 offset:swizzle(BROADCAST,32,15)
; GFX11-GISEL-NEXT: s_waitcnt lgkmcnt(0)
; GFX11-GISEL-NEXT: v_max_u32_e32 v1, v1, v2
-; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX11-GISEL-NEXT: v_readlane_b32 s2, v1, 31
; GFX11-GISEL-NEXT: s_mov_b32 exec_lo, s1
; GFX11-GISEL-NEXT: v_mov_b32_e32 v0, 22
@@ -1775,10 +1781,11 @@ define void @test_dynamic_stackalloc_device_divergent() #0 {
; GFX11-SDAG-NEXT: scratch_store_b32 off, v0, s33
; GFX11-SDAG-NEXT: scratch_store_b32 off, v1, s33 offset:4
; GFX11-SDAG-NEXT: s_mov_b32 exec_lo, s0
+; GFX11-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-SDAG-NEXT: v_and_b32_e32 v2, 0x3ff, v31
; GFX11-SDAG-NEXT: s_add_i32 s32, s32, 16
-; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-SDAG-NEXT: v_lshl_add_u32 v2, v2, 2, 15
+; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-SDAG-NEXT: v_and_b32_e32 v2, 0x1ff0, v2
; GFX11-SDAG-NEXT: s_or_saveexec_b32 s0, -1
; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
@@ -1821,12 +1828,12 @@ define void @test_dynamic_stackalloc_device_divergent() #0 {
; GFX11-GISEL-NEXT: scratch_store_b32 off, v0, s33
; GFX11-GISEL-NEXT: scratch_store_b32 off, v1, s33 offset:4
; GFX11-GISEL-NEXT: s_mov_b32 exec_lo, s0
+; GFX11-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX11-GISEL-NEXT: v_and_b32_e32 v2, 0x3ff, v31
; GFX11-GISEL-NEXT: s_add_i32 s32, s32, 16
-; GFX11-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-GISEL-NEXT: s_mov_b32 s0, s32
+; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-GISEL-NEXT: v_lshl_add_u32 v2, v2, 2, 15
-; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-GISEL-NEXT: v_and_b32_e32 v2, -16, v2
; GFX11-GISEL-NEXT: s_or_saveexec_b32 s1, -1
; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
@@ -1840,7 +1847,7 @@ define void @test_dynamic_stackalloc_device_divergent() #0 {
; GFX11-GISEL-NEXT: ds_swizzle_b32 v1, v0 offset:swizzle(BROADCAST,32,15)
; GFX11-GISEL-NEXT: s_waitcnt lgkmcnt(0)
; GFX11-GISEL-NEXT: v_max_u32_e32 v0, v0, v1
-; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX11-GISEL-NEXT: v_readlane_b32 s2, v0, 31
; GFX11-GISEL-NEXT: s_mov_b32 exec_lo, s1
; GFX11-GISEL-NEXT: v_mov_b32_e32 v2, 0x7b
@@ -1971,12 +1978,13 @@ define void @test_dynamic_stackalloc_device_divergent_over_aligned() #0 {
; GFX11-SDAG-NEXT: scratch_store_b32 off, v0, s33
; GFX11-SDAG-NEXT: scratch_store_b32 off, v1, s33 offset:4
; GFX11-SDAG-NEXT: s_mov_b32 exec_lo, s0
+; GFX11-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_3) | instid1(VALU_DEP_1)
; GFX11-SDAG-NEXT: v_and_b32_e32 v2, 0x3ff, v31
; GFX11-SDAG-NEXT: s_mov_b32 s3, s34
; GFX11-SDAG-NEXT: s_mov_b32 s34, s32
; GFX11-SDAG-NEXT: s_addk_i32 s32, 0x100
-; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-SDAG-NEXT: v_lshl_add_u32 v2, v2, 2, 15
+; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-SDAG-NEXT: v_and_b32_e32 v2, 0x1ff0, v2
; GFX11-SDAG-NEXT: s_or_saveexec_b32 s0, -1
; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
@@ -2023,12 +2031,13 @@ define void @test_dynamic_stackalloc_device_divergent_over_aligned() #0 {
; GFX11-GISEL-NEXT: scratch_store_b32 off, v0, s33
; GFX11-GISEL-NEXT: scratch_store_b32 off, v1, s33 offset:4
; GFX11-GISEL-NEXT: s_mov_b32 exec_lo, s0
+; GFX11-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_3) | instid1(VALU_DEP_1)
; GFX11-GISEL-NEXT: v_and_b32_e32 v2, 0x3ff, v31
; GFX11-GISEL-NEXT: s_mov_b32 s3, s34
; GFX11-GISEL-NEXT: s_mov_b32 s34, s32
; GFX11-GISEL-NEXT: s_addk_i32 s32, 0x100
-; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-GISEL-NEXT: v_lshl_add_u32 v2, v2, 2, 15
+; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-GISEL-NEXT: v_and_b32_e32 v2, -16, v2
; GFX11-GISEL-NEXT: s_or_saveexec_b32 s0, -1
; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
@@ -2042,13 +2051,13 @@ define void @test_dynamic_stackalloc_device_divergent_over_aligned() #0 {
; GFX11-GISEL-NEXT: ds_swizzle_b32 v1, v0 offset:swizzle(BROADCAST,32,15)
; GFX11-GISEL-NEXT: s_waitcnt lgkmcnt(0)
; GFX11-GISEL-NEXT: v_max_u32_e32 v0, v0, v1
-; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_3) | instid1(SALU_CYCLE_1)
+; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX11-GISEL-NEXT: v_readlane_b32 s1, v0, 31
; GFX11-GISEL-NEXT: s_mov_b32 exec_lo, s0
; GFX11-GISEL-NEXT: v_mov_b32_e32 v2, 0x1bc
; GFX11-GISEL-NEXT: s_add_u32 s0, s32, 0x7f
+; GFX11-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-GISEL-NEXT: s_and_b32 s0, s0, 0xffffff80
-; GFX11-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-GISEL-NEXT: s_add_u32 s32, s0, s1
; GFX11-GISEL-NEXT: scratch_store_b32 off, v2, s0 dlc
; GFX11-GISEL-NEXT: s_waitcnt_vscnt null, 0x0
@@ -2165,10 +2174,11 @@ define void @test_dynamic_stackalloc_device_divergent_under_aligned() #0 {
; GFX11-SDAG-NEXT: scratch_store_b32 off, v0, s33
; GFX11-SDAG-NEXT: scratch_store_b32 off, v1, s33 offset:4
; GFX11-SDAG-NEXT: s_mov_b32 exec_lo, s0
+; GFX11-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-SDAG-NEXT: v_and_b32_e32 v2, 0x3ff, v31
; GFX11-SDAG-NEXT: s_add_i32 s32, s32, 16
-; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-SDAG-NEXT: v_lshl_add_u32 v2, v2, 2, 15
+; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-SDAG-NEXT: v_and_b32_e32 v2, 0x1ff0, v2
; GFX11-SDAG-NEXT: s_or_saveexec_b32 s0, -1
; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
@@ -2211,12 +2221,12 @@ define void @test_dynamic_stackalloc_device_divergent_under_aligned() #0 {
; GFX11-GISEL-NEXT: scratch_store_b32 off, v0, s33
; GFX11-GISEL-NEXT: scratch_store_b32 off, v1, s33 offset:4
; GFX11-GISEL-NEXT: s_mov_b32 exec_lo, s0
+; GFX11-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX11-GISEL-NEXT: v_and_b32_e32 v2, 0x3ff, v31
; GFX11-GISEL-NEXT: s_add_i32 s32, s32, 16
-; GFX11-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-GISEL-NEXT: s_mov_b32 s0, s32
+; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-GISEL-NEXT: v_lshl_add_u32 v2, v2, 2, 15
-; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-GISEL-NEXT: v_and_b32_e32 v2, -16, v2
; GFX11-GISEL-NEXT: s_or_saveexec_b32 s1, -1
; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
@@ -2230,7 +2240,7 @@ define void @test_dynamic_stackalloc_device_divergent_under_aligned() #0 {
; GFX11-GISEL-NEXT: ds_swizzle_b32 v1, v0 offset:swizzle(BROADCAST,32,15)
; GFX11-GISEL-NEXT: s_waitcnt lgkmcnt(0)
; GFX11-GISEL-NEXT: v_max_u32_e32 v0, v0, v1
-; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX11-GISEL-NEXT: v_readlane_b32 s2, v0, 31
; GFX11-GISEL-NEXT: s_mov_b32 exec_lo, s1
; GFX11-GISEL-NEXT: v_mov_b32_e32 v2, 0x29a
@@ -2486,6 +2496,7 @@ define void @test_dynamic_stackalloc_device_multiple_allocas(i32 %n, i32 %m) #0
; GFX11-SDAG-NEXT: s_mov_b32 s34, s32
; GFX11-SDAG-NEXT: s_addk_i32 s32, 0xc0
; GFX11-SDAG-NEXT: v_cmpx_eq_u32_e32 0, v0
+; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-SDAG-NEXT: s_cbranch_execz .LBB14_2
; GFX11-SDAG-NEXT: ; %bb.1: ; %bb.0
; GFX11-SDAG-NEXT: v_lshl_add_u32 v1, v1, 2, 15
@@ -2497,22 +2508,23 @@ define void @test_dynamic_stackalloc_device_multiple_allocas(i32 %n, i32 %m) #0
; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_2) | instid1(SALU_CYCLE_1)
; GFX11-SDAG-NEXT: v_cndmask_b32_e64 v2, 0, v1, s1
; GFX11-SDAG-NEXT: s_mov_b32 exec_lo, s1
-; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_2)
+; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_2) | instid1(SALU_CYCLE_1)
; GFX11-SDAG-NEXT: v_and_b32_e32 v1, 0x1ff0, v6
; GFX11-SDAG-NEXT: s_or_saveexec_b32 s1, -1
+; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-SDAG-NEXT: v_max_u32_dpp v2, v2, v2 row_shr:1 row_mask:0xf bank_mask:0xf
-; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX11-SDAG-NEXT: v_cndmask_b32_e64 v3, 0, v1, s1
; GFX11-SDAG-NEXT: s_add_i32 s3, s32, 63
+; GFX11-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-SDAG-NEXT: s_and_not1_b32 s3, s3, 63
-; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-SDAG-NEXT: v_max_u32_dpp v2, v2, v2 row_shr:2 row_mask:0xf bank_mask:0xf
-; GFX11-SDAG-NEXT: v_max_u32_dpp v3, v3, v3 row_shr:1 row_mask:0xf bank_mask:0xf
; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX11-SDAG-NEXT: v_max_u32_dpp v3, v3, v3 row_shr:1 row_mask:0xf bank_mask:0xf
; GFX11-SDAG-NEXT: v_max_u32_dpp v2, v2, v2 row_shr:4 row_mask:0xf bank_mask:0xf
-; GFX11-SDAG-NEXT: v_max_u32_dpp v3, v3, v3 row_shr:2 row_mask:0xf bank_mask:0xf
; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX11-SDAG-NEXT: v_max_u32_dpp v3, v3, v3 row_shr:2 row_mask:0xf bank_mask:0xf
; GFX11-SDAG-NEXT: v_max_u32_dpp v2, v2, v2 row_shr:8 row_mask:0xf bank_mask:0xf
+; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX11-SDAG-NEXT: v_max_u32_dpp v3, v3, v3 row_shr:4 row_mask:0xf bank_mask:0xf
; GFX11-SDAG-NEXT: ds_swizzle_b32 v4, v2 offset:swizzle(BROADCAST,32,15)
; GFX11-SDAG-NEXT: v_max_u32_dpp v3, v3, v3 row_shr:8 row_mask:0xf bank_mask:0xf
@@ -2525,9 +2537,9 @@ define void @test_dynamic_stackalloc_device_multiple_allocas(i32 %n, i32 %m) #0
; GFX11-SDAG-NEXT: v_max_u32_e32 v2, v3, v5
; GFX11-SDAG-NEXT: v_readlane_b32 s4, v2, 31
; GFX11-SDAG-NEXT: s_mov_b32 exec_lo, s1
+; GFX11-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX11-SDAG-NEXT: v_add_nc_u32_e64 v1, s3, s2
; GFX11-SDAG-NEXT: v_dual_mov_b32 v6, 3 :: v_dual_mov_b32 v7, 4
-; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX11-SDAG-NEXT: v_readfirstlane_b32 s32, v1
; GFX11-SDAG-NEXT: s_mov_b32 s1, s32
; GFX11-SDAG-NEXT: scratch_store_b32 off, v6, s3 dlc
@@ -2539,8 +2551,8 @@ define void @test_dynamic_stackalloc_device_multiple_allocas(i32 %n, i32 %m) #0
; GFX11-SDAG-NEXT: v_readfirstlane_b32 s32, v1
; GFX11-SDAG-NEXT: .LBB14_2: ; %bb.1
; GFX11-SDAG-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-SDAG-NEXT: v_lshl_add_u32 v0, v0, 2, 15
-; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-SDAG-NEXT: v_and_b32_e32 v0, -16, v0
; GFX11-SDAG-NEXT: s_or_saveexec_b32 s0, -1
; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
@@ -2597,6 +2609,7 @@ define void @test_dynamic_stackalloc_device_multiple_allocas(i32 %n, i32 %m) #0
; GFX11-GISEL-NEXT: s_mov_b32 s34, s32
; GFX11-GISEL-NEXT: s_addk_i32 s32, 0xc0
; GFX11-GISEL-NEXT: v_cmpx_eq_u32_e32 0, v0
+; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-GISEL-NEXT: s_cbranch_execz .LBB14_2
; GFX11-GISEL-NEXT: ; %bb.1: ; %bb.0
; GFX11-GISEL-NEXT: v_and_b32_e32 v6, 0x3ff, v31
@@ -2646,9 +2659,9 @@ define void @test_dynamic_stackalloc_device_multiple_allocas(i32 %n, i32 %m) #0
; GFX11-GISEL-NEXT: s_waitcnt_vscnt null, 0x0
; GFX11-GISEL-NEXT: .LBB14_2: ; %bb.1
; GFX11-GISEL-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-GISEL-NEXT: v_lshl_add_u32 v0, v0, 2, 15
; GFX11-GISEL-NEXT: s_mov_b32 s0, s32
-; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-GISEL-NEXT: v_and_b32_e32 v0, -16, v0
; GFX11-GISEL-NEXT: s_or_saveexec_b32 s1, -1
; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
@@ -2662,7 +2675,7 @@ define void @test_dynamic_stackalloc_device_multiple_allocas(i32 %n, i32 %m) #0
; GFX11-GISEL-NEXT: ds_swizzle_b32 v3, v2 offset:swizzle(BROADCAST,32,15)
; GFX11-GISEL-NEXT: s_waitcnt lgkmcnt(0)
; GFX11-GISEL-NEXT: v_max_u32_e32 v2, v2, v3
-; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX11-GISEL-NEXT: v_readlane_b32 s2, v2, 31
; GFX11-GISEL-NEXT: s_mov_b32 exec_lo, s1
; GFX11-GISEL-NEXT: v_dual_mov_b32 v0, 1 :: v_dual_mov_b32 v1, 2
@@ -2889,6 +2902,7 @@ define void @test_dynamic_stackalloc_device_control_flow(i32 %n, i32 %m) #0 {
; GFX11-SDAG-NEXT: s_mov_b32 s34, s32
; GFX11-SDAG-NEXT: s_addk_i32 s32, 0x80
; GFX11-SDAG-NEXT: v_cmpx_ne_u32_e32 0, v0
+; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-SDAG-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX11-SDAG-NEXT: s_cbranch_execz .LBB15_2
; GFX11-SDAG-NEXT: ; %bb.1: ; %bb.1
@@ -2977,6 +2991,7 @@ define void @test_dynamic_stackalloc_device_control_flow(i32 %n, i32 %m) #0 {
; GFX11-GISEL-NEXT: s_mov_b32 s34, s32
; GFX11-GISEL-NEXT: s_addk_i32 s32, 0x80
; GFX11-GISEL-NEXT: v_cmpx_ne_u32_e32 0, v0
+; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-GISEL-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX11-GISEL-NEXT: s_cbranch_execz .LBB15_2
; GFX11-GISEL-NEXT: ; %bb.1: ; %bb.1
@@ -2995,14 +3010,14 @@ define void @test_dynamic_stackalloc_device_control_flow(i32 %n, i32 %m) #0 {
; GFX11-GISEL-NEXT: ds_swizzle_b32 v3, v2 offset:swizzle(BROADCAST,32,15)
; GFX11-GISEL-NEXT: s_waitcnt lgkmcnt(0)
; GFX11-GISEL-NEXT: v_max_u32_e32 v2, v2, v3
-; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_3) | instid1(SALU_CYCLE_1)
+; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX11-GISEL-NEXT: v_readlane_b32 s2, v2, 31
; GFX11-GISEL-NEXT: s_mov_b32 exec_lo, s1
; GFX11-GISEL-NEXT: v_mov_b32_e32 v0, 2
; GFX11-GISEL-NEXT: s_add_u32 s1, s32, 63
; GFX11-GISEL-NEXT: ; implicit-def: $vgpr31
+; GFX11-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-GISEL-NEXT: s_and_not1_b32 s1, s1, 63
-; GFX11-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-GISEL-NEXT: s_add_u32 s32, s1, s2
; GFX11-GISEL-NEXT: scratch_store_b32 off, v0, s1 dlc
; GFX11-GISEL-NEXT: s_waitcnt_vscnt null, 0x0
@@ -3027,7 +3042,7 @@ define void @test_dynamic_stackalloc_device_control_flow(i32 %n, i32 %m) #0 {
; GFX11-GISEL-NEXT: ds_swizzle_b32 v3, v2 offset:swizzle(BROADCAST,32,15)
; GFX11-GISEL-NEXT: s_waitcnt lgkmcnt(0)
; GFX11-GISEL-NEXT: v_max_u32_e32 v2, v2, v3
-; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX11-GISEL-NEXT: v_readlane_b32 s3, v2, 31
; GFX11-GISEL-NEXT: s_mov_b32 exec_lo, s2
; GFX11-GISEL-NEXT: v_mov_b32_e32 v0, 1
@@ -3159,10 +3174,11 @@ define void @test_dynamic_stackalloc_device_divergent_non_standard_size_i16(i16
; GFX11-SDAG-NEXT: scratch_store_b32 off, v1, s33
; GFX11-SDAG-NEXT: scratch_store_b32 off, v2, s33 offset:4
; GFX11-SDAG-NEXT: s_mov_b32 exec_lo, s0
+; GFX11-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-SDAG-NEXT: v_cvt_u32_u16_e32 v0, v0.l
; GFX11-SDAG-NEXT: s_add_i32 s32, s32, 16
-; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-SDAG-NEXT: v_lshl_add_u32 v0, v0, 2, 15
+; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-SDAG-NEXT: v_and_b32_e32 v0, 0x7fff0, v0
; GFX11-SDAG-NEXT: s_or_saveexec_b32 s0, -1
; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
@@ -3205,12 +3221,12 @@ define void @test_dynamic_stackalloc_device_divergent_non_standard_size_i16(i16
; GFX11-GISEL-NEXT: scratch_store_b32 off, v1, s33
; GFX11-GISEL-NEXT: scratch_store_b32 off, v2, s33 offset:4
; GFX11-GISEL-NEXT: s_mov_b32 exec_lo, s0
+; GFX11-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX11-GISEL-NEXT: v_and_b32_e32 v0, 0xffff, v0
; GFX11-GISEL-NEXT: s_add_i32 s32, s32, 16
-; GFX11-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-GISEL-NEXT: s_mov_b32 s0, s32
+; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-GISEL-NEXT: v_lshl_add_u32 v0, v0, 2, 15
-; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-GISEL-NEXT: v_and_b32_e32 v0, -16, v0
; GFX11-GISEL-NEXT: s_or_saveexec_b32 s1, -1
; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
@@ -3224,7 +3240,7 @@ define void @test_dynamic_stackalloc_device_divergent_non_standard_size_i16(i16
; GFX11-GISEL-NEXT: ds_swizzle_b32 v2, v1 offset:swizzle(BROADCAST,32,15)
; GFX11-GISEL-NEXT: s_waitcnt lgkmcnt(0)
; GFX11-GISEL-NEXT: v_max_u32_e32 v1, v1, v2
-; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX11-GISEL-NEXT: v_readlane_b32 s2, v1, 31
; GFX11-GISEL-NEXT: s_mov_b32 exec_lo, s1
; GFX11-GISEL-NEXT: v_mov_b32_e32 v0, 0x29a
@@ -3340,9 +3356,9 @@ define void @test_dynamic_stackalloc_device_divergent_non_standard_size_i64(i64
; GFX11-SDAG-NEXT: scratch_store_b32 off, v1, s33
; GFX11-SDAG-NEXT: scratch_store_b32 off, v2, s33 offset:4
; GFX11-SDAG-NEXT: s_mov_b32 exec_lo, s0
+; GFX11-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-SDAG-NEXT: v_lshl_add_u32 v0, v0, 2, 15
; GFX11-SDAG-NEXT: s_add_i32 s32, s32, 16
-; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-SDAG-NEXT: v_and_b32_e32 v0, -16, v0
; GFX11-SDAG-NEXT: s_or_saveexec_b32 s0, -1
; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
@@ -3385,10 +3401,11 @@ define void @test_dynamic_stackalloc_device_divergent_non_standard_size_i64(i64
; GFX11-GISEL-NEXT: scratch_store_b32 off, v1, s33
; GFX11-GISEL-NEXT: scratch_store_b32 off, v2, s33 offset:4
; GFX11-GISEL-NEXT: s_mov_b32 exec_lo, s0
+; GFX11-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX11-GISEL-NEXT: v_lshl_add_u32 v0, v0, 2, 15
; GFX11-GISEL-NEXT: s_add_i32 s32, s32, 16
-; GFX11-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-GISEL-NEXT: s_mov_b32 s0, s32
+; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-GISEL-NEXT: v_and_b32_e32 v0, -16, v0
; GFX11-GISEL-NEXT: s_or_saveexec_b32 s1, -1
; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
@@ -3402,7 +3419,7 @@ define void @test_dynamic_stackalloc_device_divergent_non_standard_size_i64(i64
; GFX11-GISEL-NEXT: ds_swizzle_b32 v2, v1 offset:swizzle(BROADCAST,32,15)
; GFX11-GISEL-NEXT: s_waitcnt lgkmcnt(0)
; GFX11-GISEL-NEXT: v_max_u32_e32 v1, v1, v2
-; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX11-GISEL-NEXT: v_readlane_b32 s2, v1, 31
; GFX11-GISEL-NEXT: s_mov_b32 exec_lo, s1
; GFX11-GISEL-NEXT: v_mov_b32_e32 v0, 0x29a
diff --git a/llvm/test/CodeGen/AMDGPU/expand-mov-b64-globaladdr.ll b/llvm/test/CodeGen/AMDGPU/expand-mov-b64-globaladdr.ll
index 6316bd2c2d4507..bd316f099c259d 100644
--- a/llvm/test/CodeGen/AMDGPU/expand-mov-b64-globaladdr.ll
+++ b/llvm/test/CodeGen/AMDGPU/expand-mov-b64-globaladdr.ll
@@ -38,9 +38,11 @@ define amdgpu_ps ptr addrspace(4) @v_mov_b64_pseudo_globaladdr(i1 %cond) {
; GFX11-NEXT: v_mov_b32_e32 v0, gv at abs32@lo
; GFX11-NEXT: v_mov_b32_e32 v1, gv at abs32@hi
; GFX11-NEXT: ; %bb.2: ; %join
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: v_readfirstlane_b32 s0, v0
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX11-NEXT: v_readfirstlane_b32 s1, v1
; GFX11-NEXT: ; return to shader part epilog
entry:
@@ -73,8 +75,8 @@ define amdgpu_ps ptr addrspace(4) @s_mov_b64_imm_pseudo_globaladdr(i1 inreg %con
; GFX11-LABEL: s_mov_b64_imm_pseudo_globaladdr:
; GFX11: ; %bb.0: ; %entry
; GFX11-NEXT: s_bitcmp1_b32 s0, 0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_cselect_b32 s0, -1, 0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_and_b32 vcc_lo, exec_lo, s0
; GFX11-NEXT: s_mov_b64 s[0:1], 0
; GFX11-NEXT: s_cbranch_vccnz .LBB1_2
diff --git a/llvm/test/CodeGen/AMDGPU/expand-scalar-carry-out-select-user.ll b/llvm/test/CodeGen/AMDGPU/expand-scalar-carry-out-select-user.ll
index 837162f8f70a38..81f0f7ac920cc0 100644
--- a/llvm/test/CodeGen/AMDGPU/expand-scalar-carry-out-select-user.ll
+++ b/llvm/test/CodeGen/AMDGPU/expand-scalar-carry-out-select-user.ll
@@ -62,11 +62,12 @@ define i32 @s_add_co_select_user() {
; GFX11-NEXT: s_add_u32 s1, s0, s0
; GFX11-NEXT: s_addc_u32 s2, s0, 0
; GFX11-NEXT: s_cselect_b32 s3, -1, 0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_3) | instid1(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_b32 s3, s3, exec_lo
; GFX11-NEXT: s_cselect_b32 s2, s2, 0
; GFX11-NEXT: s_cmp_gt_u32 s0, 31
; GFX11-NEXT: s_cselect_b32 s0, s1, s2
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: v_mov_b32_e32 v0, s0
; GFX11-NEXT: s_setpc_b64 s[30:31]
bb:
@@ -173,6 +174,7 @@ define amdgpu_kernel void @s_add_co_br_user(i32 %i) {
; GFX11-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB1_2
; GFX11-NEXT: ; %bb.1: ; %bb0
; GFX11-NEXT: v_dual_mov_b32 v0, 0 :: v_dual_mov_b32 v1, 0
diff --git a/llvm/test/CodeGen/AMDGPU/extract-subvector-16bit.ll b/llvm/test/CodeGen/AMDGPU/extract-subvector-16bit.ll
index 176dffac4abc5a..732a4646790363 100644
--- a/llvm/test/CodeGen/AMDGPU/extract-subvector-16bit.ll
+++ b/llvm/test/CodeGen/AMDGPU/extract-subvector-16bit.ll
@@ -144,6 +144,7 @@ define <4 x i16> @vec_8xi16_extract_4xi16(ptr addrspace(1) %p0, ptr addrspace(1)
; GFX11-TRUE16-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-TRUE16-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-TRUE16-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_cbranch_scc1 .LBB0_5
; GFX11-TRUE16-NEXT: ; %bb.4: ; %T
; GFX11-TRUE16-NEXT: global_load_b128 v[2:5], v[0:1], off glc dlc
@@ -181,6 +182,7 @@ define <4 x i16> @vec_8xi16_extract_4xi16(ptr addrspace(1) %p0, ptr addrspace(1)
; GFX11-FAKE16-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-FAKE16-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-FAKE16-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_cbranch_scc1 .LBB0_5
; GFX11-FAKE16-NEXT: ; %bb.4: ; %T
; GFX11-FAKE16-NEXT: global_load_b128 v[2:5], v[0:1], off glc dlc
@@ -360,6 +362,7 @@ define <4 x i16> @vec_8xi16_extract_4xi16_2(ptr addrspace(1) %p0, ptr addrspace(
; GFX11-TRUE16-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-TRUE16-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-TRUE16-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_cbranch_scc1 .LBB1_5
; GFX11-TRUE16-NEXT: ; %bb.4: ; %T
; GFX11-TRUE16-NEXT: global_load_b128 v[2:5], v[0:1], off glc dlc
@@ -395,6 +398,7 @@ define <4 x i16> @vec_8xi16_extract_4xi16_2(ptr addrspace(1) %p0, ptr addrspace(
; GFX11-FAKE16-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-FAKE16-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-FAKE16-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_cbranch_scc1 .LBB1_5
; GFX11-FAKE16-NEXT: ; %bb.4: ; %T
; GFX11-FAKE16-NEXT: global_load_b128 v[2:5], v[0:1], off glc dlc
@@ -574,6 +578,7 @@ define <4 x half> @vec_8xf16_extract_4xf16(ptr addrspace(1) %p0, ptr addrspace(1
; GFX11-TRUE16-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-TRUE16-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-TRUE16-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_cbranch_scc1 .LBB2_5
; GFX11-TRUE16-NEXT: ; %bb.4: ; %T
; GFX11-TRUE16-NEXT: global_load_b128 v[2:5], v[0:1], off glc dlc
@@ -585,10 +590,9 @@ define <4 x half> @vec_8xf16_extract_4xf16(ptr addrspace(1) %p0, ptr addrspace(1
; GFX11-TRUE16-NEXT: v_cmp_nge_f16_e64 s0, 0.5, v3.l
; GFX11-TRUE16-NEXT: v_cmp_ge_f16_e64 s2, 0.5, v3.l
; GFX11-TRUE16-NEXT: v_cmp_ge_f16_e64 s1, 0.5, v0.l
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_4)
; GFX11-TRUE16-NEXT: v_cndmask_b16 v0.l, v1.l, 0x3d00, vcc_lo
; GFX11-TRUE16-NEXT: v_cndmask_b16 v1.h, 0x3d00, v1.l, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX11-TRUE16-NEXT: v_cndmask_b16 v0.h, v1.l, 0x3d00, s1
; GFX11-TRUE16-NEXT: v_cndmask_b16 v1.l, v1.l, 0x3d00, s2
; GFX11-TRUE16-NEXT: s_setpc_b64 s[30:31]
@@ -611,6 +615,7 @@ define <4 x half> @vec_8xf16_extract_4xf16(ptr addrspace(1) %p0, ptr addrspace(1
; GFX11-FAKE16-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-FAKE16-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-FAKE16-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_cbranch_scc1 .LBB2_5
; GFX11-FAKE16-NEXT: ; %bb.4: ; %T
; GFX11-FAKE16-NEXT: global_load_b128 v[2:5], v[0:1], off glc dlc
@@ -831,6 +836,7 @@ define <4 x i16> @vec_16xi16_extract_4xi16(ptr addrspace(1) %p0, ptr addrspace(1
; GFX11-TRUE16-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-TRUE16-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-TRUE16-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_cbranch_scc1 .LBB3_5
; GFX11-TRUE16-NEXT: ; %bb.4: ; %T
; GFX11-TRUE16-NEXT: global_load_b128 v[2:5], v[0:1], off offset:16 glc dlc
@@ -872,6 +878,7 @@ define <4 x i16> @vec_16xi16_extract_4xi16(ptr addrspace(1) %p0, ptr addrspace(1
; GFX11-FAKE16-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-FAKE16-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-FAKE16-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_cbranch_scc1 .LBB3_5
; GFX11-FAKE16-NEXT: ; %bb.4: ; %T
; GFX11-FAKE16-NEXT: global_load_b128 v[2:5], v[0:1], off offset:16 glc dlc
@@ -1094,6 +1101,7 @@ define <4 x i16> @vec_16xi16_extract_4xi16_2(ptr addrspace(1) %p0, ptr addrspace
; GFX11-TRUE16-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-TRUE16-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-TRUE16-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_cbranch_scc1 .LBB4_5
; GFX11-TRUE16-NEXT: ; %bb.4: ; %T
; GFX11-TRUE16-NEXT: global_load_b128 v[2:5], v[0:1], off offset:16 glc dlc
@@ -1133,6 +1141,7 @@ define <4 x i16> @vec_16xi16_extract_4xi16_2(ptr addrspace(1) %p0, ptr addrspace
; GFX11-FAKE16-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-FAKE16-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-FAKE16-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_cbranch_scc1 .LBB4_5
; GFX11-FAKE16-NEXT: ; %bb.4: ; %T
; GFX11-FAKE16-NEXT: global_load_b128 v[2:5], v[0:1], off offset:16 glc dlc
@@ -1355,6 +1364,7 @@ define <4 x half> @vec_16xf16_extract_4xf16(ptr addrspace(1) %p0, ptr addrspace(
; GFX11-TRUE16-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-TRUE16-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-TRUE16-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_cbranch_scc1 .LBB5_5
; GFX11-TRUE16-NEXT: ; %bb.4: ; %T
; GFX11-TRUE16-NEXT: global_load_b128 v[2:5], v[0:1], off offset:16 glc dlc
@@ -1368,10 +1378,9 @@ define <4 x half> @vec_16xf16_extract_4xf16(ptr addrspace(1) %p0, ptr addrspace(
; GFX11-TRUE16-NEXT: v_cmp_nge_f16_e64 s0, 0.5, v3.l
; GFX11-TRUE16-NEXT: v_cmp_ge_f16_e64 s2, 0.5, v3.l
; GFX11-TRUE16-NEXT: v_cmp_ge_f16_e64 s1, 0.5, v0.l
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_4)
; GFX11-TRUE16-NEXT: v_cndmask_b16 v0.l, v1.l, 0x3d00, vcc_lo
; GFX11-TRUE16-NEXT: v_cndmask_b16 v1.h, 0x3d00, v1.l, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX11-TRUE16-NEXT: v_cndmask_b16 v0.h, v1.l, 0x3d00, s1
; GFX11-TRUE16-NEXT: v_cndmask_b16 v1.l, v1.l, 0x3d00, s2
; GFX11-TRUE16-NEXT: s_setpc_b64 s[30:31]
@@ -1396,6 +1405,7 @@ define <4 x half> @vec_16xf16_extract_4xf16(ptr addrspace(1) %p0, ptr addrspace(
; GFX11-FAKE16-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-FAKE16-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-FAKE16-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_cbranch_scc1 .LBB5_5
; GFX11-FAKE16-NEXT: ; %bb.4: ; %T
; GFX11-FAKE16-NEXT: global_load_b128 v[2:5], v[0:1], off offset:16 glc dlc
@@ -1751,6 +1761,7 @@ define amdgpu_gfx <8 x i16> @vec_16xi16_extract_8xi16_0(i1 inreg %cond, ptr addr
; GFX11-TRUE16-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-TRUE16-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-TRUE16-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_cbranch_scc1 .LBB7_5
; GFX11-TRUE16-NEXT: ; %bb.4: ; %T
; GFX11-TRUE16-NEXT: global_load_b128 v[2:5], v[0:1], off offset:16 glc dlc
@@ -1806,6 +1817,7 @@ define amdgpu_gfx <8 x i16> @vec_16xi16_extract_8xi16_0(i1 inreg %cond, ptr addr
; GFX11-FAKE16-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-FAKE16-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-FAKE16-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_cbranch_scc1 .LBB7_5
; GFX11-FAKE16-NEXT: ; %bb.4: ; %T
; GFX11-FAKE16-NEXT: global_load_b128 v[2:5], v[0:1], off offset:16 glc dlc
@@ -2093,6 +2105,7 @@ define amdgpu_gfx <8 x half> @vec_16xf16_extract_8xf16_0(i1 inreg %cond, ptr add
; GFX11-TRUE16-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-TRUE16-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-TRUE16-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_cbranch_scc1 .LBB8_5
; GFX11-TRUE16-NEXT: ; %bb.4: ; %T
; GFX11-TRUE16-NEXT: global_load_b128 v[2:5], v[0:1], off offset:16 glc dlc
@@ -2148,6 +2161,7 @@ define amdgpu_gfx <8 x half> @vec_16xf16_extract_8xf16_0(i1 inreg %cond, ptr add
; GFX11-FAKE16-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-FAKE16-NEXT: s_cselect_b32 s0, 1, 0
; GFX11-FAKE16-NEXT: s_cmp_lg_u32 s0, 1
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_cbranch_scc1 .LBB8_5
; GFX11-FAKE16-NEXT: ; %bb.4: ; %T
; GFX11-FAKE16-NEXT: global_load_b128 v[2:5], v[0:1], off offset:16 glc dlc
diff --git a/llvm/test/CodeGen/AMDGPU/extract_vector_elt-f16.ll b/llvm/test/CodeGen/AMDGPU/extract_vector_elt-f16.ll
index e95de144525cc7..827b1916666005 100644
--- a/llvm/test/CodeGen/AMDGPU/extract_vector_elt-f16.ll
+++ b/llvm/test/CodeGen/AMDGPU/extract_vector_elt-f16.ll
@@ -662,7 +662,7 @@ define amdgpu_kernel void @v_extractelement_v8f16_dynamic_sgpr(ptr addrspace(1)
; GFX11-TRUE16-NEXT: s_load_b128 s[0:3], s[4:5], 0x24
; GFX11-TRUE16-NEXT: v_and_b32_e32 v4, 0x3ff, v0
; GFX11-TRUE16-NEXT: s_load_b32 s4, s[4:5], 0x34
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_3) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_lshlrev_b32_e32 v0, 4, v4
; GFX11-TRUE16-NEXT: s_waitcnt lgkmcnt(0)
; GFX11-TRUE16-NEXT: global_load_b128 v[0:3], v0, s[2:3]
@@ -705,7 +705,7 @@ define amdgpu_kernel void @v_extractelement_v8f16_dynamic_sgpr(ptr addrspace(1)
; GFX11-FAKE16-NEXT: s_load_b128 s[0:3], s[4:5], 0x24
; GFX11-FAKE16-NEXT: v_and_b32_e32 v4, 0x3ff, v0
; GFX11-FAKE16-NEXT: s_load_b32 s4, s[4:5], 0x34
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_3) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v0, 4, v4
; GFX11-FAKE16-NEXT: s_waitcnt lgkmcnt(0)
; GFX11-FAKE16-NEXT: global_load_b128 v[0:3], v0, s[2:3]
@@ -738,7 +738,7 @@ define amdgpu_kernel void @v_extractelement_v8f16_dynamic_sgpr(ptr addrspace(1)
; GFX11-FAKE16-NEXT: s_cmp_eq_u32 s4, 7
; GFX11-FAKE16-NEXT: v_cndmask_b32_e32 v0, v0, v3, vcc_lo
; GFX11-FAKE16-NEXT: s_cselect_b32 vcc_lo, -1, 0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_cndmask_b32_e32 v0, v0, v1, vcc_lo
; GFX11-FAKE16-NEXT: global_store_b16 v2, v0, s[0:1]
; GFX11-FAKE16-NEXT: s_endpgm
@@ -909,36 +909,36 @@ define amdgpu_kernel void @v_extractelement_v16f16_dynamic_sgpr(ptr addrspace(1)
; GFX11-TRUE16-NEXT: global_load_b128 v[0:3], v4, s[2:3]
; GFX11-TRUE16-NEXT: global_load_b128 v[4:7], v4, s[2:3] offset:16
; GFX11-TRUE16-NEXT: s_cmp_eq_u32 s4, 1
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_3) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: s_cselect_b32 s2, -1, 0
; GFX11-TRUE16-NEXT: s_cmp_eq_u32 s4, 2
; GFX11-TRUE16-NEXT: s_waitcnt vmcnt(1)
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v9, 16, v0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_3) | instid1(VALU_DEP_2)
; GFX11-TRUE16-NEXT: v_cndmask_b16 v0.l, v0.l, v9.l, s2
; GFX11-TRUE16-NEXT: s_cselect_b32 s2, -1, 0
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v9, 16, v1
; GFX11-TRUE16-NEXT: s_cmp_eq_u32 s4, 3
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_3) | instid1(VALU_DEP_2)
; GFX11-TRUE16-NEXT: v_cndmask_b16 v0.l, v0.l, v1.l, s2
; GFX11-TRUE16-NEXT: s_cselect_b32 s2, -1, 0
; GFX11-TRUE16-NEXT: s_cmp_eq_u32 s4, 4
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v1, 16, v2
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_2) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_cndmask_b16 v0.l, v0.l, v9.l, s2
; GFX11-TRUE16-NEXT: s_cselect_b32 s2, -1, 0
; GFX11-TRUE16-NEXT: s_cmp_eq_u32 s4, 5
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_3) | instid1(VALU_DEP_2)
; GFX11-TRUE16-NEXT: v_cndmask_b16 v0.l, v0.l, v2.l, s2
; GFX11-TRUE16-NEXT: s_cselect_b32 s2, -1, 0
; GFX11-TRUE16-NEXT: s_cmp_eq_u32 s4, 6
; GFX11-TRUE16-NEXT: v_lshlrev_b32_e32 v2, 1, v8
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_3) | instid1(VALU_DEP_2)
; GFX11-TRUE16-NEXT: v_cndmask_b16 v0.l, v0.l, v1.l, s2
; GFX11-TRUE16-NEXT: s_cselect_b32 s2, -1, 0
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v1, 16, v3
; GFX11-TRUE16-NEXT: s_cmp_eq_u32 s4, 7
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_2) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_cndmask_b16 v0.l, v0.l, v3.l, s2
; GFX11-TRUE16-NEXT: s_cselect_b32 s2, -1, 0
; GFX11-TRUE16-NEXT: s_cmp_eq_u32 s4, 8
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_cndmask_b16 v0.l, v0.l, v1.l, s2
; GFX11-TRUE16-NEXT: s_cselect_b32 s2, -1, 0
; GFX11-TRUE16-NEXT: s_waitcnt vmcnt(0)
@@ -987,36 +987,36 @@ define amdgpu_kernel void @v_extractelement_v16f16_dynamic_sgpr(ptr addrspace(1)
; GFX11-FAKE16-NEXT: global_load_b128 v[0:3], v4, s[2:3]
; GFX11-FAKE16-NEXT: global_load_b128 v[4:7], v4, s[2:3] offset:16
; GFX11-FAKE16-NEXT: s_cmp_eq_u32 s4, 1
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_3) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: s_cselect_b32 vcc_lo, -1, 0
; GFX11-FAKE16-NEXT: s_cmp_eq_u32 s4, 2
; GFX11-FAKE16-NEXT: s_waitcnt vmcnt(1)
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v9, 16, v0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_3) | instid1(VALU_DEP_2)
; GFX11-FAKE16-NEXT: v_cndmask_b32_e32 v0, v0, v9, vcc_lo
; GFX11-FAKE16-NEXT: s_cselect_b32 vcc_lo, -1, 0
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v9, 16, v1
; GFX11-FAKE16-NEXT: s_cmp_eq_u32 s4, 3
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_3) | instid1(VALU_DEP_2)
; GFX11-FAKE16-NEXT: v_cndmask_b32_e32 v0, v0, v1, vcc_lo
; GFX11-FAKE16-NEXT: s_cselect_b32 vcc_lo, -1, 0
; GFX11-FAKE16-NEXT: s_cmp_eq_u32 s4, 4
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v1, 16, v2
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_2) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_cndmask_b32_e32 v0, v0, v9, vcc_lo
; GFX11-FAKE16-NEXT: s_cselect_b32 vcc_lo, -1, 0
; GFX11-FAKE16-NEXT: s_cmp_eq_u32 s4, 5
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_3) | instid1(VALU_DEP_2)
; GFX11-FAKE16-NEXT: v_cndmask_b32_e32 v0, v0, v2, vcc_lo
; GFX11-FAKE16-NEXT: s_cselect_b32 vcc_lo, -1, 0
; GFX11-FAKE16-NEXT: s_cmp_eq_u32 s4, 6
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v2, 1, v8
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_3) | instid1(VALU_DEP_2)
; GFX11-FAKE16-NEXT: v_cndmask_b32_e32 v0, v0, v1, vcc_lo
; GFX11-FAKE16-NEXT: s_cselect_b32 vcc_lo, -1, 0
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v1, 16, v3
; GFX11-FAKE16-NEXT: s_cmp_eq_u32 s4, 7
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_2) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_cndmask_b32_e32 v0, v0, v3, vcc_lo
; GFX11-FAKE16-NEXT: s_cselect_b32 vcc_lo, -1, 0
; GFX11-FAKE16-NEXT: s_cmp_eq_u32 s4, 8
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_cndmask_b32_e32 v0, v0, v1, vcc_lo
; GFX11-FAKE16-NEXT: s_cselect_b32 vcc_lo, -1, 0
; GFX11-FAKE16-NEXT: s_waitcnt vmcnt(0)
@@ -1048,7 +1048,7 @@ define amdgpu_kernel void @v_extractelement_v16f16_dynamic_sgpr(ptr addrspace(1)
; GFX11-FAKE16-NEXT: s_cmp_eq_u32 s4, 15
; GFX11-FAKE16-NEXT: v_cndmask_b32_e32 v0, v0, v7, vcc_lo
; GFX11-FAKE16-NEXT: s_cselect_b32 vcc_lo, -1, 0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_cndmask_b32_e32 v0, v0, v1, vcc_lo
; GFX11-FAKE16-NEXT: global_store_b16 v2, v0, s[0:1]
; GFX11-FAKE16-NEXT: s_endpgm
diff --git a/llvm/test/CodeGen/AMDGPU/fcopysign.bf16.ll b/llvm/test/CodeGen/AMDGPU/fcopysign.bf16.ll
index 04632801c6ed14..52061df6b8b4da 100644
--- a/llvm/test/CodeGen/AMDGPU/fcopysign.bf16.ll
+++ b/llvm/test/CodeGen/AMDGPU/fcopysign.bf16.ll
@@ -3696,12 +3696,11 @@ define <2 x bfloat> @v_copysign_out_v2bf16_mag_v2f64_sign_v2bf16(<2 x double> %m
; GFX11TRUE16-NEXT: v_cmp_gt_f64_e64 s1, |v[2:3]|, |v[5:6]|
; GFX11TRUE16-NEXT: v_cmp_nlg_f64_e32 vcc_lo, v[2:3], v[5:6]
; GFX11TRUE16-NEXT: v_cmp_nlg_f64_e64 s0, v[0:1], v[7:8]
-; GFX11TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX11TRUE16-NEXT: v_cndmask_b32_e64 v5, -1, 1, s1
; GFX11TRUE16-NEXT: v_cmp_gt_f64_e64 s1, |v[0:1]|, |v[7:8]|
+; GFX11TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11TRUE16-NEXT: v_add_nc_u32_e32 v5, v9, v5
; GFX11TRUE16-NEXT: v_and_b32_e32 v6, 1, v10
-; GFX11TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_4)
; GFX11TRUE16-NEXT: v_cmp_eq_u32_e64 s2, 1, v6
; GFX11TRUE16-NEXT: v_cndmask_b32_e64 v7, -1, 1, s1
; GFX11TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
@@ -3709,6 +3708,7 @@ define <2 x bfloat> @v_copysign_out_v2bf16_mag_v2f64_sign_v2bf16(<2 x double> %m
; GFX11TRUE16-NEXT: v_and_b32_e32 v11, 1, v9
; GFX11TRUE16-NEXT: v_cmp_eq_u32_e64 s1, 1, v11
; GFX11TRUE16-NEXT: s_or_b32 vcc_lo, vcc_lo, s1
+; GFX11TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX11TRUE16-NEXT: v_cndmask_b32_e32 v5, v5, v9, vcc_lo
; GFX11TRUE16-NEXT: s_or_b32 vcc_lo, s0, s2
; GFX11TRUE16-NEXT: v_cndmask_b32_e32 v6, v6, v10, vcc_lo
@@ -3745,36 +3745,35 @@ define <2 x bfloat> @v_copysign_out_v2bf16_mag_v2f64_sign_v2bf16(<2 x double> %m
; GFX11FAKE16-NEXT: v_cmp_gt_f64_e64 s1, |v[0:1]|, |v[5:6]|
; GFX11FAKE16-NEXT: v_cmp_nlg_f64_e32 vcc_lo, v[0:1], v[5:6]
; GFX11FAKE16-NEXT: v_cmp_nlg_f64_e64 s0, v[2:3], v[7:8]
-; GFX11FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX11FAKE16-NEXT: v_cndmask_b32_e64 v5, -1, 1, s1
; GFX11FAKE16-NEXT: v_cmp_gt_f64_e64 s1, |v[2:3]|, |v[7:8]|
+; GFX11FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11FAKE16-NEXT: v_add_nc_u32_e32 v5, v9, v5
-; GFX11FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11FAKE16-NEXT: v_cndmask_b32_e64 v6, -1, 1, s1
; GFX11FAKE16-NEXT: v_add_nc_u32_e32 v6, v10, v6
; GFX11FAKE16-NEXT: v_and_b32_e32 v11, 1, v9
-; GFX11FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_2) | instid1(VALU_DEP_1)
+; GFX11FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX11FAKE16-NEXT: v_cmp_eq_u32_e64 s1, 1, v11
; GFX11FAKE16-NEXT: s_or_b32 vcc_lo, vcc_lo, s1
; GFX11FAKE16-NEXT: v_dual_cndmask_b32 v5, v5, v9 :: v_dual_and_b32 v12, 1, v10
+; GFX11FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11FAKE16-NEXT: v_cmp_eq_u32_e64 s2, 1, v12
-; GFX11FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_2) | instid1(VALU_DEP_2)
; GFX11FAKE16-NEXT: v_bfe_u32 v7, v5, 16, 1
; GFX11FAKE16-NEXT: v_or_b32_e32 v9, 0x400000, v5
; GFX11FAKE16-NEXT: s_or_b32 vcc_lo, s0, s2
+; GFX11FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_2) | instid1(VALU_DEP_2)
; GFX11FAKE16-NEXT: v_add3_u32 v5, v7, v5, 0x7fff
; GFX11FAKE16-NEXT: v_cndmask_b32_e32 v6, v6, v10, vcc_lo
; GFX11FAKE16-NEXT: v_cmp_u_f64_e32 vcc_lo, v[0:1], v[0:1]
-; GFX11FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX11FAKE16-NEXT: v_bfe_u32 v8, v6, 16, 1
; GFX11FAKE16-NEXT: v_or_b32_e32 v7, 0x400000, v6
+; GFX11FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_2) | instid1(VALU_DEP_3)
; GFX11FAKE16-NEXT: v_add3_u32 v6, v8, v6, 0x7fff
; GFX11FAKE16-NEXT: v_cndmask_b32_e32 v0, v5, v9, vcc_lo
; GFX11FAKE16-NEXT: v_cmp_u_f64_e32 vcc_lo, v[2:3], v[2:3]
-; GFX11FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11FAKE16-NEXT: v_cndmask_b32_e32 v1, v6, v7, vcc_lo
+; GFX11FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11FAKE16-NEXT: v_perm_b32 v0, v1, v0, 0x7060302
-; GFX11FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11FAKE16-NEXT: v_bfi_b32 v0, 0x7fff7fff, v0, v4
; GFX11FAKE16-NEXT: s_setpc_b64 s[30:31]
%mag.trunc = fptrunc <2 x double> %mag to <2 x bfloat>
@@ -4642,29 +4641,29 @@ define amdgpu_ps i32 @s_copysign_out_v2bf16_mag_v2f64_sign_v2bf16(<2 x double> i
; GFX11-NEXT: s_addk_i32 s5, 0x7fff
; GFX11-NEXT: s_and_b32 s3, s3, exec_lo
; GFX11-NEXT: s_cselect_b32 s1, s1, s5
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_lshr_b32 s1, s1, 16
; GFX11-NEXT: s_bitcmp1_b32 s7, 0
; GFX11-NEXT: s_cselect_b32 s3, -1, 0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_or_b32 s2, s2, s3
; GFX11-NEXT: s_and_b32 s3, s6, exec_lo
; GFX11-NEXT: s_cselect_b32 s3, 1, -1
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_add_i32 s3, s7, s3
; GFX11-NEXT: s_and_b32 s2, s2, exec_lo
; GFX11-NEXT: s_cselect_b32 s2, s7, s3
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_bfe_u32 s3, s2, 0x10010
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_4) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_add_i32 s3, s3, s2
; GFX11-NEXT: s_bitset1_b32 s2, 22
; GFX11-NEXT: s_addk_i32 s3, 0x7fff
; GFX11-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-NEXT: s_cselect_b32 s0, s2, s3
-; GFX11-NEXT: s_lshr_b32 s0, s0, 16
; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
+; GFX11-NEXT: s_lshr_b32 s0, s0, 16
; GFX11-NEXT: s_pack_ll_b32_b16 s0, s0, s1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_bfi_b32 v0, 0x7fff7fff, s0, v0
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_readfirstlane_b32 s0, v0
; GFX11-NEXT: ; return to shader part epilog
%mag.trunc = fptrunc <2 x double> %mag to <2 x bfloat>
@@ -5571,12 +5570,12 @@ define <3 x bfloat> @v_copysign_out_v3bf16_mag_v3f64_sign_v3bf16(<3 x double> %m
; GFX11TRUE16-NEXT: v_cmp_gt_f64_e64 s3, |v[0:1]|, |v[10:11]|
; GFX11TRUE16-NEXT: v_cndmask_b32_e64 v10, -1, 1, s4
; GFX11TRUE16-NEXT: v_cndmask_b32_e64 v8, -1, 1, s2
-; GFX11TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_3)
; GFX11TRUE16-NEXT: v_cndmask_b32_e64 v9, -1, 1, s3
+; GFX11TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_1) | instid1(VALU_DEP_4)
; GFX11TRUE16-NEXT: v_add_nc_u32_e32 v10, v16, v10
; GFX11TRUE16-NEXT: v_and_b32_e32 v17, 1, v14
-; GFX11TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11TRUE16-NEXT: v_add_nc_u32_e32 v8, v14, v8
+; GFX11TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX11TRUE16-NEXT: v_cmp_eq_u32_e64 s2, 1, v17
; GFX11TRUE16-NEXT: s_or_b32 vcc_lo, vcc_lo, s2
; GFX11TRUE16-NEXT: v_dual_cndmask_b32 v8, v8, v14 :: v_dual_and_b32 v19, 1, v15
@@ -5633,29 +5632,28 @@ define <3 x bfloat> @v_copysign_out_v3bf16_mag_v3f64_sign_v3bf16(<3 x double> %m
; GFX11FAKE16-NEXT: v_cmp_nlg_f64_e32 vcc_lo, v[4:5], v[8:9]
; GFX11FAKE16-NEXT: v_cmp_nlg_f64_e64 s0, v[0:1], v[10:11]
; GFX11FAKE16-NEXT: v_cmp_nlg_f64_e64 s1, v[2:3], v[12:13]
-; GFX11FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11FAKE16-NEXT: v_cndmask_b32_e64 v8, -1, 1, s3
; GFX11FAKE16-NEXT: v_cmp_gt_f64_e64 s3, |v[0:1]|, |v[10:11]|
; GFX11FAKE16-NEXT: v_cndmask_b32_e64 v9, -1, 1, s3
; GFX11FAKE16-NEXT: v_cmp_gt_f64_e64 s3, |v[2:3]|, |v[12:13]|
-; GFX11FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_3)
+; GFX11FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_3) | instid1(VALU_DEP_2)
; GFX11FAKE16-NEXT: v_add_nc_u32_e32 v9, v15, v9
; GFX11FAKE16-NEXT: v_add_nc_u32_e32 v8, v14, v8
; GFX11FAKE16-NEXT: v_cndmask_b32_e64 v10, -1, 1, s3
; GFX11FAKE16-NEXT: v_cmp_eq_u32_e64 s3, 1, v18
-; GFX11FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11FAKE16-NEXT: v_add_nc_u32_e32 v10, v16, v10
; GFX11FAKE16-NEXT: v_and_b32_e32 v17, 1, v14
+; GFX11FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX11FAKE16-NEXT: v_cmp_eq_u32_e64 s2, 1, v17
; GFX11FAKE16-NEXT: s_or_b32 vcc_lo, vcc_lo, s2
; GFX11FAKE16-NEXT: v_dual_cndmask_b32 v8, v8, v14 :: v_dual_and_b32 v19, 1, v16
; GFX11FAKE16-NEXT: s_or_b32 vcc_lo, s0, s3
+; GFX11FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11FAKE16-NEXT: v_cndmask_b32_e32 v9, v9, v15, vcc_lo
-; GFX11FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_3)
; GFX11FAKE16-NEXT: v_cmp_eq_u32_e64 s4, 1, v19
+; GFX11FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_1) | instid1(VALU_DEP_4)
; GFX11FAKE16-NEXT: v_bfe_u32 v12, v8, 16, 1
; GFX11FAKE16-NEXT: v_or_b32_e32 v14, 0x400000, v8
-; GFX11FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_4)
; GFX11FAKE16-NEXT: v_bfe_u32 v11, v9, 16, 1
; GFX11FAKE16-NEXT: s_or_b32 vcc_lo, s1, s4
; GFX11FAKE16-NEXT: v_or_b32_e32 v15, 0x400000, v9
@@ -6817,13 +6815,13 @@ define <4 x bfloat> @v_copysign_out_v4bf16_mag_v4f64_sign_v4bf16(<4 x double> %m
; GFX11TRUE16-NEXT: v_cmp_nlg_f64_e64 s2, v[0:1], v[16:17]
; GFX11TRUE16-NEXT: v_cndmask_b32_e64 v10, -1, 1, s6
; GFX11TRUE16-NEXT: v_cmp_gt_f64_e64 s6, |v[4:5]|, |v[12:13]|
-; GFX11TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX11TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_2) | instid1(VALU_DEP_2)
; GFX11TRUE16-NEXT: v_add_nc_u32_e32 v10, v18, v10
; GFX11TRUE16-NEXT: v_cndmask_b32_e64 v11, -1, 1, s6
; GFX11TRUE16-NEXT: v_cmp_gt_f64_e64 s6, |v[2:3]|, |v[14:15]|
-; GFX11TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11TRUE16-NEXT: v_add_nc_u32_e32 v11, v19, v11
; GFX11TRUE16-NEXT: v_and_b32_e32 v22, 1, v18
+; GFX11TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX11TRUE16-NEXT: v_cmp_eq_u32_e64 s3, 1, v22
; GFX11TRUE16-NEXT: s_or_b32 vcc_lo, vcc_lo, s3
; GFX11TRUE16-NEXT: v_dual_cndmask_b32 v10, v10, v18 :: v_dual_and_b32 v23, 1, v19
@@ -6848,19 +6846,19 @@ define <4 x bfloat> @v_copysign_out_v4bf16_mag_v4f64_sign_v4bf16(<4 x double> %m
; GFX11TRUE16-NEXT: v_and_b32_e32 v24, 1, v20
; GFX11TRUE16-NEXT: v_cmp_eq_u32_e64 s5, 1, v24
; GFX11TRUE16-NEXT: s_or_b32 vcc_lo, s1, s5
+; GFX11TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11TRUE16-NEXT: v_dual_cndmask_b32 v12, v12, v20 :: v_dual_and_b32 v25, 1, v21
-; GFX11TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11TRUE16-NEXT: v_cmp_eq_u32_e64 s6, 1, v25
+; GFX11TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_2) | instid1(VALU_DEP_2)
; GFX11TRUE16-NEXT: v_bfe_u32 v18, v12, 16, 1
; GFX11TRUE16-NEXT: v_or_b32_e32 v19, 0x400000, v12
; GFX11TRUE16-NEXT: s_or_b32 vcc_lo, s2, s6
-; GFX11TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_2) | instid1(VALU_DEP_2)
; GFX11TRUE16-NEXT: v_add3_u32 v12, v18, v12, 0x7fff
; GFX11TRUE16-NEXT: v_cndmask_b32_e32 v13, v13, v21, vcc_lo
; GFX11TRUE16-NEXT: v_cmp_u_f64_e32 vcc_lo, v[6:7], v[6:7]
+; GFX11TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX11TRUE16-NEXT: v_bfe_u32 v20, v13, 16, 1
; GFX11TRUE16-NEXT: v_or_b32_e32 v21, 0x400000, v13
-; GFX11TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX11TRUE16-NEXT: v_add3_u32 v13, v20, v13, 0x7fff
; GFX11TRUE16-NEXT: v_cndmask_b32_e32 v6, v10, v15, vcc_lo
; GFX11TRUE16-NEXT: v_cmp_u_f64_e32 vcc_lo, v[2:3], v[2:3]
@@ -6903,13 +6901,13 @@ define <4 x bfloat> @v_copysign_out_v4bf16_mag_v4f64_sign_v4bf16(<4 x double> %m
; GFX11FAKE16-NEXT: v_cmp_nlg_f64_e64 s2, v[2:3], v[16:17]
; GFX11FAKE16-NEXT: v_cndmask_b32_e64 v10, -1, 1, s6
; GFX11FAKE16-NEXT: v_cmp_gt_f64_e64 s6, |v[6:7]|, |v[12:13]|
-; GFX11FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX11FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_2) | instid1(VALU_DEP_2)
; GFX11FAKE16-NEXT: v_add_nc_u32_e32 v10, v18, v10
; GFX11FAKE16-NEXT: v_cndmask_b32_e64 v11, -1, 1, s6
; GFX11FAKE16-NEXT: v_cmp_gt_f64_e64 s6, |v[0:1]|, |v[14:15]|
-; GFX11FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11FAKE16-NEXT: v_add_nc_u32_e32 v11, v19, v11
; GFX11FAKE16-NEXT: v_and_b32_e32 v22, 1, v18
+; GFX11FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX11FAKE16-NEXT: v_cmp_eq_u32_e64 s3, 1, v22
; GFX11FAKE16-NEXT: s_or_b32 vcc_lo, vcc_lo, s3
; GFX11FAKE16-NEXT: v_dual_cndmask_b32 v10, v10, v18 :: v_dual_and_b32 v23, 1, v19
@@ -6934,19 +6932,19 @@ define <4 x bfloat> @v_copysign_out_v4bf16_mag_v4f64_sign_v4bf16(<4 x double> %m
; GFX11FAKE16-NEXT: v_and_b32_e32 v24, 1, v20
; GFX11FAKE16-NEXT: v_cmp_eq_u32_e64 s5, 1, v24
; GFX11FAKE16-NEXT: s_or_b32 vcc_lo, s1, s5
+; GFX11FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11FAKE16-NEXT: v_dual_cndmask_b32 v12, v12, v20 :: v_dual_and_b32 v25, 1, v21
-; GFX11FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11FAKE16-NEXT: v_cmp_eq_u32_e64 s6, 1, v25
+; GFX11FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_2) | instid1(VALU_DEP_2)
; GFX11FAKE16-NEXT: v_bfe_u32 v18, v12, 16, 1
; GFX11FAKE16-NEXT: v_or_b32_e32 v19, 0x400000, v12
; GFX11FAKE16-NEXT: s_or_b32 vcc_lo, s2, s6
-; GFX11FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_2) | instid1(VALU_DEP_2)
; GFX11FAKE16-NEXT: v_add3_u32 v12, v18, v12, 0x7fff
; GFX11FAKE16-NEXT: v_cndmask_b32_e32 v13, v13, v21, vcc_lo
; GFX11FAKE16-NEXT: v_cmp_u_f64_e32 vcc_lo, v[4:5], v[4:5]
+; GFX11FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX11FAKE16-NEXT: v_bfe_u32 v20, v13, 16, 1
; GFX11FAKE16-NEXT: v_or_b32_e32 v21, 0x400000, v13
-; GFX11FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX11FAKE16-NEXT: v_add3_u32 v13, v20, v13, 0x7fff
; GFX11FAKE16-NEXT: v_cndmask_b32_e32 v4, v10, v15, vcc_lo
; GFX11FAKE16-NEXT: v_cmp_u_f64_e32 vcc_lo, v[0:1], v[0:1]
@@ -7616,23 +7614,23 @@ define amdgpu_ps i32 @s_copysign_bf16_0_f64(double inreg %sign) {
; GFX11-NEXT: v_cmp_u_f64_e64 s0, s[0:1], s[0:1]
; GFX11-NEXT: v_readfirstlane_b32 s1, v2
; GFX11-NEXT: s_bitcmp1_b32 s1, 0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_cselect_b32 s3, -1, 0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_or_b32 s3, vcc_lo, s3
; GFX11-NEXT: s_and_b32 s2, s2, exec_lo
; GFX11-NEXT: s_cselect_b32 s2, 1, -1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_add_i32 s2, s1, s2
; GFX11-NEXT: s_and_b32 s3, s3, exec_lo
; GFX11-NEXT: s_cselect_b32 s1, s1, s2
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_bfe_u32 s2, s1, 0x10010
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_add_i32 s2, s2, s1
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_addk_i32 s2, 0x7fff
; GFX11-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-NEXT: s_cselect_b32 s0, s1, s2
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_lshr_b32 s0, s0, 16
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_and_b32 s0, s0, 0x8000
; GFX11-NEXT: ; return to shader part epilog
%sign.trunc = fptrunc double %sign to bfloat
diff --git a/llvm/test/CodeGen/AMDGPU/fcopysign.f16.ll b/llvm/test/CodeGen/AMDGPU/fcopysign.f16.ll
index 9db4987fc82108..6e64babeb3e5bc 100644
--- a/llvm/test/CodeGen/AMDGPU/fcopysign.f16.ll
+++ b/llvm/test/CodeGen/AMDGPU/fcopysign.f16.ll
@@ -1236,28 +1236,30 @@ define amdgpu_ps i16 @s_copysign_out_f16_mag_f64_sign_f16(double inreg %mag, hal
; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_lshl_b32 s4, s5, s4
; GFX11-TRUE16-NEXT: s_cmp_lg_u32 s4, s1
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_3) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_cselect_b32 s1, 1, 0
; GFX11-TRUE16-NEXT: s_addk_i32 s3, 0xfc10
; GFX11-TRUE16-NEXT: s_or_b32 s1, s5, s1
; GFX11-TRUE16-NEXT: s_lshl_b32 s4, s3, 12
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_or_b32 s4, s0, s4
; GFX11-TRUE16-NEXT: s_cmp_lt_i32 s3, 1
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_cselect_b32 s1, s1, s4
; GFX11-TRUE16-NEXT: s_and_b32 s4, s1, 7
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_cmp_gt_i32 s4, 5
; GFX11-TRUE16-NEXT: s_cselect_b32 s5, 1, 0
; GFX11-TRUE16-NEXT: s_cmp_eq_u32 s4, 3
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_cselect_b32 s4, 1, 0
; GFX11-TRUE16-NEXT: s_lshr_b32 s1, s1, 2
; GFX11-TRUE16-NEXT: s_or_b32 s4, s4, s5
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_add_i32 s1, s1, s4
; GFX11-TRUE16-NEXT: s_cmp_lt_i32 s3, 31
; GFX11-TRUE16-NEXT: s_movk_i32 s4, 0x7e00
; GFX11-TRUE16-NEXT: s_cselect_b32 s1, s1, 0x7c00
; GFX11-TRUE16-NEXT: s_cmp_lg_u32 s0, 0
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_cselect_b32 s0, s4, 0x7c00
; GFX11-TRUE16-NEXT: s_cmpk_eq_i32 s3, 0x40f
; GFX11-TRUE16-NEXT: s_cselect_b32 s0, s0, s1
@@ -1288,28 +1290,30 @@ define amdgpu_ps i16 @s_copysign_out_f16_mag_f64_sign_f16(double inreg %mag, hal
; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_lshl_b32 s4, s5, s4
; GFX11-FAKE16-NEXT: s_cmp_lg_u32 s4, s1
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_3) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_cselect_b32 s1, 1, 0
; GFX11-FAKE16-NEXT: s_addk_i32 s3, 0xfc10
; GFX11-FAKE16-NEXT: s_or_b32 s1, s5, s1
; GFX11-FAKE16-NEXT: s_lshl_b32 s4, s3, 12
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_or_b32 s4, s0, s4
; GFX11-FAKE16-NEXT: s_cmp_lt_i32 s3, 1
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_cselect_b32 s1, s1, s4
; GFX11-FAKE16-NEXT: s_and_b32 s4, s1, 7
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_cmp_gt_i32 s4, 5
; GFX11-FAKE16-NEXT: s_cselect_b32 s5, 1, 0
; GFX11-FAKE16-NEXT: s_cmp_eq_u32 s4, 3
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_cselect_b32 s4, 1, 0
; GFX11-FAKE16-NEXT: s_lshr_b32 s1, s1, 2
; GFX11-FAKE16-NEXT: s_or_b32 s4, s4, s5
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_add_i32 s1, s1, s4
; GFX11-FAKE16-NEXT: s_cmp_lt_i32 s3, 31
; GFX11-FAKE16-NEXT: s_movk_i32 s4, 0x7e00
; GFX11-FAKE16-NEXT: s_cselect_b32 s1, s1, 0x7c00
; GFX11-FAKE16-NEXT: s_cmp_lg_u32 s0, 0
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_cselect_b32 s0, s4, 0x7c00
; GFX11-FAKE16-NEXT: s_cmpk_eq_i32 s3, 0x40f
; GFX11-FAKE16-NEXT: s_cselect_b32 s0, s0, s1
@@ -4035,24 +4039,26 @@ define amdgpu_ps i32 @s_copysign_out_v2f16_mag_v2f64_sign_v2f16(<2 x double> inr
; GFX11-NEXT: v_readfirstlane_b32 s7, v0
; GFX11-NEXT: s_lshr_b32 s8, s5, s7
; GFX11-NEXT: s_lshl_b32 s7, s8, s7
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_4) | instid1(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_cmp_lg_u32 s7, s5
; GFX11-NEXT: s_cselect_b32 s5, 1, 0
; GFX11-NEXT: s_addk_i32 s6, 0xfc10
; GFX11-NEXT: s_or_b32 s5, s8, s5
; GFX11-NEXT: s_lshl_b32 s7, s6, 12
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_or_b32 s7, s2, s7
; GFX11-NEXT: s_cmp_lt_i32 s6, 1
; GFX11-NEXT: s_cselect_b32 s5, s5, s7
; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_b32 s7, s5, 7
; GFX11-NEXT: s_cmp_gt_i32 s7, 5
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_cselect_b32 s8, 1, 0
; GFX11-NEXT: s_cmp_eq_u32 s7, 3
; GFX11-NEXT: s_cselect_b32 s7, 1, 0
; GFX11-NEXT: s_lshr_b32 s5, s5, 2
; GFX11-NEXT: s_or_b32 s7, s7, s8
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_4) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_add_i32 s5, s5, s7
; GFX11-NEXT: s_cmp_lt_i32 s6, 31
; GFX11-NEXT: s_movk_i32 s7, 0x7e00
@@ -4060,6 +4066,7 @@ define amdgpu_ps i32 @s_copysign_out_v2f16_mag_v2f64_sign_v2f16(<2 x double> inr
; GFX11-NEXT: s_cmp_lg_u32 s2, 0
; GFX11-NEXT: s_cselect_b32 s2, s7, 0x7c00
; GFX11-NEXT: s_cmpk_eq_i32 s6, 0x40f
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cselect_b32 s2, s2, s5
; GFX11-NEXT: s_lshr_b32 s3, s3, 16
; GFX11-NEXT: s_lshr_b32 s5, s1, 8
@@ -4079,28 +4086,31 @@ define amdgpu_ps i32 @s_copysign_out_v2f16_mag_v2f64_sign_v2f16(<2 x double> inr
; GFX11-NEXT: v_mov_b32_e32 v0, s4
; GFX11-NEXT: s_lshr_b32 s8, s5, s6
; GFX11-NEXT: s_lshl_b32 s6, s8, s6
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_4) | instid1(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_cmp_lg_u32 s6, s5
; GFX11-NEXT: s_cselect_b32 s5, 1, 0
; GFX11-NEXT: s_addk_i32 s3, 0xfc10
; GFX11-NEXT: s_or_b32 s5, s8, s5
; GFX11-NEXT: s_lshl_b32 s6, s3, 12
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_or_b32 s6, s0, s6
; GFX11-NEXT: s_cmp_lt_i32 s3, 1
; GFX11-NEXT: s_cselect_b32 s5, s5, s6
; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_b32 s6, s5, 7
; GFX11-NEXT: s_cmp_gt_i32 s6, 5
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_cselect_b32 s8, 1, 0
; GFX11-NEXT: s_cmp_eq_u32 s6, 3
; GFX11-NEXT: s_cselect_b32 s6, 1, 0
; GFX11-NEXT: s_lshr_b32 s5, s5, 2
; GFX11-NEXT: s_or_b32 s6, s6, s8
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_add_i32 s5, s5, s6
; GFX11-NEXT: s_cmp_lt_i32 s3, 31
; GFX11-NEXT: s_cselect_b32 s5, s5, 0x7c00
; GFX11-NEXT: s_cmp_lg_u32 s0, 0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_cselect_b32 s0, s7, 0x7c00
; GFX11-NEXT: s_cmpk_eq_i32 s3, 0x40f
; GFX11-NEXT: s_cselect_b32 s0, s0, s5
@@ -6916,28 +6926,31 @@ define amdgpu_ps i32 @s_copysign_f16_0_f64(double inreg %sign) {
; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_lshr_b32 s5, s3, s4
; GFX11-NEXT: s_lshl_b32 s4, s5, s4
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_4) | instid1(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_cmp_lg_u32 s4, s3
; GFX11-NEXT: s_cselect_b32 s3, 1, 0
; GFX11-NEXT: s_addk_i32 s2, 0xfc10
; GFX11-NEXT: s_or_b32 s3, s5, s3
; GFX11-NEXT: s_lshl_b32 s4, s2, 12
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_or_b32 s0, s0, s4
; GFX11-NEXT: s_cmp_lt_i32 s2, 1
; GFX11-NEXT: s_cselect_b32 s0, s3, s0
; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_b32 s3, s0, 7
; GFX11-NEXT: s_cmp_gt_i32 s3, 5
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_cselect_b32 s4, 1, 0
; GFX11-NEXT: s_cmp_eq_u32 s3, 3
; GFX11-NEXT: s_cselect_b32 s3, 1, 0
; GFX11-NEXT: s_lshr_b32 s0, s0, 2
; GFX11-NEXT: s_or_b32 s3, s3, s4
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_4) | instid1(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_add_i32 s0, s0, s3
; GFX11-NEXT: s_cmp_lt_i32 s2, 31
; GFX11-NEXT: s_cselect_b32 s3, -1, 0
; GFX11-NEXT: s_cmpk_lg_i32 s2, 0x40f
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_cselect_b32 s2, -1, 0
; GFX11-NEXT: s_and_b32 s2, s2, s3
; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
diff --git a/llvm/test/CodeGen/AMDGPU/fdiv.bf16.ll b/llvm/test/CodeGen/AMDGPU/fdiv.bf16.ll
index d3a526dfbe4259..bf26e58489875f 100644
--- a/llvm/test/CodeGen/AMDGPU/fdiv.bf16.ll
+++ b/llvm/test/CodeGen/AMDGPU/fdiv.bf16.ll
@@ -24,10 +24,10 @@ define bfloat @v_fdiv_bf16(bfloat %x, bfloat %y) {
; GFX1250-TRUE16-NEXT: v_fmac_f32_e32 v5, v6, v3
; GFX1250-TRUE16-NEXT: v_fma_f32 v2, -v2, v5, v4
; GFX1250-TRUE16-NEXT: s_denorm_mode 12
-; GFX1250-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX1250-TRUE16-NEXT: v_div_fmas_f32 v2, v2, v3, v5
+; GFX1250-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1250-TRUE16-NEXT: v_div_fixup_f32 v0, v2, v1, v0
-; GFX1250-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1250-TRUE16-NEXT: v_cvt_pk_bf16_f32 v0, v0, s0
; GFX1250-TRUE16-NEXT: s_set_pc_i64 s[30:31]
;
@@ -52,10 +52,10 @@ define bfloat @v_fdiv_bf16(bfloat %x, bfloat %y) {
; GFX1250-FAKE16-NEXT: v_fmac_f32_e32 v5, v6, v3
; GFX1250-FAKE16-NEXT: v_fma_f32 v2, -v2, v5, v4
; GFX1250-FAKE16-NEXT: s_denorm_mode 12
-; GFX1250-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX1250-FAKE16-NEXT: v_div_fmas_f32 v2, v2, v3, v5
+; GFX1250-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1250-FAKE16-NEXT: v_div_fixup_f32 v0, v2, v1, v0
-; GFX1250-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1250-FAKE16-NEXT: v_cvt_pk_bf16_f32 v0, v0, s0
; GFX1250-FAKE16-NEXT: s_set_pc_i64 s[30:31]
%fdiv = fdiv bfloat %x, %y
diff --git a/llvm/test/CodeGen/AMDGPU/flat-atomicrmw-fadd.ll b/llvm/test/CodeGen/AMDGPU/flat-atomicrmw-fadd.ll
index 760308608c7d80..74badd0d5b4c84 100644
--- a/llvm/test/CodeGen/AMDGPU/flat-atomicrmw-fadd.ll
+++ b/llvm/test/CodeGen/AMDGPU/flat-atomicrmw-fadd.ll
@@ -409,7 +409,6 @@ define float @flat_agent_atomic_fadd_ret_f32__offset12b_neg__amdgpu_no_fine_grai
; GFX11: ; %bb.0:
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: v_add_co_u32 v0, vcc_lo, 0xfffff800, v0
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, -1, v1, vcc_lo
; GFX11-NEXT: s_waitcnt_vscnt null, 0x0
; GFX11-NEXT: flat_atomic_add_f32 v0, v[0:1], v2 glc
@@ -1031,7 +1030,6 @@ define void @flat_agent_atomic_fadd_noret_f32__offset12b_neg__amdgpu_no_fine_gra
; GFX11: ; %bb.0:
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: v_add_co_u32 v0, vcc_lo, 0xfffff800, v0
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, -1, v1, vcc_lo
; GFX11-NEXT: s_waitcnt_vscnt null, 0x0
; GFX11-NEXT: flat_atomic_add_f32 v[0:1], v2
@@ -1678,7 +1676,7 @@ define void @flat_agent_atomic_fadd_noret_f32_maybe_remote(ptr %ptr, float %val)
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-NEXT: v_mov_b32_e32 v4, v3
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB8_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -2213,7 +2211,7 @@ define void @flat_agent_atomic_fadd_noret_f32_amdgpu_ignore_denormal_mode(ptr %p
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-NEXT: v_mov_b32_e32 v4, v3
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB11_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -2736,7 +2734,6 @@ define float @flat_agent_atomic_fadd_ret_f32__offset12b_neg__ftz__amdgpu_no_fine
; GFX11: ; %bb.0:
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: v_add_co_u32 v0, vcc_lo, 0xfffff800, v0
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, -1, v1, vcc_lo
; GFX11-NEXT: s_waitcnt_vscnt null, 0x0
; GFX11-NEXT: flat_atomic_add_f32 v0, v[0:1], v2 glc
@@ -3358,7 +3355,6 @@ define void @flat_agent_atomic_fadd_noret_f32__offset12b_neg__ftz__amdgpu_no_fin
; GFX11: ; %bb.0:
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: v_add_co_u32 v0, vcc_lo, 0xfffff800, v0
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, -1, v1, vcc_lo
; GFX11-NEXT: s_waitcnt_vscnt null, 0x0
; GFX11-NEXT: flat_atomic_add_f32 v[0:1], v2
@@ -4419,11 +4415,12 @@ define float @flat_agent_atomic_fadd_ret_f32__amdgpu_no_remote_memory__amdgpu_ig
; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB22_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: v_mov_b32_e32 v0, v3
; GFX11-NEXT: s_setpc_b64 s[30:31]
;
@@ -4583,7 +4580,7 @@ define void @flat_agent_atomic_fadd_noret_f32__amdgpu_no_remote_memory__amdgpu_i
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-NEXT: v_mov_b32_e32 v4, v3
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB23_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -4742,11 +4739,12 @@ define float @flat_agent_atomic_fadd_ret_f32__amdgpu_no_remote_memory(ptr %ptr,
; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB24_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: v_mov_b32_e32 v0, v3
; GFX11-NEXT: s_setpc_b64 s[30:31]
;
@@ -4906,7 +4904,7 @@ define void @flat_agent_atomic_fadd_noret_f32__amdgpu_no_remote_memory(ptr %ptr,
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-NEXT: v_mov_b32_e32 v4, v3
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB25_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -5718,6 +5716,7 @@ define double @flat_agent_atomic_fadd_ret_f64__amdgpu_no_fine_grained_memory(ptr
; GFX12-NEXT: ; implicit-def: $vgpr4_vgpr5
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: v_cmpx_ne_u32_e32 s1, v1
+; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX12-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX12-NEXT: s_cbranch_execz .LBB30_4
; GFX12-NEXT: ; %bb.1: ; %atomicrmw.global
@@ -5738,6 +5737,7 @@ define double @flat_agent_atomic_fadd_ret_f64__amdgpu_no_fine_grained_memory(ptr
; GFX12-NEXT: s_or_b32 s1, vcc_lo, s1
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB30_2
; GFX12-NEXT: ; %bb.3: ; %Flow
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s1
@@ -5758,6 +5758,7 @@ define double @flat_agent_atomic_fadd_ret_f64__amdgpu_no_fine_grained_memory(ptr
; GFX12-NEXT: .LBB30_6: ; %atomicrmw.phi
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: v_dual_mov_b32 v0, v4 :: v_dual_mov_b32 v1, v5
; GFX12-NEXT: s_setpc_b64 s[30:31]
;
@@ -5826,6 +5827,7 @@ define double @flat_agent_atomic_fadd_ret_f64__amdgpu_no_fine_grained_memory(ptr
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: ; implicit-def: $vgpr4_vgpr5
; GFX11-NEXT: v_cmpx_ne_u32_e32 s1, v1
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX11-NEXT: s_cbranch_execz .LBB30_4
; GFX11-NEXT: ; %bb.1: ; %atomicrmw.global
@@ -5844,7 +5846,7 @@ define double @flat_agent_atomic_fadd_ret_f64__amdgpu_no_fine_grained_memory(ptr
; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: v_cmp_eq_u64_e32 vcc_lo, v[4:5], v[6:7]
; GFX11-NEXT: s_or_b32 s1, vcc_lo, s1
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
; GFX11-NEXT: s_cbranch_execnz .LBB30_2
; GFX11-NEXT: ; %bb.3: ; %Flow
@@ -5852,6 +5854,7 @@ define double @flat_agent_atomic_fadd_ret_f64__amdgpu_no_fine_grained_memory(ptr
; GFX11-NEXT: ; implicit-def: $vgpr0_vgpr1
; GFX11-NEXT: ; implicit-def: $vgpr2_vgpr3
; GFX11-NEXT: .LBB30_4: ; %Flow3
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB30_6
; GFX11-NEXT: ; %bb.5: ; %atomicrmw.private
@@ -5863,6 +5866,7 @@ define double @flat_agent_atomic_fadd_ret_f64__amdgpu_no_fine_grained_memory(ptr
; GFX11-NEXT: scratch_store_b64 v6, v[0:1], off
; GFX11-NEXT: .LBB30_6: ; %atomicrmw.phi
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: v_dual_mov_b32 v0, v4 :: v_dual_mov_b32 v1, v5
; GFX11-NEXT: s_setpc_b64 s[30:31]
;
@@ -6154,6 +6158,7 @@ define double @flat_agent_atomic_fadd_ret_f64__offset12b_pos__amdgpu_no_fine_gra
; GFX12-NEXT: ; implicit-def: $vgpr0_vgpr1
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: v_cmpx_ne_u32_e32 s1, v5
+; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX12-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX12-NEXT: s_cbranch_execnz .LBB31_3
; GFX12-NEXT: ; %bb.1: ; %Flow3
@@ -6182,11 +6187,13 @@ define double @flat_agent_atomic_fadd_ret_f64__offset12b_pos__amdgpu_no_fine_gra
; GFX12-NEXT: s_or_b32 s1, vcc_lo, s1
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB31_4
; GFX12-NEXT: ; %bb.5: ; %Flow
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s1
; GFX12-NEXT: ; implicit-def: $vgpr4_vgpr5
; GFX12-NEXT: ; implicit-def: $vgpr2_vgpr3
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX12-NEXT: s_cbranch_execz .LBB31_2
; GFX12-NEXT: .LBB31_6: ; %atomicrmw.private
@@ -6263,12 +6270,12 @@ define double @flat_agent_atomic_fadd_ret_f64__offset12b_pos__amdgpu_no_fine_gra
; GFX11: ; %bb.0:
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: v_add_co_u32 v4, vcc_lo, 0x7f8, v0
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v5, null, 0, v1, vcc_lo
; GFX11-NEXT: s_mov_b64 s[0:1], src_private_base
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: ; implicit-def: $vgpr0_vgpr1
; GFX11-NEXT: v_cmpx_ne_u32_e32 s1, v5
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB31_3
; GFX11-NEXT: ; %bb.1: ; %Flow3
@@ -6293,13 +6300,14 @@ define double @flat_agent_atomic_fadd_ret_f64__offset12b_pos__amdgpu_no_fine_gra
; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: v_cmp_eq_u64_e32 vcc_lo, v[0:1], v[8:9]
; GFX11-NEXT: s_or_b32 s1, vcc_lo, s1
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
; GFX11-NEXT: s_cbranch_execnz .LBB31_4
; GFX11-NEXT: ; %bb.5: ; %Flow
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s1
; GFX11-NEXT: ; implicit-def: $vgpr4_vgpr5
; GFX11-NEXT: ; implicit-def: $vgpr2_vgpr3
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB31_2
; GFX11-NEXT: .LBB31_6: ; %atomicrmw.private
@@ -6613,6 +6621,7 @@ define double @flat_agent_atomic_fadd_ret_f64__offset12b_neg__amdgpu_no_fine_gra
; GFX12-NEXT: ; implicit-def: $vgpr0_vgpr1
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: v_cmpx_ne_u32_e32 s1, v5
+; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX12-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX12-NEXT: s_cbranch_execnz .LBB32_3
; GFX12-NEXT: ; %bb.1: ; %Flow3
@@ -6641,11 +6650,13 @@ define double @flat_agent_atomic_fadd_ret_f64__offset12b_neg__amdgpu_no_fine_gra
; GFX12-NEXT: s_or_b32 s1, vcc_lo, s1
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB32_4
; GFX12-NEXT: ; %bb.5: ; %Flow
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s1
; GFX12-NEXT: ; implicit-def: $vgpr4_vgpr5
; GFX12-NEXT: ; implicit-def: $vgpr2_vgpr3
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX12-NEXT: s_cbranch_execz .LBB32_2
; GFX12-NEXT: .LBB32_6: ; %atomicrmw.private
@@ -6723,12 +6734,12 @@ define double @flat_agent_atomic_fadd_ret_f64__offset12b_neg__amdgpu_no_fine_gra
; GFX11: ; %bb.0:
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: v_add_co_u32 v4, vcc_lo, 0xfffff800, v0
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v5, null, -1, v1, vcc_lo
; GFX11-NEXT: s_mov_b64 s[0:1], src_private_base
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: ; implicit-def: $vgpr0_vgpr1
; GFX11-NEXT: v_cmpx_ne_u32_e32 s1, v5
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB32_3
; GFX11-NEXT: ; %bb.1: ; %Flow3
@@ -6753,13 +6764,14 @@ define double @flat_agent_atomic_fadd_ret_f64__offset12b_neg__amdgpu_no_fine_gra
; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: v_cmp_eq_u64_e32 vcc_lo, v[0:1], v[8:9]
; GFX11-NEXT: s_or_b32 s1, vcc_lo, s1
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
; GFX11-NEXT: s_cbranch_execnz .LBB32_4
; GFX11-NEXT: ; %bb.5: ; %Flow
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s1
; GFX11-NEXT: ; implicit-def: $vgpr4_vgpr5
; GFX11-NEXT: ; implicit-def: $vgpr2_vgpr3
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB32_2
; GFX11-NEXT: .LBB32_6: ; %atomicrmw.private
@@ -7069,6 +7081,7 @@ define void @flat_agent_atomic_fadd_noret_f64__amdgpu_no_fine_grained_memory(ptr
; GFX12-NEXT: s_mov_b32 s0, exec_lo
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: v_cmpx_ne_u32_e32 s1, v1
+; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX12-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX12-NEXT: s_cbranch_execnz .LBB33_3
; GFX12-NEXT: ; %bb.1: ; %Flow3
@@ -7096,11 +7109,13 @@ define void @flat_agent_atomic_fadd_noret_f64__amdgpu_no_fine_grained_memory(ptr
; GFX12-NEXT: s_or_b32 s1, vcc_lo, s1
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB33_4
; GFX12-NEXT: ; %bb.5: ; %Flow
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s1
; GFX12-NEXT: ; implicit-def: $vgpr0_vgpr1
; GFX12-NEXT: ; implicit-def: $vgpr2_vgpr3
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX12-NEXT: s_cbranch_execz .LBB33_2
; GFX12-NEXT: .LBB33_6: ; %atomicrmw.private
@@ -7175,6 +7190,7 @@ define void @flat_agent_atomic_fadd_noret_f64__amdgpu_no_fine_grained_memory(ptr
; GFX11-NEXT: s_mov_b64 s[0:1], src_private_base
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 s1, v1
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB33_3
; GFX11-NEXT: ; %bb.1: ; %Flow3
@@ -7198,13 +7214,14 @@ define void @flat_agent_atomic_fadd_noret_f64__amdgpu_no_fine_grained_memory(ptr
; GFX11-NEXT: v_cmp_eq_u64_e32 vcc_lo, v[4:5], v[6:7]
; GFX11-NEXT: v_dual_mov_b32 v7, v5 :: v_dual_mov_b32 v6, v4
; GFX11-NEXT: s_or_b32 s1, vcc_lo, s1
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
; GFX11-NEXT: s_cbranch_execnz .LBB33_4
; GFX11-NEXT: ; %bb.5: ; %Flow
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s1
; GFX11-NEXT: ; implicit-def: $vgpr0_vgpr1
; GFX11-NEXT: ; implicit-def: $vgpr2_vgpr3
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB33_2
; GFX11-NEXT: .LBB33_6: ; %atomicrmw.private
@@ -7500,6 +7517,7 @@ define void @flat_agent_atomic_fadd_noret_f64__offset12b_pos__amdgpu_no_fine_gra
; GFX12-NEXT: s_mov_b32 s0, exec_lo
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: v_cmpx_ne_u32_e32 s1, v1
+; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX12-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX12-NEXT: s_cbranch_execnz .LBB34_3
; GFX12-NEXT: ; %bb.1: ; %Flow3
@@ -7527,11 +7545,13 @@ define void @flat_agent_atomic_fadd_noret_f64__offset12b_pos__amdgpu_no_fine_gra
; GFX12-NEXT: s_or_b32 s1, vcc_lo, s1
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB34_4
; GFX12-NEXT: ; %bb.5: ; %Flow
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s1
; GFX12-NEXT: ; implicit-def: $vgpr0_vgpr1
; GFX12-NEXT: ; implicit-def: $vgpr2_vgpr3
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX12-NEXT: s_cbranch_execz .LBB34_2
; GFX12-NEXT: .LBB34_6: ; %atomicrmw.private
@@ -7606,11 +7626,11 @@ define void @flat_agent_atomic_fadd_noret_f64__offset12b_pos__amdgpu_no_fine_gra
; GFX11: ; %bb.0:
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: v_add_co_u32 v0, vcc_lo, 0x7f8, v0
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-NEXT: s_mov_b64 s[0:1], src_private_base
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 s1, v1
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB34_3
; GFX11-NEXT: ; %bb.1: ; %Flow3
@@ -7634,13 +7654,14 @@ define void @flat_agent_atomic_fadd_noret_f64__offset12b_pos__amdgpu_no_fine_gra
; GFX11-NEXT: v_cmp_eq_u64_e32 vcc_lo, v[4:5], v[6:7]
; GFX11-NEXT: v_dual_mov_b32 v7, v5 :: v_dual_mov_b32 v6, v4
; GFX11-NEXT: s_or_b32 s1, vcc_lo, s1
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
; GFX11-NEXT: s_cbranch_execnz .LBB34_4
; GFX11-NEXT: ; %bb.5: ; %Flow
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s1
; GFX11-NEXT: ; implicit-def: $vgpr0_vgpr1
; GFX11-NEXT: ; implicit-def: $vgpr2_vgpr3
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB34_2
; GFX11-NEXT: .LBB34_6: ; %atomicrmw.private
@@ -7947,6 +7968,7 @@ define void @flat_agent_atomic_fadd_noret_f64__offset12b_neg__amdgpu_no_fine_gra
; GFX12-NEXT: s_mov_b32 s0, exec_lo
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: v_cmpx_ne_u32_e32 s1, v1
+; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX12-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX12-NEXT: s_cbranch_execnz .LBB35_3
; GFX12-NEXT: ; %bb.1: ; %Flow3
@@ -7974,11 +7996,13 @@ define void @flat_agent_atomic_fadd_noret_f64__offset12b_neg__amdgpu_no_fine_gra
; GFX12-NEXT: s_or_b32 s1, vcc_lo, s1
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB35_4
; GFX12-NEXT: ; %bb.5: ; %Flow
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s1
; GFX12-NEXT: ; implicit-def: $vgpr0_vgpr1
; GFX12-NEXT: ; implicit-def: $vgpr2_vgpr3
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX12-NEXT: s_cbranch_execz .LBB35_2
; GFX12-NEXT: .LBB35_6: ; %atomicrmw.private
@@ -8054,11 +8078,11 @@ define void @flat_agent_atomic_fadd_noret_f64__offset12b_neg__amdgpu_no_fine_gra
; GFX11: ; %bb.0:
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: v_add_co_u32 v0, vcc_lo, 0xfffff800, v0
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, -1, v1, vcc_lo
; GFX11-NEXT: s_mov_b64 s[0:1], src_private_base
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 s1, v1
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB35_3
; GFX11-NEXT: ; %bb.1: ; %Flow3
@@ -8082,13 +8106,14 @@ define void @flat_agent_atomic_fadd_noret_f64__offset12b_neg__amdgpu_no_fine_gra
; GFX11-NEXT: v_cmp_eq_u64_e32 vcc_lo, v[4:5], v[6:7]
; GFX11-NEXT: v_dual_mov_b32 v7, v5 :: v_dual_mov_b32 v6, v4
; GFX11-NEXT: s_or_b32 s1, vcc_lo, s1
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
; GFX11-NEXT: s_cbranch_execnz .LBB35_4
; GFX11-NEXT: ; %bb.5: ; %Flow
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s1
; GFX11-NEXT: ; implicit-def: $vgpr0_vgpr1
; GFX11-NEXT: ; implicit-def: $vgpr2_vgpr3
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB35_2
; GFX11-NEXT: .LBB35_6: ; %atomicrmw.private
@@ -8423,9 +8448,11 @@ define half @flat_agent_atomic_fadd_ret_f16__amdgpu_no_fine_grained_memory(ptr %
; GFX12-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_cbranch_execnz .LBB36_1
; GFX12-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: v_lshrrev_b32_e32 v0, v3, v5
; GFX12-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -8467,9 +8494,11 @@ define half @flat_agent_atomic_fadd_ret_f16__amdgpu_no_fine_grained_memory(ptr %
; GFX12-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: s_cbranch_execnz .LBB36_1
; GFX12-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: v_lshrrev_b32_e32 v0, v3, v5
; GFX12-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -8539,11 +8568,12 @@ define half @flat_agent_atomic_fadd_ret_f16__amdgpu_no_fine_grained_memory(ptr %
; GFX11-TRUE16-NEXT: buffer_gl0_inv
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v5, v6
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB36_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v0, v3, v5
; GFX11-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -8579,11 +8609,12 @@ define half @flat_agent_atomic_fadd_ret_f16__amdgpu_no_fine_grained_memory(ptr %
; GFX11-FAKE16-NEXT: buffer_gl0_inv
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v5, v6
; GFX11-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB36_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v0, v3, v5
; GFX11-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -8795,9 +8826,11 @@ define half @flat_agent_atomic_fadd_ret_f16__offset12b_pos__amdgpu_no_fine_grain
; GFX12-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_cbranch_execnz .LBB37_1
; GFX12-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: v_lshrrev_b32_e32 v0, v3, v5
; GFX12-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -8840,9 +8873,11 @@ define half @flat_agent_atomic_fadd_ret_f16__offset12b_pos__amdgpu_no_fine_grain
; GFX12-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: s_cbranch_execnz .LBB37_1
; GFX12-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: v_lshrrev_b32_e32 v0, v3, v5
; GFX12-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -8886,7 +8921,6 @@ define half @flat_agent_atomic_fadd_ret_f16__offset12b_pos__amdgpu_no_fine_grain
; GFX11-TRUE16: ; %bb.0:
; GFX11-TRUE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-TRUE16-NEXT: v_add_co_u32 v3, vcc_lo, 0x7fe, v0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-TRUE16-NEXT: s_mov_b32 s0, 0
; GFX11-TRUE16-NEXT: v_and_b32_e32 v0, -4, v3
@@ -8915,11 +8949,12 @@ define half @flat_agent_atomic_fadd_ret_f16__offset12b_pos__amdgpu_no_fine_grain
; GFX11-TRUE16-NEXT: buffer_gl0_inv
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v5, v6
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB37_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v0, v3, v5
; GFX11-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -8927,7 +8962,6 @@ define half @flat_agent_atomic_fadd_ret_f16__offset12b_pos__amdgpu_no_fine_grain
; GFX11-FAKE16: ; %bb.0:
; GFX11-FAKE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-FAKE16-NEXT: v_add_co_u32 v3, vcc_lo, 0x7fe, v0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-FAKE16-NEXT: s_mov_b32 s0, 0
; GFX11-FAKE16-NEXT: v_and_b32_e32 v0, -4, v3
@@ -8956,11 +8990,12 @@ define half @flat_agent_atomic_fadd_ret_f16__offset12b_pos__amdgpu_no_fine_grain
; GFX11-FAKE16-NEXT: buffer_gl0_inv
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v5, v6
; GFX11-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB37_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v0, v3, v5
; GFX11-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -9178,9 +9213,11 @@ define half @flat_agent_atomic_fadd_ret_f16__offset12b_neg__amdgpu_no_fine_grain
; GFX12-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_cbranch_execnz .LBB38_1
; GFX12-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: v_lshrrev_b32_e32 v0, v3, v5
; GFX12-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -9223,9 +9260,11 @@ define half @flat_agent_atomic_fadd_ret_f16__offset12b_neg__amdgpu_no_fine_grain
; GFX12-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: s_cbranch_execnz .LBB38_1
; GFX12-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: v_lshrrev_b32_e32 v0, v3, v5
; GFX12-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -9270,7 +9309,6 @@ define half @flat_agent_atomic_fadd_ret_f16__offset12b_neg__amdgpu_no_fine_grain
; GFX11-TRUE16: ; %bb.0:
; GFX11-TRUE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-TRUE16-NEXT: v_add_co_u32 v3, vcc_lo, 0xfffff800, v0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v1, null, -1, v1, vcc_lo
; GFX11-TRUE16-NEXT: s_mov_b32 s0, 0
; GFX11-TRUE16-NEXT: v_and_b32_e32 v0, -4, v3
@@ -9299,11 +9337,12 @@ define half @flat_agent_atomic_fadd_ret_f16__offset12b_neg__amdgpu_no_fine_grain
; GFX11-TRUE16-NEXT: buffer_gl0_inv
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v5, v6
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB38_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v0, v3, v5
; GFX11-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -9311,7 +9350,6 @@ define half @flat_agent_atomic_fadd_ret_f16__offset12b_neg__amdgpu_no_fine_grain
; GFX11-FAKE16: ; %bb.0:
; GFX11-FAKE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-FAKE16-NEXT: v_add_co_u32 v3, vcc_lo, 0xfffff800, v0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v1, null, -1, v1, vcc_lo
; GFX11-FAKE16-NEXT: s_mov_b32 s0, 0
; GFX11-FAKE16-NEXT: v_and_b32_e32 v0, -4, v3
@@ -9340,11 +9378,12 @@ define half @flat_agent_atomic_fadd_ret_f16__offset12b_neg__amdgpu_no_fine_grain
; GFX11-FAKE16-NEXT: buffer_gl0_inv
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v5, v6
; GFX11-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB38_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v0, v3, v5
; GFX11-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -9560,6 +9599,7 @@ define void @flat_agent_atomic_fadd_noret_f16__amdgpu_no_fine_grained_memory(ptr
; GFX12-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_cbranch_execnz .LBB39_1
; GFX12-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -9602,6 +9642,7 @@ define void @flat_agent_atomic_fadd_noret_f16__amdgpu_no_fine_grained_memory(ptr
; GFX12-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: s_cbranch_execnz .LBB39_1
; GFX12-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -9671,7 +9712,7 @@ define void @flat_agent_atomic_fadd_noret_f16__amdgpu_no_fine_grained_memory(ptr
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-TRUE16-NEXT: v_mov_b32_e32 v4, v3
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB39_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -9709,7 +9750,7 @@ define void @flat_agent_atomic_fadd_noret_f16__amdgpu_no_fine_grained_memory(ptr
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v4, v3
; GFX11-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB39_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -9918,6 +9959,7 @@ define void @flat_agent_atomic_fadd_noret_f16__offset12b_pos__amdgpu_no_fine_gra
; GFX12-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_cbranch_execnz .LBB40_1
; GFX12-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -9961,6 +10003,7 @@ define void @flat_agent_atomic_fadd_noret_f16__offset12b_pos__amdgpu_no_fine_gra
; GFX12-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: s_cbranch_execnz .LBB40_1
; GFX12-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -10005,7 +10048,6 @@ define void @flat_agent_atomic_fadd_noret_f16__offset12b_pos__amdgpu_no_fine_gra
; GFX11-TRUE16: ; %bb.0:
; GFX11-TRUE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-TRUE16-NEXT: v_add_co_u32 v3, vcc_lo, 0x7fe, v0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-TRUE16-NEXT: s_mov_b32 s0, 0
; GFX11-TRUE16-NEXT: v_and_b32_e32 v0, -4, v3
@@ -10033,7 +10075,7 @@ define void @flat_agent_atomic_fadd_noret_f16__offset12b_pos__amdgpu_no_fine_gra
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-TRUE16-NEXT: v_mov_b32_e32 v4, v3
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB40_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -10044,7 +10086,6 @@ define void @flat_agent_atomic_fadd_noret_f16__offset12b_pos__amdgpu_no_fine_gra
; GFX11-FAKE16: ; %bb.0:
; GFX11-FAKE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-FAKE16-NEXT: v_add_co_u32 v3, vcc_lo, 0x7fe, v0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-FAKE16-NEXT: s_mov_b32 s0, 0
; GFX11-FAKE16-NEXT: v_and_b32_e32 v0, -4, v3
@@ -10072,7 +10113,7 @@ define void @flat_agent_atomic_fadd_noret_f16__offset12b_pos__amdgpu_no_fine_gra
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v4, v3
; GFX11-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB40_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -10287,6 +10328,7 @@ define void @flat_agent_atomic_fadd_noret_f16__offset12b_neg__amdgpu_no_fine_gra
; GFX12-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_cbranch_execnz .LBB41_1
; GFX12-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -10330,6 +10372,7 @@ define void @flat_agent_atomic_fadd_noret_f16__offset12b_neg__amdgpu_no_fine_gra
; GFX12-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: s_cbranch_execnz .LBB41_1
; GFX12-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -10375,7 +10418,6 @@ define void @flat_agent_atomic_fadd_noret_f16__offset12b_neg__amdgpu_no_fine_gra
; GFX11-TRUE16: ; %bb.0:
; GFX11-TRUE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-TRUE16-NEXT: v_add_co_u32 v3, vcc_lo, 0xfffff800, v0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v1, null, -1, v1, vcc_lo
; GFX11-TRUE16-NEXT: s_mov_b32 s0, 0
; GFX11-TRUE16-NEXT: v_and_b32_e32 v0, -4, v3
@@ -10403,7 +10445,7 @@ define void @flat_agent_atomic_fadd_noret_f16__offset12b_neg__amdgpu_no_fine_gra
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-TRUE16-NEXT: v_mov_b32_e32 v4, v3
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB41_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -10414,7 +10456,6 @@ define void @flat_agent_atomic_fadd_noret_f16__offset12b_neg__amdgpu_no_fine_gra
; GFX11-FAKE16: ; %bb.0:
; GFX11-FAKE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-FAKE16-NEXT: v_add_co_u32 v3, vcc_lo, 0xfffff800, v0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v1, null, -1, v1, vcc_lo
; GFX11-FAKE16-NEXT: s_mov_b32 s0, 0
; GFX11-FAKE16-NEXT: v_and_b32_e32 v0, -4, v3
@@ -10442,7 +10483,7 @@ define void @flat_agent_atomic_fadd_noret_f16__offset12b_neg__amdgpu_no_fine_gra
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v4, v3
; GFX11-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB41_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -10645,6 +10686,7 @@ define void @flat_agent_atomic_fadd_noret_f16__offset12b__align4_pos__amdgpu_no_
; GFX12-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_cbranch_execnz .LBB42_1
; GFX12-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -10676,6 +10718,7 @@ define void @flat_agent_atomic_fadd_noret_f16__offset12b__align4_pos__amdgpu_no_
; GFX12-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: s_cbranch_execnz .LBB42_1
; GFX12-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -10726,7 +10769,7 @@ define void @flat_agent_atomic_fadd_noret_f16__offset12b__align4_pos__amdgpu_no_
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-TRUE16-NEXT: v_mov_b32_e32 v4, v3
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB42_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -10753,7 +10796,7 @@ define void @flat_agent_atomic_fadd_noret_f16__offset12b__align4_pos__amdgpu_no_
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v4, v3
; GFX11-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB42_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -10918,9 +10961,11 @@ define half @flat_agent_atomic_fadd_ret_f16__offset12b_pos__align4__amdgpu_no_fi
; GFX12-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_cbranch_execnz .LBB43_1
; GFX12-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: v_mov_b16_e32 v0.l, v3.l
; GFX12-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -10951,9 +10996,11 @@ define half @flat_agent_atomic_fadd_ret_f16__offset12b_pos__align4__amdgpu_no_fi
; GFX12-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: s_cbranch_execnz .LBB43_1
; GFX12-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: v_mov_b32_e32 v0, v3
; GFX12-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -11004,11 +11051,12 @@ define half @flat_agent_atomic_fadd_ret_f16__offset12b_pos__align4__amdgpu_no_fi
; GFX11-TRUE16-NEXT: buffer_gl0_inv
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB43_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v0.l, v3.l
; GFX11-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -11033,11 +11081,12 @@ define half @flat_agent_atomic_fadd_ret_f16__offset12b_pos__align4__amdgpu_no_fi
; GFX11-FAKE16-NEXT: buffer_gl0_inv
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB43_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v0, v3
; GFX11-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -11214,9 +11263,11 @@ define half @flat_system_atomic_fadd_ret_f16__offset12b_pos__amdgpu_no_fine_grai
; GFX12-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_cbranch_execnz .LBB44_1
; GFX12-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: v_lshrrev_b32_e32 v0, v3, v5
; GFX12-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -11260,9 +11311,11 @@ define half @flat_system_atomic_fadd_ret_f16__offset12b_pos__amdgpu_no_fine_grai
; GFX12-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: s_cbranch_execnz .LBB44_1
; GFX12-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: v_lshrrev_b32_e32 v0, v3, v5
; GFX12-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -11306,7 +11359,6 @@ define half @flat_system_atomic_fadd_ret_f16__offset12b_pos__amdgpu_no_fine_grai
; GFX11-TRUE16: ; %bb.0:
; GFX11-TRUE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-TRUE16-NEXT: v_add_co_u32 v3, vcc_lo, 0x7fe, v0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-TRUE16-NEXT: s_mov_b32 s0, 0
; GFX11-TRUE16-NEXT: v_and_b32_e32 v0, -4, v3
@@ -11335,11 +11387,12 @@ define half @flat_system_atomic_fadd_ret_f16__offset12b_pos__amdgpu_no_fine_grai
; GFX11-TRUE16-NEXT: buffer_gl0_inv
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v5, v6
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB44_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v0, v3, v5
; GFX11-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -11347,7 +11400,6 @@ define half @flat_system_atomic_fadd_ret_f16__offset12b_pos__amdgpu_no_fine_grai
; GFX11-FAKE16: ; %bb.0:
; GFX11-FAKE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-FAKE16-NEXT: v_add_co_u32 v3, vcc_lo, 0x7fe, v0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-FAKE16-NEXT: s_mov_b32 s0, 0
; GFX11-FAKE16-NEXT: v_and_b32_e32 v0, -4, v3
@@ -11376,11 +11428,12 @@ define half @flat_system_atomic_fadd_ret_f16__offset12b_pos__amdgpu_no_fine_grai
; GFX11-FAKE16-NEXT: buffer_gl0_inv
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v5, v6
; GFX11-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB44_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v0, v3, v5
; GFX11-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -11601,6 +11654,7 @@ define void @flat_system_atomic_fadd_noret_f16__offset12b_pos__amdgpu_no_fine_gr
; GFX12-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_cbranch_execnz .LBB45_1
; GFX12-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -11645,6 +11699,7 @@ define void @flat_system_atomic_fadd_noret_f16__offset12b_pos__amdgpu_no_fine_gr
; GFX12-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: s_cbranch_execnz .LBB45_1
; GFX12-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -11689,7 +11744,6 @@ define void @flat_system_atomic_fadd_noret_f16__offset12b_pos__amdgpu_no_fine_gr
; GFX11-TRUE16: ; %bb.0:
; GFX11-TRUE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-TRUE16-NEXT: v_add_co_u32 v3, vcc_lo, 0x7fe, v0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-TRUE16-NEXT: s_mov_b32 s0, 0
; GFX11-TRUE16-NEXT: v_and_b32_e32 v0, -4, v3
@@ -11717,7 +11771,7 @@ define void @flat_system_atomic_fadd_noret_f16__offset12b_pos__amdgpu_no_fine_gr
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-TRUE16-NEXT: v_mov_b32_e32 v4, v3
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB45_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -11728,7 +11782,6 @@ define void @flat_system_atomic_fadd_noret_f16__offset12b_pos__amdgpu_no_fine_gr
; GFX11-FAKE16: ; %bb.0:
; GFX11-FAKE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-FAKE16-NEXT: v_add_co_u32 v3, vcc_lo, 0x7fe, v0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-FAKE16-NEXT: s_mov_b32 s0, 0
; GFX11-FAKE16-NEXT: v_and_b32_e32 v0, -4, v3
@@ -11756,7 +11809,7 @@ define void @flat_system_atomic_fadd_noret_f16__offset12b_pos__amdgpu_no_fine_gr
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v4, v3
; GFX11-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB45_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -11987,9 +12040,11 @@ define bfloat @flat_agent_atomic_fadd_ret_bf16__amdgpu_no_fine_grained_memory(pt
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB46_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: v_lshrrev_b32_e32 v0, v3, v5
; GFX12-NEXT: s_setpc_b64 s[30:31]
;
@@ -12077,11 +12132,12 @@ define bfloat @flat_agent_atomic_fadd_ret_bf16__amdgpu_no_fine_grained_memory(pt
; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v5, v6
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB46_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: v_lshrrev_b32_e32 v0, v3, v5
; GFX11-NEXT: s_setpc_b64 s[30:31]
;
@@ -12333,9 +12389,11 @@ define bfloat @flat_agent_atomic_fadd_ret_bf16__offset12b_pos__amdgpu_no_fine_gr
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB47_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: v_lshrrev_b32_e32 v0, v3, v5
; GFX12-NEXT: s_setpc_b64 s[30:31]
;
@@ -12388,16 +12446,16 @@ define bfloat @flat_agent_atomic_fadd_ret_bf16__offset12b_pos__amdgpu_no_fine_gr
; GFX11: ; %bb.0:
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: v_add_co_u32 v3, vcc_lo, 0x7fe, v0
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_3)
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-NEXT: v_lshlrev_b32_e32 v2, 16, v2
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_4) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_and_b32_e32 v0, -4, v3
; GFX11-NEXT: v_and_b32_e32 v3, 3, v3
; GFX11-NEXT: s_mov_b32 s0, 0
; GFX11-NEXT: flat_load_b32 v5, v[0:1]
; GFX11-NEXT: v_lshlrev_b32_e32 v3, 3, v3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_lshlrev_b32_e64 v4, v3, 0xffff
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_not_b32_e32 v4, v4
; GFX11-NEXT: .p2align 6
; GFX11-NEXT: .LBB47_1: ; %atomicrmw.start
@@ -12427,11 +12485,12 @@ define bfloat @flat_agent_atomic_fadd_ret_bf16__offset12b_pos__amdgpu_no_fine_gr
; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v5, v6
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB47_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: v_lshrrev_b32_e32 v0, v3, v5
; GFX11-NEXT: s_setpc_b64 s[30:31]
;
@@ -12689,9 +12748,11 @@ define bfloat @flat_agent_atomic_fadd_ret_bf16__offset12b_neg__amdgpu_no_fine_gr
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB48_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: v_lshrrev_b32_e32 v0, v3, v5
; GFX12-NEXT: s_setpc_b64 s[30:31]
;
@@ -12745,16 +12806,16 @@ define bfloat @flat_agent_atomic_fadd_ret_bf16__offset12b_neg__amdgpu_no_fine_gr
; GFX11: ; %bb.0:
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: v_add_co_u32 v3, vcc_lo, 0xfffff800, v0
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_3)
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, -1, v1, vcc_lo
; GFX11-NEXT: v_lshlrev_b32_e32 v2, 16, v2
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_4) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_and_b32_e32 v0, -4, v3
; GFX11-NEXT: v_and_b32_e32 v3, 3, v3
; GFX11-NEXT: s_mov_b32 s0, 0
; GFX11-NEXT: flat_load_b32 v5, v[0:1]
; GFX11-NEXT: v_lshlrev_b32_e32 v3, 3, v3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_lshlrev_b32_e64 v4, v3, 0xffff
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_not_b32_e32 v4, v4
; GFX11-NEXT: .p2align 6
; GFX11-NEXT: .LBB48_1: ; %atomicrmw.start
@@ -12784,11 +12845,12 @@ define bfloat @flat_agent_atomic_fadd_ret_bf16__offset12b_neg__amdgpu_no_fine_gr
; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v5, v6
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB48_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: v_lshrrev_b32_e32 v0, v3, v5
; GFX11-NEXT: s_setpc_b64 s[30:31]
;
@@ -13045,6 +13107,7 @@ define void @flat_agent_atomic_fadd_noret_bf16__offset12b_pos__amdgpu_no_fine_gr
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB49_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -13098,16 +13161,16 @@ define void @flat_agent_atomic_fadd_noret_bf16__offset12b_pos__amdgpu_no_fine_gr
; GFX11: ; %bb.0:
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: v_add_co_u32 v4, vcc_lo, 0x7fe, v0
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_3)
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-NEXT: v_lshlrev_b32_e32 v6, 16, v2
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_4) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_and_b32_e32 v0, -4, v4
; GFX11-NEXT: v_and_b32_e32 v4, 3, v4
; GFX11-NEXT: s_mov_b32 s0, 0
; GFX11-NEXT: flat_load_b32 v3, v[0:1]
; GFX11-NEXT: v_lshlrev_b32_e32 v4, 3, v4
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_lshlrev_b32_e64 v5, v4, 0xffff
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_not_b32_e32 v5, v5
; GFX11-NEXT: .p2align 6
; GFX11-NEXT: .LBB49_1: ; %atomicrmw.start
@@ -13136,7 +13199,7 @@ define void @flat_agent_atomic_fadd_noret_bf16__offset12b_pos__amdgpu_no_fine_gr
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v3
; GFX11-NEXT: v_mov_b32_e32 v3, v2
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB49_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -13391,6 +13454,7 @@ define void @flat_agent_atomic_fadd_noret_bf16__offset12b_neg__amdgpu_no_fine_gr
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB50_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -13445,16 +13509,16 @@ define void @flat_agent_atomic_fadd_noret_bf16__offset12b_neg__amdgpu_no_fine_gr
; GFX11: ; %bb.0:
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: v_add_co_u32 v4, vcc_lo, 0xfffff800, v0
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_3)
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, -1, v1, vcc_lo
; GFX11-NEXT: v_lshlrev_b32_e32 v6, 16, v2
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_4) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_and_b32_e32 v0, -4, v4
; GFX11-NEXT: v_and_b32_e32 v4, 3, v4
; GFX11-NEXT: s_mov_b32 s0, 0
; GFX11-NEXT: flat_load_b32 v3, v[0:1]
; GFX11-NEXT: v_lshlrev_b32_e32 v4, 3, v4
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_lshlrev_b32_e64 v5, v4, 0xffff
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_not_b32_e32 v5, v5
; GFX11-NEXT: .p2align 6
; GFX11-NEXT: .LBB50_1: ; %atomicrmw.start
@@ -13483,7 +13547,7 @@ define void @flat_agent_atomic_fadd_noret_bf16__offset12b_neg__amdgpu_no_fine_gr
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v3
; GFX11-NEXT: v_mov_b32_e32 v3, v2
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB50_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -13726,9 +13790,11 @@ define bfloat @flat_agent_atomic_fadd_ret_bf16__offset12b_pos__align4__amdgpu_no
; GFX12-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_cbranch_execnz .LBB51_1
; GFX12-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: v_mov_b16_e32 v0.l, v3.l
; GFX12-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -13769,9 +13835,11 @@ define bfloat @flat_agent_atomic_fadd_ret_bf16__offset12b_pos__align4__amdgpu_no
; GFX12-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: s_cbranch_execnz .LBB51_1
; GFX12-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: v_mov_b32_e32 v0, v3
; GFX12-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -13842,11 +13910,12 @@ define bfloat @flat_agent_atomic_fadd_ret_bf16__offset12b_pos__align4__amdgpu_no
; GFX11-TRUE16-NEXT: buffer_gl0_inv
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB51_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v0.l, v3.l
; GFX11-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -13881,11 +13950,12 @@ define bfloat @flat_agent_atomic_fadd_ret_bf16__offset12b_pos__align4__amdgpu_no
; GFX11-FAKE16-NEXT: buffer_gl0_inv
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB51_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v0, v3
; GFX11-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -14093,6 +14163,7 @@ define void @flat_agent_atomic_fadd_noret_bf16__offset12b__align4_pos__amdgpu_no
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB52_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -14163,7 +14234,7 @@ define void @flat_agent_atomic_fadd_noret_bf16__offset12b__align4_pos__amdgpu_no
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v3
; GFX11-NEXT: v_mov_b32_e32 v3, v2
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB52_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -14382,6 +14453,7 @@ define void @flat_agent_atomic_fadd_noret_bf16__amdgpu_no_fine_grained_memory(pt
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB53_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -14469,7 +14541,7 @@ define void @flat_agent_atomic_fadd_noret_bf16__amdgpu_no_fine_grained_memory(pt
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-NEXT: v_mov_b32_e32 v4, v3
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB53_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -14720,9 +14792,11 @@ define bfloat @flat_system_atomic_fadd_ret_bf16__offset12b_pos__amdgpu_no_fine_g
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB54_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: v_lshrrev_b32_e32 v0, v3, v5
; GFX12-NEXT: s_setpc_b64 s[30:31]
;
@@ -14775,16 +14849,16 @@ define bfloat @flat_system_atomic_fadd_ret_bf16__offset12b_pos__amdgpu_no_fine_g
; GFX11: ; %bb.0:
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: v_add_co_u32 v3, vcc_lo, 0x7fe, v0
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_3)
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-NEXT: v_lshlrev_b32_e32 v2, 16, v2
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_4) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_and_b32_e32 v0, -4, v3
; GFX11-NEXT: v_and_b32_e32 v3, 3, v3
; GFX11-NEXT: s_mov_b32 s0, 0
; GFX11-NEXT: flat_load_b32 v5, v[0:1]
; GFX11-NEXT: v_lshlrev_b32_e32 v3, 3, v3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_lshlrev_b32_e64 v4, v3, 0xffff
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_not_b32_e32 v4, v4
; GFX11-NEXT: .p2align 6
; GFX11-NEXT: .LBB54_1: ; %atomicrmw.start
@@ -14814,11 +14888,12 @@ define bfloat @flat_system_atomic_fadd_ret_bf16__offset12b_pos__amdgpu_no_fine_g
; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v5, v6
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB54_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: v_lshrrev_b32_e32 v0, v3, v5
; GFX11-NEXT: s_setpc_b64 s[30:31]
;
@@ -15079,6 +15154,7 @@ define void @flat_system_atomic_fadd_noret_bf16__offset12b_pos__amdgpu_no_fine_g
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB55_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -15132,16 +15208,16 @@ define void @flat_system_atomic_fadd_noret_bf16__offset12b_pos__amdgpu_no_fine_g
; GFX11: ; %bb.0:
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: v_add_co_u32 v4, vcc_lo, 0x7fe, v0
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_3)
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-NEXT: v_lshlrev_b32_e32 v6, 16, v2
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_4) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_and_b32_e32 v0, -4, v4
; GFX11-NEXT: v_and_b32_e32 v4, 3, v4
; GFX11-NEXT: s_mov_b32 s0, 0
; GFX11-NEXT: flat_load_b32 v3, v[0:1]
; GFX11-NEXT: v_lshlrev_b32_e32 v4, 3, v4
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_lshlrev_b32_e64 v5, v4, 0xffff
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_not_b32_e32 v5, v5
; GFX11-NEXT: .p2align 6
; GFX11-NEXT: .LBB55_1: ; %atomicrmw.start
@@ -15170,7 +15246,7 @@ define void @flat_system_atomic_fadd_noret_bf16__offset12b_pos__amdgpu_no_fine_g
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v3
; GFX11-NEXT: v_mov_b32_e32 v3, v2
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB55_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -15424,11 +15500,12 @@ define <2 x half> @flat_agent_atomic_fadd_ret_v2f16__amdgpu_no_fine_grained_memo
; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB56_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: v_mov_b32_e32 v0, v3
; GFX11-NEXT: s_setpc_b64 s[30:31]
;
@@ -15607,11 +15684,12 @@ define <2 x half> @flat_agent_atomic_fadd_ret_v2f16__offset12b_pos__amdgpu_no_fi
; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB57_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: v_mov_b32_e32 v0, v3
; GFX11-NEXT: s_setpc_b64 s[30:31]
;
@@ -15784,7 +15862,6 @@ define <2 x half> @flat_agent_atomic_fadd_ret_v2f16__offset12b_neg__amdgpu_no_fi
; GFX11: ; %bb.0:
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: v_add_co_u32 v3, vcc_lo, 0xfffff800, v0
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v4, null, -1, v1, vcc_lo
; GFX11-NEXT: s_mov_b32 s0, 0
; GFX11-NEXT: flat_load_b32 v0, v[3:4]
@@ -15801,7 +15878,7 @@ define <2 x half> @flat_agent_atomic_fadd_ret_v2f16__offset12b_neg__amdgpu_no_fi
; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v0, v1
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB58_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -15995,7 +16072,7 @@ define void @flat_agent_atomic_fadd_noret_v2f16__amdgpu_no_fine_grained_memory(p
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-NEXT: v_mov_b32_e32 v4, v3
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB59_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -16169,7 +16246,7 @@ define void @flat_agent_atomic_fadd_noret_v2f16__offset12b_pos__amdgpu_no_fine_g
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-NEXT: v_mov_b32_e32 v4, v3
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB60_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -16340,7 +16417,6 @@ define void @flat_agent_atomic_fadd_noret_v2f16__offset12b_neg__amdgpu_no_fine_g
; GFX11: ; %bb.0:
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: v_add_co_u32 v0, vcc_lo, 0xfffff800, v0
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, -1, v1, vcc_lo
; GFX11-NEXT: s_mov_b32 s0, 0
; GFX11-NEXT: flat_load_b32 v4, v[0:1]
@@ -16356,7 +16432,7 @@ define void @flat_agent_atomic_fadd_noret_v2f16__offset12b_neg__amdgpu_no_fine_g
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-NEXT: v_mov_b32_e32 v4, v3
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB61_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -16549,11 +16625,12 @@ define <2 x half> @flat_system_atomic_fadd_ret_v2f16__offset12b_pos__amdgpu_no_f
; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB62_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: v_mov_b32_e32 v0, v3
; GFX11-NEXT: s_setpc_b64 s[30:31]
;
@@ -16740,7 +16817,7 @@ define void @flat_system_atomic_fadd_noret_v2f16__offset12b_pos__amdgpu_no_fine_
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-NEXT: v_mov_b32_e32 v4, v3
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB63_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -16925,11 +17002,12 @@ define <2 x half> @flat_agent_atomic_fadd_ret_v2f16__amdgpu_no_remote_memory(ptr
; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB64_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: v_mov_b32_e32 v0, v3
; GFX11-NEXT: s_setpc_b64 s[30:31]
;
@@ -17107,7 +17185,7 @@ define void @flat_agent_atomic_fadd_noret_v2f16__amdgpu_no_remote_memory(ptr %pt
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-NEXT: v_mov_b32_e32 v4, v3
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB65_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -17282,11 +17360,12 @@ define <2 x half> @flat_agent_atomic_fadd_ret_v2f16__amdgpu_no_fine_grained_memo
; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB66_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: v_mov_b32_e32 v0, v3
; GFX11-NEXT: s_setpc_b64 s[30:31]
;
@@ -17464,7 +17543,7 @@ define void @flat_agent_atomic_fadd_noret_v2f16__amdgpu_no_fine_grained_memory__
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-NEXT: v_mov_b32_e32 v4, v3
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB67_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -17668,12 +17747,13 @@ define <2 x bfloat> @flat_agent_atomic_fadd_ret_v2bf16__amdgpu_no_fine_grained_m
; GFX11-TRUE16-NEXT: buffer_gl0_inv
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v6
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB68_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-TRUE16-NEXT: s_set_inst_prefetch_distance 0x2
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_mov_b32_e32 v0, v3
; GFX11-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -17717,12 +17797,13 @@ define <2 x bfloat> @flat_agent_atomic_fadd_ret_v2bf16__amdgpu_no_fine_grained_m
; GFX11-FAKE16-NEXT: buffer_gl0_inv
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v6
; GFX11-FAKE16-NEXT: s_or_b32 s1, vcc_lo, s1
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB68_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-FAKE16-NEXT: s_set_inst_prefetch_distance 0x2
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s1
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v0, v3
; GFX11-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -18000,12 +18081,13 @@ define <2 x bfloat> @flat_agent_atomic_fadd_ret_v2bf16__offset12b_pos__amdgpu_no
; GFX11-TRUE16-NEXT: buffer_gl0_inv
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v6
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB69_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-TRUE16-NEXT: s_set_inst_prefetch_distance 0x2
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_mov_b32_e32 v0, v3
; GFX11-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -18049,12 +18131,13 @@ define <2 x bfloat> @flat_agent_atomic_fadd_ret_v2bf16__offset12b_pos__amdgpu_no
; GFX11-FAKE16-NEXT: buffer_gl0_inv
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v6
; GFX11-FAKE16-NEXT: s_or_b32 s1, vcc_lo, s1
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB69_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-FAKE16-NEXT: s_set_inst_prefetch_distance 0x2
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s1
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v0, v3
; GFX11-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -18301,7 +18384,6 @@ define <2 x bfloat> @flat_agent_atomic_fadd_ret_v2bf16__offset12b_neg__amdgpu_no
; GFX11-TRUE16: ; %bb.0:
; GFX11-TRUE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-TRUE16-NEXT: v_add_co_u32 v3, vcc_lo, 0xfffff800, v0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v4, null, -1, v1, vcc_lo
; GFX11-TRUE16-NEXT: v_and_b32_e32 v1, 0xffff0000, v2
; GFX11-TRUE16-NEXT: v_lshlrev_b32_e32 v2, 16, v2
@@ -18343,7 +18425,7 @@ define <2 x bfloat> @flat_agent_atomic_fadd_ret_v2bf16__offset12b_neg__amdgpu_no
; GFX11-TRUE16-NEXT: buffer_gl0_inv
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v0, v6
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB70_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -18355,7 +18437,6 @@ define <2 x bfloat> @flat_agent_atomic_fadd_ret_v2bf16__offset12b_neg__amdgpu_no
; GFX11-FAKE16: ; %bb.0:
; GFX11-FAKE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-FAKE16-NEXT: v_add_co_u32 v3, vcc_lo, 0xfffff800, v0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v4, null, -1, v1, vcc_lo
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v1, 16, v2
; GFX11-FAKE16-NEXT: v_and_b32_e32 v2, 0xffff0000, v2
@@ -18394,7 +18475,7 @@ define <2 x bfloat> @flat_agent_atomic_fadd_ret_v2bf16__offset12b_neg__amdgpu_no
; GFX11-FAKE16-NEXT: buffer_gl0_inv
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v0, v6
; GFX11-FAKE16-NEXT: s_or_b32 s1, vcc_lo, s1
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB70_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -18688,7 +18769,7 @@ define void @flat_agent_atomic_fadd_noret_v2bf16__amdgpu_no_fine_grained_memory(
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v3
; GFX11-TRUE16-NEXT: v_mov_b32_e32 v3, v2
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB71_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -18722,7 +18803,7 @@ define void @flat_agent_atomic_fadd_noret_v2bf16__amdgpu_no_fine_grained_memory(
; GFX11-FAKE16-NEXT: v_add3_u32 v7, v7, v2, 0x7fff
; GFX11-FAKE16-NEXT: v_add3_u32 v8, v8, v6, 0x7fff
; GFX11-FAKE16-NEXT: v_cmp_u_f32_e64 s0, v2, v2
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_4)
; GFX11-FAKE16-NEXT: v_cndmask_b32_e32 v6, v8, v10, vcc_lo
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v2, v7, v9, s0
; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
@@ -18735,7 +18816,7 @@ define void @flat_agent_atomic_fadd_noret_v2bf16__amdgpu_no_fine_grained_memory(
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v3
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v3, v2
; GFX11-FAKE16-NEXT: s_or_b32 s1, vcc_lo, s1
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB71_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -19008,7 +19089,7 @@ define void @flat_agent_atomic_fadd_noret_v2bf16__offset12b_pos__amdgpu_no_fine_
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v3
; GFX11-TRUE16-NEXT: v_mov_b32_e32 v3, v2
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB72_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -19042,7 +19123,7 @@ define void @flat_agent_atomic_fadd_noret_v2bf16__offset12b_pos__amdgpu_no_fine_
; GFX11-FAKE16-NEXT: v_add3_u32 v7, v7, v2, 0x7fff
; GFX11-FAKE16-NEXT: v_add3_u32 v8, v8, v6, 0x7fff
; GFX11-FAKE16-NEXT: v_cmp_u_f32_e64 s0, v2, v2
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_4)
; GFX11-FAKE16-NEXT: v_cndmask_b32_e32 v6, v8, v10, vcc_lo
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v2, v7, v9, s0
; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
@@ -19055,7 +19136,7 @@ define void @flat_agent_atomic_fadd_noret_v2bf16__offset12b_pos__amdgpu_no_fine_
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v3
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v3, v2
; GFX11-FAKE16-NEXT: s_or_b32 s1, vcc_lo, s1
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB72_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -19300,7 +19381,6 @@ define void @flat_agent_atomic_fadd_noret_v2bf16__offset12b_neg__amdgpu_no_fine_
; GFX11-TRUE16: ; %bb.0:
; GFX11-TRUE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-TRUE16-NEXT: v_add_co_u32 v0, vcc_lo, 0xfffff800, v0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v1, null, -1, v1, vcc_lo
; GFX11-TRUE16-NEXT: v_and_b32_e32 v4, 0xffff0000, v2
; GFX11-TRUE16-NEXT: v_lshlrev_b32_e32 v5, 16, v2
@@ -19341,7 +19421,7 @@ define void @flat_agent_atomic_fadd_noret_v2bf16__offset12b_neg__amdgpu_no_fine_
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v3
; GFX11-TRUE16-NEXT: v_mov_b32_e32 v3, v2
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB73_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -19353,7 +19433,6 @@ define void @flat_agent_atomic_fadd_noret_v2bf16__offset12b_neg__amdgpu_no_fine_
; GFX11-FAKE16: ; %bb.0:
; GFX11-FAKE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-FAKE16-NEXT: v_add_co_u32 v0, vcc_lo, 0xfffff800, v0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v1, null, -1, v1, vcc_lo
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v4, 16, v2
; GFX11-FAKE16-NEXT: v_and_b32_e32 v5, 0xffff0000, v2
@@ -19378,7 +19457,7 @@ define void @flat_agent_atomic_fadd_noret_v2bf16__offset12b_neg__amdgpu_no_fine_
; GFX11-FAKE16-NEXT: v_add3_u32 v7, v7, v2, 0x7fff
; GFX11-FAKE16-NEXT: v_add3_u32 v8, v8, v6, 0x7fff
; GFX11-FAKE16-NEXT: v_cmp_u_f32_e64 s0, v2, v2
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_4)
; GFX11-FAKE16-NEXT: v_cndmask_b32_e32 v6, v8, v10, vcc_lo
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v2, v7, v9, s0
; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
@@ -19391,7 +19470,7 @@ define void @flat_agent_atomic_fadd_noret_v2bf16__offset12b_neg__amdgpu_no_fine_
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v3
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v3, v2
; GFX11-FAKE16-NEXT: s_or_b32 s1, vcc_lo, s1
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB73_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -19683,12 +19762,13 @@ define <2 x bfloat> @flat_system_atomic_fadd_ret_v2bf16__offset12b_pos__amdgpu_n
; GFX11-TRUE16-NEXT: buffer_gl0_inv
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v6
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB74_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-TRUE16-NEXT: s_set_inst_prefetch_distance 0x2
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_mov_b32_e32 v0, v3
; GFX11-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -19732,12 +19812,13 @@ define <2 x bfloat> @flat_system_atomic_fadd_ret_v2bf16__offset12b_pos__amdgpu_n
; GFX11-FAKE16-NEXT: buffer_gl0_inv
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v6
; GFX11-FAKE16-NEXT: s_or_b32 s1, vcc_lo, s1
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB74_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-FAKE16-NEXT: s_set_inst_prefetch_distance 0x2
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s1
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v0, v3
; GFX11-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -20023,7 +20104,7 @@ define void @flat_system_atomic_fadd_noret_v2bf16__offset12b_pos__amdgpu_no_fine
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v3
; GFX11-TRUE16-NEXT: v_mov_b32_e32 v3, v2
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB75_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -20057,7 +20138,7 @@ define void @flat_system_atomic_fadd_noret_v2bf16__offset12b_pos__amdgpu_no_fine
; GFX11-FAKE16-NEXT: v_add3_u32 v7, v7, v2, 0x7fff
; GFX11-FAKE16-NEXT: v_add3_u32 v8, v8, v6, 0x7fff
; GFX11-FAKE16-NEXT: v_cmp_u_f32_e64 s0, v2, v2
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_4)
; GFX11-FAKE16-NEXT: v_cndmask_b32_e32 v6, v8, v10, vcc_lo
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v2, v7, v9, s0
; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
@@ -20070,7 +20151,7 @@ define void @flat_system_atomic_fadd_noret_v2bf16__offset12b_pos__amdgpu_no_fine
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v3
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v3, v2
; GFX11-FAKE16-NEXT: s_or_b32 s1, vcc_lo, s1
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB75_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -20354,12 +20435,13 @@ define <2 x bfloat> @flat_agent_atomic_fadd_ret_v2bf16__amdgpu_no_remote_memory(
; GFX11-TRUE16-NEXT: buffer_gl0_inv
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v6
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB76_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-TRUE16-NEXT: s_set_inst_prefetch_distance 0x2
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_mov_b32_e32 v0, v3
; GFX11-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -20403,12 +20485,13 @@ define <2 x bfloat> @flat_agent_atomic_fadd_ret_v2bf16__amdgpu_no_remote_memory(
; GFX11-FAKE16-NEXT: buffer_gl0_inv
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v6
; GFX11-FAKE16-NEXT: s_or_b32 s1, vcc_lo, s1
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB76_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-FAKE16-NEXT: s_set_inst_prefetch_distance 0x2
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s1
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v0, v3
; GFX11-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -20685,7 +20768,7 @@ define void @flat_agent_atomic_fadd_noret_v2bf16__amdgpu_no_remote_memory(ptr %p
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v3
; GFX11-TRUE16-NEXT: v_mov_b32_e32 v3, v2
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB77_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -20719,7 +20802,7 @@ define void @flat_agent_atomic_fadd_noret_v2bf16__amdgpu_no_remote_memory(ptr %p
; GFX11-FAKE16-NEXT: v_add3_u32 v7, v7, v2, 0x7fff
; GFX11-FAKE16-NEXT: v_add3_u32 v8, v8, v6, 0x7fff
; GFX11-FAKE16-NEXT: v_cmp_u_f32_e64 s0, v2, v2
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_4)
; GFX11-FAKE16-NEXT: v_cndmask_b32_e32 v6, v8, v10, vcc_lo
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v2, v7, v9, s0
; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
@@ -20732,7 +20815,7 @@ define void @flat_agent_atomic_fadd_noret_v2bf16__amdgpu_no_remote_memory(ptr %p
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v3
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v3, v2
; GFX11-FAKE16-NEXT: s_or_b32 s1, vcc_lo, s1
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB77_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -21006,12 +21089,13 @@ define <2 x bfloat> @flat_agent_atomic_fadd_ret_v2bf16__amdgpu_no_fine_grained_m
; GFX11-TRUE16-NEXT: buffer_gl0_inv
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v6
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB78_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-TRUE16-NEXT: s_set_inst_prefetch_distance 0x2
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_mov_b32_e32 v0, v3
; GFX11-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -21055,12 +21139,13 @@ define <2 x bfloat> @flat_agent_atomic_fadd_ret_v2bf16__amdgpu_no_fine_grained_m
; GFX11-FAKE16-NEXT: buffer_gl0_inv
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v6
; GFX11-FAKE16-NEXT: s_or_b32 s1, vcc_lo, s1
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB78_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-FAKE16-NEXT: s_set_inst_prefetch_distance 0x2
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s1
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v0, v3
; GFX11-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -21337,7 +21422,7 @@ define void @flat_agent_atomic_fadd_noret_v2bf16__amdgpu_no_fine_grained_memory_
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v3
; GFX11-TRUE16-NEXT: v_mov_b32_e32 v3, v2
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB79_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -21371,7 +21456,7 @@ define void @flat_agent_atomic_fadd_noret_v2bf16__amdgpu_no_fine_grained_memory_
; GFX11-FAKE16-NEXT: v_add3_u32 v7, v7, v2, 0x7fff
; GFX11-FAKE16-NEXT: v_add3_u32 v8, v8, v6, 0x7fff
; GFX11-FAKE16-NEXT: v_cmp_u_f32_e64 s0, v2, v2
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_4)
; GFX11-FAKE16-NEXT: v_cndmask_b32_e32 v6, v8, v10, vcc_lo
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v2, v7, v9, s0
; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
@@ -21384,7 +21469,7 @@ define void @flat_agent_atomic_fadd_noret_v2bf16__amdgpu_no_fine_grained_memory_
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v3
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v3, v2
; GFX11-FAKE16-NEXT: s_or_b32 s1, vcc_lo, s1
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB79_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
diff --git a/llvm/test/CodeGen/AMDGPU/flat-atomicrmw-fmax.ll b/llvm/test/CodeGen/AMDGPU/flat-atomicrmw-fmax.ll
index 91d75d70772591..331e0200f10ed4 100644
--- a/llvm/test/CodeGen/AMDGPU/flat-atomicrmw-fmax.ll
+++ b/llvm/test/CodeGen/AMDGPU/flat-atomicrmw-fmax.ll
@@ -355,7 +355,6 @@ define float @flat_agent_atomic_fmax_ret_f32__offset12b_neg__amdgpu_no_fine_grai
; GFX11: ; %bb.0:
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: v_add_co_u32 v0, vcc_lo, 0xfffff800, v0
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, -1, v1, vcc_lo
; GFX11-NEXT: s_waitcnt_vscnt null, 0x0
; GFX11-NEXT: flat_atomic_max_f32 v0, v[0:1], v2 glc
@@ -809,7 +808,6 @@ define void @flat_agent_atomic_fmax_noret_f32__offset12b_neg__amdgpu_no_fine_gra
; GFX11: ; %bb.0:
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: v_add_co_u32 v0, vcc_lo, 0xfffff800, v0
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, -1, v1, vcc_lo
; GFX11-NEXT: s_waitcnt_vscnt null, 0x0
; GFX11-NEXT: flat_atomic_max_f32 v[0:1], v2
@@ -1292,11 +1290,12 @@ define float @flat_agent_atomic_fmax_ret_f32__amdgpu_no_remote_memory(ptr %ptr,
; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB8_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: v_mov_b32_e32 v0, v3
; GFX11-NEXT: s_setpc_b64 s[30:31]
;
@@ -1913,7 +1912,6 @@ define float @flat_agent_atomic_fmax_ret_f32__offset12b_neg__ftz__amdgpu_no_fine
; GFX11: ; %bb.0:
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: v_add_co_u32 v0, vcc_lo, 0xfffff800, v0
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, -1, v1, vcc_lo
; GFX11-NEXT: s_waitcnt_vscnt null, 0x0
; GFX11-NEXT: flat_atomic_max_f32 v0, v[0:1], v2 glc
@@ -2367,7 +2365,6 @@ define void @flat_agent_atomic_fmax_noret_f32__offset12b_neg__ftz__amdgpu_no_fin
; GFX11: ; %bb.0:
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: v_add_co_u32 v0, vcc_lo, 0xfffff800, v0
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, -1, v1, vcc_lo
; GFX11-NEXT: s_waitcnt_vscnt null, 0x0
; GFX11-NEXT: flat_atomic_max_f32 v[0:1], v2
@@ -2808,6 +2805,7 @@ define double @flat_agent_atomic_fmax_ret_f64__amdgpu_no_fine_grained_memory(ptr
; GFX12-NEXT: ; implicit-def: $vgpr2_vgpr3
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: v_cmpx_ne_u32_e32 s1, v1
+; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX12-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX12-NEXT: s_cbranch_execz .LBB18_4
; GFX12-NEXT: ; %bb.1: ; %atomicrmw.global
@@ -2829,6 +2827,7 @@ define double @flat_agent_atomic_fmax_ret_f64__amdgpu_no_fine_grained_memory(ptr
; GFX12-NEXT: s_or_b32 s1, vcc_lo, s1
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB18_2
; GFX12-NEXT: ; %bb.3: ; %Flow
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s1
@@ -2851,6 +2850,7 @@ define double @flat_agent_atomic_fmax_ret_f64__amdgpu_no_fine_grained_memory(ptr
; GFX12-NEXT: .LBB18_6: ; %atomicrmw.phi
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: v_dual_mov_b32 v0, v2 :: v_dual_mov_b32 v1, v3
; GFX12-NEXT: s_setpc_b64 s[30:31]
;
@@ -2903,6 +2903,7 @@ define double @flat_agent_atomic_fmax_ret_f64__amdgpu_no_fine_grained_memory(ptr
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: ; implicit-def: $vgpr2_vgpr3
; GFX11-NEXT: v_cmpx_ne_u32_e32 s1, v1
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX11-NEXT: s_cbranch_execz .LBB18_4
; GFX11-NEXT: ; %bb.1: ; %atomicrmw.global
@@ -2922,7 +2923,7 @@ define double @flat_agent_atomic_fmax_ret_f64__amdgpu_no_fine_grained_memory(ptr
; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: v_cmp_eq_u64_e32 vcc_lo, v[2:3], v[8:9]
; GFX11-NEXT: s_or_b32 s1, vcc_lo, s1
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
; GFX11-NEXT: s_cbranch_execnz .LBB18_2
; GFX11-NEXT: ; %bb.3: ; %Flow
@@ -2930,6 +2931,7 @@ define double @flat_agent_atomic_fmax_ret_f64__amdgpu_no_fine_grained_memory(ptr
; GFX11-NEXT: ; implicit-def: $vgpr0_vgpr1
; GFX11-NEXT: ; implicit-def: $vgpr4_vgpr5
; GFX11-NEXT: .LBB18_4: ; %Flow2
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB18_6
; GFX11-NEXT: ; %bb.5: ; %atomicrmw.private
@@ -2943,6 +2945,7 @@ define double @flat_agent_atomic_fmax_ret_f64__amdgpu_no_fine_grained_memory(ptr
; GFX11-NEXT: scratch_store_b64 v6, v[0:1], off
; GFX11-NEXT: .LBB18_6: ; %atomicrmw.phi
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: v_dual_mov_b32 v0, v2 :: v_dual_mov_b32 v1, v3
; GFX11-NEXT: s_setpc_b64 s[30:31]
;
@@ -3201,6 +3204,7 @@ define double @flat_agent_atomic_fmax_ret_f64__offset12b_pos__amdgpu_no_fine_gra
; GFX12-NEXT: ; implicit-def: $vgpr0_vgpr1
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: v_cmpx_ne_u32_e32 s1, v5
+; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX12-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX12-NEXT: s_cbranch_execnz .LBB19_3
; GFX12-NEXT: ; %bb.1: ; %Flow2
@@ -3230,11 +3234,13 @@ define double @flat_agent_atomic_fmax_ret_f64__offset12b_pos__amdgpu_no_fine_gra
; GFX12-NEXT: s_or_b32 s1, vcc_lo, s1
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB19_4
; GFX12-NEXT: ; %bb.5: ; %Flow
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s1
; GFX12-NEXT: ; implicit-def: $vgpr4_vgpr5
; GFX12-NEXT: ; implicit-def: $vgpr2_vgpr3
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX12-NEXT: s_cbranch_execz .LBB19_2
; GFX12-NEXT: .LBB19_6: ; %atomicrmw.private
@@ -3297,12 +3303,12 @@ define double @flat_agent_atomic_fmax_ret_f64__offset12b_pos__amdgpu_no_fine_gra
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: v_max_f64 v[2:3], v[2:3], v[2:3]
; GFX11-NEXT: v_add_co_u32 v4, vcc_lo, 0x7f8, v0
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v5, null, 0, v1, vcc_lo
; GFX11-NEXT: s_mov_b64 s[0:1], src_private_base
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: ; implicit-def: $vgpr0_vgpr1
; GFX11-NEXT: v_cmpx_ne_u32_e32 s1, v5
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB19_3
; GFX11-NEXT: ; %bb.1: ; %Flow2
@@ -3328,13 +3334,14 @@ define double @flat_agent_atomic_fmax_ret_f64__offset12b_pos__amdgpu_no_fine_gra
; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: v_cmp_eq_u64_e32 vcc_lo, v[0:1], v[8:9]
; GFX11-NEXT: s_or_b32 s1, vcc_lo, s1
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
; GFX11-NEXT: s_cbranch_execnz .LBB19_4
; GFX11-NEXT: ; %bb.5: ; %Flow
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s1
; GFX11-NEXT: ; implicit-def: $vgpr4_vgpr5
; GFX11-NEXT: ; implicit-def: $vgpr2_vgpr3
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB19_2
; GFX11-NEXT: .LBB19_6: ; %atomicrmw.private
@@ -3613,6 +3620,7 @@ define double @flat_agent_atomic_fmax_ret_f64__offset12b_neg__amdgpu_no_fine_gra
; GFX12-NEXT: ; implicit-def: $vgpr0_vgpr1
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: v_cmpx_ne_u32_e32 s1, v5
+; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX12-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX12-NEXT: s_cbranch_execnz .LBB20_3
; GFX12-NEXT: ; %bb.1: ; %Flow2
@@ -3642,11 +3650,13 @@ define double @flat_agent_atomic_fmax_ret_f64__offset12b_neg__amdgpu_no_fine_gra
; GFX12-NEXT: s_or_b32 s1, vcc_lo, s1
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB20_4
; GFX12-NEXT: ; %bb.5: ; %Flow
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s1
; GFX12-NEXT: ; implicit-def: $vgpr4_vgpr5
; GFX12-NEXT: ; implicit-def: $vgpr2_vgpr3
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX12-NEXT: s_cbranch_execz .LBB20_2
; GFX12-NEXT: .LBB20_6: ; %atomicrmw.private
@@ -3710,12 +3720,12 @@ define double @flat_agent_atomic_fmax_ret_f64__offset12b_neg__amdgpu_no_fine_gra
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: v_max_f64 v[2:3], v[2:3], v[2:3]
; GFX11-NEXT: v_add_co_u32 v4, vcc_lo, 0xfffff800, v0
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v5, null, -1, v1, vcc_lo
; GFX11-NEXT: s_mov_b64 s[0:1], src_private_base
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: ; implicit-def: $vgpr0_vgpr1
; GFX11-NEXT: v_cmpx_ne_u32_e32 s1, v5
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB20_3
; GFX11-NEXT: ; %bb.1: ; %Flow2
@@ -3741,13 +3751,14 @@ define double @flat_agent_atomic_fmax_ret_f64__offset12b_neg__amdgpu_no_fine_gra
; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: v_cmp_eq_u64_e32 vcc_lo, v[0:1], v[8:9]
; GFX11-NEXT: s_or_b32 s1, vcc_lo, s1
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
; GFX11-NEXT: s_cbranch_execnz .LBB20_4
; GFX11-NEXT: ; %bb.5: ; %Flow
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s1
; GFX11-NEXT: ; implicit-def: $vgpr4_vgpr5
; GFX11-NEXT: ; implicit-def: $vgpr2_vgpr3
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB20_2
; GFX11-NEXT: .LBB20_6: ; %atomicrmw.private
@@ -4022,6 +4033,7 @@ define void @flat_agent_atomic_fmax_noret_f64__amdgpu_no_fine_grained_memory(ptr
; GFX12-NEXT: s_mov_b32 s0, exec_lo
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: v_cmpx_ne_u32_e32 s1, v1
+; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX12-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX12-NEXT: s_cbranch_execnz .LBB21_3
; GFX12-NEXT: ; %bb.1: ; %Flow2
@@ -4051,11 +4063,13 @@ define void @flat_agent_atomic_fmax_noret_f64__amdgpu_no_fine_grained_memory(ptr
; GFX12-NEXT: s_or_b32 s1, vcc_lo, s1
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB21_4
; GFX12-NEXT: ; %bb.5: ; %Flow
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s1
; GFX12-NEXT: ; implicit-def: $vgpr0_vgpr1
; GFX12-NEXT: ; implicit-def: $vgpr6_vgpr7
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX12-NEXT: s_cbranch_execz .LBB21_2
; GFX12-NEXT: .LBB21_6: ; %atomicrmw.private
@@ -4117,6 +4131,7 @@ define void @flat_agent_atomic_fmax_noret_f64__amdgpu_no_fine_grained_memory(ptr
; GFX11-NEXT: s_mov_b64 s[0:1], src_private_base
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 s1, v1
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB21_3
; GFX11-NEXT: ; %bb.1: ; %Flow2
@@ -4142,13 +4157,14 @@ define void @flat_agent_atomic_fmax_noret_f64__amdgpu_no_fine_grained_memory(ptr
; GFX11-NEXT: v_cmp_eq_u64_e32 vcc_lo, v[2:3], v[4:5]
; GFX11-NEXT: v_dual_mov_b32 v5, v3 :: v_dual_mov_b32 v4, v2
; GFX11-NEXT: s_or_b32 s1, vcc_lo, s1
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
; GFX11-NEXT: s_cbranch_execnz .LBB21_4
; GFX11-NEXT: ; %bb.5: ; %Flow
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s1
; GFX11-NEXT: ; implicit-def: $vgpr0_vgpr1
; GFX11-NEXT: ; implicit-def: $vgpr6_vgpr7
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB21_2
; GFX11-NEXT: .LBB21_6: ; %atomicrmw.private
@@ -4411,6 +4427,7 @@ define void @flat_agent_atomic_fmax_noret_f64__offset12b_pos__amdgpu_no_fine_gra
; GFX12-NEXT: s_mov_b32 s0, exec_lo
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: v_cmpx_ne_u32_e32 s1, v7
+; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX12-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX12-NEXT: s_cbranch_execnz .LBB22_3
; GFX12-NEXT: ; %bb.1: ; %Flow2
@@ -4440,11 +4457,13 @@ define void @flat_agent_atomic_fmax_noret_f64__offset12b_pos__amdgpu_no_fine_gra
; GFX12-NEXT: s_or_b32 s1, vcc_lo, s1
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB22_4
; GFX12-NEXT: ; %bb.5: ; %Flow
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s1
; GFX12-NEXT: ; implicit-def: $vgpr6_vgpr7
; GFX12-NEXT: ; implicit-def: $vgpr4_vgpr5
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX12-NEXT: s_cbranch_execz .LBB22_2
; GFX12-NEXT: .LBB22_6: ; %atomicrmw.private
@@ -4506,11 +4525,11 @@ define void @flat_agent_atomic_fmax_noret_f64__offset12b_pos__amdgpu_no_fine_gra
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: v_max_f64 v[4:5], v[2:3], v[2:3]
; GFX11-NEXT: v_add_co_u32 v6, vcc_lo, 0x7f8, v0
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v7, null, 0, v1, vcc_lo
; GFX11-NEXT: s_mov_b64 s[0:1], src_private_base
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 s1, v7
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB22_3
; GFX11-NEXT: ; %bb.1: ; %Flow2
@@ -4536,13 +4555,14 @@ define void @flat_agent_atomic_fmax_noret_f64__offset12b_pos__amdgpu_no_fine_gra
; GFX11-NEXT: v_cmp_eq_u64_e32 vcc_lo, v[0:1], v[2:3]
; GFX11-NEXT: v_dual_mov_b32 v3, v1 :: v_dual_mov_b32 v2, v0
; GFX11-NEXT: s_or_b32 s1, vcc_lo, s1
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
; GFX11-NEXT: s_cbranch_execnz .LBB22_4
; GFX11-NEXT: ; %bb.5: ; %Flow
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s1
; GFX11-NEXT: ; implicit-def: $vgpr6_vgpr7
; GFX11-NEXT: ; implicit-def: $vgpr4_vgpr5
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB22_2
; GFX11-NEXT: .LBB22_6: ; %atomicrmw.private
@@ -4816,6 +4836,7 @@ define void @flat_agent_atomic_fmax_noret_f64__offset12b_neg__amdgpu_no_fine_gra
; GFX12-NEXT: s_mov_b32 s0, exec_lo
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: v_cmpx_ne_u32_e32 s1, v7
+; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX12-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX12-NEXT: s_cbranch_execnz .LBB23_3
; GFX12-NEXT: ; %bb.1: ; %Flow2
@@ -4845,11 +4866,13 @@ define void @flat_agent_atomic_fmax_noret_f64__offset12b_neg__amdgpu_no_fine_gra
; GFX12-NEXT: s_or_b32 s1, vcc_lo, s1
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB23_4
; GFX12-NEXT: ; %bb.5: ; %Flow
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s1
; GFX12-NEXT: ; implicit-def: $vgpr6_vgpr7
; GFX12-NEXT: ; implicit-def: $vgpr4_vgpr5
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX12-NEXT: s_cbranch_execz .LBB23_2
; GFX12-NEXT: .LBB23_6: ; %atomicrmw.private
@@ -4912,11 +4935,11 @@ define void @flat_agent_atomic_fmax_noret_f64__offset12b_neg__amdgpu_no_fine_gra
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: v_max_f64 v[4:5], v[2:3], v[2:3]
; GFX11-NEXT: v_add_co_u32 v6, vcc_lo, 0xfffff800, v0
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v7, null, -1, v1, vcc_lo
; GFX11-NEXT: s_mov_b64 s[0:1], src_private_base
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 s1, v7
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB23_3
; GFX11-NEXT: ; %bb.1: ; %Flow2
@@ -4942,13 +4965,14 @@ define void @flat_agent_atomic_fmax_noret_f64__offset12b_neg__amdgpu_no_fine_gra
; GFX11-NEXT: v_cmp_eq_u64_e32 vcc_lo, v[0:1], v[2:3]
; GFX11-NEXT: v_dual_mov_b32 v3, v1 :: v_dual_mov_b32 v2, v0
; GFX11-NEXT: s_or_b32 s1, vcc_lo, s1
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
; GFX11-NEXT: s_cbranch_execnz .LBB23_4
; GFX11-NEXT: ; %bb.5: ; %Flow
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s1
; GFX11-NEXT: ; implicit-def: $vgpr6_vgpr7
; GFX11-NEXT: ; implicit-def: $vgpr4_vgpr5
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB23_2
; GFX11-NEXT: .LBB23_6: ; %atomicrmw.private
@@ -5220,6 +5244,7 @@ define double @flat_agent_atomic_fmax_ret_f64__amdgpu_no_remote_memory(ptr %ptr,
; GFX12-NEXT: ; implicit-def: $vgpr2_vgpr3
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: v_cmpx_ne_u32_e32 s1, v1
+; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX12-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX12-NEXT: s_cbranch_execz .LBB24_4
; GFX12-NEXT: ; %bb.1: ; %atomicrmw.global
@@ -5241,6 +5266,7 @@ define double @flat_agent_atomic_fmax_ret_f64__amdgpu_no_remote_memory(ptr %ptr,
; GFX12-NEXT: s_or_b32 s1, vcc_lo, s1
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB24_2
; GFX12-NEXT: ; %bb.3: ; %Flow
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s1
@@ -5263,6 +5289,7 @@ define double @flat_agent_atomic_fmax_ret_f64__amdgpu_no_remote_memory(ptr %ptr,
; GFX12-NEXT: .LBB24_6: ; %atomicrmw.phi
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: v_dual_mov_b32 v0, v2 :: v_dual_mov_b32 v1, v3
; GFX12-NEXT: s_setpc_b64 s[30:31]
;
@@ -5315,6 +5342,7 @@ define double @flat_agent_atomic_fmax_ret_f64__amdgpu_no_remote_memory(ptr %ptr,
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: ; implicit-def: $vgpr2_vgpr3
; GFX11-NEXT: v_cmpx_ne_u32_e32 s1, v1
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX11-NEXT: s_cbranch_execz .LBB24_4
; GFX11-NEXT: ; %bb.1: ; %atomicrmw.global
@@ -5334,7 +5362,7 @@ define double @flat_agent_atomic_fmax_ret_f64__amdgpu_no_remote_memory(ptr %ptr,
; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: v_cmp_eq_u64_e32 vcc_lo, v[2:3], v[8:9]
; GFX11-NEXT: s_or_b32 s1, vcc_lo, s1
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
; GFX11-NEXT: s_cbranch_execnz .LBB24_2
; GFX11-NEXT: ; %bb.3: ; %Flow
@@ -5342,6 +5370,7 @@ define double @flat_agent_atomic_fmax_ret_f64__amdgpu_no_remote_memory(ptr %ptr,
; GFX11-NEXT: ; implicit-def: $vgpr0_vgpr1
; GFX11-NEXT: ; implicit-def: $vgpr4_vgpr5
; GFX11-NEXT: .LBB24_4: ; %Flow2
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB24_6
; GFX11-NEXT: ; %bb.5: ; %atomicrmw.private
@@ -5355,6 +5384,7 @@ define double @flat_agent_atomic_fmax_ret_f64__amdgpu_no_remote_memory(ptr %ptr,
; GFX11-NEXT: scratch_store_b64 v6, v[0:1], off
; GFX11-NEXT: .LBB24_6: ; %atomicrmw.phi
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: v_dual_mov_b32 v0, v2 :: v_dual_mov_b32 v1, v3
; GFX11-NEXT: s_setpc_b64 s[30:31]
;
@@ -5644,6 +5674,7 @@ define double @flat_agent_atomic_fmax_ret_f64__amdgpu_no_fine_grained_memory__am
; GFX12-NEXT: ; implicit-def: $vgpr2_vgpr3
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: v_cmpx_ne_u32_e32 s1, v1
+; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX12-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX12-NEXT: s_cbranch_execz .LBB25_4
; GFX12-NEXT: ; %bb.1: ; %atomicrmw.global
@@ -5665,6 +5696,7 @@ define double @flat_agent_atomic_fmax_ret_f64__amdgpu_no_fine_grained_memory__am
; GFX12-NEXT: s_or_b32 s1, vcc_lo, s1
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB25_2
; GFX12-NEXT: ; %bb.3: ; %Flow
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s1
@@ -5687,6 +5719,7 @@ define double @flat_agent_atomic_fmax_ret_f64__amdgpu_no_fine_grained_memory__am
; GFX12-NEXT: .LBB25_6: ; %atomicrmw.phi
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: v_dual_mov_b32 v0, v2 :: v_dual_mov_b32 v1, v3
; GFX12-NEXT: s_setpc_b64 s[30:31]
;
@@ -5739,6 +5772,7 @@ define double @flat_agent_atomic_fmax_ret_f64__amdgpu_no_fine_grained_memory__am
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: ; implicit-def: $vgpr2_vgpr3
; GFX11-NEXT: v_cmpx_ne_u32_e32 s1, v1
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX11-NEXT: s_cbranch_execz .LBB25_4
; GFX11-NEXT: ; %bb.1: ; %atomicrmw.global
@@ -5758,7 +5792,7 @@ define double @flat_agent_atomic_fmax_ret_f64__amdgpu_no_fine_grained_memory__am
; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: v_cmp_eq_u64_e32 vcc_lo, v[2:3], v[8:9]
; GFX11-NEXT: s_or_b32 s1, vcc_lo, s1
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
; GFX11-NEXT: s_cbranch_execnz .LBB25_2
; GFX11-NEXT: ; %bb.3: ; %Flow
@@ -5766,6 +5800,7 @@ define double @flat_agent_atomic_fmax_ret_f64__amdgpu_no_fine_grained_memory__am
; GFX11-NEXT: ; implicit-def: $vgpr0_vgpr1
; GFX11-NEXT: ; implicit-def: $vgpr4_vgpr5
; GFX11-NEXT: .LBB25_4: ; %Flow2
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB25_6
; GFX11-NEXT: ; %bb.5: ; %atomicrmw.private
@@ -5779,6 +5814,7 @@ define double @flat_agent_atomic_fmax_ret_f64__amdgpu_no_fine_grained_memory__am
; GFX11-NEXT: scratch_store_b64 v6, v[0:1], off
; GFX11-NEXT: .LBB25_6: ; %atomicrmw.phi
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: v_dual_mov_b32 v0, v2 :: v_dual_mov_b32 v1, v3
; GFX11-NEXT: s_setpc_b64 s[30:31]
;
@@ -6065,9 +6101,11 @@ define half @flat_agent_atomic_fmax_ret_f16__amdgpu_no_fine_grained_memory(ptr %
; GFX12-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_cbranch_execnz .LBB26_1
; GFX12-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: v_lshrrev_b32_e32 v0, v3, v5
; GFX12-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -6111,9 +6149,11 @@ define half @flat_agent_atomic_fmax_ret_f16__amdgpu_no_fine_grained_memory(ptr %
; GFX12-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: s_cbranch_execnz .LBB26_1
; GFX12-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: v_lshrrev_b32_e32 v0, v3, v5
; GFX12-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -6187,11 +6227,12 @@ define half @flat_agent_atomic_fmax_ret_f16__amdgpu_no_fine_grained_memory(ptr %
; GFX11-TRUE16-NEXT: buffer_gl0_inv
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v5, v6
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB26_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v0, v3, v5
; GFX11-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -6229,11 +6270,12 @@ define half @flat_agent_atomic_fmax_ret_f16__amdgpu_no_fine_grained_memory(ptr %
; GFX11-FAKE16-NEXT: buffer_gl0_inv
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v5, v6
; GFX11-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB26_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v0, v3, v5
; GFX11-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -6457,9 +6499,11 @@ define half @flat_agent_atomic_fmax_ret_f16__offset12b_pos__amdgpu_no_fine_grain
; GFX12-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_cbranch_execnz .LBB27_1
; GFX12-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: v_lshrrev_b32_e32 v0, v1, v5
; GFX12-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -6505,9 +6549,11 @@ define half @flat_agent_atomic_fmax_ret_f16__offset12b_pos__amdgpu_no_fine_grain
; GFX12-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: s_cbranch_execnz .LBB27_1
; GFX12-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: v_lshrrev_b32_e32 v0, v3, v5
; GFX12-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -6553,7 +6599,6 @@ define half @flat_agent_atomic_fmax_ret_f16__offset12b_pos__amdgpu_no_fine_grain
; GFX11-TRUE16: ; %bb.0:
; GFX11-TRUE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-TRUE16-NEXT: v_add_co_u32 v0, vcc_lo, 0x7fe, v0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v4, null, 0, v1, vcc_lo
; GFX11-TRUE16-NEXT: s_mov_b32 s0, 0
; GFX11-TRUE16-NEXT: v_and_b32_e32 v3, -4, v0
@@ -6586,11 +6631,12 @@ define half @flat_agent_atomic_fmax_ret_f16__offset12b_pos__amdgpu_no_fine_grain
; GFX11-TRUE16-NEXT: buffer_gl0_inv
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v5, v6
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB27_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v0, v1, v5
; GFX11-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -6598,16 +6644,16 @@ define half @flat_agent_atomic_fmax_ret_f16__offset12b_pos__amdgpu_no_fine_grain
; GFX11-FAKE16: ; %bb.0:
; GFX11-FAKE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-FAKE16-NEXT: v_add_co_u32 v3, vcc_lo, 0x7fe, v0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-FAKE16-NEXT: v_max_f16_e32 v2, v2, v2
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_4) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_and_b32_e32 v0, -4, v3
; GFX11-FAKE16-NEXT: v_and_b32_e32 v3, 3, v3
; GFX11-FAKE16-NEXT: s_mov_b32 s0, 0
; GFX11-FAKE16-NEXT: flat_load_b32 v5, v[0:1]
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v3, 3, v3
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e64 v4, v3, 0xffff
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_not_b32_e32 v4, v4
; GFX11-FAKE16-NEXT: .LBB27_1: ; %atomicrmw.start
; GFX11-FAKE16-NEXT: ; =>This Inner Loop Header: Depth=1
@@ -6629,11 +6675,12 @@ define half @flat_agent_atomic_fmax_ret_f16__offset12b_pos__amdgpu_no_fine_grain
; GFX11-FAKE16-NEXT: buffer_gl0_inv
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v5, v6
; GFX11-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB27_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v0, v3, v5
; GFX11-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -6863,9 +6910,11 @@ define half @flat_agent_atomic_fmax_ret_f16__offset12b_neg__amdgpu_no_fine_grain
; GFX12-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_cbranch_execnz .LBB28_1
; GFX12-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: v_lshrrev_b32_e32 v0, v1, v5
; GFX12-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -6911,9 +6960,11 @@ define half @flat_agent_atomic_fmax_ret_f16__offset12b_neg__amdgpu_no_fine_grain
; GFX12-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: s_cbranch_execnz .LBB28_1
; GFX12-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: v_lshrrev_b32_e32 v0, v3, v5
; GFX12-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -6960,7 +7011,6 @@ define half @flat_agent_atomic_fmax_ret_f16__offset12b_neg__amdgpu_no_fine_grain
; GFX11-TRUE16: ; %bb.0:
; GFX11-TRUE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-TRUE16-NEXT: v_add_co_u32 v0, vcc_lo, 0xfffff800, v0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v4, null, -1, v1, vcc_lo
; GFX11-TRUE16-NEXT: s_mov_b32 s0, 0
; GFX11-TRUE16-NEXT: v_and_b32_e32 v3, -4, v0
@@ -6993,11 +7043,12 @@ define half @flat_agent_atomic_fmax_ret_f16__offset12b_neg__amdgpu_no_fine_grain
; GFX11-TRUE16-NEXT: buffer_gl0_inv
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v5, v6
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB28_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v0, v1, v5
; GFX11-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -7005,16 +7056,16 @@ define half @flat_agent_atomic_fmax_ret_f16__offset12b_neg__amdgpu_no_fine_grain
; GFX11-FAKE16: ; %bb.0:
; GFX11-FAKE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-FAKE16-NEXT: v_add_co_u32 v3, vcc_lo, 0xfffff800, v0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v1, null, -1, v1, vcc_lo
; GFX11-FAKE16-NEXT: v_max_f16_e32 v2, v2, v2
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_4) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_and_b32_e32 v0, -4, v3
; GFX11-FAKE16-NEXT: v_and_b32_e32 v3, 3, v3
; GFX11-FAKE16-NEXT: s_mov_b32 s0, 0
; GFX11-FAKE16-NEXT: flat_load_b32 v5, v[0:1]
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v3, 3, v3
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e64 v4, v3, 0xffff
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_not_b32_e32 v4, v4
; GFX11-FAKE16-NEXT: .LBB28_1: ; %atomicrmw.start
; GFX11-FAKE16-NEXT: ; =>This Inner Loop Header: Depth=1
@@ -7036,11 +7087,12 @@ define half @flat_agent_atomic_fmax_ret_f16__offset12b_neg__amdgpu_no_fine_grain
; GFX11-FAKE16-NEXT: buffer_gl0_inv
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v5, v6
; GFX11-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB28_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v0, v3, v5
; GFX11-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -7267,6 +7319,7 @@ define void @flat_agent_atomic_fmax_noret_f16__amdgpu_no_fine_grained_memory(ptr
; GFX12-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_cbranch_execnz .LBB29_1
; GFX12-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -7312,6 +7365,7 @@ define void @flat_agent_atomic_fmax_noret_f16__amdgpu_no_fine_grained_memory(ptr
; GFX12-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: s_cbranch_execnz .LBB29_1
; GFX12-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -7386,7 +7440,7 @@ define void @flat_agent_atomic_fmax_noret_f16__amdgpu_no_fine_grained_memory(ptr
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-TRUE16-NEXT: v_mov_b32_e32 v4, v3
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB29_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -7427,7 +7481,7 @@ define void @flat_agent_atomic_fmax_noret_f16__amdgpu_no_fine_grained_memory(ptr
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v4, v3
; GFX11-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB29_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -7649,6 +7703,7 @@ define void @flat_agent_atomic_fmax_noret_f16__offset12b_pos__amdgpu_no_fine_gra
; GFX12-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_cbranch_execnz .LBB30_1
; GFX12-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -7696,6 +7751,7 @@ define void @flat_agent_atomic_fmax_noret_f16__offset12b_pos__amdgpu_no_fine_gra
; GFX12-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: s_cbranch_execnz .LBB30_1
; GFX12-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -7742,7 +7798,6 @@ define void @flat_agent_atomic_fmax_noret_f16__offset12b_pos__amdgpu_no_fine_gra
; GFX11-TRUE16: ; %bb.0:
; GFX11-TRUE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-TRUE16-NEXT: v_add_co_u32 v0, vcc_lo, 0x7fe, v0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v4, null, 0, v1, vcc_lo
; GFX11-TRUE16-NEXT: s_mov_b32 s0, 0
; GFX11-TRUE16-NEXT: v_and_b32_e32 v3, -4, v0
@@ -7775,7 +7830,7 @@ define void @flat_agent_atomic_fmax_noret_f16__offset12b_pos__amdgpu_no_fine_gra
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v5, v6
; GFX11-TRUE16-NEXT: v_mov_b32_e32 v6, v5
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB30_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -7786,16 +7841,16 @@ define void @flat_agent_atomic_fmax_noret_f16__offset12b_pos__amdgpu_no_fine_gra
; GFX11-FAKE16: ; %bb.0:
; GFX11-FAKE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-FAKE16-NEXT: v_add_co_u32 v4, vcc_lo, 0x7fe, v0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-FAKE16-NEXT: v_max_f16_e32 v6, v2, v2
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_4) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_and_b32_e32 v0, -4, v4
; GFX11-FAKE16-NEXT: v_and_b32_e32 v4, 3, v4
; GFX11-FAKE16-NEXT: s_mov_b32 s0, 0
; GFX11-FAKE16-NEXT: flat_load_b32 v3, v[0:1]
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v4, 3, v4
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e64 v5, v4, 0xffff
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_not_b32_e32 v5, v5
; GFX11-FAKE16-NEXT: .LBB30_1: ; %atomicrmw.start
; GFX11-FAKE16-NEXT: ; =>This Inner Loop Header: Depth=1
@@ -7817,7 +7872,7 @@ define void @flat_agent_atomic_fmax_noret_f16__offset12b_pos__amdgpu_no_fine_gra
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v3
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v3, v2
; GFX11-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB30_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -8045,6 +8100,7 @@ define void @flat_agent_atomic_fmax_noret_f16__offset12b_neg__amdgpu_no_fine_gra
; GFX12-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_cbranch_execnz .LBB31_1
; GFX12-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -8092,6 +8148,7 @@ define void @flat_agent_atomic_fmax_noret_f16__offset12b_neg__amdgpu_no_fine_gra
; GFX12-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: s_cbranch_execnz .LBB31_1
; GFX12-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -8139,7 +8196,6 @@ define void @flat_agent_atomic_fmax_noret_f16__offset12b_neg__amdgpu_no_fine_gra
; GFX11-TRUE16: ; %bb.0:
; GFX11-TRUE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-TRUE16-NEXT: v_add_co_u32 v0, vcc_lo, 0xfffff800, v0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v4, null, -1, v1, vcc_lo
; GFX11-TRUE16-NEXT: s_mov_b32 s0, 0
; GFX11-TRUE16-NEXT: v_and_b32_e32 v3, -4, v0
@@ -8172,7 +8228,7 @@ define void @flat_agent_atomic_fmax_noret_f16__offset12b_neg__amdgpu_no_fine_gra
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v5, v6
; GFX11-TRUE16-NEXT: v_mov_b32_e32 v6, v5
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB31_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -8183,16 +8239,16 @@ define void @flat_agent_atomic_fmax_noret_f16__offset12b_neg__amdgpu_no_fine_gra
; GFX11-FAKE16: ; %bb.0:
; GFX11-FAKE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-FAKE16-NEXT: v_add_co_u32 v4, vcc_lo, 0xfffff800, v0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v1, null, -1, v1, vcc_lo
; GFX11-FAKE16-NEXT: v_max_f16_e32 v6, v2, v2
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_4) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_and_b32_e32 v0, -4, v4
; GFX11-FAKE16-NEXT: v_and_b32_e32 v4, 3, v4
; GFX11-FAKE16-NEXT: s_mov_b32 s0, 0
; GFX11-FAKE16-NEXT: flat_load_b32 v3, v[0:1]
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v4, 3, v4
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e64 v5, v4, 0xffff
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_not_b32_e32 v5, v5
; GFX11-FAKE16-NEXT: .LBB31_1: ; %atomicrmw.start
; GFX11-FAKE16-NEXT: ; =>This Inner Loop Header: Depth=1
@@ -8214,7 +8270,7 @@ define void @flat_agent_atomic_fmax_noret_f16__offset12b_neg__amdgpu_no_fine_gra
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v3
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v3, v2
; GFX11-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB31_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -8428,9 +8484,11 @@ define half @flat_agent_atomic_fmax_ret_f16__offset12b_pos__align4__amdgpu_no_fi
; GFX12-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_cbranch_execnz .LBB32_1
; GFX12-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: v_mov_b16_e32 v0.l, v3.l
; GFX12-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -8463,9 +8521,11 @@ define half @flat_agent_atomic_fmax_ret_f16__offset12b_pos__align4__amdgpu_no_fi
; GFX12-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: s_cbranch_execnz .LBB32_1
; GFX12-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: v_mov_b32_e32 v0, v3
; GFX12-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -8520,11 +8580,12 @@ define half @flat_agent_atomic_fmax_ret_f16__offset12b_pos__align4__amdgpu_no_fi
; GFX11-TRUE16-NEXT: buffer_gl0_inv
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB32_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v0.l, v3.l
; GFX11-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -8551,11 +8612,12 @@ define half @flat_agent_atomic_fmax_ret_f16__offset12b_pos__align4__amdgpu_no_fi
; GFX11-FAKE16-NEXT: buffer_gl0_inv
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB32_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v0, v3
; GFX11-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -8729,6 +8791,7 @@ define void @flat_agent_atomic_fmax_noret_f16__offset12b__align4_pos__amdgpu_no_
; GFX12-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_cbranch_execnz .LBB33_1
; GFX12-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -8763,6 +8826,7 @@ define void @flat_agent_atomic_fmax_noret_f16__offset12b__align4_pos__amdgpu_no_
; GFX12-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: s_cbranch_execnz .LBB33_1
; GFX12-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -8818,7 +8882,7 @@ define void @flat_agent_atomic_fmax_noret_f16__offset12b__align4_pos__amdgpu_no_
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-TRUE16-NEXT: v_mov_b32_e32 v4, v3
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB33_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -8848,7 +8912,7 @@ define void @flat_agent_atomic_fmax_noret_f16__offset12b__align4_pos__amdgpu_no_
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v3
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v3, v2
; GFX11-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB33_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -9038,9 +9102,11 @@ define half @flat_system_atomic_fmax_ret_f16__offset12b_pos__amdgpu_no_fine_grai
; GFX12-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_cbranch_execnz .LBB34_1
; GFX12-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: v_lshrrev_b32_e32 v0, v1, v5
; GFX12-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -9087,9 +9153,11 @@ define half @flat_system_atomic_fmax_ret_f16__offset12b_pos__amdgpu_no_fine_grai
; GFX12-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: s_cbranch_execnz .LBB34_1
; GFX12-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: v_lshrrev_b32_e32 v0, v3, v5
; GFX12-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -9135,7 +9203,6 @@ define half @flat_system_atomic_fmax_ret_f16__offset12b_pos__amdgpu_no_fine_grai
; GFX11-TRUE16: ; %bb.0:
; GFX11-TRUE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-TRUE16-NEXT: v_add_co_u32 v0, vcc_lo, 0x7fe, v0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v4, null, 0, v1, vcc_lo
; GFX11-TRUE16-NEXT: s_mov_b32 s0, 0
; GFX11-TRUE16-NEXT: v_and_b32_e32 v3, -4, v0
@@ -9168,11 +9235,12 @@ define half @flat_system_atomic_fmax_ret_f16__offset12b_pos__amdgpu_no_fine_grai
; GFX11-TRUE16-NEXT: buffer_gl0_inv
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v5, v6
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB34_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v0, v1, v5
; GFX11-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -9180,16 +9248,16 @@ define half @flat_system_atomic_fmax_ret_f16__offset12b_pos__amdgpu_no_fine_grai
; GFX11-FAKE16: ; %bb.0:
; GFX11-FAKE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-FAKE16-NEXT: v_add_co_u32 v3, vcc_lo, 0x7fe, v0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-FAKE16-NEXT: v_max_f16_e32 v2, v2, v2
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_4) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_and_b32_e32 v0, -4, v3
; GFX11-FAKE16-NEXT: v_and_b32_e32 v3, 3, v3
; GFX11-FAKE16-NEXT: s_mov_b32 s0, 0
; GFX11-FAKE16-NEXT: flat_load_b32 v5, v[0:1]
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v3, 3, v3
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e64 v4, v3, 0xffff
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_not_b32_e32 v4, v4
; GFX11-FAKE16-NEXT: .LBB34_1: ; %atomicrmw.start
; GFX11-FAKE16-NEXT: ; =>This Inner Loop Header: Depth=1
@@ -9211,11 +9279,12 @@ define half @flat_system_atomic_fmax_ret_f16__offset12b_pos__amdgpu_no_fine_grai
; GFX11-FAKE16-NEXT: buffer_gl0_inv
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v5, v6
; GFX11-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB34_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v0, v3, v5
; GFX11-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -9449,6 +9518,7 @@ define void @flat_system_atomic_fmax_noret_f16__offset12b_pos__amdgpu_no_fine_gr
; GFX12-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_cbranch_execnz .LBB35_1
; GFX12-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -9497,6 +9567,7 @@ define void @flat_system_atomic_fmax_noret_f16__offset12b_pos__amdgpu_no_fine_gr
; GFX12-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: s_cbranch_execnz .LBB35_1
; GFX12-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -9543,7 +9614,6 @@ define void @flat_system_atomic_fmax_noret_f16__offset12b_pos__amdgpu_no_fine_gr
; GFX11-TRUE16: ; %bb.0:
; GFX11-TRUE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-TRUE16-NEXT: v_add_co_u32 v0, vcc_lo, 0x7fe, v0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v4, null, 0, v1, vcc_lo
; GFX11-TRUE16-NEXT: s_mov_b32 s0, 0
; GFX11-TRUE16-NEXT: v_and_b32_e32 v3, -4, v0
@@ -9576,7 +9646,7 @@ define void @flat_system_atomic_fmax_noret_f16__offset12b_pos__amdgpu_no_fine_gr
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v5, v6
; GFX11-TRUE16-NEXT: v_mov_b32_e32 v6, v5
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB35_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -9587,16 +9657,16 @@ define void @flat_system_atomic_fmax_noret_f16__offset12b_pos__amdgpu_no_fine_gr
; GFX11-FAKE16: ; %bb.0:
; GFX11-FAKE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-FAKE16-NEXT: v_add_co_u32 v4, vcc_lo, 0x7fe, v0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-FAKE16-NEXT: v_max_f16_e32 v6, v2, v2
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_4) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_and_b32_e32 v0, -4, v4
; GFX11-FAKE16-NEXT: v_and_b32_e32 v4, 3, v4
; GFX11-FAKE16-NEXT: s_mov_b32 s0, 0
; GFX11-FAKE16-NEXT: flat_load_b32 v3, v[0:1]
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v4, 3, v4
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e64 v5, v4, 0xffff
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_not_b32_e32 v5, v5
; GFX11-FAKE16-NEXT: .LBB35_1: ; %atomicrmw.start
; GFX11-FAKE16-NEXT: ; =>This Inner Loop Header: Depth=1
@@ -9618,7 +9688,7 @@ define void @flat_system_atomic_fmax_noret_f16__offset12b_pos__amdgpu_no_fine_gr
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v3
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v3, v2
; GFX11-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB35_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -9857,9 +9927,11 @@ define bfloat @flat_agent_atomic_fmax_ret_bf16__amdgpu_no_fine_grained_memory(pt
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB36_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: v_lshrrev_b32_e32 v0, v3, v5
; GFX12-NEXT: s_setpc_b64 s[30:31]
;
@@ -9947,11 +10019,12 @@ define bfloat @flat_agent_atomic_fmax_ret_bf16__amdgpu_no_fine_grained_memory(pt
; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v5, v6
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB36_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: v_lshrrev_b32_e32 v0, v3, v5
; GFX11-NEXT: s_setpc_b64 s[30:31]
;
@@ -10204,9 +10277,11 @@ define bfloat @flat_agent_atomic_fmax_ret_bf16__offset12b_pos__amdgpu_no_fine_gr
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB37_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: v_lshrrev_b32_e32 v0, v3, v5
; GFX12-NEXT: s_setpc_b64 s[30:31]
;
@@ -10259,16 +10334,16 @@ define bfloat @flat_agent_atomic_fmax_ret_bf16__offset12b_pos__amdgpu_no_fine_gr
; GFX11: ; %bb.0:
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: v_add_co_u32 v3, vcc_lo, 0x7fe, v0
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_3)
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-NEXT: v_lshlrev_b32_e32 v2, 16, v2
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_4) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_and_b32_e32 v0, -4, v3
; GFX11-NEXT: v_and_b32_e32 v3, 3, v3
; GFX11-NEXT: s_mov_b32 s0, 0
; GFX11-NEXT: flat_load_b32 v5, v[0:1]
; GFX11-NEXT: v_lshlrev_b32_e32 v3, 3, v3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_lshlrev_b32_e64 v4, v3, 0xffff
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_not_b32_e32 v4, v4
; GFX11-NEXT: .p2align 6
; GFX11-NEXT: .LBB37_1: ; %atomicrmw.start
@@ -10298,11 +10373,12 @@ define bfloat @flat_agent_atomic_fmax_ret_bf16__offset12b_pos__amdgpu_no_fine_gr
; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v5, v6
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB37_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: v_lshrrev_b32_e32 v0, v3, v5
; GFX11-NEXT: s_setpc_b64 s[30:31]
;
@@ -10561,9 +10637,11 @@ define bfloat @flat_agent_atomic_fmax_ret_bf16__offset12b_neg__amdgpu_no_fine_gr
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB38_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: v_lshrrev_b32_e32 v0, v3, v5
; GFX12-NEXT: s_setpc_b64 s[30:31]
;
@@ -10617,16 +10695,16 @@ define bfloat @flat_agent_atomic_fmax_ret_bf16__offset12b_neg__amdgpu_no_fine_gr
; GFX11: ; %bb.0:
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: v_add_co_u32 v3, vcc_lo, 0xfffff800, v0
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_3)
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, -1, v1, vcc_lo
; GFX11-NEXT: v_lshlrev_b32_e32 v2, 16, v2
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_4) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_and_b32_e32 v0, -4, v3
; GFX11-NEXT: v_and_b32_e32 v3, 3, v3
; GFX11-NEXT: s_mov_b32 s0, 0
; GFX11-NEXT: flat_load_b32 v5, v[0:1]
; GFX11-NEXT: v_lshlrev_b32_e32 v3, 3, v3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_lshlrev_b32_e64 v4, v3, 0xffff
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_not_b32_e32 v4, v4
; GFX11-NEXT: .p2align 6
; GFX11-NEXT: .LBB38_1: ; %atomicrmw.start
@@ -10656,11 +10734,12 @@ define bfloat @flat_agent_atomic_fmax_ret_bf16__offset12b_neg__amdgpu_no_fine_gr
; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v5, v6
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB38_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: v_lshrrev_b32_e32 v0, v3, v5
; GFX11-NEXT: s_setpc_b64 s[30:31]
;
@@ -10915,6 +10994,7 @@ define void @flat_agent_atomic_fmax_noret_bf16__amdgpu_no_fine_grained_memory(pt
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB39_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -11002,7 +11082,7 @@ define void @flat_agent_atomic_fmax_noret_bf16__amdgpu_no_fine_grained_memory(pt
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-NEXT: v_mov_b32_e32 v4, v3
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB39_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -11252,6 +11332,7 @@ define void @flat_agent_atomic_fmax_noret_bf16__offset12b_pos__amdgpu_no_fine_gr
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB40_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -11305,16 +11386,16 @@ define void @flat_agent_atomic_fmax_noret_bf16__offset12b_pos__amdgpu_no_fine_gr
; GFX11: ; %bb.0:
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: v_add_co_u32 v4, vcc_lo, 0x7fe, v0
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_3)
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-NEXT: v_lshlrev_b32_e32 v6, 16, v2
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_4) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_and_b32_e32 v0, -4, v4
; GFX11-NEXT: v_and_b32_e32 v4, 3, v4
; GFX11-NEXT: s_mov_b32 s0, 0
; GFX11-NEXT: flat_load_b32 v3, v[0:1]
; GFX11-NEXT: v_lshlrev_b32_e32 v4, 3, v4
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_lshlrev_b32_e64 v5, v4, 0xffff
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_not_b32_e32 v5, v5
; GFX11-NEXT: .p2align 6
; GFX11-NEXT: .LBB40_1: ; %atomicrmw.start
@@ -11343,7 +11424,7 @@ define void @flat_agent_atomic_fmax_noret_bf16__offset12b_pos__amdgpu_no_fine_gr
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v3
; GFX11-NEXT: v_mov_b32_e32 v3, v2
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB40_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -11599,6 +11680,7 @@ define void @flat_agent_atomic_fmax_noret_bf16__offset12b_neg__amdgpu_no_fine_gr
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB41_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -11653,16 +11735,16 @@ define void @flat_agent_atomic_fmax_noret_bf16__offset12b_neg__amdgpu_no_fine_gr
; GFX11: ; %bb.0:
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: v_add_co_u32 v4, vcc_lo, 0xfffff800, v0
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_3)
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, -1, v1, vcc_lo
; GFX11-NEXT: v_lshlrev_b32_e32 v6, 16, v2
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_4) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_and_b32_e32 v0, -4, v4
; GFX11-NEXT: v_and_b32_e32 v4, 3, v4
; GFX11-NEXT: s_mov_b32 s0, 0
; GFX11-NEXT: flat_load_b32 v3, v[0:1]
; GFX11-NEXT: v_lshlrev_b32_e32 v4, 3, v4
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_lshlrev_b32_e64 v5, v4, 0xffff
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_not_b32_e32 v5, v5
; GFX11-NEXT: .p2align 6
; GFX11-NEXT: .LBB41_1: ; %atomicrmw.start
@@ -11691,7 +11773,7 @@ define void @flat_agent_atomic_fmax_noret_bf16__offset12b_neg__amdgpu_no_fine_gr
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v3
; GFX11-NEXT: v_mov_b32_e32 v3, v2
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB41_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -11935,9 +12017,11 @@ define bfloat @flat_agent_atomic_fmax_ret_bf16__offset12b_pos__align4__amdgpu_no
; GFX12-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_cbranch_execnz .LBB42_1
; GFX12-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: v_mov_b16_e32 v0.l, v3.l
; GFX12-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -11978,9 +12062,11 @@ define bfloat @flat_agent_atomic_fmax_ret_bf16__offset12b_pos__align4__amdgpu_no
; GFX12-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: s_cbranch_execnz .LBB42_1
; GFX12-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: v_mov_b32_e32 v0, v3
; GFX12-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -12051,11 +12137,12 @@ define bfloat @flat_agent_atomic_fmax_ret_bf16__offset12b_pos__align4__amdgpu_no
; GFX11-TRUE16-NEXT: buffer_gl0_inv
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB42_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v0.l, v3.l
; GFX11-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -12090,11 +12177,12 @@ define bfloat @flat_agent_atomic_fmax_ret_bf16__offset12b_pos__align4__amdgpu_no
; GFX11-FAKE16-NEXT: buffer_gl0_inv
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB42_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v0, v3
; GFX11-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -12303,6 +12391,7 @@ define void @flat_agent_atomic_fmax_noret_bf16__offset12b__align4_pos__amdgpu_no
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB43_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -12373,7 +12462,7 @@ define void @flat_agent_atomic_fmax_noret_bf16__offset12b__align4_pos__amdgpu_no
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v3
; GFX11-NEXT: v_mov_b32_e32 v3, v2
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB43_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -12598,9 +12687,11 @@ define bfloat @flat_system_atomic_fmax_ret_bf16__offset12b_pos__amdgpu_no_fine_g
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB44_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: v_lshrrev_b32_e32 v0, v3, v5
; GFX12-NEXT: s_setpc_b64 s[30:31]
;
@@ -12653,16 +12744,16 @@ define bfloat @flat_system_atomic_fmax_ret_bf16__offset12b_pos__amdgpu_no_fine_g
; GFX11: ; %bb.0:
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: v_add_co_u32 v3, vcc_lo, 0x7fe, v0
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_3)
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-NEXT: v_lshlrev_b32_e32 v2, 16, v2
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_4) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_and_b32_e32 v0, -4, v3
; GFX11-NEXT: v_and_b32_e32 v3, 3, v3
; GFX11-NEXT: s_mov_b32 s0, 0
; GFX11-NEXT: flat_load_b32 v5, v[0:1]
; GFX11-NEXT: v_lshlrev_b32_e32 v3, 3, v3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_lshlrev_b32_e64 v4, v3, 0xffff
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_not_b32_e32 v4, v4
; GFX11-NEXT: .p2align 6
; GFX11-NEXT: .LBB44_1: ; %atomicrmw.start
@@ -12692,11 +12783,12 @@ define bfloat @flat_system_atomic_fmax_ret_bf16__offset12b_pos__amdgpu_no_fine_g
; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v5, v6
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB44_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: v_lshrrev_b32_e32 v0, v3, v5
; GFX11-NEXT: s_setpc_b64 s[30:31]
;
@@ -12958,6 +13050,7 @@ define void @flat_system_atomic_fmax_noret_bf16__offset12b_pos__amdgpu_no_fine_g
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB45_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -13011,16 +13104,16 @@ define void @flat_system_atomic_fmax_noret_bf16__offset12b_pos__amdgpu_no_fine_g
; GFX11: ; %bb.0:
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: v_add_co_u32 v4, vcc_lo, 0x7fe, v0
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_3)
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-NEXT: v_lshlrev_b32_e32 v6, 16, v2
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_4) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_and_b32_e32 v0, -4, v4
; GFX11-NEXT: v_and_b32_e32 v4, 3, v4
; GFX11-NEXT: s_mov_b32 s0, 0
; GFX11-NEXT: flat_load_b32 v3, v[0:1]
; GFX11-NEXT: v_lshlrev_b32_e32 v4, 3, v4
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_lshlrev_b32_e64 v5, v4, 0xffff
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_not_b32_e32 v5, v5
; GFX11-NEXT: .p2align 6
; GFX11-NEXT: .LBB45_1: ; %atomicrmw.start
@@ -13049,7 +13142,7 @@ define void @flat_system_atomic_fmax_noret_bf16__offset12b_pos__amdgpu_no_fine_g
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v3
; GFX11-NEXT: v_mov_b32_e32 v3, v2
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB45_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -13289,9 +13382,11 @@ define <2 x half> @flat_agent_atomic_fmax_ret_v2f16__amdgpu_no_fine_grained_memo
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB46_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: v_mov_b32_e32 v0, v3
; GFX12-NEXT: s_setpc_b64 s[30:31]
;
@@ -13342,11 +13437,12 @@ define <2 x half> @flat_agent_atomic_fmax_ret_v2f16__amdgpu_no_fine_grained_memo
; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB46_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: v_mov_b32_e32 v0, v3
; GFX11-NEXT: s_setpc_b64 s[30:31]
;
@@ -13520,9 +13616,11 @@ define <2 x half> @flat_agent_atomic_fmax_ret_v2f16__offset12b_pos__amdgpu_no_fi
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB47_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: v_mov_b32_e32 v0, v3
; GFX12-NEXT: s_setpc_b64 s[30:31]
;
@@ -13573,11 +13671,12 @@ define <2 x half> @flat_agent_atomic_fmax_ret_v2f16__offset12b_pos__amdgpu_no_fi
; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB47_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: v_mov_b32_e32 v0, v3
; GFX11-NEXT: s_setpc_b64 s[30:31]
;
@@ -13756,9 +13855,11 @@ define <2 x half> @flat_agent_atomic_fmax_ret_v2f16__offset12b_neg__amdgpu_no_fi
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB48_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: v_mov_b32_e32 v0, v3
; GFX12-NEXT: s_setpc_b64 s[30:31]
;
@@ -13798,7 +13899,6 @@ define <2 x half> @flat_agent_atomic_fmax_ret_v2f16__offset12b_neg__amdgpu_no_fi
; GFX11: ; %bb.0:
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: v_add_co_u32 v3, vcc_lo, 0xfffff800, v0
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v4, null, -1, v1, vcc_lo
; GFX11-NEXT: v_pk_max_f16 v1, v2, v2
; GFX11-NEXT: s_mov_b32 s0, 0
@@ -13817,7 +13917,7 @@ define <2 x half> @flat_agent_atomic_fmax_ret_v2f16__offset12b_neg__amdgpu_no_fi
; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v0, v6
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB48_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -14007,6 +14107,7 @@ define void @flat_agent_atomic_fmax_noret_v2f16__amdgpu_no_fine_grained_memory(p
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB49_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -14058,7 +14159,7 @@ define void @flat_agent_atomic_fmax_noret_v2f16__amdgpu_no_fine_grained_memory(p
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v3
; GFX11-NEXT: v_mov_b32_e32 v3, v2
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB49_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -14228,6 +14329,7 @@ define void @flat_agent_atomic_fmax_noret_v2f16__offset12b_pos__amdgpu_no_fine_g
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB50_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -14279,7 +14381,7 @@ define void @flat_agent_atomic_fmax_noret_v2f16__offset12b_pos__amdgpu_no_fine_g
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v3
; GFX11-NEXT: v_mov_b32_e32 v3, v2
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB50_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -14456,6 +14558,7 @@ define void @flat_agent_atomic_fmax_noret_v2f16__offset12b_neg__amdgpu_no_fine_g
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB51_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -14497,7 +14600,6 @@ define void @flat_agent_atomic_fmax_noret_v2f16__offset12b_neg__amdgpu_no_fine_g
; GFX11: ; %bb.0:
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: v_add_co_u32 v0, vcc_lo, 0xfffff800, v0
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, -1, v1, vcc_lo
; GFX11-NEXT: v_pk_max_f16 v4, v2, v2
; GFX11-NEXT: s_mov_b32 s0, 0
@@ -14516,7 +14618,7 @@ define void @flat_agent_atomic_fmax_noret_v2f16__offset12b_neg__amdgpu_no_fine_g
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v3
; GFX11-NEXT: v_mov_b32_e32 v3, v2
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB51_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -14704,9 +14806,11 @@ define <2 x half> @flat_system_atomic_fmax_ret_v2f16__offset12b_pos__amdgpu_no_f
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB52_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: v_mov_b32_e32 v0, v3
; GFX12-NEXT: s_setpc_b64 s[30:31]
;
@@ -14757,11 +14861,12 @@ define <2 x half> @flat_system_atomic_fmax_ret_v2f16__offset12b_pos__amdgpu_no_f
; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB52_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: v_mov_b32_e32 v0, v3
; GFX11-NEXT: s_setpc_b64 s[30:31]
;
@@ -14944,6 +15049,7 @@ define void @flat_system_atomic_fmax_noret_v2f16__offset12b_pos__amdgpu_no_fine_
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB53_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -14995,7 +15101,7 @@ define void @flat_system_atomic_fmax_noret_v2f16__offset12b_pos__amdgpu_no_fine_
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v3
; GFX11-NEXT: v_mov_b32_e32 v3, v2
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB53_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -15202,9 +15308,11 @@ define <2 x bfloat> @flat_agent_atomic_fmax_ret_v2bf16__amdgpu_no_fine_grained_m
; GFX12-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_cbranch_execnz .LBB54_1
; GFX12-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: v_mov_b32_e32 v0, v3
; GFX12-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -15253,9 +15361,11 @@ define <2 x bfloat> @flat_agent_atomic_fmax_ret_v2bf16__amdgpu_no_fine_grained_m
; GFX12-FAKE16-NEXT: s_or_b32 s1, vcc_lo, s1
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: s_cbranch_execnz .LBB54_1
; GFX12-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s1
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: v_mov_b32_e32 v0, v3
; GFX12-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -15345,12 +15455,13 @@ define <2 x bfloat> @flat_agent_atomic_fmax_ret_v2bf16__amdgpu_no_fine_grained_m
; GFX11-TRUE16-NEXT: buffer_gl0_inv
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v6
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB54_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-TRUE16-NEXT: s_set_inst_prefetch_distance 0x2
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_mov_b32_e32 v0, v3
; GFX11-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -15394,12 +15505,13 @@ define <2 x bfloat> @flat_agent_atomic_fmax_ret_v2bf16__amdgpu_no_fine_grained_m
; GFX11-FAKE16-NEXT: buffer_gl0_inv
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v6
; GFX11-FAKE16-NEXT: s_or_b32 s1, vcc_lo, s1
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB54_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-FAKE16-NEXT: s_set_inst_prefetch_distance 0x2
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s1
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v0, v3
; GFX11-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -15660,9 +15772,11 @@ define <2 x bfloat> @flat_agent_atomic_fmax_ret_v2bf16__offset12b_pos__amdgpu_no
; GFX12-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_cbranch_execnz .LBB55_1
; GFX12-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: v_mov_b32_e32 v0, v3
; GFX12-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -15711,9 +15825,11 @@ define <2 x bfloat> @flat_agent_atomic_fmax_ret_v2bf16__offset12b_pos__amdgpu_no
; GFX12-FAKE16-NEXT: s_or_b32 s1, vcc_lo, s1
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: s_cbranch_execnz .LBB55_1
; GFX12-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s1
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: v_mov_b32_e32 v0, v3
; GFX12-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -15803,12 +15919,13 @@ define <2 x bfloat> @flat_agent_atomic_fmax_ret_v2bf16__offset12b_pos__amdgpu_no
; GFX11-TRUE16-NEXT: buffer_gl0_inv
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v6
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB55_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-TRUE16-NEXT: s_set_inst_prefetch_distance 0x2
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_mov_b32_e32 v0, v3
; GFX11-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -15852,12 +15969,13 @@ define <2 x bfloat> @flat_agent_atomic_fmax_ret_v2bf16__offset12b_pos__amdgpu_no
; GFX11-FAKE16-NEXT: buffer_gl0_inv
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v6
; GFX11-FAKE16-NEXT: s_or_b32 s1, vcc_lo, s1
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB55_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-FAKE16-NEXT: s_set_inst_prefetch_distance 0x2
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s1
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v0, v3
; GFX11-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -16123,9 +16241,11 @@ define <2 x bfloat> @flat_agent_atomic_fmax_ret_v2bf16__offset12b_neg__amdgpu_no
; GFX12-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_cbranch_execnz .LBB56_1
; GFX12-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: v_mov_b32_e32 v0, v3
; GFX12-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -16174,9 +16294,11 @@ define <2 x bfloat> @flat_agent_atomic_fmax_ret_v2bf16__offset12b_neg__amdgpu_no
; GFX12-FAKE16-NEXT: s_or_b32 s1, vcc_lo, s1
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: s_cbranch_execnz .LBB56_1
; GFX12-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s1
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: v_mov_b32_e32 v0, v3
; GFX12-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -16232,7 +16354,6 @@ define <2 x bfloat> @flat_agent_atomic_fmax_ret_v2bf16__offset12b_neg__amdgpu_no
; GFX11-TRUE16: ; %bb.0:
; GFX11-TRUE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-TRUE16-NEXT: v_add_co_u32 v3, vcc_lo, 0xfffff800, v0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v4, null, -1, v1, vcc_lo
; GFX11-TRUE16-NEXT: v_and_b32_e32 v1, 0xffff0000, v2
; GFX11-TRUE16-NEXT: v_lshlrev_b32_e32 v2, 16, v2
@@ -16274,7 +16395,7 @@ define <2 x bfloat> @flat_agent_atomic_fmax_ret_v2bf16__offset12b_neg__amdgpu_no
; GFX11-TRUE16-NEXT: buffer_gl0_inv
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v0, v6
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB56_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -16286,7 +16407,6 @@ define <2 x bfloat> @flat_agent_atomic_fmax_ret_v2bf16__offset12b_neg__amdgpu_no
; GFX11-FAKE16: ; %bb.0:
; GFX11-FAKE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-FAKE16-NEXT: v_add_co_u32 v3, vcc_lo, 0xfffff800, v0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v4, null, -1, v1, vcc_lo
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v1, 16, v2
; GFX11-FAKE16-NEXT: v_and_b32_e32 v2, 0xffff0000, v2
@@ -16325,7 +16445,7 @@ define <2 x bfloat> @flat_agent_atomic_fmax_ret_v2bf16__offset12b_neg__amdgpu_no
; GFX11-FAKE16-NEXT: buffer_gl0_inv
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v0, v6
; GFX11-FAKE16-NEXT: s_or_b32 s1, vcc_lo, s1
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB56_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -16601,6 +16721,7 @@ define void @flat_agent_atomic_fmax_noret_v2bf16__amdgpu_no_fine_grained_memory(
; GFX12-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_cbranch_execnz .LBB57_1
; GFX12-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -16635,11 +16756,10 @@ define void @flat_agent_atomic_fmax_noret_v2bf16__amdgpu_no_fine_grained_memory(
; GFX12-FAKE16-NEXT: v_add3_u32 v8, v8, v6, 0x7fff
; GFX12-FAKE16-NEXT: v_cmp_u_f32_e64 s0, v2, v2
; GFX12-FAKE16-NEXT: s_wait_alu depctr_va_vcc(0)
-; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_2)
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_2) | instid1(VALU_DEP_1)
; GFX12-FAKE16-NEXT: v_cndmask_b32_e32 v6, v8, v10, vcc_lo
; GFX12-FAKE16-NEXT: s_wait_alu depctr_va_sdst(0)
; GFX12-FAKE16-NEXT: v_cndmask_b32_e64 v2, v7, v9, s0
-; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-FAKE16-NEXT: v_perm_b32 v2, v6, v2, 0x7060302
; GFX12-FAKE16-NEXT: s_wait_storecnt 0x0
; GFX12-FAKE16-NEXT: flat_atomic_cmpswap_b32 v2, v[0:1], v[2:3] th:TH_ATOMIC_RETURN scope:SCOPE_DEV
@@ -16651,6 +16771,7 @@ define void @flat_agent_atomic_fmax_noret_v2bf16__amdgpu_no_fine_grained_memory(
; GFX12-FAKE16-NEXT: s_or_b32 s1, vcc_lo, s1
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: s_cbranch_execnz .LBB57_1
; GFX12-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s1
@@ -16740,7 +16861,7 @@ define void @flat_agent_atomic_fmax_noret_v2bf16__amdgpu_no_fine_grained_memory(
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v3
; GFX11-TRUE16-NEXT: v_mov_b32_e32 v3, v2
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB57_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -16774,7 +16895,7 @@ define void @flat_agent_atomic_fmax_noret_v2bf16__amdgpu_no_fine_grained_memory(
; GFX11-FAKE16-NEXT: v_add3_u32 v7, v7, v2, 0x7fff
; GFX11-FAKE16-NEXT: v_add3_u32 v8, v8, v6, 0x7fff
; GFX11-FAKE16-NEXT: v_cmp_u_f32_e64 s0, v2, v2
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_4)
; GFX11-FAKE16-NEXT: v_cndmask_b32_e32 v6, v8, v10, vcc_lo
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v2, v7, v9, s0
; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
@@ -16787,7 +16908,7 @@ define void @flat_agent_atomic_fmax_noret_v2bf16__amdgpu_no_fine_grained_memory(
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v3
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v3, v2
; GFX11-FAKE16-NEXT: s_or_b32 s1, vcc_lo, s1
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB57_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -17042,6 +17163,7 @@ define void @flat_agent_atomic_fmax_noret_v2bf16__offset12b_pos__amdgpu_no_fine_
; GFX12-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_cbranch_execnz .LBB58_1
; GFX12-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -17076,11 +17198,10 @@ define void @flat_agent_atomic_fmax_noret_v2bf16__offset12b_pos__amdgpu_no_fine_
; GFX12-FAKE16-NEXT: v_add3_u32 v8, v8, v6, 0x7fff
; GFX12-FAKE16-NEXT: v_cmp_u_f32_e64 s0, v2, v2
; GFX12-FAKE16-NEXT: s_wait_alu depctr_va_vcc(0)
-; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_2)
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_2) | instid1(VALU_DEP_1)
; GFX12-FAKE16-NEXT: v_cndmask_b32_e32 v6, v8, v10, vcc_lo
; GFX12-FAKE16-NEXT: s_wait_alu depctr_va_sdst(0)
; GFX12-FAKE16-NEXT: v_cndmask_b32_e64 v2, v7, v9, s0
-; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-FAKE16-NEXT: v_perm_b32 v2, v6, v2, 0x7060302
; GFX12-FAKE16-NEXT: s_wait_storecnt 0x0
; GFX12-FAKE16-NEXT: flat_atomic_cmpswap_b32 v2, v[0:1], v[2:3] offset:2044 th:TH_ATOMIC_RETURN scope:SCOPE_DEV
@@ -17092,6 +17213,7 @@ define void @flat_agent_atomic_fmax_noret_v2bf16__offset12b_pos__amdgpu_no_fine_
; GFX12-FAKE16-NEXT: s_or_b32 s1, vcc_lo, s1
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: s_cbranch_execnz .LBB58_1
; GFX12-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s1
@@ -17181,7 +17303,7 @@ define void @flat_agent_atomic_fmax_noret_v2bf16__offset12b_pos__amdgpu_no_fine_
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v3
; GFX11-TRUE16-NEXT: v_mov_b32_e32 v3, v2
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB58_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -17215,7 +17337,7 @@ define void @flat_agent_atomic_fmax_noret_v2bf16__offset12b_pos__amdgpu_no_fine_
; GFX11-FAKE16-NEXT: v_add3_u32 v7, v7, v2, 0x7fff
; GFX11-FAKE16-NEXT: v_add3_u32 v8, v8, v6, 0x7fff
; GFX11-FAKE16-NEXT: v_cmp_u_f32_e64 s0, v2, v2
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_4)
; GFX11-FAKE16-NEXT: v_cndmask_b32_e32 v6, v8, v10, vcc_lo
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v2, v7, v9, s0
; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
@@ -17228,7 +17350,7 @@ define void @flat_agent_atomic_fmax_noret_v2bf16__offset12b_pos__amdgpu_no_fine_
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v3
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v3, v2
; GFX11-FAKE16-NEXT: s_or_b32 s1, vcc_lo, s1
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB58_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -17490,6 +17612,7 @@ define void @flat_agent_atomic_fmax_noret_v2bf16__offset12b_neg__amdgpu_no_fine_
; GFX12-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_cbranch_execnz .LBB59_1
; GFX12-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -17524,11 +17647,10 @@ define void @flat_agent_atomic_fmax_noret_v2bf16__offset12b_neg__amdgpu_no_fine_
; GFX12-FAKE16-NEXT: v_add3_u32 v8, v8, v6, 0x7fff
; GFX12-FAKE16-NEXT: v_cmp_u_f32_e64 s0, v2, v2
; GFX12-FAKE16-NEXT: s_wait_alu depctr_va_vcc(0)
-; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_2)
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_2) | instid1(VALU_DEP_1)
; GFX12-FAKE16-NEXT: v_cndmask_b32_e32 v6, v8, v10, vcc_lo
; GFX12-FAKE16-NEXT: s_wait_alu depctr_va_sdst(0)
; GFX12-FAKE16-NEXT: v_cndmask_b32_e64 v2, v7, v9, s0
-; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-FAKE16-NEXT: v_perm_b32 v2, v6, v2, 0x7060302
; GFX12-FAKE16-NEXT: s_wait_storecnt 0x0
; GFX12-FAKE16-NEXT: flat_atomic_cmpswap_b32 v2, v[0:1], v[2:3] offset:-2048 th:TH_ATOMIC_RETURN scope:SCOPE_DEV
@@ -17540,6 +17662,7 @@ define void @flat_agent_atomic_fmax_noret_v2bf16__offset12b_neg__amdgpu_no_fine_
; GFX12-FAKE16-NEXT: s_or_b32 s1, vcc_lo, s1
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: s_cbranch_execnz .LBB59_1
; GFX12-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s1
@@ -17597,7 +17720,6 @@ define void @flat_agent_atomic_fmax_noret_v2bf16__offset12b_neg__amdgpu_no_fine_
; GFX11-TRUE16: ; %bb.0:
; GFX11-TRUE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-TRUE16-NEXT: v_add_co_u32 v0, vcc_lo, 0xfffff800, v0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v1, null, -1, v1, vcc_lo
; GFX11-TRUE16-NEXT: v_and_b32_e32 v4, 0xffff0000, v2
; GFX11-TRUE16-NEXT: v_lshlrev_b32_e32 v5, 16, v2
@@ -17638,7 +17760,7 @@ define void @flat_agent_atomic_fmax_noret_v2bf16__offset12b_neg__amdgpu_no_fine_
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v3
; GFX11-TRUE16-NEXT: v_mov_b32_e32 v3, v2
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB59_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -17650,7 +17772,6 @@ define void @flat_agent_atomic_fmax_noret_v2bf16__offset12b_neg__amdgpu_no_fine_
; GFX11-FAKE16: ; %bb.0:
; GFX11-FAKE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-FAKE16-NEXT: v_add_co_u32 v0, vcc_lo, 0xfffff800, v0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v1, null, -1, v1, vcc_lo
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v4, 16, v2
; GFX11-FAKE16-NEXT: v_and_b32_e32 v5, 0xffff0000, v2
@@ -17675,7 +17796,7 @@ define void @flat_agent_atomic_fmax_noret_v2bf16__offset12b_neg__amdgpu_no_fine_
; GFX11-FAKE16-NEXT: v_add3_u32 v7, v7, v2, 0x7fff
; GFX11-FAKE16-NEXT: v_add3_u32 v8, v8, v6, 0x7fff
; GFX11-FAKE16-NEXT: v_cmp_u_f32_e64 s0, v2, v2
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_4)
; GFX11-FAKE16-NEXT: v_cndmask_b32_e32 v6, v8, v10, vcc_lo
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v2, v7, v9, s0
; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
@@ -17688,7 +17809,7 @@ define void @flat_agent_atomic_fmax_noret_v2bf16__offset12b_neg__amdgpu_no_fine_
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v3
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v3, v2
; GFX11-FAKE16-NEXT: s_or_b32 s1, vcc_lo, s1
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB59_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -17963,9 +18084,11 @@ define <2 x bfloat> @flat_system_atomic_fmax_ret_v2bf16__offset12b_pos__amdgpu_n
; GFX12-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_cbranch_execnz .LBB60_1
; GFX12-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: v_mov_b32_e32 v0, v3
; GFX12-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -18015,9 +18138,11 @@ define <2 x bfloat> @flat_system_atomic_fmax_ret_v2bf16__offset12b_pos__amdgpu_n
; GFX12-FAKE16-NEXT: s_or_b32 s1, vcc_lo, s1
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: s_cbranch_execnz .LBB60_1
; GFX12-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s1
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: v_mov_b32_e32 v0, v3
; GFX12-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -18107,12 +18232,13 @@ define <2 x bfloat> @flat_system_atomic_fmax_ret_v2bf16__offset12b_pos__amdgpu_n
; GFX11-TRUE16-NEXT: buffer_gl0_inv
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v6
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB60_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-TRUE16-NEXT: s_set_inst_prefetch_distance 0x2
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_mov_b32_e32 v0, v3
; GFX11-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -18156,12 +18282,13 @@ define <2 x bfloat> @flat_system_atomic_fmax_ret_v2bf16__offset12b_pos__amdgpu_n
; GFX11-FAKE16-NEXT: buffer_gl0_inv
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v6
; GFX11-FAKE16-NEXT: s_or_b32 s1, vcc_lo, s1
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB60_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-FAKE16-NEXT: s_set_inst_prefetch_distance 0x2
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s1
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v0, v3
; GFX11-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -18429,6 +18556,7 @@ define void @flat_system_atomic_fmax_noret_v2bf16__offset12b_pos__amdgpu_no_fine
; GFX12-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_cbranch_execnz .LBB61_1
; GFX12-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -18463,11 +18591,10 @@ define void @flat_system_atomic_fmax_noret_v2bf16__offset12b_pos__amdgpu_no_fine
; GFX12-FAKE16-NEXT: v_add3_u32 v8, v8, v6, 0x7fff
; GFX12-FAKE16-NEXT: v_cmp_u_f32_e64 s0, v2, v2
; GFX12-FAKE16-NEXT: s_wait_alu depctr_va_vcc(0)
-; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_2)
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_2) | instid1(VALU_DEP_1)
; GFX12-FAKE16-NEXT: v_cndmask_b32_e32 v6, v8, v10, vcc_lo
; GFX12-FAKE16-NEXT: s_wait_alu depctr_va_sdst(0)
; GFX12-FAKE16-NEXT: v_cndmask_b32_e64 v2, v7, v9, s0
-; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-FAKE16-NEXT: v_perm_b32 v2, v6, v2, 0x7060302
; GFX12-FAKE16-NEXT: global_wb scope:SCOPE_SYS
; GFX12-FAKE16-NEXT: s_wait_storecnt 0x0
@@ -18480,6 +18607,7 @@ define void @flat_system_atomic_fmax_noret_v2bf16__offset12b_pos__amdgpu_no_fine
; GFX12-FAKE16-NEXT: s_or_b32 s1, vcc_lo, s1
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: s_cbranch_execnz .LBB61_1
; GFX12-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s1
@@ -18569,7 +18697,7 @@ define void @flat_system_atomic_fmax_noret_v2bf16__offset12b_pos__amdgpu_no_fine
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v3
; GFX11-TRUE16-NEXT: v_mov_b32_e32 v3, v2
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB61_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -18603,7 +18731,7 @@ define void @flat_system_atomic_fmax_noret_v2bf16__offset12b_pos__amdgpu_no_fine
; GFX11-FAKE16-NEXT: v_add3_u32 v7, v7, v2, 0x7fff
; GFX11-FAKE16-NEXT: v_add3_u32 v8, v8, v6, 0x7fff
; GFX11-FAKE16-NEXT: v_cmp_u_f32_e64 s0, v2, v2
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_4)
; GFX11-FAKE16-NEXT: v_cndmask_b32_e32 v6, v8, v10, vcc_lo
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v2, v7, v9, s0
; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
@@ -18616,7 +18744,7 @@ define void @flat_system_atomic_fmax_noret_v2bf16__offset12b_pos__amdgpu_no_fine
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v3
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v3, v2
; GFX11-FAKE16-NEXT: s_or_b32 s1, vcc_lo, s1
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB61_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
diff --git a/llvm/test/CodeGen/AMDGPU/flat-atomicrmw-fmin.ll b/llvm/test/CodeGen/AMDGPU/flat-atomicrmw-fmin.ll
index f5564f14eee24f..48a5831f9d0f0e 100644
--- a/llvm/test/CodeGen/AMDGPU/flat-atomicrmw-fmin.ll
+++ b/llvm/test/CodeGen/AMDGPU/flat-atomicrmw-fmin.ll
@@ -355,7 +355,6 @@ define float @flat_agent_atomic_fmin_ret_f32__offset12b_neg__amdgpu_no_fine_grai
; GFX11: ; %bb.0:
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: v_add_co_u32 v0, vcc_lo, 0xfffff800, v0
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, -1, v1, vcc_lo
; GFX11-NEXT: s_waitcnt_vscnt null, 0x0
; GFX11-NEXT: flat_atomic_min_f32 v0, v[0:1], v2 glc
@@ -809,7 +808,6 @@ define void @flat_agent_atomic_fmin_noret_f32__offset12b_neg__amdgpu_no_fine_gra
; GFX11: ; %bb.0:
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: v_add_co_u32 v0, vcc_lo, 0xfffff800, v0
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, -1, v1, vcc_lo
; GFX11-NEXT: s_waitcnt_vscnt null, 0x0
; GFX11-NEXT: flat_atomic_min_f32 v[0:1], v2
@@ -1292,11 +1290,12 @@ define float @flat_agent_atomic_fmin_ret_f32__amdgpu_no_remote_memory(ptr %ptr,
; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB8_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: v_mov_b32_e32 v0, v3
; GFX11-NEXT: s_setpc_b64 s[30:31]
;
@@ -1913,7 +1912,6 @@ define float @flat_agent_atomic_fmin_ret_f32__offset12b_neg__ftz__amdgpu_no_fine
; GFX11: ; %bb.0:
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: v_add_co_u32 v0, vcc_lo, 0xfffff800, v0
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, -1, v1, vcc_lo
; GFX11-NEXT: s_waitcnt_vscnt null, 0x0
; GFX11-NEXT: flat_atomic_min_f32 v0, v[0:1], v2 glc
@@ -2367,7 +2365,6 @@ define void @flat_agent_atomic_fmin_noret_f32__offset12b_neg__ftz__amdgpu_no_fin
; GFX11: ; %bb.0:
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: v_add_co_u32 v0, vcc_lo, 0xfffff800, v0
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, -1, v1, vcc_lo
; GFX11-NEXT: s_waitcnt_vscnt null, 0x0
; GFX11-NEXT: flat_atomic_min_f32 v[0:1], v2
@@ -2808,6 +2805,7 @@ define double @flat_agent_atomic_fmin_ret_f64__amdgpu_no_fine_grained_memory(ptr
; GFX12-NEXT: ; implicit-def: $vgpr2_vgpr3
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: v_cmpx_ne_u32_e32 s1, v1
+; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX12-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX12-NEXT: s_cbranch_execz .LBB18_4
; GFX12-NEXT: ; %bb.1: ; %atomicrmw.global
@@ -2829,6 +2827,7 @@ define double @flat_agent_atomic_fmin_ret_f64__amdgpu_no_fine_grained_memory(ptr
; GFX12-NEXT: s_or_b32 s1, vcc_lo, s1
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB18_2
; GFX12-NEXT: ; %bb.3: ; %Flow
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s1
@@ -2851,6 +2850,7 @@ define double @flat_agent_atomic_fmin_ret_f64__amdgpu_no_fine_grained_memory(ptr
; GFX12-NEXT: .LBB18_6: ; %atomicrmw.phi
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: v_dual_mov_b32 v0, v2 :: v_dual_mov_b32 v1, v3
; GFX12-NEXT: s_setpc_b64 s[30:31]
;
@@ -2903,6 +2903,7 @@ define double @flat_agent_atomic_fmin_ret_f64__amdgpu_no_fine_grained_memory(ptr
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: ; implicit-def: $vgpr2_vgpr3
; GFX11-NEXT: v_cmpx_ne_u32_e32 s1, v1
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX11-NEXT: s_cbranch_execz .LBB18_4
; GFX11-NEXT: ; %bb.1: ; %atomicrmw.global
@@ -2922,7 +2923,7 @@ define double @flat_agent_atomic_fmin_ret_f64__amdgpu_no_fine_grained_memory(ptr
; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: v_cmp_eq_u64_e32 vcc_lo, v[2:3], v[8:9]
; GFX11-NEXT: s_or_b32 s1, vcc_lo, s1
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
; GFX11-NEXT: s_cbranch_execnz .LBB18_2
; GFX11-NEXT: ; %bb.3: ; %Flow
@@ -2930,6 +2931,7 @@ define double @flat_agent_atomic_fmin_ret_f64__amdgpu_no_fine_grained_memory(ptr
; GFX11-NEXT: ; implicit-def: $vgpr0_vgpr1
; GFX11-NEXT: ; implicit-def: $vgpr4_vgpr5
; GFX11-NEXT: .LBB18_4: ; %Flow2
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB18_6
; GFX11-NEXT: ; %bb.5: ; %atomicrmw.private
@@ -2943,6 +2945,7 @@ define double @flat_agent_atomic_fmin_ret_f64__amdgpu_no_fine_grained_memory(ptr
; GFX11-NEXT: scratch_store_b64 v6, v[0:1], off
; GFX11-NEXT: .LBB18_6: ; %atomicrmw.phi
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: v_dual_mov_b32 v0, v2 :: v_dual_mov_b32 v1, v3
; GFX11-NEXT: s_setpc_b64 s[30:31]
;
@@ -3201,6 +3204,7 @@ define double @flat_agent_atomic_fmin_ret_f64__offset12b_pos__amdgpu_no_fine_gra
; GFX12-NEXT: ; implicit-def: $vgpr0_vgpr1
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: v_cmpx_ne_u32_e32 s1, v5
+; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX12-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX12-NEXT: s_cbranch_execnz .LBB19_3
; GFX12-NEXT: ; %bb.1: ; %Flow2
@@ -3230,11 +3234,13 @@ define double @flat_agent_atomic_fmin_ret_f64__offset12b_pos__amdgpu_no_fine_gra
; GFX12-NEXT: s_or_b32 s1, vcc_lo, s1
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB19_4
; GFX12-NEXT: ; %bb.5: ; %Flow
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s1
; GFX12-NEXT: ; implicit-def: $vgpr4_vgpr5
; GFX12-NEXT: ; implicit-def: $vgpr2_vgpr3
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX12-NEXT: s_cbranch_execz .LBB19_2
; GFX12-NEXT: .LBB19_6: ; %atomicrmw.private
@@ -3297,12 +3303,12 @@ define double @flat_agent_atomic_fmin_ret_f64__offset12b_pos__amdgpu_no_fine_gra
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: v_max_f64 v[2:3], v[2:3], v[2:3]
; GFX11-NEXT: v_add_co_u32 v4, vcc_lo, 0x7f8, v0
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v5, null, 0, v1, vcc_lo
; GFX11-NEXT: s_mov_b64 s[0:1], src_private_base
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: ; implicit-def: $vgpr0_vgpr1
; GFX11-NEXT: v_cmpx_ne_u32_e32 s1, v5
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB19_3
; GFX11-NEXT: ; %bb.1: ; %Flow2
@@ -3328,13 +3334,14 @@ define double @flat_agent_atomic_fmin_ret_f64__offset12b_pos__amdgpu_no_fine_gra
; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: v_cmp_eq_u64_e32 vcc_lo, v[0:1], v[8:9]
; GFX11-NEXT: s_or_b32 s1, vcc_lo, s1
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
; GFX11-NEXT: s_cbranch_execnz .LBB19_4
; GFX11-NEXT: ; %bb.5: ; %Flow
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s1
; GFX11-NEXT: ; implicit-def: $vgpr4_vgpr5
; GFX11-NEXT: ; implicit-def: $vgpr2_vgpr3
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB19_2
; GFX11-NEXT: .LBB19_6: ; %atomicrmw.private
@@ -3613,6 +3620,7 @@ define double @flat_agent_atomic_fmin_ret_f64__offset12b_neg__amdgpu_no_fine_gra
; GFX12-NEXT: ; implicit-def: $vgpr0_vgpr1
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: v_cmpx_ne_u32_e32 s1, v5
+; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX12-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX12-NEXT: s_cbranch_execnz .LBB20_3
; GFX12-NEXT: ; %bb.1: ; %Flow2
@@ -3642,11 +3650,13 @@ define double @flat_agent_atomic_fmin_ret_f64__offset12b_neg__amdgpu_no_fine_gra
; GFX12-NEXT: s_or_b32 s1, vcc_lo, s1
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB20_4
; GFX12-NEXT: ; %bb.5: ; %Flow
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s1
; GFX12-NEXT: ; implicit-def: $vgpr4_vgpr5
; GFX12-NEXT: ; implicit-def: $vgpr2_vgpr3
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX12-NEXT: s_cbranch_execz .LBB20_2
; GFX12-NEXT: .LBB20_6: ; %atomicrmw.private
@@ -3710,12 +3720,12 @@ define double @flat_agent_atomic_fmin_ret_f64__offset12b_neg__amdgpu_no_fine_gra
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: v_max_f64 v[2:3], v[2:3], v[2:3]
; GFX11-NEXT: v_add_co_u32 v4, vcc_lo, 0xfffff800, v0
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v5, null, -1, v1, vcc_lo
; GFX11-NEXT: s_mov_b64 s[0:1], src_private_base
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: ; implicit-def: $vgpr0_vgpr1
; GFX11-NEXT: v_cmpx_ne_u32_e32 s1, v5
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB20_3
; GFX11-NEXT: ; %bb.1: ; %Flow2
@@ -3741,13 +3751,14 @@ define double @flat_agent_atomic_fmin_ret_f64__offset12b_neg__amdgpu_no_fine_gra
; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: v_cmp_eq_u64_e32 vcc_lo, v[0:1], v[8:9]
; GFX11-NEXT: s_or_b32 s1, vcc_lo, s1
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
; GFX11-NEXT: s_cbranch_execnz .LBB20_4
; GFX11-NEXT: ; %bb.5: ; %Flow
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s1
; GFX11-NEXT: ; implicit-def: $vgpr4_vgpr5
; GFX11-NEXT: ; implicit-def: $vgpr2_vgpr3
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB20_2
; GFX11-NEXT: .LBB20_6: ; %atomicrmw.private
@@ -4022,6 +4033,7 @@ define void @flat_agent_atomic_fmin_noret_f64__amdgpu_no_fine_grained_memory(ptr
; GFX12-NEXT: s_mov_b32 s0, exec_lo
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: v_cmpx_ne_u32_e32 s1, v1
+; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX12-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX12-NEXT: s_cbranch_execnz .LBB21_3
; GFX12-NEXT: ; %bb.1: ; %Flow2
@@ -4051,11 +4063,13 @@ define void @flat_agent_atomic_fmin_noret_f64__amdgpu_no_fine_grained_memory(ptr
; GFX12-NEXT: s_or_b32 s1, vcc_lo, s1
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB21_4
; GFX12-NEXT: ; %bb.5: ; %Flow
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s1
; GFX12-NEXT: ; implicit-def: $vgpr0_vgpr1
; GFX12-NEXT: ; implicit-def: $vgpr6_vgpr7
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX12-NEXT: s_cbranch_execz .LBB21_2
; GFX12-NEXT: .LBB21_6: ; %atomicrmw.private
@@ -4117,6 +4131,7 @@ define void @flat_agent_atomic_fmin_noret_f64__amdgpu_no_fine_grained_memory(ptr
; GFX11-NEXT: s_mov_b64 s[0:1], src_private_base
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 s1, v1
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB21_3
; GFX11-NEXT: ; %bb.1: ; %Flow2
@@ -4142,13 +4157,14 @@ define void @flat_agent_atomic_fmin_noret_f64__amdgpu_no_fine_grained_memory(ptr
; GFX11-NEXT: v_cmp_eq_u64_e32 vcc_lo, v[2:3], v[4:5]
; GFX11-NEXT: v_dual_mov_b32 v5, v3 :: v_dual_mov_b32 v4, v2
; GFX11-NEXT: s_or_b32 s1, vcc_lo, s1
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
; GFX11-NEXT: s_cbranch_execnz .LBB21_4
; GFX11-NEXT: ; %bb.5: ; %Flow
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s1
; GFX11-NEXT: ; implicit-def: $vgpr0_vgpr1
; GFX11-NEXT: ; implicit-def: $vgpr6_vgpr7
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB21_2
; GFX11-NEXT: .LBB21_6: ; %atomicrmw.private
@@ -4411,6 +4427,7 @@ define void @flat_agent_atomic_fmin_noret_f64__offset12b_pos__amdgpu_no_fine_gra
; GFX12-NEXT: s_mov_b32 s0, exec_lo
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: v_cmpx_ne_u32_e32 s1, v7
+; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX12-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX12-NEXT: s_cbranch_execnz .LBB22_3
; GFX12-NEXT: ; %bb.1: ; %Flow2
@@ -4440,11 +4457,13 @@ define void @flat_agent_atomic_fmin_noret_f64__offset12b_pos__amdgpu_no_fine_gra
; GFX12-NEXT: s_or_b32 s1, vcc_lo, s1
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB22_4
; GFX12-NEXT: ; %bb.5: ; %Flow
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s1
; GFX12-NEXT: ; implicit-def: $vgpr6_vgpr7
; GFX12-NEXT: ; implicit-def: $vgpr4_vgpr5
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX12-NEXT: s_cbranch_execz .LBB22_2
; GFX12-NEXT: .LBB22_6: ; %atomicrmw.private
@@ -4506,11 +4525,11 @@ define void @flat_agent_atomic_fmin_noret_f64__offset12b_pos__amdgpu_no_fine_gra
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: v_max_f64 v[4:5], v[2:3], v[2:3]
; GFX11-NEXT: v_add_co_u32 v6, vcc_lo, 0x7f8, v0
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v7, null, 0, v1, vcc_lo
; GFX11-NEXT: s_mov_b64 s[0:1], src_private_base
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 s1, v7
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB22_3
; GFX11-NEXT: ; %bb.1: ; %Flow2
@@ -4536,13 +4555,14 @@ define void @flat_agent_atomic_fmin_noret_f64__offset12b_pos__amdgpu_no_fine_gra
; GFX11-NEXT: v_cmp_eq_u64_e32 vcc_lo, v[0:1], v[2:3]
; GFX11-NEXT: v_dual_mov_b32 v3, v1 :: v_dual_mov_b32 v2, v0
; GFX11-NEXT: s_or_b32 s1, vcc_lo, s1
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
; GFX11-NEXT: s_cbranch_execnz .LBB22_4
; GFX11-NEXT: ; %bb.5: ; %Flow
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s1
; GFX11-NEXT: ; implicit-def: $vgpr6_vgpr7
; GFX11-NEXT: ; implicit-def: $vgpr4_vgpr5
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB22_2
; GFX11-NEXT: .LBB22_6: ; %atomicrmw.private
@@ -4816,6 +4836,7 @@ define void @flat_agent_atomic_fmin_noret_f64__offset12b_neg__amdgpu_no_fine_gra
; GFX12-NEXT: s_mov_b32 s0, exec_lo
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: v_cmpx_ne_u32_e32 s1, v7
+; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX12-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX12-NEXT: s_cbranch_execnz .LBB23_3
; GFX12-NEXT: ; %bb.1: ; %Flow2
@@ -4845,11 +4866,13 @@ define void @flat_agent_atomic_fmin_noret_f64__offset12b_neg__amdgpu_no_fine_gra
; GFX12-NEXT: s_or_b32 s1, vcc_lo, s1
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB23_4
; GFX12-NEXT: ; %bb.5: ; %Flow
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s1
; GFX12-NEXT: ; implicit-def: $vgpr6_vgpr7
; GFX12-NEXT: ; implicit-def: $vgpr4_vgpr5
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX12-NEXT: s_cbranch_execz .LBB23_2
; GFX12-NEXT: .LBB23_6: ; %atomicrmw.private
@@ -4912,11 +4935,11 @@ define void @flat_agent_atomic_fmin_noret_f64__offset12b_neg__amdgpu_no_fine_gra
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: v_max_f64 v[4:5], v[2:3], v[2:3]
; GFX11-NEXT: v_add_co_u32 v6, vcc_lo, 0xfffff800, v0
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v7, null, -1, v1, vcc_lo
; GFX11-NEXT: s_mov_b64 s[0:1], src_private_base
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 s1, v7
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB23_3
; GFX11-NEXT: ; %bb.1: ; %Flow2
@@ -4942,13 +4965,14 @@ define void @flat_agent_atomic_fmin_noret_f64__offset12b_neg__amdgpu_no_fine_gra
; GFX11-NEXT: v_cmp_eq_u64_e32 vcc_lo, v[0:1], v[2:3]
; GFX11-NEXT: v_dual_mov_b32 v3, v1 :: v_dual_mov_b32 v2, v0
; GFX11-NEXT: s_or_b32 s1, vcc_lo, s1
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
; GFX11-NEXT: s_cbranch_execnz .LBB23_4
; GFX11-NEXT: ; %bb.5: ; %Flow
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s1
; GFX11-NEXT: ; implicit-def: $vgpr6_vgpr7
; GFX11-NEXT: ; implicit-def: $vgpr4_vgpr5
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB23_2
; GFX11-NEXT: .LBB23_6: ; %atomicrmw.private
@@ -5220,6 +5244,7 @@ define double @flat_agent_atomic_fmin_ret_f64__amdgpu_no_remote_memory(ptr %ptr,
; GFX12-NEXT: ; implicit-def: $vgpr2_vgpr3
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: v_cmpx_ne_u32_e32 s1, v1
+; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX12-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX12-NEXT: s_cbranch_execz .LBB24_4
; GFX12-NEXT: ; %bb.1: ; %atomicrmw.global
@@ -5241,6 +5266,7 @@ define double @flat_agent_atomic_fmin_ret_f64__amdgpu_no_remote_memory(ptr %ptr,
; GFX12-NEXT: s_or_b32 s1, vcc_lo, s1
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB24_2
; GFX12-NEXT: ; %bb.3: ; %Flow
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s1
@@ -5263,6 +5289,7 @@ define double @flat_agent_atomic_fmin_ret_f64__amdgpu_no_remote_memory(ptr %ptr,
; GFX12-NEXT: .LBB24_6: ; %atomicrmw.phi
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: v_dual_mov_b32 v0, v2 :: v_dual_mov_b32 v1, v3
; GFX12-NEXT: s_setpc_b64 s[30:31]
;
@@ -5315,6 +5342,7 @@ define double @flat_agent_atomic_fmin_ret_f64__amdgpu_no_remote_memory(ptr %ptr,
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: ; implicit-def: $vgpr2_vgpr3
; GFX11-NEXT: v_cmpx_ne_u32_e32 s1, v1
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX11-NEXT: s_cbranch_execz .LBB24_4
; GFX11-NEXT: ; %bb.1: ; %atomicrmw.global
@@ -5334,7 +5362,7 @@ define double @flat_agent_atomic_fmin_ret_f64__amdgpu_no_remote_memory(ptr %ptr,
; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: v_cmp_eq_u64_e32 vcc_lo, v[2:3], v[8:9]
; GFX11-NEXT: s_or_b32 s1, vcc_lo, s1
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
; GFX11-NEXT: s_cbranch_execnz .LBB24_2
; GFX11-NEXT: ; %bb.3: ; %Flow
@@ -5342,6 +5370,7 @@ define double @flat_agent_atomic_fmin_ret_f64__amdgpu_no_remote_memory(ptr %ptr,
; GFX11-NEXT: ; implicit-def: $vgpr0_vgpr1
; GFX11-NEXT: ; implicit-def: $vgpr4_vgpr5
; GFX11-NEXT: .LBB24_4: ; %Flow2
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB24_6
; GFX11-NEXT: ; %bb.5: ; %atomicrmw.private
@@ -5355,6 +5384,7 @@ define double @flat_agent_atomic_fmin_ret_f64__amdgpu_no_remote_memory(ptr %ptr,
; GFX11-NEXT: scratch_store_b64 v6, v[0:1], off
; GFX11-NEXT: .LBB24_6: ; %atomicrmw.phi
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: v_dual_mov_b32 v0, v2 :: v_dual_mov_b32 v1, v3
; GFX11-NEXT: s_setpc_b64 s[30:31]
;
@@ -5644,6 +5674,7 @@ define double @flat_agent_atomic_fmin_ret_f64__amdgpu_no_fine_grained_memory__am
; GFX12-NEXT: ; implicit-def: $vgpr2_vgpr3
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: v_cmpx_ne_u32_e32 s1, v1
+; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX12-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX12-NEXT: s_cbranch_execz .LBB25_4
; GFX12-NEXT: ; %bb.1: ; %atomicrmw.global
@@ -5665,6 +5696,7 @@ define double @flat_agent_atomic_fmin_ret_f64__amdgpu_no_fine_grained_memory__am
; GFX12-NEXT: s_or_b32 s1, vcc_lo, s1
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB25_2
; GFX12-NEXT: ; %bb.3: ; %Flow
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s1
@@ -5687,6 +5719,7 @@ define double @flat_agent_atomic_fmin_ret_f64__amdgpu_no_fine_grained_memory__am
; GFX12-NEXT: .LBB25_6: ; %atomicrmw.phi
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: v_dual_mov_b32 v0, v2 :: v_dual_mov_b32 v1, v3
; GFX12-NEXT: s_setpc_b64 s[30:31]
;
@@ -5739,6 +5772,7 @@ define double @flat_agent_atomic_fmin_ret_f64__amdgpu_no_fine_grained_memory__am
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: ; implicit-def: $vgpr2_vgpr3
; GFX11-NEXT: v_cmpx_ne_u32_e32 s1, v1
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX11-NEXT: s_cbranch_execz .LBB25_4
; GFX11-NEXT: ; %bb.1: ; %atomicrmw.global
@@ -5758,7 +5792,7 @@ define double @flat_agent_atomic_fmin_ret_f64__amdgpu_no_fine_grained_memory__am
; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: v_cmp_eq_u64_e32 vcc_lo, v[2:3], v[8:9]
; GFX11-NEXT: s_or_b32 s1, vcc_lo, s1
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
; GFX11-NEXT: s_cbranch_execnz .LBB25_2
; GFX11-NEXT: ; %bb.3: ; %Flow
@@ -5766,6 +5800,7 @@ define double @flat_agent_atomic_fmin_ret_f64__amdgpu_no_fine_grained_memory__am
; GFX11-NEXT: ; implicit-def: $vgpr0_vgpr1
; GFX11-NEXT: ; implicit-def: $vgpr4_vgpr5
; GFX11-NEXT: .LBB25_4: ; %Flow2
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB25_6
; GFX11-NEXT: ; %bb.5: ; %atomicrmw.private
@@ -5779,6 +5814,7 @@ define double @flat_agent_atomic_fmin_ret_f64__amdgpu_no_fine_grained_memory__am
; GFX11-NEXT: scratch_store_b64 v6, v[0:1], off
; GFX11-NEXT: .LBB25_6: ; %atomicrmw.phi
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: v_dual_mov_b32 v0, v2 :: v_dual_mov_b32 v1, v3
; GFX11-NEXT: s_setpc_b64 s[30:31]
;
@@ -6065,9 +6101,11 @@ define half @flat_agent_atomic_fmin_ret_f16__amdgpu_no_fine_grained_memory(ptr %
; GFX12-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_cbranch_execnz .LBB26_1
; GFX12-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: v_lshrrev_b32_e32 v0, v3, v5
; GFX12-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -6111,9 +6149,11 @@ define half @flat_agent_atomic_fmin_ret_f16__amdgpu_no_fine_grained_memory(ptr %
; GFX12-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: s_cbranch_execnz .LBB26_1
; GFX12-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: v_lshrrev_b32_e32 v0, v3, v5
; GFX12-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -6187,11 +6227,12 @@ define half @flat_agent_atomic_fmin_ret_f16__amdgpu_no_fine_grained_memory(ptr %
; GFX11-TRUE16-NEXT: buffer_gl0_inv
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v5, v6
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB26_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v0, v3, v5
; GFX11-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -6229,11 +6270,12 @@ define half @flat_agent_atomic_fmin_ret_f16__amdgpu_no_fine_grained_memory(ptr %
; GFX11-FAKE16-NEXT: buffer_gl0_inv
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v5, v6
; GFX11-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB26_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v0, v3, v5
; GFX11-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -6457,9 +6499,11 @@ define half @flat_agent_atomic_fmin_ret_f16__offset12b_pos__amdgpu_no_fine_grain
; GFX12-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_cbranch_execnz .LBB27_1
; GFX12-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: v_lshrrev_b32_e32 v0, v1, v5
; GFX12-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -6505,9 +6549,11 @@ define half @flat_agent_atomic_fmin_ret_f16__offset12b_pos__amdgpu_no_fine_grain
; GFX12-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: s_cbranch_execnz .LBB27_1
; GFX12-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: v_lshrrev_b32_e32 v0, v3, v5
; GFX12-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -6553,7 +6599,6 @@ define half @flat_agent_atomic_fmin_ret_f16__offset12b_pos__amdgpu_no_fine_grain
; GFX11-TRUE16: ; %bb.0:
; GFX11-TRUE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-TRUE16-NEXT: v_add_co_u32 v0, vcc_lo, 0x7fe, v0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v4, null, 0, v1, vcc_lo
; GFX11-TRUE16-NEXT: s_mov_b32 s0, 0
; GFX11-TRUE16-NEXT: v_and_b32_e32 v3, -4, v0
@@ -6586,11 +6631,12 @@ define half @flat_agent_atomic_fmin_ret_f16__offset12b_pos__amdgpu_no_fine_grain
; GFX11-TRUE16-NEXT: buffer_gl0_inv
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v5, v6
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB27_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v0, v1, v5
; GFX11-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -6598,16 +6644,16 @@ define half @flat_agent_atomic_fmin_ret_f16__offset12b_pos__amdgpu_no_fine_grain
; GFX11-FAKE16: ; %bb.0:
; GFX11-FAKE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-FAKE16-NEXT: v_add_co_u32 v3, vcc_lo, 0x7fe, v0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-FAKE16-NEXT: v_max_f16_e32 v2, v2, v2
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_4) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_and_b32_e32 v0, -4, v3
; GFX11-FAKE16-NEXT: v_and_b32_e32 v3, 3, v3
; GFX11-FAKE16-NEXT: s_mov_b32 s0, 0
; GFX11-FAKE16-NEXT: flat_load_b32 v5, v[0:1]
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v3, 3, v3
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e64 v4, v3, 0xffff
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_not_b32_e32 v4, v4
; GFX11-FAKE16-NEXT: .LBB27_1: ; %atomicrmw.start
; GFX11-FAKE16-NEXT: ; =>This Inner Loop Header: Depth=1
@@ -6629,11 +6675,12 @@ define half @flat_agent_atomic_fmin_ret_f16__offset12b_pos__amdgpu_no_fine_grain
; GFX11-FAKE16-NEXT: buffer_gl0_inv
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v5, v6
; GFX11-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB27_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v0, v3, v5
; GFX11-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -6863,9 +6910,11 @@ define half @flat_agent_atomic_fmin_ret_f16__offset12b_neg__amdgpu_no_fine_grain
; GFX12-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_cbranch_execnz .LBB28_1
; GFX12-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: v_lshrrev_b32_e32 v0, v1, v5
; GFX12-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -6911,9 +6960,11 @@ define half @flat_agent_atomic_fmin_ret_f16__offset12b_neg__amdgpu_no_fine_grain
; GFX12-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: s_cbranch_execnz .LBB28_1
; GFX12-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: v_lshrrev_b32_e32 v0, v3, v5
; GFX12-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -6960,7 +7011,6 @@ define half @flat_agent_atomic_fmin_ret_f16__offset12b_neg__amdgpu_no_fine_grain
; GFX11-TRUE16: ; %bb.0:
; GFX11-TRUE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-TRUE16-NEXT: v_add_co_u32 v0, vcc_lo, 0xfffff800, v0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v4, null, -1, v1, vcc_lo
; GFX11-TRUE16-NEXT: s_mov_b32 s0, 0
; GFX11-TRUE16-NEXT: v_and_b32_e32 v3, -4, v0
@@ -6993,11 +7043,12 @@ define half @flat_agent_atomic_fmin_ret_f16__offset12b_neg__amdgpu_no_fine_grain
; GFX11-TRUE16-NEXT: buffer_gl0_inv
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v5, v6
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB28_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v0, v1, v5
; GFX11-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -7005,16 +7056,16 @@ define half @flat_agent_atomic_fmin_ret_f16__offset12b_neg__amdgpu_no_fine_grain
; GFX11-FAKE16: ; %bb.0:
; GFX11-FAKE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-FAKE16-NEXT: v_add_co_u32 v3, vcc_lo, 0xfffff800, v0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v1, null, -1, v1, vcc_lo
; GFX11-FAKE16-NEXT: v_max_f16_e32 v2, v2, v2
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_4) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_and_b32_e32 v0, -4, v3
; GFX11-FAKE16-NEXT: v_and_b32_e32 v3, 3, v3
; GFX11-FAKE16-NEXT: s_mov_b32 s0, 0
; GFX11-FAKE16-NEXT: flat_load_b32 v5, v[0:1]
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v3, 3, v3
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e64 v4, v3, 0xffff
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_not_b32_e32 v4, v4
; GFX11-FAKE16-NEXT: .LBB28_1: ; %atomicrmw.start
; GFX11-FAKE16-NEXT: ; =>This Inner Loop Header: Depth=1
@@ -7036,11 +7087,12 @@ define half @flat_agent_atomic_fmin_ret_f16__offset12b_neg__amdgpu_no_fine_grain
; GFX11-FAKE16-NEXT: buffer_gl0_inv
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v5, v6
; GFX11-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB28_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v0, v3, v5
; GFX11-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -7267,6 +7319,7 @@ define void @flat_agent_atomic_fmin_noret_f16__amdgpu_no_fine_grained_memory(ptr
; GFX12-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_cbranch_execnz .LBB29_1
; GFX12-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -7312,6 +7365,7 @@ define void @flat_agent_atomic_fmin_noret_f16__amdgpu_no_fine_grained_memory(ptr
; GFX12-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: s_cbranch_execnz .LBB29_1
; GFX12-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -7386,7 +7440,7 @@ define void @flat_agent_atomic_fmin_noret_f16__amdgpu_no_fine_grained_memory(ptr
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-TRUE16-NEXT: v_mov_b32_e32 v4, v3
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB29_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -7427,7 +7481,7 @@ define void @flat_agent_atomic_fmin_noret_f16__amdgpu_no_fine_grained_memory(ptr
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v4, v3
; GFX11-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB29_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -7649,6 +7703,7 @@ define void @flat_agent_atomic_fmin_noret_f16__offset12b_pos__amdgpu_no_fine_gra
; GFX12-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_cbranch_execnz .LBB30_1
; GFX12-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -7696,6 +7751,7 @@ define void @flat_agent_atomic_fmin_noret_f16__offset12b_pos__amdgpu_no_fine_gra
; GFX12-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: s_cbranch_execnz .LBB30_1
; GFX12-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -7742,7 +7798,6 @@ define void @flat_agent_atomic_fmin_noret_f16__offset12b_pos__amdgpu_no_fine_gra
; GFX11-TRUE16: ; %bb.0:
; GFX11-TRUE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-TRUE16-NEXT: v_add_co_u32 v0, vcc_lo, 0x7fe, v0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v4, null, 0, v1, vcc_lo
; GFX11-TRUE16-NEXT: s_mov_b32 s0, 0
; GFX11-TRUE16-NEXT: v_and_b32_e32 v3, -4, v0
@@ -7775,7 +7830,7 @@ define void @flat_agent_atomic_fmin_noret_f16__offset12b_pos__amdgpu_no_fine_gra
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v5, v6
; GFX11-TRUE16-NEXT: v_mov_b32_e32 v6, v5
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB30_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -7786,16 +7841,16 @@ define void @flat_agent_atomic_fmin_noret_f16__offset12b_pos__amdgpu_no_fine_gra
; GFX11-FAKE16: ; %bb.0:
; GFX11-FAKE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-FAKE16-NEXT: v_add_co_u32 v4, vcc_lo, 0x7fe, v0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-FAKE16-NEXT: v_max_f16_e32 v6, v2, v2
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_4) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_and_b32_e32 v0, -4, v4
; GFX11-FAKE16-NEXT: v_and_b32_e32 v4, 3, v4
; GFX11-FAKE16-NEXT: s_mov_b32 s0, 0
; GFX11-FAKE16-NEXT: flat_load_b32 v3, v[0:1]
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v4, 3, v4
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e64 v5, v4, 0xffff
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_not_b32_e32 v5, v5
; GFX11-FAKE16-NEXT: .LBB30_1: ; %atomicrmw.start
; GFX11-FAKE16-NEXT: ; =>This Inner Loop Header: Depth=1
@@ -7817,7 +7872,7 @@ define void @flat_agent_atomic_fmin_noret_f16__offset12b_pos__amdgpu_no_fine_gra
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v3
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v3, v2
; GFX11-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB30_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -8045,6 +8100,7 @@ define void @flat_agent_atomic_fmin_noret_f16__offset12b_neg__amdgpu_no_fine_gra
; GFX12-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_cbranch_execnz .LBB31_1
; GFX12-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -8092,6 +8148,7 @@ define void @flat_agent_atomic_fmin_noret_f16__offset12b_neg__amdgpu_no_fine_gra
; GFX12-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: s_cbranch_execnz .LBB31_1
; GFX12-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -8139,7 +8196,6 @@ define void @flat_agent_atomic_fmin_noret_f16__offset12b_neg__amdgpu_no_fine_gra
; GFX11-TRUE16: ; %bb.0:
; GFX11-TRUE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-TRUE16-NEXT: v_add_co_u32 v0, vcc_lo, 0xfffff800, v0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v4, null, -1, v1, vcc_lo
; GFX11-TRUE16-NEXT: s_mov_b32 s0, 0
; GFX11-TRUE16-NEXT: v_and_b32_e32 v3, -4, v0
@@ -8172,7 +8228,7 @@ define void @flat_agent_atomic_fmin_noret_f16__offset12b_neg__amdgpu_no_fine_gra
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v5, v6
; GFX11-TRUE16-NEXT: v_mov_b32_e32 v6, v5
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB31_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -8183,16 +8239,16 @@ define void @flat_agent_atomic_fmin_noret_f16__offset12b_neg__amdgpu_no_fine_gra
; GFX11-FAKE16: ; %bb.0:
; GFX11-FAKE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-FAKE16-NEXT: v_add_co_u32 v4, vcc_lo, 0xfffff800, v0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v1, null, -1, v1, vcc_lo
; GFX11-FAKE16-NEXT: v_max_f16_e32 v6, v2, v2
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_4) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_and_b32_e32 v0, -4, v4
; GFX11-FAKE16-NEXT: v_and_b32_e32 v4, 3, v4
; GFX11-FAKE16-NEXT: s_mov_b32 s0, 0
; GFX11-FAKE16-NEXT: flat_load_b32 v3, v[0:1]
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v4, 3, v4
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e64 v5, v4, 0xffff
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_not_b32_e32 v5, v5
; GFX11-FAKE16-NEXT: .LBB31_1: ; %atomicrmw.start
; GFX11-FAKE16-NEXT: ; =>This Inner Loop Header: Depth=1
@@ -8214,7 +8270,7 @@ define void @flat_agent_atomic_fmin_noret_f16__offset12b_neg__amdgpu_no_fine_gra
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v3
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v3, v2
; GFX11-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB31_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -8428,9 +8484,11 @@ define half @flat_agent_atomic_fmin_ret_f16__offset12b_pos__align4__amdgpu_no_fi
; GFX12-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_cbranch_execnz .LBB32_1
; GFX12-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: v_mov_b16_e32 v0.l, v3.l
; GFX12-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -8463,9 +8521,11 @@ define half @flat_agent_atomic_fmin_ret_f16__offset12b_pos__align4__amdgpu_no_fi
; GFX12-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: s_cbranch_execnz .LBB32_1
; GFX12-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: v_mov_b32_e32 v0, v3
; GFX12-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -8520,11 +8580,12 @@ define half @flat_agent_atomic_fmin_ret_f16__offset12b_pos__align4__amdgpu_no_fi
; GFX11-TRUE16-NEXT: buffer_gl0_inv
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB32_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v0.l, v3.l
; GFX11-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -8551,11 +8612,12 @@ define half @flat_agent_atomic_fmin_ret_f16__offset12b_pos__align4__amdgpu_no_fi
; GFX11-FAKE16-NEXT: buffer_gl0_inv
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB32_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v0, v3
; GFX11-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -8729,6 +8791,7 @@ define void @flat_agent_atomic_fmin_noret_f16__offset12b__align4_pos__amdgpu_no_
; GFX12-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_cbranch_execnz .LBB33_1
; GFX12-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -8763,6 +8826,7 @@ define void @flat_agent_atomic_fmin_noret_f16__offset12b__align4_pos__amdgpu_no_
; GFX12-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: s_cbranch_execnz .LBB33_1
; GFX12-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -8818,7 +8882,7 @@ define void @flat_agent_atomic_fmin_noret_f16__offset12b__align4_pos__amdgpu_no_
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-TRUE16-NEXT: v_mov_b32_e32 v4, v3
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB33_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -8848,7 +8912,7 @@ define void @flat_agent_atomic_fmin_noret_f16__offset12b__align4_pos__amdgpu_no_
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v3
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v3, v2
; GFX11-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB33_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -9038,9 +9102,11 @@ define half @flat_system_atomic_fmin_ret_f16__offset12b_pos__amdgpu_no_fine_grai
; GFX12-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_cbranch_execnz .LBB34_1
; GFX12-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: v_lshrrev_b32_e32 v0, v1, v5
; GFX12-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -9087,9 +9153,11 @@ define half @flat_system_atomic_fmin_ret_f16__offset12b_pos__amdgpu_no_fine_grai
; GFX12-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: s_cbranch_execnz .LBB34_1
; GFX12-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: v_lshrrev_b32_e32 v0, v3, v5
; GFX12-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -9135,7 +9203,6 @@ define half @flat_system_atomic_fmin_ret_f16__offset12b_pos__amdgpu_no_fine_grai
; GFX11-TRUE16: ; %bb.0:
; GFX11-TRUE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-TRUE16-NEXT: v_add_co_u32 v0, vcc_lo, 0x7fe, v0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v4, null, 0, v1, vcc_lo
; GFX11-TRUE16-NEXT: s_mov_b32 s0, 0
; GFX11-TRUE16-NEXT: v_and_b32_e32 v3, -4, v0
@@ -9168,11 +9235,12 @@ define half @flat_system_atomic_fmin_ret_f16__offset12b_pos__amdgpu_no_fine_grai
; GFX11-TRUE16-NEXT: buffer_gl0_inv
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v5, v6
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB34_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v0, v1, v5
; GFX11-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -9180,16 +9248,16 @@ define half @flat_system_atomic_fmin_ret_f16__offset12b_pos__amdgpu_no_fine_grai
; GFX11-FAKE16: ; %bb.0:
; GFX11-FAKE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-FAKE16-NEXT: v_add_co_u32 v3, vcc_lo, 0x7fe, v0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-FAKE16-NEXT: v_max_f16_e32 v2, v2, v2
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_4) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_and_b32_e32 v0, -4, v3
; GFX11-FAKE16-NEXT: v_and_b32_e32 v3, 3, v3
; GFX11-FAKE16-NEXT: s_mov_b32 s0, 0
; GFX11-FAKE16-NEXT: flat_load_b32 v5, v[0:1]
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v3, 3, v3
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e64 v4, v3, 0xffff
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_not_b32_e32 v4, v4
; GFX11-FAKE16-NEXT: .LBB34_1: ; %atomicrmw.start
; GFX11-FAKE16-NEXT: ; =>This Inner Loop Header: Depth=1
@@ -9211,11 +9279,12 @@ define half @flat_system_atomic_fmin_ret_f16__offset12b_pos__amdgpu_no_fine_grai
; GFX11-FAKE16-NEXT: buffer_gl0_inv
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v5, v6
; GFX11-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB34_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v0, v3, v5
; GFX11-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -9449,6 +9518,7 @@ define void @flat_system_atomic_fmin_noret_f16__offset12b_pos__amdgpu_no_fine_gr
; GFX12-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_cbranch_execnz .LBB35_1
; GFX12-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -9497,6 +9567,7 @@ define void @flat_system_atomic_fmin_noret_f16__offset12b_pos__amdgpu_no_fine_gr
; GFX12-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: s_cbranch_execnz .LBB35_1
; GFX12-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -9543,7 +9614,6 @@ define void @flat_system_atomic_fmin_noret_f16__offset12b_pos__amdgpu_no_fine_gr
; GFX11-TRUE16: ; %bb.0:
; GFX11-TRUE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-TRUE16-NEXT: v_add_co_u32 v0, vcc_lo, 0x7fe, v0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v4, null, 0, v1, vcc_lo
; GFX11-TRUE16-NEXT: s_mov_b32 s0, 0
; GFX11-TRUE16-NEXT: v_and_b32_e32 v3, -4, v0
@@ -9576,7 +9646,7 @@ define void @flat_system_atomic_fmin_noret_f16__offset12b_pos__amdgpu_no_fine_gr
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v5, v6
; GFX11-TRUE16-NEXT: v_mov_b32_e32 v6, v5
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB35_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -9587,16 +9657,16 @@ define void @flat_system_atomic_fmin_noret_f16__offset12b_pos__amdgpu_no_fine_gr
; GFX11-FAKE16: ; %bb.0:
; GFX11-FAKE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-FAKE16-NEXT: v_add_co_u32 v4, vcc_lo, 0x7fe, v0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-FAKE16-NEXT: v_max_f16_e32 v6, v2, v2
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_4) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_and_b32_e32 v0, -4, v4
; GFX11-FAKE16-NEXT: v_and_b32_e32 v4, 3, v4
; GFX11-FAKE16-NEXT: s_mov_b32 s0, 0
; GFX11-FAKE16-NEXT: flat_load_b32 v3, v[0:1]
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v4, 3, v4
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e64 v5, v4, 0xffff
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_not_b32_e32 v5, v5
; GFX11-FAKE16-NEXT: .LBB35_1: ; %atomicrmw.start
; GFX11-FAKE16-NEXT: ; =>This Inner Loop Header: Depth=1
@@ -9618,7 +9688,7 @@ define void @flat_system_atomic_fmin_noret_f16__offset12b_pos__amdgpu_no_fine_gr
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v3
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v3, v2
; GFX11-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB35_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -9857,9 +9927,11 @@ define bfloat @flat_agent_atomic_fmin_ret_bf16__amdgpu_no_fine_grained_memory(pt
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB36_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: v_lshrrev_b32_e32 v0, v3, v5
; GFX12-NEXT: s_setpc_b64 s[30:31]
;
@@ -9947,11 +10019,12 @@ define bfloat @flat_agent_atomic_fmin_ret_bf16__amdgpu_no_fine_grained_memory(pt
; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v5, v6
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB36_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: v_lshrrev_b32_e32 v0, v3, v5
; GFX11-NEXT: s_setpc_b64 s[30:31]
;
@@ -10204,9 +10277,11 @@ define bfloat @flat_agent_atomic_fmin_ret_bf16__offset12b_pos__amdgpu_no_fine_gr
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB37_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: v_lshrrev_b32_e32 v0, v3, v5
; GFX12-NEXT: s_setpc_b64 s[30:31]
;
@@ -10259,16 +10334,16 @@ define bfloat @flat_agent_atomic_fmin_ret_bf16__offset12b_pos__amdgpu_no_fine_gr
; GFX11: ; %bb.0:
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: v_add_co_u32 v3, vcc_lo, 0x7fe, v0
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_3)
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-NEXT: v_lshlrev_b32_e32 v2, 16, v2
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_4) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_and_b32_e32 v0, -4, v3
; GFX11-NEXT: v_and_b32_e32 v3, 3, v3
; GFX11-NEXT: s_mov_b32 s0, 0
; GFX11-NEXT: flat_load_b32 v5, v[0:1]
; GFX11-NEXT: v_lshlrev_b32_e32 v3, 3, v3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_lshlrev_b32_e64 v4, v3, 0xffff
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_not_b32_e32 v4, v4
; GFX11-NEXT: .p2align 6
; GFX11-NEXT: .LBB37_1: ; %atomicrmw.start
@@ -10298,11 +10373,12 @@ define bfloat @flat_agent_atomic_fmin_ret_bf16__offset12b_pos__amdgpu_no_fine_gr
; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v5, v6
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB37_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: v_lshrrev_b32_e32 v0, v3, v5
; GFX11-NEXT: s_setpc_b64 s[30:31]
;
@@ -10561,9 +10637,11 @@ define bfloat @flat_agent_atomic_fmin_ret_bf16__offset12b_neg__amdgpu_no_fine_gr
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB38_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: v_lshrrev_b32_e32 v0, v3, v5
; GFX12-NEXT: s_setpc_b64 s[30:31]
;
@@ -10617,16 +10695,16 @@ define bfloat @flat_agent_atomic_fmin_ret_bf16__offset12b_neg__amdgpu_no_fine_gr
; GFX11: ; %bb.0:
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: v_add_co_u32 v3, vcc_lo, 0xfffff800, v0
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_3)
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, -1, v1, vcc_lo
; GFX11-NEXT: v_lshlrev_b32_e32 v2, 16, v2
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_4) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_and_b32_e32 v0, -4, v3
; GFX11-NEXT: v_and_b32_e32 v3, 3, v3
; GFX11-NEXT: s_mov_b32 s0, 0
; GFX11-NEXT: flat_load_b32 v5, v[0:1]
; GFX11-NEXT: v_lshlrev_b32_e32 v3, 3, v3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_lshlrev_b32_e64 v4, v3, 0xffff
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_not_b32_e32 v4, v4
; GFX11-NEXT: .p2align 6
; GFX11-NEXT: .LBB38_1: ; %atomicrmw.start
@@ -10656,11 +10734,12 @@ define bfloat @flat_agent_atomic_fmin_ret_bf16__offset12b_neg__amdgpu_no_fine_gr
; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v5, v6
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB38_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: v_lshrrev_b32_e32 v0, v3, v5
; GFX11-NEXT: s_setpc_b64 s[30:31]
;
@@ -10915,6 +10994,7 @@ define void @flat_agent_atomic_fmin_noret_bf16__amdgpu_no_fine_grained_memory(pt
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB39_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -11002,7 +11082,7 @@ define void @flat_agent_atomic_fmin_noret_bf16__amdgpu_no_fine_grained_memory(pt
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-NEXT: v_mov_b32_e32 v4, v3
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB39_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -11252,6 +11332,7 @@ define void @flat_agent_atomic_fmin_noret_bf16__offset12b_pos__amdgpu_no_fine_gr
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB40_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -11305,16 +11386,16 @@ define void @flat_agent_atomic_fmin_noret_bf16__offset12b_pos__amdgpu_no_fine_gr
; GFX11: ; %bb.0:
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: v_add_co_u32 v4, vcc_lo, 0x7fe, v0
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_3)
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-NEXT: v_lshlrev_b32_e32 v6, 16, v2
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_4) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_and_b32_e32 v0, -4, v4
; GFX11-NEXT: v_and_b32_e32 v4, 3, v4
; GFX11-NEXT: s_mov_b32 s0, 0
; GFX11-NEXT: flat_load_b32 v3, v[0:1]
; GFX11-NEXT: v_lshlrev_b32_e32 v4, 3, v4
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_lshlrev_b32_e64 v5, v4, 0xffff
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_not_b32_e32 v5, v5
; GFX11-NEXT: .p2align 6
; GFX11-NEXT: .LBB40_1: ; %atomicrmw.start
@@ -11343,7 +11424,7 @@ define void @flat_agent_atomic_fmin_noret_bf16__offset12b_pos__amdgpu_no_fine_gr
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v3
; GFX11-NEXT: v_mov_b32_e32 v3, v2
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB40_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -11599,6 +11680,7 @@ define void @flat_agent_atomic_fmin_noret_bf16__offset12b_neg__amdgpu_no_fine_gr
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB41_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -11653,16 +11735,16 @@ define void @flat_agent_atomic_fmin_noret_bf16__offset12b_neg__amdgpu_no_fine_gr
; GFX11: ; %bb.0:
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: v_add_co_u32 v4, vcc_lo, 0xfffff800, v0
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_3)
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, -1, v1, vcc_lo
; GFX11-NEXT: v_lshlrev_b32_e32 v6, 16, v2
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_4) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_and_b32_e32 v0, -4, v4
; GFX11-NEXT: v_and_b32_e32 v4, 3, v4
; GFX11-NEXT: s_mov_b32 s0, 0
; GFX11-NEXT: flat_load_b32 v3, v[0:1]
; GFX11-NEXT: v_lshlrev_b32_e32 v4, 3, v4
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_lshlrev_b32_e64 v5, v4, 0xffff
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_not_b32_e32 v5, v5
; GFX11-NEXT: .p2align 6
; GFX11-NEXT: .LBB41_1: ; %atomicrmw.start
@@ -11691,7 +11773,7 @@ define void @flat_agent_atomic_fmin_noret_bf16__offset12b_neg__amdgpu_no_fine_gr
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v3
; GFX11-NEXT: v_mov_b32_e32 v3, v2
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB41_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -11935,9 +12017,11 @@ define bfloat @flat_agent_atomic_fmin_ret_bf16__offset12b_pos__align4__amdgpu_no
; GFX12-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_cbranch_execnz .LBB42_1
; GFX12-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: v_mov_b16_e32 v0.l, v3.l
; GFX12-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -11978,9 +12062,11 @@ define bfloat @flat_agent_atomic_fmin_ret_bf16__offset12b_pos__align4__amdgpu_no
; GFX12-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: s_cbranch_execnz .LBB42_1
; GFX12-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: v_mov_b32_e32 v0, v3
; GFX12-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -12051,11 +12137,12 @@ define bfloat @flat_agent_atomic_fmin_ret_bf16__offset12b_pos__align4__amdgpu_no
; GFX11-TRUE16-NEXT: buffer_gl0_inv
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB42_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v0.l, v3.l
; GFX11-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -12090,11 +12177,12 @@ define bfloat @flat_agent_atomic_fmin_ret_bf16__offset12b_pos__align4__amdgpu_no
; GFX11-FAKE16-NEXT: buffer_gl0_inv
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB42_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v0, v3
; GFX11-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -12303,6 +12391,7 @@ define void @flat_agent_atomic_fmin_noret_bf16__offset12b__align4_pos__amdgpu_no
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB43_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -12373,7 +12462,7 @@ define void @flat_agent_atomic_fmin_noret_bf16__offset12b__align4_pos__amdgpu_no
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v3
; GFX11-NEXT: v_mov_b32_e32 v3, v2
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB43_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -12598,9 +12687,11 @@ define bfloat @flat_system_atomic_fmin_ret_bf16__offset12b_pos__amdgpu_no_fine_g
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB44_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: v_lshrrev_b32_e32 v0, v3, v5
; GFX12-NEXT: s_setpc_b64 s[30:31]
;
@@ -12653,16 +12744,16 @@ define bfloat @flat_system_atomic_fmin_ret_bf16__offset12b_pos__amdgpu_no_fine_g
; GFX11: ; %bb.0:
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: v_add_co_u32 v3, vcc_lo, 0x7fe, v0
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_3)
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-NEXT: v_lshlrev_b32_e32 v2, 16, v2
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_4) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_and_b32_e32 v0, -4, v3
; GFX11-NEXT: v_and_b32_e32 v3, 3, v3
; GFX11-NEXT: s_mov_b32 s0, 0
; GFX11-NEXT: flat_load_b32 v5, v[0:1]
; GFX11-NEXT: v_lshlrev_b32_e32 v3, 3, v3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_lshlrev_b32_e64 v4, v3, 0xffff
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_not_b32_e32 v4, v4
; GFX11-NEXT: .p2align 6
; GFX11-NEXT: .LBB44_1: ; %atomicrmw.start
@@ -12692,11 +12783,12 @@ define bfloat @flat_system_atomic_fmin_ret_bf16__offset12b_pos__amdgpu_no_fine_g
; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v5, v6
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB44_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: v_lshrrev_b32_e32 v0, v3, v5
; GFX11-NEXT: s_setpc_b64 s[30:31]
;
@@ -12958,6 +13050,7 @@ define void @flat_system_atomic_fmin_noret_bf16__offset12b_pos__amdgpu_no_fine_g
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB45_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -13011,16 +13104,16 @@ define void @flat_system_atomic_fmin_noret_bf16__offset12b_pos__amdgpu_no_fine_g
; GFX11: ; %bb.0:
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: v_add_co_u32 v4, vcc_lo, 0x7fe, v0
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_3)
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-NEXT: v_lshlrev_b32_e32 v6, 16, v2
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_4) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_and_b32_e32 v0, -4, v4
; GFX11-NEXT: v_and_b32_e32 v4, 3, v4
; GFX11-NEXT: s_mov_b32 s0, 0
; GFX11-NEXT: flat_load_b32 v3, v[0:1]
; GFX11-NEXT: v_lshlrev_b32_e32 v4, 3, v4
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_lshlrev_b32_e64 v5, v4, 0xffff
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_not_b32_e32 v5, v5
; GFX11-NEXT: .p2align 6
; GFX11-NEXT: .LBB45_1: ; %atomicrmw.start
@@ -13049,7 +13142,7 @@ define void @flat_system_atomic_fmin_noret_bf16__offset12b_pos__amdgpu_no_fine_g
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v3
; GFX11-NEXT: v_mov_b32_e32 v3, v2
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB45_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -13289,9 +13382,11 @@ define <2 x half> @flat_agent_atomic_fmin_ret_v2f16__amdgpu_no_fine_grained_memo
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB46_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: v_mov_b32_e32 v0, v3
; GFX12-NEXT: s_setpc_b64 s[30:31]
;
@@ -13342,11 +13437,12 @@ define <2 x half> @flat_agent_atomic_fmin_ret_v2f16__amdgpu_no_fine_grained_memo
; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB46_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: v_mov_b32_e32 v0, v3
; GFX11-NEXT: s_setpc_b64 s[30:31]
;
@@ -13520,9 +13616,11 @@ define <2 x half> @flat_agent_atomic_fmin_ret_v2f16__offset12b_pos__amdgpu_no_fi
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB47_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: v_mov_b32_e32 v0, v3
; GFX12-NEXT: s_setpc_b64 s[30:31]
;
@@ -13573,11 +13671,12 @@ define <2 x half> @flat_agent_atomic_fmin_ret_v2f16__offset12b_pos__amdgpu_no_fi
; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB47_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: v_mov_b32_e32 v0, v3
; GFX11-NEXT: s_setpc_b64 s[30:31]
;
@@ -13756,9 +13855,11 @@ define <2 x half> @flat_agent_atomic_fmin_ret_v2f16__offset12b_neg__amdgpu_no_fi
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB48_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: v_mov_b32_e32 v0, v3
; GFX12-NEXT: s_setpc_b64 s[30:31]
;
@@ -13798,7 +13899,6 @@ define <2 x half> @flat_agent_atomic_fmin_ret_v2f16__offset12b_neg__amdgpu_no_fi
; GFX11: ; %bb.0:
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: v_add_co_u32 v3, vcc_lo, 0xfffff800, v0
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v4, null, -1, v1, vcc_lo
; GFX11-NEXT: v_pk_max_f16 v1, v2, v2
; GFX11-NEXT: s_mov_b32 s0, 0
@@ -13817,7 +13917,7 @@ define <2 x half> @flat_agent_atomic_fmin_ret_v2f16__offset12b_neg__amdgpu_no_fi
; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v0, v6
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB48_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -14007,6 +14107,7 @@ define void @flat_agent_atomic_fmin_noret_v2f16__amdgpu_no_fine_grained_memory(p
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB49_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -14058,7 +14159,7 @@ define void @flat_agent_atomic_fmin_noret_v2f16__amdgpu_no_fine_grained_memory(p
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v3
; GFX11-NEXT: v_mov_b32_e32 v3, v2
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB49_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -14228,6 +14329,7 @@ define void @flat_agent_atomic_fmin_noret_v2f16__offset12b_pos__amdgpu_no_fine_g
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB50_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -14279,7 +14381,7 @@ define void @flat_agent_atomic_fmin_noret_v2f16__offset12b_pos__amdgpu_no_fine_g
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v3
; GFX11-NEXT: v_mov_b32_e32 v3, v2
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB50_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -14456,6 +14558,7 @@ define void @flat_agent_atomic_fmin_noret_v2f16__offset12b_neg__amdgpu_no_fine_g
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB51_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -14497,7 +14600,6 @@ define void @flat_agent_atomic_fmin_noret_v2f16__offset12b_neg__amdgpu_no_fine_g
; GFX11: ; %bb.0:
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: v_add_co_u32 v0, vcc_lo, 0xfffff800, v0
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, -1, v1, vcc_lo
; GFX11-NEXT: v_pk_max_f16 v4, v2, v2
; GFX11-NEXT: s_mov_b32 s0, 0
@@ -14516,7 +14618,7 @@ define void @flat_agent_atomic_fmin_noret_v2f16__offset12b_neg__amdgpu_no_fine_g
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v3
; GFX11-NEXT: v_mov_b32_e32 v3, v2
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB51_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -14704,9 +14806,11 @@ define <2 x half> @flat_system_atomic_fmin_ret_v2f16__offset12b_pos__amdgpu_no_f
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB52_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: v_mov_b32_e32 v0, v3
; GFX12-NEXT: s_setpc_b64 s[30:31]
;
@@ -14757,11 +14861,12 @@ define <2 x half> @flat_system_atomic_fmin_ret_v2f16__offset12b_pos__amdgpu_no_f
; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB52_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: v_mov_b32_e32 v0, v3
; GFX11-NEXT: s_setpc_b64 s[30:31]
;
@@ -14944,6 +15049,7 @@ define void @flat_system_atomic_fmin_noret_v2f16__offset12b_pos__amdgpu_no_fine_
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB53_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -14995,7 +15101,7 @@ define void @flat_system_atomic_fmin_noret_v2f16__offset12b_pos__amdgpu_no_fine_
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v3
; GFX11-NEXT: v_mov_b32_e32 v3, v2
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB53_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -15202,9 +15308,11 @@ define <2 x bfloat> @flat_agent_atomic_fmin_ret_v2bf16__amdgpu_no_fine_grained_m
; GFX12-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_cbranch_execnz .LBB54_1
; GFX12-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: v_mov_b32_e32 v0, v3
; GFX12-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -15253,9 +15361,11 @@ define <2 x bfloat> @flat_agent_atomic_fmin_ret_v2bf16__amdgpu_no_fine_grained_m
; GFX12-FAKE16-NEXT: s_or_b32 s1, vcc_lo, s1
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: s_cbranch_execnz .LBB54_1
; GFX12-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s1
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: v_mov_b32_e32 v0, v3
; GFX12-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -15345,12 +15455,13 @@ define <2 x bfloat> @flat_agent_atomic_fmin_ret_v2bf16__amdgpu_no_fine_grained_m
; GFX11-TRUE16-NEXT: buffer_gl0_inv
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v6
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB54_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-TRUE16-NEXT: s_set_inst_prefetch_distance 0x2
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_mov_b32_e32 v0, v3
; GFX11-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -15394,12 +15505,13 @@ define <2 x bfloat> @flat_agent_atomic_fmin_ret_v2bf16__amdgpu_no_fine_grained_m
; GFX11-FAKE16-NEXT: buffer_gl0_inv
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v6
; GFX11-FAKE16-NEXT: s_or_b32 s1, vcc_lo, s1
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB54_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-FAKE16-NEXT: s_set_inst_prefetch_distance 0x2
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s1
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v0, v3
; GFX11-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -15660,9 +15772,11 @@ define <2 x bfloat> @flat_agent_atomic_fmin_ret_v2bf16__offset12b_pos__amdgpu_no
; GFX12-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_cbranch_execnz .LBB55_1
; GFX12-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: v_mov_b32_e32 v0, v3
; GFX12-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -15711,9 +15825,11 @@ define <2 x bfloat> @flat_agent_atomic_fmin_ret_v2bf16__offset12b_pos__amdgpu_no
; GFX12-FAKE16-NEXT: s_or_b32 s1, vcc_lo, s1
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: s_cbranch_execnz .LBB55_1
; GFX12-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s1
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: v_mov_b32_e32 v0, v3
; GFX12-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -15803,12 +15919,13 @@ define <2 x bfloat> @flat_agent_atomic_fmin_ret_v2bf16__offset12b_pos__amdgpu_no
; GFX11-TRUE16-NEXT: buffer_gl0_inv
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v6
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB55_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-TRUE16-NEXT: s_set_inst_prefetch_distance 0x2
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_mov_b32_e32 v0, v3
; GFX11-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -15852,12 +15969,13 @@ define <2 x bfloat> @flat_agent_atomic_fmin_ret_v2bf16__offset12b_pos__amdgpu_no
; GFX11-FAKE16-NEXT: buffer_gl0_inv
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v6
; GFX11-FAKE16-NEXT: s_or_b32 s1, vcc_lo, s1
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB55_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-FAKE16-NEXT: s_set_inst_prefetch_distance 0x2
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s1
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v0, v3
; GFX11-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -16123,9 +16241,11 @@ define <2 x bfloat> @flat_agent_atomic_fmin_ret_v2bf16__offset12b_neg__amdgpu_no
; GFX12-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_cbranch_execnz .LBB56_1
; GFX12-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: v_mov_b32_e32 v0, v3
; GFX12-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -16174,9 +16294,11 @@ define <2 x bfloat> @flat_agent_atomic_fmin_ret_v2bf16__offset12b_neg__amdgpu_no
; GFX12-FAKE16-NEXT: s_or_b32 s1, vcc_lo, s1
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: s_cbranch_execnz .LBB56_1
; GFX12-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s1
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: v_mov_b32_e32 v0, v3
; GFX12-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -16232,7 +16354,6 @@ define <2 x bfloat> @flat_agent_atomic_fmin_ret_v2bf16__offset12b_neg__amdgpu_no
; GFX11-TRUE16: ; %bb.0:
; GFX11-TRUE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-TRUE16-NEXT: v_add_co_u32 v3, vcc_lo, 0xfffff800, v0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v4, null, -1, v1, vcc_lo
; GFX11-TRUE16-NEXT: v_and_b32_e32 v1, 0xffff0000, v2
; GFX11-TRUE16-NEXT: v_lshlrev_b32_e32 v2, 16, v2
@@ -16274,7 +16395,7 @@ define <2 x bfloat> @flat_agent_atomic_fmin_ret_v2bf16__offset12b_neg__amdgpu_no
; GFX11-TRUE16-NEXT: buffer_gl0_inv
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v0, v6
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB56_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -16286,7 +16407,6 @@ define <2 x bfloat> @flat_agent_atomic_fmin_ret_v2bf16__offset12b_neg__amdgpu_no
; GFX11-FAKE16: ; %bb.0:
; GFX11-FAKE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-FAKE16-NEXT: v_add_co_u32 v3, vcc_lo, 0xfffff800, v0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v4, null, -1, v1, vcc_lo
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v1, 16, v2
; GFX11-FAKE16-NEXT: v_and_b32_e32 v2, 0xffff0000, v2
@@ -16325,7 +16445,7 @@ define <2 x bfloat> @flat_agent_atomic_fmin_ret_v2bf16__offset12b_neg__amdgpu_no
; GFX11-FAKE16-NEXT: buffer_gl0_inv
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v0, v6
; GFX11-FAKE16-NEXT: s_or_b32 s1, vcc_lo, s1
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB56_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -16601,6 +16721,7 @@ define void @flat_agent_atomic_fmin_noret_v2bf16__amdgpu_no_fine_grained_memory(
; GFX12-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_cbranch_execnz .LBB57_1
; GFX12-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -16635,11 +16756,10 @@ define void @flat_agent_atomic_fmin_noret_v2bf16__amdgpu_no_fine_grained_memory(
; GFX12-FAKE16-NEXT: v_add3_u32 v8, v8, v6, 0x7fff
; GFX12-FAKE16-NEXT: v_cmp_u_f32_e64 s0, v2, v2
; GFX12-FAKE16-NEXT: s_wait_alu depctr_va_vcc(0)
-; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_2)
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_2) | instid1(VALU_DEP_1)
; GFX12-FAKE16-NEXT: v_cndmask_b32_e32 v6, v8, v10, vcc_lo
; GFX12-FAKE16-NEXT: s_wait_alu depctr_va_sdst(0)
; GFX12-FAKE16-NEXT: v_cndmask_b32_e64 v2, v7, v9, s0
-; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-FAKE16-NEXT: v_perm_b32 v2, v6, v2, 0x7060302
; GFX12-FAKE16-NEXT: s_wait_storecnt 0x0
; GFX12-FAKE16-NEXT: flat_atomic_cmpswap_b32 v2, v[0:1], v[2:3] th:TH_ATOMIC_RETURN scope:SCOPE_DEV
@@ -16651,6 +16771,7 @@ define void @flat_agent_atomic_fmin_noret_v2bf16__amdgpu_no_fine_grained_memory(
; GFX12-FAKE16-NEXT: s_or_b32 s1, vcc_lo, s1
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: s_cbranch_execnz .LBB57_1
; GFX12-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s1
@@ -16740,7 +16861,7 @@ define void @flat_agent_atomic_fmin_noret_v2bf16__amdgpu_no_fine_grained_memory(
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v3
; GFX11-TRUE16-NEXT: v_mov_b32_e32 v3, v2
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB57_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -16774,7 +16895,7 @@ define void @flat_agent_atomic_fmin_noret_v2bf16__amdgpu_no_fine_grained_memory(
; GFX11-FAKE16-NEXT: v_add3_u32 v7, v7, v2, 0x7fff
; GFX11-FAKE16-NEXT: v_add3_u32 v8, v8, v6, 0x7fff
; GFX11-FAKE16-NEXT: v_cmp_u_f32_e64 s0, v2, v2
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_4)
; GFX11-FAKE16-NEXT: v_cndmask_b32_e32 v6, v8, v10, vcc_lo
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v2, v7, v9, s0
; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
@@ -16787,7 +16908,7 @@ define void @flat_agent_atomic_fmin_noret_v2bf16__amdgpu_no_fine_grained_memory(
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v3
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v3, v2
; GFX11-FAKE16-NEXT: s_or_b32 s1, vcc_lo, s1
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB57_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -17042,6 +17163,7 @@ define void @flat_agent_atomic_fmin_noret_v2bf16__offset12b_pos__amdgpu_no_fine_
; GFX12-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_cbranch_execnz .LBB58_1
; GFX12-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -17076,11 +17198,10 @@ define void @flat_agent_atomic_fmin_noret_v2bf16__offset12b_pos__amdgpu_no_fine_
; GFX12-FAKE16-NEXT: v_add3_u32 v8, v8, v6, 0x7fff
; GFX12-FAKE16-NEXT: v_cmp_u_f32_e64 s0, v2, v2
; GFX12-FAKE16-NEXT: s_wait_alu depctr_va_vcc(0)
-; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_2)
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_2) | instid1(VALU_DEP_1)
; GFX12-FAKE16-NEXT: v_cndmask_b32_e32 v6, v8, v10, vcc_lo
; GFX12-FAKE16-NEXT: s_wait_alu depctr_va_sdst(0)
; GFX12-FAKE16-NEXT: v_cndmask_b32_e64 v2, v7, v9, s0
-; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-FAKE16-NEXT: v_perm_b32 v2, v6, v2, 0x7060302
; GFX12-FAKE16-NEXT: s_wait_storecnt 0x0
; GFX12-FAKE16-NEXT: flat_atomic_cmpswap_b32 v2, v[0:1], v[2:3] offset:2044 th:TH_ATOMIC_RETURN scope:SCOPE_DEV
@@ -17092,6 +17213,7 @@ define void @flat_agent_atomic_fmin_noret_v2bf16__offset12b_pos__amdgpu_no_fine_
; GFX12-FAKE16-NEXT: s_or_b32 s1, vcc_lo, s1
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: s_cbranch_execnz .LBB58_1
; GFX12-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s1
@@ -17181,7 +17303,7 @@ define void @flat_agent_atomic_fmin_noret_v2bf16__offset12b_pos__amdgpu_no_fine_
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v3
; GFX11-TRUE16-NEXT: v_mov_b32_e32 v3, v2
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB58_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -17215,7 +17337,7 @@ define void @flat_agent_atomic_fmin_noret_v2bf16__offset12b_pos__amdgpu_no_fine_
; GFX11-FAKE16-NEXT: v_add3_u32 v7, v7, v2, 0x7fff
; GFX11-FAKE16-NEXT: v_add3_u32 v8, v8, v6, 0x7fff
; GFX11-FAKE16-NEXT: v_cmp_u_f32_e64 s0, v2, v2
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_4)
; GFX11-FAKE16-NEXT: v_cndmask_b32_e32 v6, v8, v10, vcc_lo
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v2, v7, v9, s0
; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
@@ -17228,7 +17350,7 @@ define void @flat_agent_atomic_fmin_noret_v2bf16__offset12b_pos__amdgpu_no_fine_
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v3
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v3, v2
; GFX11-FAKE16-NEXT: s_or_b32 s1, vcc_lo, s1
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB58_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -17490,6 +17612,7 @@ define void @flat_agent_atomic_fmin_noret_v2bf16__offset12b_neg__amdgpu_no_fine_
; GFX12-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_cbranch_execnz .LBB59_1
; GFX12-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -17524,11 +17647,10 @@ define void @flat_agent_atomic_fmin_noret_v2bf16__offset12b_neg__amdgpu_no_fine_
; GFX12-FAKE16-NEXT: v_add3_u32 v8, v8, v6, 0x7fff
; GFX12-FAKE16-NEXT: v_cmp_u_f32_e64 s0, v2, v2
; GFX12-FAKE16-NEXT: s_wait_alu depctr_va_vcc(0)
-; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_2)
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_2) | instid1(VALU_DEP_1)
; GFX12-FAKE16-NEXT: v_cndmask_b32_e32 v6, v8, v10, vcc_lo
; GFX12-FAKE16-NEXT: s_wait_alu depctr_va_sdst(0)
; GFX12-FAKE16-NEXT: v_cndmask_b32_e64 v2, v7, v9, s0
-; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-FAKE16-NEXT: v_perm_b32 v2, v6, v2, 0x7060302
; GFX12-FAKE16-NEXT: s_wait_storecnt 0x0
; GFX12-FAKE16-NEXT: flat_atomic_cmpswap_b32 v2, v[0:1], v[2:3] offset:-2048 th:TH_ATOMIC_RETURN scope:SCOPE_DEV
@@ -17540,6 +17662,7 @@ define void @flat_agent_atomic_fmin_noret_v2bf16__offset12b_neg__amdgpu_no_fine_
; GFX12-FAKE16-NEXT: s_or_b32 s1, vcc_lo, s1
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: s_cbranch_execnz .LBB59_1
; GFX12-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s1
@@ -17597,7 +17720,6 @@ define void @flat_agent_atomic_fmin_noret_v2bf16__offset12b_neg__amdgpu_no_fine_
; GFX11-TRUE16: ; %bb.0:
; GFX11-TRUE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-TRUE16-NEXT: v_add_co_u32 v0, vcc_lo, 0xfffff800, v0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v1, null, -1, v1, vcc_lo
; GFX11-TRUE16-NEXT: v_and_b32_e32 v4, 0xffff0000, v2
; GFX11-TRUE16-NEXT: v_lshlrev_b32_e32 v5, 16, v2
@@ -17638,7 +17760,7 @@ define void @flat_agent_atomic_fmin_noret_v2bf16__offset12b_neg__amdgpu_no_fine_
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v3
; GFX11-TRUE16-NEXT: v_mov_b32_e32 v3, v2
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB59_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -17650,7 +17772,6 @@ define void @flat_agent_atomic_fmin_noret_v2bf16__offset12b_neg__amdgpu_no_fine_
; GFX11-FAKE16: ; %bb.0:
; GFX11-FAKE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-FAKE16-NEXT: v_add_co_u32 v0, vcc_lo, 0xfffff800, v0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v1, null, -1, v1, vcc_lo
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v4, 16, v2
; GFX11-FAKE16-NEXT: v_and_b32_e32 v5, 0xffff0000, v2
@@ -17675,7 +17796,7 @@ define void @flat_agent_atomic_fmin_noret_v2bf16__offset12b_neg__amdgpu_no_fine_
; GFX11-FAKE16-NEXT: v_add3_u32 v7, v7, v2, 0x7fff
; GFX11-FAKE16-NEXT: v_add3_u32 v8, v8, v6, 0x7fff
; GFX11-FAKE16-NEXT: v_cmp_u_f32_e64 s0, v2, v2
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_4)
; GFX11-FAKE16-NEXT: v_cndmask_b32_e32 v6, v8, v10, vcc_lo
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v2, v7, v9, s0
; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
@@ -17688,7 +17809,7 @@ define void @flat_agent_atomic_fmin_noret_v2bf16__offset12b_neg__amdgpu_no_fine_
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v3
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v3, v2
; GFX11-FAKE16-NEXT: s_or_b32 s1, vcc_lo, s1
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB59_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -17963,9 +18084,11 @@ define <2 x bfloat> @flat_system_atomic_fmin_ret_v2bf16__offset12b_pos__amdgpu_n
; GFX12-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_cbranch_execnz .LBB60_1
; GFX12-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: v_mov_b32_e32 v0, v3
; GFX12-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -18015,9 +18138,11 @@ define <2 x bfloat> @flat_system_atomic_fmin_ret_v2bf16__offset12b_pos__amdgpu_n
; GFX12-FAKE16-NEXT: s_or_b32 s1, vcc_lo, s1
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: s_cbranch_execnz .LBB60_1
; GFX12-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s1
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: v_mov_b32_e32 v0, v3
; GFX12-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -18107,12 +18232,13 @@ define <2 x bfloat> @flat_system_atomic_fmin_ret_v2bf16__offset12b_pos__amdgpu_n
; GFX11-TRUE16-NEXT: buffer_gl0_inv
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v6
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB60_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-TRUE16-NEXT: s_set_inst_prefetch_distance 0x2
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_mov_b32_e32 v0, v3
; GFX11-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -18156,12 +18282,13 @@ define <2 x bfloat> @flat_system_atomic_fmin_ret_v2bf16__offset12b_pos__amdgpu_n
; GFX11-FAKE16-NEXT: buffer_gl0_inv
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v6
; GFX11-FAKE16-NEXT: s_or_b32 s1, vcc_lo, s1
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB60_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-FAKE16-NEXT: s_set_inst_prefetch_distance 0x2
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s1
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v0, v3
; GFX11-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -18429,6 +18556,7 @@ define void @flat_system_atomic_fmin_noret_v2bf16__offset12b_pos__amdgpu_no_fine
; GFX12-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_cbranch_execnz .LBB61_1
; GFX12-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -18463,11 +18591,10 @@ define void @flat_system_atomic_fmin_noret_v2bf16__offset12b_pos__amdgpu_no_fine
; GFX12-FAKE16-NEXT: v_add3_u32 v8, v8, v6, 0x7fff
; GFX12-FAKE16-NEXT: v_cmp_u_f32_e64 s0, v2, v2
; GFX12-FAKE16-NEXT: s_wait_alu depctr_va_vcc(0)
-; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_2)
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_2) | instid1(VALU_DEP_1)
; GFX12-FAKE16-NEXT: v_cndmask_b32_e32 v6, v8, v10, vcc_lo
; GFX12-FAKE16-NEXT: s_wait_alu depctr_va_sdst(0)
; GFX12-FAKE16-NEXT: v_cndmask_b32_e64 v2, v7, v9, s0
-; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-FAKE16-NEXT: v_perm_b32 v2, v6, v2, 0x7060302
; GFX12-FAKE16-NEXT: global_wb scope:SCOPE_SYS
; GFX12-FAKE16-NEXT: s_wait_storecnt 0x0
@@ -18480,6 +18607,7 @@ define void @flat_system_atomic_fmin_noret_v2bf16__offset12b_pos__amdgpu_no_fine
; GFX12-FAKE16-NEXT: s_or_b32 s1, vcc_lo, s1
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: s_cbranch_execnz .LBB61_1
; GFX12-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s1
@@ -18569,7 +18697,7 @@ define void @flat_system_atomic_fmin_noret_v2bf16__offset12b_pos__amdgpu_no_fine
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v3
; GFX11-TRUE16-NEXT: v_mov_b32_e32 v3, v2
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB61_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -18603,7 +18731,7 @@ define void @flat_system_atomic_fmin_noret_v2bf16__offset12b_pos__amdgpu_no_fine
; GFX11-FAKE16-NEXT: v_add3_u32 v7, v7, v2, 0x7fff
; GFX11-FAKE16-NEXT: v_add3_u32 v8, v8, v6, 0x7fff
; GFX11-FAKE16-NEXT: v_cmp_u_f32_e64 s0, v2, v2
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_4)
; GFX11-FAKE16-NEXT: v_cndmask_b32_e32 v6, v8, v10, vcc_lo
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v2, v7, v9, s0
; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
@@ -18616,7 +18744,7 @@ define void @flat_system_atomic_fmin_noret_v2bf16__offset12b_pos__amdgpu_no_fine
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v3
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v3, v2
; GFX11-FAKE16-NEXT: s_or_b32 s1, vcc_lo, s1
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB61_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
diff --git a/llvm/test/CodeGen/AMDGPU/flat-atomicrmw-fsub.ll b/llvm/test/CodeGen/AMDGPU/flat-atomicrmw-fsub.ll
index 5f078cb96c5780..8bff68b311b56b 100644
--- a/llvm/test/CodeGen/AMDGPU/flat-atomicrmw-fsub.ll
+++ b/llvm/test/CodeGen/AMDGPU/flat-atomicrmw-fsub.ll
@@ -39,9 +39,11 @@ define float @flat_agent_atomic_fsub_ret_f32(ptr %ptr, float %val) #0 {
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB0_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: v_mov_b32_e32 v0, v3
; GFX12-NEXT: s_setpc_b64 s[30:31]
;
@@ -87,11 +89,12 @@ define float @flat_agent_atomic_fsub_ret_f32(ptr %ptr, float %val) #0 {
; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB0_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: v_mov_b32_e32 v0, v3
; GFX11-NEXT: s_setpc_b64 s[30:31]
;
@@ -235,9 +238,11 @@ define float @flat_agent_atomic_fsub_ret_f32__offset12b_pos(ptr %ptr, float %val
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB1_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: v_mov_b32_e32 v0, v3
; GFX12-NEXT: s_setpc_b64 s[30:31]
;
@@ -283,11 +288,12 @@ define float @flat_agent_atomic_fsub_ret_f32__offset12b_pos(ptr %ptr, float %val
; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB1_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: v_mov_b32_e32 v0, v3
; GFX11-NEXT: s_setpc_b64 s[30:31]
;
@@ -435,9 +441,11 @@ define float @flat_agent_atomic_fsub_ret_f32__offset12b_neg(ptr %ptr, float %val
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB2_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: v_mov_b32_e32 v0, v3
; GFX12-NEXT: s_setpc_b64 s[30:31]
;
@@ -474,7 +482,6 @@ define float @flat_agent_atomic_fsub_ret_f32__offset12b_neg(ptr %ptr, float %val
; GFX11: ; %bb.0:
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: v_add_co_u32 v3, vcc_lo, 0xfffff800, v0
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v4, null, -1, v1, vcc_lo
; GFX11-NEXT: s_mov_b32 s0, 0
; GFX11-NEXT: flat_load_b32 v0, v[3:4]
@@ -491,7 +498,7 @@ define float @flat_agent_atomic_fsub_ret_f32__offset12b_neg(ptr %ptr, float %val
; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v0, v1
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB2_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -649,6 +656,7 @@ define void @flat_agent_atomic_fsub_noret_f32(ptr %ptr, float %val) #0 {
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB3_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -694,7 +702,7 @@ define void @flat_agent_atomic_fsub_noret_f32(ptr %ptr, float %val) #0 {
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-NEXT: v_mov_b32_e32 v4, v3
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB3_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -835,6 +843,7 @@ define void @flat_agent_atomic_fsub_noret_f32__offset12b_pos(ptr %ptr, float %va
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB4_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -880,7 +889,7 @@ define void @flat_agent_atomic_fsub_noret_f32__offset12b_pos(ptr %ptr, float %va
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-NEXT: v_mov_b32_e32 v4, v3
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB4_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -1028,6 +1037,7 @@ define void @flat_agent_atomic_fsub_noret_f32__offset12b_neg(ptr %ptr, float %va
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB5_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -1066,7 +1076,6 @@ define void @flat_agent_atomic_fsub_noret_f32__offset12b_neg(ptr %ptr, float %va
; GFX11: ; %bb.0:
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: v_add_co_u32 v0, vcc_lo, 0xfffff800, v0
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, -1, v1, vcc_lo
; GFX11-NEXT: s_mov_b32 s0, 0
; GFX11-NEXT: flat_load_b32 v4, v[0:1]
@@ -1082,7 +1091,7 @@ define void @flat_agent_atomic_fsub_noret_f32__offset12b_neg(ptr %ptr, float %va
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-NEXT: v_mov_b32_e32 v4, v3
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB5_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -1242,9 +1251,11 @@ define float @flat_system_atomic_fsub_ret_f32__offset12b_pos(ptr %ptr, float %va
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB6_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: v_mov_b32_e32 v0, v3
; GFX12-NEXT: s_setpc_b64 s[30:31]
;
@@ -1290,11 +1301,12 @@ define float @flat_system_atomic_fsub_ret_f32__offset12b_pos(ptr %ptr, float %va
; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB6_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: v_mov_b32_e32 v0, v3
; GFX11-NEXT: s_setpc_b64 s[30:31]
;
@@ -1445,6 +1457,7 @@ define void @flat_system_atomic_fsub_noret_f32__offset12b_pos(ptr %ptr, float %v
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB7_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -1490,7 +1503,7 @@ define void @flat_system_atomic_fsub_noret_f32__offset12b_pos(ptr %ptr, float %v
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-NEXT: v_mov_b32_e32 v4, v3
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB7_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -1646,9 +1659,11 @@ define float @flat_agent_atomic_fsub_ret_f32__ftz(ptr %ptr, float %val) #1 {
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB8_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: v_mov_b32_e32 v0, v3
; GFX12-NEXT: s_setpc_b64 s[30:31]
;
@@ -1694,11 +1709,12 @@ define float @flat_agent_atomic_fsub_ret_f32__ftz(ptr %ptr, float %val) #1 {
; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB8_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: v_mov_b32_e32 v0, v3
; GFX11-NEXT: s_setpc_b64 s[30:31]
;
@@ -1842,9 +1858,11 @@ define float @flat_agent_atomic_fsub_ret_f32__offset12b_pos__ftz(ptr %ptr, float
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB9_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: v_mov_b32_e32 v0, v3
; GFX12-NEXT: s_setpc_b64 s[30:31]
;
@@ -1890,11 +1908,12 @@ define float @flat_agent_atomic_fsub_ret_f32__offset12b_pos__ftz(ptr %ptr, float
; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB9_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: v_mov_b32_e32 v0, v3
; GFX11-NEXT: s_setpc_b64 s[30:31]
;
@@ -2042,9 +2061,11 @@ define float @flat_agent_atomic_fsub_ret_f32__offset12b_neg__ftz(ptr %ptr, float
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB10_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: v_mov_b32_e32 v0, v3
; GFX12-NEXT: s_setpc_b64 s[30:31]
;
@@ -2081,7 +2102,6 @@ define float @flat_agent_atomic_fsub_ret_f32__offset12b_neg__ftz(ptr %ptr, float
; GFX11: ; %bb.0:
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: v_add_co_u32 v3, vcc_lo, 0xfffff800, v0
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v4, null, -1, v1, vcc_lo
; GFX11-NEXT: s_mov_b32 s0, 0
; GFX11-NEXT: flat_load_b32 v0, v[3:4]
@@ -2098,7 +2118,7 @@ define float @flat_agent_atomic_fsub_ret_f32__offset12b_neg__ftz(ptr %ptr, float
; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v0, v1
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB10_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -2256,6 +2276,7 @@ define void @flat_agent_atomic_fsub_noret_f32__ftz(ptr %ptr, float %val) #1 {
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB11_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -2301,7 +2322,7 @@ define void @flat_agent_atomic_fsub_noret_f32__ftz(ptr %ptr, float %val) #1 {
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-NEXT: v_mov_b32_e32 v4, v3
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB11_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -2442,6 +2463,7 @@ define void @flat_agent_atomic_fsub_noret_f32__offset12b_pos__ftz(ptr %ptr, floa
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB12_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -2487,7 +2509,7 @@ define void @flat_agent_atomic_fsub_noret_f32__offset12b_pos__ftz(ptr %ptr, floa
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-NEXT: v_mov_b32_e32 v4, v3
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB12_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -2635,6 +2657,7 @@ define void @flat_agent_atomic_fsub_noret_f32__offset12b_neg__ftz(ptr %ptr, floa
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB13_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -2673,7 +2696,6 @@ define void @flat_agent_atomic_fsub_noret_f32__offset12b_neg__ftz(ptr %ptr, floa
; GFX11: ; %bb.0:
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: v_add_co_u32 v0, vcc_lo, 0xfffff800, v0
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, -1, v1, vcc_lo
; GFX11-NEXT: s_mov_b32 s0, 0
; GFX11-NEXT: flat_load_b32 v4, v[0:1]
@@ -2689,7 +2711,7 @@ define void @flat_agent_atomic_fsub_noret_f32__offset12b_neg__ftz(ptr %ptr, floa
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-NEXT: v_mov_b32_e32 v4, v3
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB13_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -2849,9 +2871,11 @@ define float @flat_system_atomic_fsub_ret_f32__offset12b_pos__ftz(ptr %ptr, floa
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB14_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: v_mov_b32_e32 v0, v3
; GFX12-NEXT: s_setpc_b64 s[30:31]
;
@@ -2897,11 +2921,12 @@ define float @flat_system_atomic_fsub_ret_f32__offset12b_pos__ftz(ptr %ptr, floa
; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB14_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: v_mov_b32_e32 v0, v3
; GFX11-NEXT: s_setpc_b64 s[30:31]
;
@@ -3052,6 +3077,7 @@ define void @flat_system_atomic_fsub_noret_f32__offset12b_pos__ftz(ptr %ptr, flo
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB15_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -3097,7 +3123,7 @@ define void @flat_system_atomic_fsub_noret_f32__offset12b_pos__ftz(ptr %ptr, flo
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-NEXT: v_mov_b32_e32 v4, v3
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB15_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -3241,6 +3267,7 @@ define double @flat_agent_atomic_fsub_ret_f64(ptr %ptr, double %val) #0 {
; GFX12-NEXT: ; implicit-def: $vgpr4_vgpr5
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: v_cmpx_ne_u32_e32 s1, v1
+; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX12-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX12-NEXT: s_cbranch_execz .LBB16_4
; GFX12-NEXT: ; %bb.1: ; %atomicrmw.global
@@ -3261,6 +3288,7 @@ define double @flat_agent_atomic_fsub_ret_f64(ptr %ptr, double %val) #0 {
; GFX12-NEXT: s_or_b32 s1, vcc_lo, s1
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB16_2
; GFX12-NEXT: ; %bb.3: ; %Flow
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s1
@@ -3281,6 +3309,7 @@ define double @flat_agent_atomic_fsub_ret_f64(ptr %ptr, double %val) #0 {
; GFX12-NEXT: .LBB16_6: ; %atomicrmw.phi
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: v_dual_mov_b32 v0, v4 :: v_dual_mov_b32 v1, v5
; GFX12-NEXT: s_setpc_b64 s[30:31]
;
@@ -3338,6 +3367,7 @@ define double @flat_agent_atomic_fsub_ret_f64(ptr %ptr, double %val) #0 {
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: ; implicit-def: $vgpr4_vgpr5
; GFX11-NEXT: v_cmpx_ne_u32_e32 s1, v1
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX11-NEXT: s_cbranch_execz .LBB16_4
; GFX11-NEXT: ; %bb.1: ; %atomicrmw.global
@@ -3356,7 +3386,7 @@ define double @flat_agent_atomic_fsub_ret_f64(ptr %ptr, double %val) #0 {
; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: v_cmp_eq_u64_e32 vcc_lo, v[4:5], v[6:7]
; GFX11-NEXT: s_or_b32 s1, vcc_lo, s1
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
; GFX11-NEXT: s_cbranch_execnz .LBB16_2
; GFX11-NEXT: ; %bb.3: ; %Flow
@@ -3364,6 +3394,7 @@ define double @flat_agent_atomic_fsub_ret_f64(ptr %ptr, double %val) #0 {
; GFX11-NEXT: ; implicit-def: $vgpr0_vgpr1
; GFX11-NEXT: ; implicit-def: $vgpr2_vgpr3
; GFX11-NEXT: .LBB16_4: ; %Flow3
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB16_6
; GFX11-NEXT: ; %bb.5: ; %atomicrmw.private
@@ -3375,6 +3406,7 @@ define double @flat_agent_atomic_fsub_ret_f64(ptr %ptr, double %val) #0 {
; GFX11-NEXT: scratch_store_b64 v6, v[0:1], off
; GFX11-NEXT: .LBB16_6: ; %atomicrmw.phi
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: v_dual_mov_b32 v0, v4 :: v_dual_mov_b32 v1, v5
; GFX11-NEXT: s_setpc_b64 s[30:31]
;
@@ -3655,6 +3687,7 @@ define double @flat_agent_atomic_fsub_ret_f64__offset12b_pos(ptr %ptr, double %v
; GFX12-NEXT: ; implicit-def: $vgpr0_vgpr1
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: v_cmpx_ne_u32_e32 s1, v5
+; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX12-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX12-NEXT: s_cbranch_execnz .LBB17_3
; GFX12-NEXT: ; %bb.1: ; %Flow3
@@ -3683,11 +3716,13 @@ define double @flat_agent_atomic_fsub_ret_f64__offset12b_pos(ptr %ptr, double %v
; GFX12-NEXT: s_or_b32 s1, vcc_lo, s1
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB17_4
; GFX12-NEXT: ; %bb.5: ; %Flow
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s1
; GFX12-NEXT: ; implicit-def: $vgpr4_vgpr5
; GFX12-NEXT: ; implicit-def: $vgpr2_vgpr3
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX12-NEXT: s_cbranch_execz .LBB17_2
; GFX12-NEXT: .LBB17_6: ; %atomicrmw.private
@@ -3758,12 +3793,12 @@ define double @flat_agent_atomic_fsub_ret_f64__offset12b_pos(ptr %ptr, double %v
; GFX11: ; %bb.0:
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: v_add_co_u32 v4, vcc_lo, 0x7f8, v0
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v5, null, 0, v1, vcc_lo
; GFX11-NEXT: s_mov_b64 s[0:1], src_private_base
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: ; implicit-def: $vgpr0_vgpr1
; GFX11-NEXT: v_cmpx_ne_u32_e32 s1, v5
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB17_3
; GFX11-NEXT: ; %bb.1: ; %Flow3
@@ -3788,13 +3823,14 @@ define double @flat_agent_atomic_fsub_ret_f64__offset12b_pos(ptr %ptr, double %v
; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: v_cmp_eq_u64_e32 vcc_lo, v[0:1], v[8:9]
; GFX11-NEXT: s_or_b32 s1, vcc_lo, s1
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
; GFX11-NEXT: s_cbranch_execnz .LBB17_4
; GFX11-NEXT: ; %bb.5: ; %Flow
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s1
; GFX11-NEXT: ; implicit-def: $vgpr4_vgpr5
; GFX11-NEXT: ; implicit-def: $vgpr2_vgpr3
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB17_2
; GFX11-NEXT: .LBB17_6: ; %atomicrmw.private
@@ -4101,6 +4137,7 @@ define double @flat_agent_atomic_fsub_ret_f64__offset12b_neg(ptr %ptr, double %v
; GFX12-NEXT: ; implicit-def: $vgpr0_vgpr1
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: v_cmpx_ne_u32_e32 s1, v5
+; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX12-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX12-NEXT: s_cbranch_execnz .LBB18_3
; GFX12-NEXT: ; %bb.1: ; %Flow3
@@ -4129,11 +4166,13 @@ define double @flat_agent_atomic_fsub_ret_f64__offset12b_neg(ptr %ptr, double %v
; GFX12-NEXT: s_or_b32 s1, vcc_lo, s1
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB18_4
; GFX12-NEXT: ; %bb.5: ; %Flow
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s1
; GFX12-NEXT: ; implicit-def: $vgpr4_vgpr5
; GFX12-NEXT: ; implicit-def: $vgpr2_vgpr3
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX12-NEXT: s_cbranch_execz .LBB18_2
; GFX12-NEXT: .LBB18_6: ; %atomicrmw.private
@@ -4205,12 +4244,12 @@ define double @flat_agent_atomic_fsub_ret_f64__offset12b_neg(ptr %ptr, double %v
; GFX11: ; %bb.0:
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: v_add_co_u32 v4, vcc_lo, 0xfffff800, v0
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v5, null, -1, v1, vcc_lo
; GFX11-NEXT: s_mov_b64 s[0:1], src_private_base
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: ; implicit-def: $vgpr0_vgpr1
; GFX11-NEXT: v_cmpx_ne_u32_e32 s1, v5
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB18_3
; GFX11-NEXT: ; %bb.1: ; %Flow3
@@ -4235,13 +4274,14 @@ define double @flat_agent_atomic_fsub_ret_f64__offset12b_neg(ptr %ptr, double %v
; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: v_cmp_eq_u64_e32 vcc_lo, v[0:1], v[8:9]
; GFX11-NEXT: s_or_b32 s1, vcc_lo, s1
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
; GFX11-NEXT: s_cbranch_execnz .LBB18_4
; GFX11-NEXT: ; %bb.5: ; %Flow
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s1
; GFX11-NEXT: ; implicit-def: $vgpr4_vgpr5
; GFX11-NEXT: ; implicit-def: $vgpr2_vgpr3
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB18_2
; GFX11-NEXT: .LBB18_6: ; %atomicrmw.private
@@ -4544,6 +4584,7 @@ define void @flat_agent_atomic_fsub_noret_f64(ptr %ptr, double %val) #0 {
; GFX12-NEXT: s_mov_b32 s0, exec_lo
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: v_cmpx_ne_u32_e32 s1, v1
+; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX12-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX12-NEXT: s_cbranch_execnz .LBB19_3
; GFX12-NEXT: ; %bb.1: ; %Flow3
@@ -4571,11 +4612,13 @@ define void @flat_agent_atomic_fsub_noret_f64(ptr %ptr, double %val) #0 {
; GFX12-NEXT: s_or_b32 s1, vcc_lo, s1
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB19_4
; GFX12-NEXT: ; %bb.5: ; %Flow
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s1
; GFX12-NEXT: ; implicit-def: $vgpr0_vgpr1
; GFX12-NEXT: ; implicit-def: $vgpr2_vgpr3
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX12-NEXT: s_cbranch_execz .LBB19_2
; GFX12-NEXT: .LBB19_6: ; %atomicrmw.private
@@ -4645,6 +4688,7 @@ define void @flat_agent_atomic_fsub_noret_f64(ptr %ptr, double %val) #0 {
; GFX11-NEXT: s_mov_b64 s[0:1], src_private_base
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 s1, v1
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB19_3
; GFX11-NEXT: ; %bb.1: ; %Flow3
@@ -4668,13 +4712,14 @@ define void @flat_agent_atomic_fsub_noret_f64(ptr %ptr, double %val) #0 {
; GFX11-NEXT: v_cmp_eq_u64_e32 vcc_lo, v[4:5], v[6:7]
; GFX11-NEXT: v_dual_mov_b32 v7, v5 :: v_dual_mov_b32 v6, v4
; GFX11-NEXT: s_or_b32 s1, vcc_lo, s1
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
; GFX11-NEXT: s_cbranch_execnz .LBB19_4
; GFX11-NEXT: ; %bb.5: ; %Flow
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s1
; GFX11-NEXT: ; implicit-def: $vgpr0_vgpr1
; GFX11-NEXT: ; implicit-def: $vgpr2_vgpr3
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB19_2
; GFX11-NEXT: .LBB19_6: ; %atomicrmw.private
@@ -4964,6 +5009,7 @@ define void @flat_agent_atomic_fsub_noret_f64__offset12b_pos(ptr %ptr, double %v
; GFX12-NEXT: s_mov_b32 s0, exec_lo
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: v_cmpx_ne_u32_e32 s1, v1
+; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX12-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX12-NEXT: s_cbranch_execnz .LBB20_3
; GFX12-NEXT: ; %bb.1: ; %Flow3
@@ -4991,11 +5037,13 @@ define void @flat_agent_atomic_fsub_noret_f64__offset12b_pos(ptr %ptr, double %v
; GFX12-NEXT: s_or_b32 s1, vcc_lo, s1
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB20_4
; GFX12-NEXT: ; %bb.5: ; %Flow
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s1
; GFX12-NEXT: ; implicit-def: $vgpr0_vgpr1
; GFX12-NEXT: ; implicit-def: $vgpr2_vgpr3
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX12-NEXT: s_cbranch_execz .LBB20_2
; GFX12-NEXT: .LBB20_6: ; %atomicrmw.private
@@ -5065,11 +5113,11 @@ define void @flat_agent_atomic_fsub_noret_f64__offset12b_pos(ptr %ptr, double %v
; GFX11: ; %bb.0:
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: v_add_co_u32 v0, vcc_lo, 0x7f8, v0
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-NEXT: s_mov_b64 s[0:1], src_private_base
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 s1, v1
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB20_3
; GFX11-NEXT: ; %bb.1: ; %Flow3
@@ -5093,13 +5141,14 @@ define void @flat_agent_atomic_fsub_noret_f64__offset12b_pos(ptr %ptr, double %v
; GFX11-NEXT: v_cmp_eq_u64_e32 vcc_lo, v[4:5], v[6:7]
; GFX11-NEXT: v_dual_mov_b32 v7, v5 :: v_dual_mov_b32 v6, v4
; GFX11-NEXT: s_or_b32 s1, vcc_lo, s1
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
; GFX11-NEXT: s_cbranch_execnz .LBB20_4
; GFX11-NEXT: ; %bb.5: ; %Flow
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s1
; GFX11-NEXT: ; implicit-def: $vgpr0_vgpr1
; GFX11-NEXT: ; implicit-def: $vgpr2_vgpr3
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB20_2
; GFX11-NEXT: .LBB20_6: ; %atomicrmw.private
@@ -5400,6 +5449,7 @@ define void @flat_agent_atomic_fsub_noret_f64__offset12b_neg(ptr %ptr, double %v
; GFX12-NEXT: s_mov_b32 s0, exec_lo
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: v_cmpx_ne_u32_e32 s1, v1
+; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX12-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX12-NEXT: s_cbranch_execnz .LBB21_3
; GFX12-NEXT: ; %bb.1: ; %Flow3
@@ -5427,11 +5477,13 @@ define void @flat_agent_atomic_fsub_noret_f64__offset12b_neg(ptr %ptr, double %v
; GFX12-NEXT: s_or_b32 s1, vcc_lo, s1
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB21_4
; GFX12-NEXT: ; %bb.5: ; %Flow
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s1
; GFX12-NEXT: ; implicit-def: $vgpr0_vgpr1
; GFX12-NEXT: ; implicit-def: $vgpr2_vgpr3
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX12-NEXT: s_cbranch_execz .LBB21_2
; GFX12-NEXT: .LBB21_6: ; %atomicrmw.private
@@ -5502,11 +5554,11 @@ define void @flat_agent_atomic_fsub_noret_f64__offset12b_neg(ptr %ptr, double %v
; GFX11: ; %bb.0:
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: v_add_co_u32 v0, vcc_lo, 0xfffff800, v0
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, -1, v1, vcc_lo
; GFX11-NEXT: s_mov_b64 s[0:1], src_private_base
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 s1, v1
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB21_3
; GFX11-NEXT: ; %bb.1: ; %Flow3
@@ -5530,13 +5582,14 @@ define void @flat_agent_atomic_fsub_noret_f64__offset12b_neg(ptr %ptr, double %v
; GFX11-NEXT: v_cmp_eq_u64_e32 vcc_lo, v[4:5], v[6:7]
; GFX11-NEXT: v_dual_mov_b32 v7, v5 :: v_dual_mov_b32 v6, v4
; GFX11-NEXT: s_or_b32 s1, vcc_lo, s1
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
; GFX11-NEXT: s_cbranch_execnz .LBB21_4
; GFX11-NEXT: ; %bb.5: ; %Flow
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s1
; GFX11-NEXT: ; implicit-def: $vgpr0_vgpr1
; GFX11-NEXT: ; implicit-def: $vgpr2_vgpr3
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_saveexec_b32 s0, s0
; GFX11-NEXT: s_cbranch_execz .LBB21_2
; GFX11-NEXT: .LBB21_6: ; %atomicrmw.private
@@ -5865,9 +5918,11 @@ define half @flat_agent_atomic_fsub_ret_f16(ptr %ptr, half %val) #0 {
; GFX12-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_cbranch_execnz .LBB22_1
; GFX12-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: v_lshrrev_b32_e32 v0, v3, v5
; GFX12-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -5909,9 +5964,11 @@ define half @flat_agent_atomic_fsub_ret_f16(ptr %ptr, half %val) #0 {
; GFX12-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: s_cbranch_execnz .LBB22_1
; GFX12-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: v_lshrrev_b32_e32 v0, v3, v5
; GFX12-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -5981,11 +6038,12 @@ define half @flat_agent_atomic_fsub_ret_f16(ptr %ptr, half %val) #0 {
; GFX11-TRUE16-NEXT: buffer_gl0_inv
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v5, v6
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB22_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v0, v3, v5
; GFX11-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -6021,11 +6079,12 @@ define half @flat_agent_atomic_fsub_ret_f16(ptr %ptr, half %val) #0 {
; GFX11-FAKE16-NEXT: buffer_gl0_inv
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v5, v6
; GFX11-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB22_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v0, v3, v5
; GFX11-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -6237,9 +6296,11 @@ define half @flat_agent_atomic_fsub_ret_f16__offset12b_pos(ptr %ptr, half %val)
; GFX12-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_cbranch_execnz .LBB23_1
; GFX12-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: v_lshrrev_b32_e32 v0, v3, v5
; GFX12-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -6282,9 +6343,11 @@ define half @flat_agent_atomic_fsub_ret_f16__offset12b_pos(ptr %ptr, half %val)
; GFX12-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: s_cbranch_execnz .LBB23_1
; GFX12-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: v_lshrrev_b32_e32 v0, v3, v5
; GFX12-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -6328,7 +6391,6 @@ define half @flat_agent_atomic_fsub_ret_f16__offset12b_pos(ptr %ptr, half %val)
; GFX11-TRUE16: ; %bb.0:
; GFX11-TRUE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-TRUE16-NEXT: v_add_co_u32 v3, vcc_lo, 0x7fe, v0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-TRUE16-NEXT: s_mov_b32 s0, 0
; GFX11-TRUE16-NEXT: v_and_b32_e32 v0, -4, v3
@@ -6357,11 +6419,12 @@ define half @flat_agent_atomic_fsub_ret_f16__offset12b_pos(ptr %ptr, half %val)
; GFX11-TRUE16-NEXT: buffer_gl0_inv
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v5, v6
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB23_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v0, v3, v5
; GFX11-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -6369,7 +6432,6 @@ define half @flat_agent_atomic_fsub_ret_f16__offset12b_pos(ptr %ptr, half %val)
; GFX11-FAKE16: ; %bb.0:
; GFX11-FAKE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-FAKE16-NEXT: v_add_co_u32 v3, vcc_lo, 0x7fe, v0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-FAKE16-NEXT: s_mov_b32 s0, 0
; GFX11-FAKE16-NEXT: v_and_b32_e32 v0, -4, v3
@@ -6398,11 +6460,12 @@ define half @flat_agent_atomic_fsub_ret_f16__offset12b_pos(ptr %ptr, half %val)
; GFX11-FAKE16-NEXT: buffer_gl0_inv
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v5, v6
; GFX11-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB23_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v0, v3, v5
; GFX11-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -6620,9 +6683,11 @@ define half @flat_agent_atomic_fsub_ret_f16__offset12b_neg(ptr %ptr, half %val)
; GFX12-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_cbranch_execnz .LBB24_1
; GFX12-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: v_lshrrev_b32_e32 v0, v3, v5
; GFX12-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -6665,9 +6730,11 @@ define half @flat_agent_atomic_fsub_ret_f16__offset12b_neg(ptr %ptr, half %val)
; GFX12-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: s_cbranch_execnz .LBB24_1
; GFX12-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: v_lshrrev_b32_e32 v0, v3, v5
; GFX12-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -6712,7 +6779,6 @@ define half @flat_agent_atomic_fsub_ret_f16__offset12b_neg(ptr %ptr, half %val)
; GFX11-TRUE16: ; %bb.0:
; GFX11-TRUE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-TRUE16-NEXT: v_add_co_u32 v3, vcc_lo, 0xfffff800, v0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v1, null, -1, v1, vcc_lo
; GFX11-TRUE16-NEXT: s_mov_b32 s0, 0
; GFX11-TRUE16-NEXT: v_and_b32_e32 v0, -4, v3
@@ -6741,11 +6807,12 @@ define half @flat_agent_atomic_fsub_ret_f16__offset12b_neg(ptr %ptr, half %val)
; GFX11-TRUE16-NEXT: buffer_gl0_inv
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v5, v6
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB24_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v0, v3, v5
; GFX11-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -6753,7 +6820,6 @@ define half @flat_agent_atomic_fsub_ret_f16__offset12b_neg(ptr %ptr, half %val)
; GFX11-FAKE16: ; %bb.0:
; GFX11-FAKE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-FAKE16-NEXT: v_add_co_u32 v3, vcc_lo, 0xfffff800, v0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v1, null, -1, v1, vcc_lo
; GFX11-FAKE16-NEXT: s_mov_b32 s0, 0
; GFX11-FAKE16-NEXT: v_and_b32_e32 v0, -4, v3
@@ -6782,11 +6848,12 @@ define half @flat_agent_atomic_fsub_ret_f16__offset12b_neg(ptr %ptr, half %val)
; GFX11-FAKE16-NEXT: buffer_gl0_inv
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v5, v6
; GFX11-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB24_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v0, v3, v5
; GFX11-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -7002,6 +7069,7 @@ define void @flat_agent_atomic_fsub_noret_f16(ptr %ptr, half %val) #0 {
; GFX12-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_cbranch_execnz .LBB25_1
; GFX12-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -7044,6 +7112,7 @@ define void @flat_agent_atomic_fsub_noret_f16(ptr %ptr, half %val) #0 {
; GFX12-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: s_cbranch_execnz .LBB25_1
; GFX12-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -7113,7 +7182,7 @@ define void @flat_agent_atomic_fsub_noret_f16(ptr %ptr, half %val) #0 {
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-TRUE16-NEXT: v_mov_b32_e32 v4, v3
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB25_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -7151,7 +7220,7 @@ define void @flat_agent_atomic_fsub_noret_f16(ptr %ptr, half %val) #0 {
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v4, v3
; GFX11-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB25_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -7360,6 +7429,7 @@ define void @flat_agent_atomic_fsub_noret_f16__offset12b_pos(ptr %ptr, half %val
; GFX12-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_cbranch_execnz .LBB26_1
; GFX12-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -7403,6 +7473,7 @@ define void @flat_agent_atomic_fsub_noret_f16__offset12b_pos(ptr %ptr, half %val
; GFX12-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: s_cbranch_execnz .LBB26_1
; GFX12-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -7447,7 +7518,6 @@ define void @flat_agent_atomic_fsub_noret_f16__offset12b_pos(ptr %ptr, half %val
; GFX11-TRUE16: ; %bb.0:
; GFX11-TRUE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-TRUE16-NEXT: v_add_co_u32 v3, vcc_lo, 0x7fe, v0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-TRUE16-NEXT: s_mov_b32 s0, 0
; GFX11-TRUE16-NEXT: v_and_b32_e32 v0, -4, v3
@@ -7475,7 +7545,7 @@ define void @flat_agent_atomic_fsub_noret_f16__offset12b_pos(ptr %ptr, half %val
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-TRUE16-NEXT: v_mov_b32_e32 v4, v3
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB26_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -7486,7 +7556,6 @@ define void @flat_agent_atomic_fsub_noret_f16__offset12b_pos(ptr %ptr, half %val
; GFX11-FAKE16: ; %bb.0:
; GFX11-FAKE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-FAKE16-NEXT: v_add_co_u32 v3, vcc_lo, 0x7fe, v0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-FAKE16-NEXT: s_mov_b32 s0, 0
; GFX11-FAKE16-NEXT: v_and_b32_e32 v0, -4, v3
@@ -7514,7 +7583,7 @@ define void @flat_agent_atomic_fsub_noret_f16__offset12b_pos(ptr %ptr, half %val
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v4, v3
; GFX11-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB26_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -7729,6 +7798,7 @@ define void @flat_agent_atomic_fsub_noret_f16__offset12b_neg(ptr %ptr, half %val
; GFX12-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_cbranch_execnz .LBB27_1
; GFX12-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -7772,6 +7842,7 @@ define void @flat_agent_atomic_fsub_noret_f16__offset12b_neg(ptr %ptr, half %val
; GFX12-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: s_cbranch_execnz .LBB27_1
; GFX12-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -7817,7 +7888,6 @@ define void @flat_agent_atomic_fsub_noret_f16__offset12b_neg(ptr %ptr, half %val
; GFX11-TRUE16: ; %bb.0:
; GFX11-TRUE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-TRUE16-NEXT: v_add_co_u32 v3, vcc_lo, 0xfffff800, v0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v1, null, -1, v1, vcc_lo
; GFX11-TRUE16-NEXT: s_mov_b32 s0, 0
; GFX11-TRUE16-NEXT: v_and_b32_e32 v0, -4, v3
@@ -7845,7 +7915,7 @@ define void @flat_agent_atomic_fsub_noret_f16__offset12b_neg(ptr %ptr, half %val
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-TRUE16-NEXT: v_mov_b32_e32 v4, v3
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB27_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -7856,7 +7926,6 @@ define void @flat_agent_atomic_fsub_noret_f16__offset12b_neg(ptr %ptr, half %val
; GFX11-FAKE16: ; %bb.0:
; GFX11-FAKE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-FAKE16-NEXT: v_add_co_u32 v3, vcc_lo, 0xfffff800, v0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v1, null, -1, v1, vcc_lo
; GFX11-FAKE16-NEXT: s_mov_b32 s0, 0
; GFX11-FAKE16-NEXT: v_and_b32_e32 v0, -4, v3
@@ -7884,7 +7953,7 @@ define void @flat_agent_atomic_fsub_noret_f16__offset12b_neg(ptr %ptr, half %val
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v4, v3
; GFX11-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB27_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -8088,9 +8157,11 @@ define half @flat_agent_atomic_fsub_ret_f16__offset12b_pos__align4(ptr %ptr, hal
; GFX12-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_cbranch_execnz .LBB28_1
; GFX12-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: v_mov_b16_e32 v0.l, v3.l
; GFX12-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -8121,9 +8192,11 @@ define half @flat_agent_atomic_fsub_ret_f16__offset12b_pos__align4(ptr %ptr, hal
; GFX12-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: s_cbranch_execnz .LBB28_1
; GFX12-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: v_mov_b32_e32 v0, v3
; GFX12-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -8174,11 +8247,12 @@ define half @flat_agent_atomic_fsub_ret_f16__offset12b_pos__align4(ptr %ptr, hal
; GFX11-TRUE16-NEXT: buffer_gl0_inv
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB28_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v0.l, v3.l
; GFX11-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -8203,11 +8277,12 @@ define half @flat_agent_atomic_fsub_ret_f16__offset12b_pos__align4(ptr %ptr, hal
; GFX11-FAKE16-NEXT: buffer_gl0_inv
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB28_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v0, v3
; GFX11-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -8370,6 +8445,7 @@ define void @flat_agent_atomic_fsub_noret_f16__offset12b__align4_pos(ptr %ptr, h
; GFX12-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_cbranch_execnz .LBB29_1
; GFX12-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -8401,6 +8477,7 @@ define void @flat_agent_atomic_fsub_noret_f16__offset12b__align4_pos(ptr %ptr, h
; GFX12-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: s_cbranch_execnz .LBB29_1
; GFX12-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -8451,7 +8528,7 @@ define void @flat_agent_atomic_fsub_noret_f16__offset12b__align4_pos(ptr %ptr, h
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-TRUE16-NEXT: v_mov_b32_e32 v4, v3
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB29_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -8478,7 +8555,7 @@ define void @flat_agent_atomic_fsub_noret_f16__offset12b__align4_pos(ptr %ptr, h
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v4, v3
; GFX11-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB29_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -8656,9 +8733,11 @@ define half @flat_system_atomic_fsub_ret_f16__offset12b_pos(ptr %ptr, half %val)
; GFX12-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_cbranch_execnz .LBB30_1
; GFX12-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: v_lshrrev_b32_e32 v0, v3, v5
; GFX12-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -8702,9 +8781,11 @@ define half @flat_system_atomic_fsub_ret_f16__offset12b_pos(ptr %ptr, half %val)
; GFX12-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: s_cbranch_execnz .LBB30_1
; GFX12-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: v_lshrrev_b32_e32 v0, v3, v5
; GFX12-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -8748,7 +8829,6 @@ define half @flat_system_atomic_fsub_ret_f16__offset12b_pos(ptr %ptr, half %val)
; GFX11-TRUE16: ; %bb.0:
; GFX11-TRUE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-TRUE16-NEXT: v_add_co_u32 v3, vcc_lo, 0x7fe, v0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-TRUE16-NEXT: s_mov_b32 s0, 0
; GFX11-TRUE16-NEXT: v_and_b32_e32 v0, -4, v3
@@ -8777,11 +8857,12 @@ define half @flat_system_atomic_fsub_ret_f16__offset12b_pos(ptr %ptr, half %val)
; GFX11-TRUE16-NEXT: buffer_gl0_inv
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v5, v6
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB30_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v0, v3, v5
; GFX11-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -8789,7 +8870,6 @@ define half @flat_system_atomic_fsub_ret_f16__offset12b_pos(ptr %ptr, half %val)
; GFX11-FAKE16: ; %bb.0:
; GFX11-FAKE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-FAKE16-NEXT: v_add_co_u32 v3, vcc_lo, 0x7fe, v0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-FAKE16-NEXT: s_mov_b32 s0, 0
; GFX11-FAKE16-NEXT: v_and_b32_e32 v0, -4, v3
@@ -8818,11 +8898,12 @@ define half @flat_system_atomic_fsub_ret_f16__offset12b_pos(ptr %ptr, half %val)
; GFX11-FAKE16-NEXT: buffer_gl0_inv
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v5, v6
; GFX11-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB30_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v0, v3, v5
; GFX11-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -9043,6 +9124,7 @@ define void @flat_system_atomic_fsub_noret_f16__offset12b_pos(ptr %ptr, half %va
; GFX12-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_cbranch_execnz .LBB31_1
; GFX12-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -9087,6 +9169,7 @@ define void @flat_system_atomic_fsub_noret_f16__offset12b_pos(ptr %ptr, half %va
; GFX12-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: s_cbranch_execnz .LBB31_1
; GFX12-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -9131,7 +9214,6 @@ define void @flat_system_atomic_fsub_noret_f16__offset12b_pos(ptr %ptr, half %va
; GFX11-TRUE16: ; %bb.0:
; GFX11-TRUE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-TRUE16-NEXT: v_add_co_u32 v3, vcc_lo, 0x7fe, v0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-TRUE16-NEXT: s_mov_b32 s0, 0
; GFX11-TRUE16-NEXT: v_and_b32_e32 v0, -4, v3
@@ -9159,7 +9241,7 @@ define void @flat_system_atomic_fsub_noret_f16__offset12b_pos(ptr %ptr, half %va
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-TRUE16-NEXT: v_mov_b32_e32 v4, v3
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB31_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -9170,7 +9252,6 @@ define void @flat_system_atomic_fsub_noret_f16__offset12b_pos(ptr %ptr, half %va
; GFX11-FAKE16: ; %bb.0:
; GFX11-FAKE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-FAKE16-NEXT: v_add_co_u32 v3, vcc_lo, 0x7fe, v0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-FAKE16-NEXT: s_mov_b32 s0, 0
; GFX11-FAKE16-NEXT: v_and_b32_e32 v0, -4, v3
@@ -9198,7 +9279,7 @@ define void @flat_system_atomic_fsub_noret_f16__offset12b_pos(ptr %ptr, half %va
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v4, v3
; GFX11-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB31_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -9429,9 +9510,11 @@ define bfloat @flat_agent_atomic_fsub_ret_bf16(ptr %ptr, bfloat %val) #0 {
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB32_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: v_lshrrev_b32_e32 v0, v3, v5
; GFX12-NEXT: s_setpc_b64 s[30:31]
;
@@ -9519,11 +9602,12 @@ define bfloat @flat_agent_atomic_fsub_ret_bf16(ptr %ptr, bfloat %val) #0 {
; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v5, v6
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB32_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: v_lshrrev_b32_e32 v0, v3, v5
; GFX11-NEXT: s_setpc_b64 s[30:31]
;
@@ -9775,9 +9859,11 @@ define bfloat @flat_agent_atomic_fsub_ret_bf16__offset12b_pos(ptr %ptr, bfloat %
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB33_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: v_lshrrev_b32_e32 v0, v3, v5
; GFX12-NEXT: s_setpc_b64 s[30:31]
;
@@ -9830,16 +9916,16 @@ define bfloat @flat_agent_atomic_fsub_ret_bf16__offset12b_pos(ptr %ptr, bfloat %
; GFX11: ; %bb.0:
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: v_add_co_u32 v3, vcc_lo, 0x7fe, v0
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_3)
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-NEXT: v_lshlrev_b32_e32 v2, 16, v2
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_4) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_and_b32_e32 v0, -4, v3
; GFX11-NEXT: v_and_b32_e32 v3, 3, v3
; GFX11-NEXT: s_mov_b32 s0, 0
; GFX11-NEXT: flat_load_b32 v5, v[0:1]
; GFX11-NEXT: v_lshlrev_b32_e32 v3, 3, v3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_lshlrev_b32_e64 v4, v3, 0xffff
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_not_b32_e32 v4, v4
; GFX11-NEXT: .p2align 6
; GFX11-NEXT: .LBB33_1: ; %atomicrmw.start
@@ -9869,11 +9955,12 @@ define bfloat @flat_agent_atomic_fsub_ret_bf16__offset12b_pos(ptr %ptr, bfloat %
; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v5, v6
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB33_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: v_lshrrev_b32_e32 v0, v3, v5
; GFX11-NEXT: s_setpc_b64 s[30:31]
;
@@ -10131,9 +10218,11 @@ define bfloat @flat_agent_atomic_fsub_ret_bf16__offset12b_neg(ptr %ptr, bfloat %
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB34_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: v_lshrrev_b32_e32 v0, v3, v5
; GFX12-NEXT: s_setpc_b64 s[30:31]
;
@@ -10187,16 +10276,16 @@ define bfloat @flat_agent_atomic_fsub_ret_bf16__offset12b_neg(ptr %ptr, bfloat %
; GFX11: ; %bb.0:
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: v_add_co_u32 v3, vcc_lo, 0xfffff800, v0
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_3)
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, -1, v1, vcc_lo
; GFX11-NEXT: v_lshlrev_b32_e32 v2, 16, v2
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_4) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_and_b32_e32 v0, -4, v3
; GFX11-NEXT: v_and_b32_e32 v3, 3, v3
; GFX11-NEXT: s_mov_b32 s0, 0
; GFX11-NEXT: flat_load_b32 v5, v[0:1]
; GFX11-NEXT: v_lshlrev_b32_e32 v3, 3, v3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_lshlrev_b32_e64 v4, v3, 0xffff
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_not_b32_e32 v4, v4
; GFX11-NEXT: .p2align 6
; GFX11-NEXT: .LBB34_1: ; %atomicrmw.start
@@ -10226,11 +10315,12 @@ define bfloat @flat_agent_atomic_fsub_ret_bf16__offset12b_neg(ptr %ptr, bfloat %
; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v5, v6
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB34_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: v_lshrrev_b32_e32 v0, v3, v5
; GFX11-NEXT: s_setpc_b64 s[30:31]
;
@@ -10484,6 +10574,7 @@ define void @flat_agent_atomic_fsub_noret_bf16(ptr %ptr, bfloat %val) #0 {
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB35_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -10571,7 +10662,7 @@ define void @flat_agent_atomic_fsub_noret_bf16(ptr %ptr, bfloat %val) #0 {
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-NEXT: v_mov_b32_e32 v4, v3
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB35_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -10820,6 +10911,7 @@ define void @flat_agent_atomic_fsub_noret_bf16__offset12b_pos(ptr %ptr, bfloat %
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB36_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -10873,16 +10965,16 @@ define void @flat_agent_atomic_fsub_noret_bf16__offset12b_pos(ptr %ptr, bfloat %
; GFX11: ; %bb.0:
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: v_add_co_u32 v4, vcc_lo, 0x7fe, v0
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_3)
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-NEXT: v_lshlrev_b32_e32 v6, 16, v2
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_4) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_and_b32_e32 v0, -4, v4
; GFX11-NEXT: v_and_b32_e32 v4, 3, v4
; GFX11-NEXT: s_mov_b32 s0, 0
; GFX11-NEXT: flat_load_b32 v3, v[0:1]
; GFX11-NEXT: v_lshlrev_b32_e32 v4, 3, v4
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_lshlrev_b32_e64 v5, v4, 0xffff
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_not_b32_e32 v5, v5
; GFX11-NEXT: .p2align 6
; GFX11-NEXT: .LBB36_1: ; %atomicrmw.start
@@ -10911,7 +11003,7 @@ define void @flat_agent_atomic_fsub_noret_bf16__offset12b_pos(ptr %ptr, bfloat %
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v3
; GFX11-NEXT: v_mov_b32_e32 v3, v2
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB36_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -11166,6 +11258,7 @@ define void @flat_agent_atomic_fsub_noret_bf16__offset12b_neg(ptr %ptr, bfloat %
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB37_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -11220,16 +11313,16 @@ define void @flat_agent_atomic_fsub_noret_bf16__offset12b_neg(ptr %ptr, bfloat %
; GFX11: ; %bb.0:
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: v_add_co_u32 v4, vcc_lo, 0xfffff800, v0
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_3)
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, -1, v1, vcc_lo
; GFX11-NEXT: v_lshlrev_b32_e32 v6, 16, v2
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_4) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_and_b32_e32 v0, -4, v4
; GFX11-NEXT: v_and_b32_e32 v4, 3, v4
; GFX11-NEXT: s_mov_b32 s0, 0
; GFX11-NEXT: flat_load_b32 v3, v[0:1]
; GFX11-NEXT: v_lshlrev_b32_e32 v4, 3, v4
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_lshlrev_b32_e64 v5, v4, 0xffff
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_not_b32_e32 v5, v5
; GFX11-NEXT: .p2align 6
; GFX11-NEXT: .LBB37_1: ; %atomicrmw.start
@@ -11258,7 +11351,7 @@ define void @flat_agent_atomic_fsub_noret_bf16__offset12b_neg(ptr %ptr, bfloat %
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v3
; GFX11-NEXT: v_mov_b32_e32 v3, v2
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB37_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -11501,9 +11594,11 @@ define bfloat @flat_agent_atomic_fsub_ret_bf16__offset12b_pos__align4(ptr %ptr,
; GFX12-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_cbranch_execnz .LBB38_1
; GFX12-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: v_mov_b16_e32 v0.l, v3.l
; GFX12-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -11544,9 +11639,11 @@ define bfloat @flat_agent_atomic_fsub_ret_bf16__offset12b_pos__align4(ptr %ptr,
; GFX12-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: s_cbranch_execnz .LBB38_1
; GFX12-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: v_mov_b32_e32 v0, v3
; GFX12-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -11617,11 +11714,12 @@ define bfloat @flat_agent_atomic_fsub_ret_bf16__offset12b_pos__align4(ptr %ptr,
; GFX11-TRUE16-NEXT: buffer_gl0_inv
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB38_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v0.l, v3.l
; GFX11-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -11656,11 +11754,12 @@ define bfloat @flat_agent_atomic_fsub_ret_bf16__offset12b_pos__align4(ptr %ptr,
; GFX11-FAKE16-NEXT: buffer_gl0_inv
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB38_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v0, v3
; GFX11-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -11868,6 +11967,7 @@ define void @flat_agent_atomic_fsub_noret_bf16__offset12b__align4_pos(ptr %ptr,
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB39_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -11938,7 +12038,7 @@ define void @flat_agent_atomic_fsub_noret_bf16__offset12b__align4_pos(ptr %ptr,
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v3
; GFX11-NEXT: v_mov_b32_e32 v3, v2
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB39_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -12162,9 +12262,11 @@ define bfloat @flat_system_atomic_fsub_ret_bf16__offset12b_pos(ptr %ptr, bfloat
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB40_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: v_lshrrev_b32_e32 v0, v3, v5
; GFX12-NEXT: s_setpc_b64 s[30:31]
;
@@ -12217,16 +12319,16 @@ define bfloat @flat_system_atomic_fsub_ret_bf16__offset12b_pos(ptr %ptr, bfloat
; GFX11: ; %bb.0:
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: v_add_co_u32 v3, vcc_lo, 0x7fe, v0
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_3)
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-NEXT: v_lshlrev_b32_e32 v2, 16, v2
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_4) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_and_b32_e32 v0, -4, v3
; GFX11-NEXT: v_and_b32_e32 v3, 3, v3
; GFX11-NEXT: s_mov_b32 s0, 0
; GFX11-NEXT: flat_load_b32 v5, v[0:1]
; GFX11-NEXT: v_lshlrev_b32_e32 v3, 3, v3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_lshlrev_b32_e64 v4, v3, 0xffff
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_not_b32_e32 v4, v4
; GFX11-NEXT: .p2align 6
; GFX11-NEXT: .LBB40_1: ; %atomicrmw.start
@@ -12256,11 +12358,12 @@ define bfloat @flat_system_atomic_fsub_ret_bf16__offset12b_pos(ptr %ptr, bfloat
; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v5, v6
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB40_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: v_lshrrev_b32_e32 v0, v3, v5
; GFX11-NEXT: s_setpc_b64 s[30:31]
;
@@ -12521,6 +12624,7 @@ define void @flat_system_atomic_fsub_noret_bf16__offset12b_pos(ptr %ptr, bfloat
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB41_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -12574,16 +12678,16 @@ define void @flat_system_atomic_fsub_noret_bf16__offset12b_pos(ptr %ptr, bfloat
; GFX11: ; %bb.0:
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: v_add_co_u32 v4, vcc_lo, 0x7fe, v0
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_3)
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-NEXT: v_lshlrev_b32_e32 v6, 16, v2
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_4) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_and_b32_e32 v0, -4, v4
; GFX11-NEXT: v_and_b32_e32 v4, 3, v4
; GFX11-NEXT: s_mov_b32 s0, 0
; GFX11-NEXT: flat_load_b32 v3, v[0:1]
; GFX11-NEXT: v_lshlrev_b32_e32 v4, 3, v4
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_lshlrev_b32_e64 v5, v4, 0xffff
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_not_b32_e32 v5, v5
; GFX11-NEXT: .p2align 6
; GFX11-NEXT: .LBB41_1: ; %atomicrmw.start
@@ -12612,7 +12716,7 @@ define void @flat_system_atomic_fsub_noret_bf16__offset12b_pos(ptr %ptr, bfloat
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v3
; GFX11-NEXT: v_mov_b32_e32 v3, v2
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB41_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -12849,9 +12953,11 @@ define <2 x half> @flat_agent_atomic_fsub_ret_v2f16(ptr %ptr, <2 x half> %val) #
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB42_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: v_mov_b32_e32 v0, v3
; GFX12-NEXT: s_setpc_b64 s[30:31]
;
@@ -12897,11 +13003,12 @@ define <2 x half> @flat_agent_atomic_fsub_ret_v2f16(ptr %ptr, <2 x half> %val) #
; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB42_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: v_mov_b32_e32 v0, v3
; GFX11-NEXT: s_setpc_b64 s[30:31]
;
@@ -13063,9 +13170,11 @@ define <2 x half> @flat_agent_atomic_fsub_ret_v2f16__offset12b_pos(ptr %ptr, <2
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB43_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: v_mov_b32_e32 v0, v3
; GFX12-NEXT: s_setpc_b64 s[30:31]
;
@@ -13111,11 +13220,12 @@ define <2 x half> @flat_agent_atomic_fsub_ret_v2f16__offset12b_pos(ptr %ptr, <2
; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB43_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: v_mov_b32_e32 v0, v3
; GFX11-NEXT: s_setpc_b64 s[30:31]
;
@@ -13282,9 +13392,11 @@ define <2 x half> @flat_agent_atomic_fsub_ret_v2f16__offset12b_neg(ptr %ptr, <2
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB44_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: v_mov_b32_e32 v0, v3
; GFX12-NEXT: s_setpc_b64 s[30:31]
;
@@ -13321,7 +13433,6 @@ define <2 x half> @flat_agent_atomic_fsub_ret_v2f16__offset12b_neg(ptr %ptr, <2
; GFX11: ; %bb.0:
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: v_add_co_u32 v3, vcc_lo, 0xfffff800, v0
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v4, null, -1, v1, vcc_lo
; GFX11-NEXT: s_mov_b32 s0, 0
; GFX11-NEXT: flat_load_b32 v0, v[3:4]
@@ -13338,7 +13449,7 @@ define <2 x half> @flat_agent_atomic_fsub_ret_v2f16__offset12b_neg(ptr %ptr, <2
; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v0, v1
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB44_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -13515,6 +13626,7 @@ define void @flat_agent_atomic_fsub_noret_v2f16(ptr %ptr, <2 x half> %val) #0 {
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB45_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -13560,7 +13672,7 @@ define void @flat_agent_atomic_fsub_noret_v2f16(ptr %ptr, <2 x half> %val) #0 {
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-NEXT: v_mov_b32_e32 v4, v3
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB45_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -13717,6 +13829,7 @@ define void @flat_agent_atomic_fsub_noret_v2f16__offset12b_pos(ptr %ptr, <2 x ha
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB46_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -13762,7 +13875,7 @@ define void @flat_agent_atomic_fsub_noret_v2f16__offset12b_pos(ptr %ptr, <2 x ha
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-NEXT: v_mov_b32_e32 v4, v3
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB46_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -13926,6 +14039,7 @@ define void @flat_agent_atomic_fsub_noret_v2f16__offset12b_neg(ptr %ptr, <2 x ha
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB47_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -13964,7 +14078,6 @@ define void @flat_agent_atomic_fsub_noret_v2f16__offset12b_neg(ptr %ptr, <2 x ha
; GFX11: ; %bb.0:
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: v_add_co_u32 v0, vcc_lo, 0xfffff800, v0
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, -1, v1, vcc_lo
; GFX11-NEXT: s_mov_b32 s0, 0
; GFX11-NEXT: flat_load_b32 v4, v[0:1]
@@ -13980,7 +14093,7 @@ define void @flat_agent_atomic_fsub_noret_v2f16__offset12b_neg(ptr %ptr, <2 x ha
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-NEXT: v_mov_b32_e32 v4, v3
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB47_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -14156,9 +14269,11 @@ define <2 x half> @flat_system_atomic_fsub_ret_v2f16__offset12b_pos(ptr %ptr, <2
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB48_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: v_mov_b32_e32 v0, v3
; GFX12-NEXT: s_setpc_b64 s[30:31]
;
@@ -14204,11 +14319,12 @@ define <2 x half> @flat_system_atomic_fsub_ret_v2f16__offset12b_pos(ptr %ptr, <2
; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB48_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: v_mov_b32_e32 v0, v3
; GFX11-NEXT: s_setpc_b64 s[30:31]
;
@@ -14378,6 +14494,7 @@ define void @flat_system_atomic_fsub_noret_v2f16__offset12b_pos(ptr %ptr, <2 x h
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB49_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -14423,7 +14540,7 @@ define void @flat_system_atomic_fsub_noret_v2f16__offset12b_pos(ptr %ptr, <2 x h
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-NEXT: v_mov_b32_e32 v4, v3
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB49_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -14620,9 +14737,11 @@ define <2 x bfloat> @flat_agent_atomic_fsub_ret_v2bf16(ptr %ptr, <2 x bfloat> %v
; GFX12-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_cbranch_execnz .LBB50_1
; GFX12-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: v_mov_b32_e32 v0, v3
; GFX12-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -14671,9 +14790,11 @@ define <2 x bfloat> @flat_agent_atomic_fsub_ret_v2bf16(ptr %ptr, <2 x bfloat> %v
; GFX12-FAKE16-NEXT: s_or_b32 s1, vcc_lo, s1
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: s_cbranch_execnz .LBB50_1
; GFX12-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s1
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: v_mov_b32_e32 v0, v3
; GFX12-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -14763,12 +14884,13 @@ define <2 x bfloat> @flat_agent_atomic_fsub_ret_v2bf16(ptr %ptr, <2 x bfloat> %v
; GFX11-TRUE16-NEXT: buffer_gl0_inv
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v6
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB50_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-TRUE16-NEXT: s_set_inst_prefetch_distance 0x2
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_mov_b32_e32 v0, v3
; GFX11-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -14812,12 +14934,13 @@ define <2 x bfloat> @flat_agent_atomic_fsub_ret_v2bf16(ptr %ptr, <2 x bfloat> %v
; GFX11-FAKE16-NEXT: buffer_gl0_inv
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v6
; GFX11-FAKE16-NEXT: s_or_b32 s1, vcc_lo, s1
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB50_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-FAKE16-NEXT: s_set_inst_prefetch_distance 0x2
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s1
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v0, v3
; GFX11-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -15078,9 +15201,11 @@ define <2 x bfloat> @flat_agent_atomic_fsub_ret_v2bf16__offset12b_pos(ptr %ptr,
; GFX12-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_cbranch_execnz .LBB51_1
; GFX12-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: v_mov_b32_e32 v0, v3
; GFX12-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -15129,9 +15254,11 @@ define <2 x bfloat> @flat_agent_atomic_fsub_ret_v2bf16__offset12b_pos(ptr %ptr,
; GFX12-FAKE16-NEXT: s_or_b32 s1, vcc_lo, s1
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: s_cbranch_execnz .LBB51_1
; GFX12-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s1
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: v_mov_b32_e32 v0, v3
; GFX12-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -15221,12 +15348,13 @@ define <2 x bfloat> @flat_agent_atomic_fsub_ret_v2bf16__offset12b_pos(ptr %ptr,
; GFX11-TRUE16-NEXT: buffer_gl0_inv
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v6
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB51_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-TRUE16-NEXT: s_set_inst_prefetch_distance 0x2
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_mov_b32_e32 v0, v3
; GFX11-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -15270,12 +15398,13 @@ define <2 x bfloat> @flat_agent_atomic_fsub_ret_v2bf16__offset12b_pos(ptr %ptr,
; GFX11-FAKE16-NEXT: buffer_gl0_inv
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v6
; GFX11-FAKE16-NEXT: s_or_b32 s1, vcc_lo, s1
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB51_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-FAKE16-NEXT: s_set_inst_prefetch_distance 0x2
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s1
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v0, v3
; GFX11-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -15541,9 +15670,11 @@ define <2 x bfloat> @flat_agent_atomic_fsub_ret_v2bf16__offset12b_neg(ptr %ptr,
; GFX12-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_cbranch_execnz .LBB52_1
; GFX12-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: v_mov_b32_e32 v0, v3
; GFX12-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -15592,9 +15723,11 @@ define <2 x bfloat> @flat_agent_atomic_fsub_ret_v2bf16__offset12b_neg(ptr %ptr,
; GFX12-FAKE16-NEXT: s_or_b32 s1, vcc_lo, s1
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: s_cbranch_execnz .LBB52_1
; GFX12-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s1
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: v_mov_b32_e32 v0, v3
; GFX12-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -15650,7 +15783,6 @@ define <2 x bfloat> @flat_agent_atomic_fsub_ret_v2bf16__offset12b_neg(ptr %ptr,
; GFX11-TRUE16: ; %bb.0:
; GFX11-TRUE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-TRUE16-NEXT: v_add_co_u32 v3, vcc_lo, 0xfffff800, v0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v4, null, -1, v1, vcc_lo
; GFX11-TRUE16-NEXT: v_and_b32_e32 v1, 0xffff0000, v2
; GFX11-TRUE16-NEXT: v_lshlrev_b32_e32 v2, 16, v2
@@ -15692,7 +15824,7 @@ define <2 x bfloat> @flat_agent_atomic_fsub_ret_v2bf16__offset12b_neg(ptr %ptr,
; GFX11-TRUE16-NEXT: buffer_gl0_inv
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v0, v6
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB52_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -15704,7 +15836,6 @@ define <2 x bfloat> @flat_agent_atomic_fsub_ret_v2bf16__offset12b_neg(ptr %ptr,
; GFX11-FAKE16: ; %bb.0:
; GFX11-FAKE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-FAKE16-NEXT: v_add_co_u32 v3, vcc_lo, 0xfffff800, v0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v4, null, -1, v1, vcc_lo
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v1, 16, v2
; GFX11-FAKE16-NEXT: v_and_b32_e32 v2, 0xffff0000, v2
@@ -15743,7 +15874,7 @@ define <2 x bfloat> @flat_agent_atomic_fsub_ret_v2bf16__offset12b_neg(ptr %ptr,
; GFX11-FAKE16-NEXT: buffer_gl0_inv
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v0, v6
; GFX11-FAKE16-NEXT: s_or_b32 s1, vcc_lo, s1
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB52_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -16019,6 +16150,7 @@ define void @flat_agent_atomic_fsub_noret_v2bf16(ptr %ptr, <2 x bfloat> %val) #0
; GFX12-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_cbranch_execnz .LBB53_1
; GFX12-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -16053,11 +16185,10 @@ define void @flat_agent_atomic_fsub_noret_v2bf16(ptr %ptr, <2 x bfloat> %val) #0
; GFX12-FAKE16-NEXT: v_add3_u32 v8, v8, v6, 0x7fff
; GFX12-FAKE16-NEXT: v_cmp_u_f32_e64 s0, v2, v2
; GFX12-FAKE16-NEXT: s_wait_alu depctr_va_vcc(0)
-; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_2)
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_2) | instid1(VALU_DEP_1)
; GFX12-FAKE16-NEXT: v_cndmask_b32_e32 v6, v8, v10, vcc_lo
; GFX12-FAKE16-NEXT: s_wait_alu depctr_va_sdst(0)
; GFX12-FAKE16-NEXT: v_cndmask_b32_e64 v2, v7, v9, s0
-; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-FAKE16-NEXT: v_perm_b32 v2, v6, v2, 0x7060302
; GFX12-FAKE16-NEXT: s_wait_storecnt 0x0
; GFX12-FAKE16-NEXT: flat_atomic_cmpswap_b32 v2, v[0:1], v[2:3] th:TH_ATOMIC_RETURN scope:SCOPE_DEV
@@ -16069,6 +16200,7 @@ define void @flat_agent_atomic_fsub_noret_v2bf16(ptr %ptr, <2 x bfloat> %val) #0
; GFX12-FAKE16-NEXT: s_or_b32 s1, vcc_lo, s1
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: s_cbranch_execnz .LBB53_1
; GFX12-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s1
@@ -16158,7 +16290,7 @@ define void @flat_agent_atomic_fsub_noret_v2bf16(ptr %ptr, <2 x bfloat> %val) #0
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v3
; GFX11-TRUE16-NEXT: v_mov_b32_e32 v3, v2
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB53_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -16192,7 +16324,7 @@ define void @flat_agent_atomic_fsub_noret_v2bf16(ptr %ptr, <2 x bfloat> %val) #0
; GFX11-FAKE16-NEXT: v_add3_u32 v7, v7, v2, 0x7fff
; GFX11-FAKE16-NEXT: v_add3_u32 v8, v8, v6, 0x7fff
; GFX11-FAKE16-NEXT: v_cmp_u_f32_e64 s0, v2, v2
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_4)
; GFX11-FAKE16-NEXT: v_cndmask_b32_e32 v6, v8, v10, vcc_lo
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v2, v7, v9, s0
; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
@@ -16205,7 +16337,7 @@ define void @flat_agent_atomic_fsub_noret_v2bf16(ptr %ptr, <2 x bfloat> %val) #0
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v3
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v3, v2
; GFX11-FAKE16-NEXT: s_or_b32 s1, vcc_lo, s1
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB53_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -16460,6 +16592,7 @@ define void @flat_agent_atomic_fsub_noret_v2bf16__offset12b_pos(ptr %ptr, <2 x b
; GFX12-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_cbranch_execnz .LBB54_1
; GFX12-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -16494,11 +16627,10 @@ define void @flat_agent_atomic_fsub_noret_v2bf16__offset12b_pos(ptr %ptr, <2 x b
; GFX12-FAKE16-NEXT: v_add3_u32 v8, v8, v6, 0x7fff
; GFX12-FAKE16-NEXT: v_cmp_u_f32_e64 s0, v2, v2
; GFX12-FAKE16-NEXT: s_wait_alu depctr_va_vcc(0)
-; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_2)
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_2) | instid1(VALU_DEP_1)
; GFX12-FAKE16-NEXT: v_cndmask_b32_e32 v6, v8, v10, vcc_lo
; GFX12-FAKE16-NEXT: s_wait_alu depctr_va_sdst(0)
; GFX12-FAKE16-NEXT: v_cndmask_b32_e64 v2, v7, v9, s0
-; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-FAKE16-NEXT: v_perm_b32 v2, v6, v2, 0x7060302
; GFX12-FAKE16-NEXT: s_wait_storecnt 0x0
; GFX12-FAKE16-NEXT: flat_atomic_cmpswap_b32 v2, v[0:1], v[2:3] offset:2044 th:TH_ATOMIC_RETURN scope:SCOPE_DEV
@@ -16510,6 +16642,7 @@ define void @flat_agent_atomic_fsub_noret_v2bf16__offset12b_pos(ptr %ptr, <2 x b
; GFX12-FAKE16-NEXT: s_or_b32 s1, vcc_lo, s1
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: s_cbranch_execnz .LBB54_1
; GFX12-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s1
@@ -16599,7 +16732,7 @@ define void @flat_agent_atomic_fsub_noret_v2bf16__offset12b_pos(ptr %ptr, <2 x b
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v3
; GFX11-TRUE16-NEXT: v_mov_b32_e32 v3, v2
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB54_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -16633,7 +16766,7 @@ define void @flat_agent_atomic_fsub_noret_v2bf16__offset12b_pos(ptr %ptr, <2 x b
; GFX11-FAKE16-NEXT: v_add3_u32 v7, v7, v2, 0x7fff
; GFX11-FAKE16-NEXT: v_add3_u32 v8, v8, v6, 0x7fff
; GFX11-FAKE16-NEXT: v_cmp_u_f32_e64 s0, v2, v2
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_4)
; GFX11-FAKE16-NEXT: v_cndmask_b32_e32 v6, v8, v10, vcc_lo
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v2, v7, v9, s0
; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
@@ -16646,7 +16779,7 @@ define void @flat_agent_atomic_fsub_noret_v2bf16__offset12b_pos(ptr %ptr, <2 x b
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v3
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v3, v2
; GFX11-FAKE16-NEXT: s_or_b32 s1, vcc_lo, s1
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB54_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -16908,6 +17041,7 @@ define void @flat_agent_atomic_fsub_noret_v2bf16__offset12b_neg(ptr %ptr, <2 x b
; GFX12-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_cbranch_execnz .LBB55_1
; GFX12-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -16942,11 +17076,10 @@ define void @flat_agent_atomic_fsub_noret_v2bf16__offset12b_neg(ptr %ptr, <2 x b
; GFX12-FAKE16-NEXT: v_add3_u32 v8, v8, v6, 0x7fff
; GFX12-FAKE16-NEXT: v_cmp_u_f32_e64 s0, v2, v2
; GFX12-FAKE16-NEXT: s_wait_alu depctr_va_vcc(0)
-; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_2)
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_2) | instid1(VALU_DEP_1)
; GFX12-FAKE16-NEXT: v_cndmask_b32_e32 v6, v8, v10, vcc_lo
; GFX12-FAKE16-NEXT: s_wait_alu depctr_va_sdst(0)
; GFX12-FAKE16-NEXT: v_cndmask_b32_e64 v2, v7, v9, s0
-; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-FAKE16-NEXT: v_perm_b32 v2, v6, v2, 0x7060302
; GFX12-FAKE16-NEXT: s_wait_storecnt 0x0
; GFX12-FAKE16-NEXT: flat_atomic_cmpswap_b32 v2, v[0:1], v[2:3] offset:-2048 th:TH_ATOMIC_RETURN scope:SCOPE_DEV
@@ -16958,6 +17091,7 @@ define void @flat_agent_atomic_fsub_noret_v2bf16__offset12b_neg(ptr %ptr, <2 x b
; GFX12-FAKE16-NEXT: s_or_b32 s1, vcc_lo, s1
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: s_cbranch_execnz .LBB55_1
; GFX12-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s1
@@ -17015,7 +17149,6 @@ define void @flat_agent_atomic_fsub_noret_v2bf16__offset12b_neg(ptr %ptr, <2 x b
; GFX11-TRUE16: ; %bb.0:
; GFX11-TRUE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-TRUE16-NEXT: v_add_co_u32 v0, vcc_lo, 0xfffff800, v0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v1, null, -1, v1, vcc_lo
; GFX11-TRUE16-NEXT: v_and_b32_e32 v4, 0xffff0000, v2
; GFX11-TRUE16-NEXT: v_lshlrev_b32_e32 v5, 16, v2
@@ -17056,7 +17189,7 @@ define void @flat_agent_atomic_fsub_noret_v2bf16__offset12b_neg(ptr %ptr, <2 x b
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v3
; GFX11-TRUE16-NEXT: v_mov_b32_e32 v3, v2
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB55_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -17068,7 +17201,6 @@ define void @flat_agent_atomic_fsub_noret_v2bf16__offset12b_neg(ptr %ptr, <2 x b
; GFX11-FAKE16: ; %bb.0:
; GFX11-FAKE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-FAKE16-NEXT: v_add_co_u32 v0, vcc_lo, 0xfffff800, v0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v1, null, -1, v1, vcc_lo
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v4, 16, v2
; GFX11-FAKE16-NEXT: v_and_b32_e32 v5, 0xffff0000, v2
@@ -17093,7 +17225,7 @@ define void @flat_agent_atomic_fsub_noret_v2bf16__offset12b_neg(ptr %ptr, <2 x b
; GFX11-FAKE16-NEXT: v_add3_u32 v7, v7, v2, 0x7fff
; GFX11-FAKE16-NEXT: v_add3_u32 v8, v8, v6, 0x7fff
; GFX11-FAKE16-NEXT: v_cmp_u_f32_e64 s0, v2, v2
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_4)
; GFX11-FAKE16-NEXT: v_cndmask_b32_e32 v6, v8, v10, vcc_lo
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v2, v7, v9, s0
; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
@@ -17106,7 +17238,7 @@ define void @flat_agent_atomic_fsub_noret_v2bf16__offset12b_neg(ptr %ptr, <2 x b
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v3
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v3, v2
; GFX11-FAKE16-NEXT: s_or_b32 s1, vcc_lo, s1
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB55_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -17381,9 +17513,11 @@ define <2 x bfloat> @flat_system_atomic_fsub_ret_v2bf16__offset12b_pos(ptr %ptr,
; GFX12-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_cbranch_execnz .LBB56_1
; GFX12-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: v_mov_b32_e32 v0, v3
; GFX12-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -17433,9 +17567,11 @@ define <2 x bfloat> @flat_system_atomic_fsub_ret_v2bf16__offset12b_pos(ptr %ptr,
; GFX12-FAKE16-NEXT: s_or_b32 s1, vcc_lo, s1
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: s_cbranch_execnz .LBB56_1
; GFX12-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s1
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: v_mov_b32_e32 v0, v3
; GFX12-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -17525,12 +17661,13 @@ define <2 x bfloat> @flat_system_atomic_fsub_ret_v2bf16__offset12b_pos(ptr %ptr,
; GFX11-TRUE16-NEXT: buffer_gl0_inv
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v6
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB56_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-TRUE16-NEXT: s_set_inst_prefetch_distance 0x2
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_mov_b32_e32 v0, v3
; GFX11-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -17574,12 +17711,13 @@ define <2 x bfloat> @flat_system_atomic_fsub_ret_v2bf16__offset12b_pos(ptr %ptr,
; GFX11-FAKE16-NEXT: buffer_gl0_inv
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v6
; GFX11-FAKE16-NEXT: s_or_b32 s1, vcc_lo, s1
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB56_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-FAKE16-NEXT: s_set_inst_prefetch_distance 0x2
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s1
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v0, v3
; GFX11-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -17847,6 +17985,7 @@ define void @flat_system_atomic_fsub_noret_v2bf16__offset12b_pos(ptr %ptr, <2 x
; GFX12-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_cbranch_execnz .LBB57_1
; GFX12-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -17881,11 +18020,10 @@ define void @flat_system_atomic_fsub_noret_v2bf16__offset12b_pos(ptr %ptr, <2 x
; GFX12-FAKE16-NEXT: v_add3_u32 v8, v8, v6, 0x7fff
; GFX12-FAKE16-NEXT: v_cmp_u_f32_e64 s0, v2, v2
; GFX12-FAKE16-NEXT: s_wait_alu depctr_va_vcc(0)
-; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_2)
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_2) | instid1(VALU_DEP_1)
; GFX12-FAKE16-NEXT: v_cndmask_b32_e32 v6, v8, v10, vcc_lo
; GFX12-FAKE16-NEXT: s_wait_alu depctr_va_sdst(0)
; GFX12-FAKE16-NEXT: v_cndmask_b32_e64 v2, v7, v9, s0
-; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-FAKE16-NEXT: v_perm_b32 v2, v6, v2, 0x7060302
; GFX12-FAKE16-NEXT: global_wb scope:SCOPE_SYS
; GFX12-FAKE16-NEXT: s_wait_storecnt 0x0
@@ -17898,6 +18036,7 @@ define void @flat_system_atomic_fsub_noret_v2bf16__offset12b_pos(ptr %ptr, <2 x
; GFX12-FAKE16-NEXT: s_or_b32 s1, vcc_lo, s1
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: s_cbranch_execnz .LBB57_1
; GFX12-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s1
@@ -17987,7 +18126,7 @@ define void @flat_system_atomic_fsub_noret_v2bf16__offset12b_pos(ptr %ptr, <2 x
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v3
; GFX11-TRUE16-NEXT: v_mov_b32_e32 v3, v2
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB57_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -18021,7 +18160,7 @@ define void @flat_system_atomic_fsub_noret_v2bf16__offset12b_pos(ptr %ptr, <2 x
; GFX11-FAKE16-NEXT: v_add3_u32 v7, v7, v2, 0x7fff
; GFX11-FAKE16-NEXT: v_add3_u32 v8, v8, v6, 0x7fff
; GFX11-FAKE16-NEXT: v_cmp_u_f32_e64 s0, v2, v2
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_4)
; GFX11-FAKE16-NEXT: v_cndmask_b32_e32 v6, v8, v10, vcc_lo
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v2, v7, v9, s0
; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
@@ -18034,7 +18173,7 @@ define void @flat_system_atomic_fsub_noret_v2bf16__offset12b_pos(ptr %ptr, <2 x
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v3
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v3, v2
; GFX11-FAKE16-NEXT: s_or_b32 s1, vcc_lo, s1
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB57_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
diff --git a/llvm/test/CodeGen/AMDGPU/flat-load-saddr-to-vaddr.ll b/llvm/test/CodeGen/AMDGPU/flat-load-saddr-to-vaddr.ll
index 4fed18d6eb0d87..f047892b5c8ba8 100644
--- a/llvm/test/CodeGen/AMDGPU/flat-load-saddr-to-vaddr.ll
+++ b/llvm/test/CodeGen/AMDGPU/flat-load-saddr-to-vaddr.ll
@@ -34,6 +34,7 @@ define amdgpu_kernel void @test_move_load_address_to_vgpr(ptr addrspace(1) nocap
; GCN-NEXT: v_add_nc_u64_e32 v[0:1], 4, v[0:1]
; GCN-NEXT: v_add_co_u32 v2, s0, v2, 1
; GCN-NEXT: s_and_b32 vcc_lo, exec_lo, s0
+; GCN-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GCN-NEXT: s_cbranch_vccz .LBB0_1
; GCN-NEXT: ; %bb.2: ; %bb2
; GCN-NEXT: s_endpgm
diff --git a/llvm/test/CodeGen/AMDGPU/flat-saddr-atomics.ll b/llvm/test/CodeGen/AMDGPU/flat-saddr-atomics.ll
index 11e0574c9f6cce..88b08321352883 100644
--- a/llvm/test/CodeGen/AMDGPU/flat-saddr-atomics.ll
+++ b/llvm/test/CodeGen/AMDGPU/flat-saddr-atomics.ll
@@ -32,7 +32,7 @@ define amdgpu_ps void @flat_xchg_saddr_i32_nortn(ptr inreg %sbase, i32 %voffset,
; GFX1250-GISEL-NEXT: v_nop
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v2, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: global_wb scope:SCOPE_DEV
@@ -99,7 +99,7 @@ define amdgpu_ps void @flat_xchg_saddr_i32_nortn_offset_2047(ptr inreg %sbase, i
; GFX1250-GISEL-NEXT: v_nop
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v2, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: global_wb scope:SCOPE_DEV
@@ -167,7 +167,7 @@ define amdgpu_ps void @flat_xchg_saddr_i32_nortn_offset_neg2048(ptr inreg %sbase
; GFX1250-GISEL-NEXT: v_nop
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v2, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: global_wb scope:SCOPE_DEV
@@ -240,7 +240,7 @@ define amdgpu_ps float @flat_xchg_saddr_i32_rtn(ptr inreg %sbase, i32 %voffset,
; GFX1250-GISEL-NEXT: v_nop
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v2, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: global_wb scope:SCOPE_DEV
@@ -307,7 +307,7 @@ define amdgpu_ps float @flat_xchg_saddr_i32_rtn_2048(ptr inreg %sbase, i32 %voff
; GFX1250-GISEL-NEXT: v_nop
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v2, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: global_wb scope:SCOPE_DEV
@@ -375,7 +375,7 @@ define amdgpu_ps float @flat_xchg_saddr_i32_rtn_neg2048(ptr inreg %sbase, i32 %v
; GFX1250-GISEL-NEXT: v_nop
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v2, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: global_wb scope:SCOPE_DEV
@@ -460,7 +460,6 @@ define amdgpu_ps float @flat_xchg_saddr_uniform_ptr_in_vgprs_rtn(i32 %voffset, i
; GFX1250-GISEL-NEXT: ds_load_b64 v[2:3], v2
; GFX1250-GISEL-NEXT: s_wait_dscnt 0x0
; GFX1250-GISEL-NEXT: v_add_co_u32 v2, vcc_lo, v2, v0
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: global_wb scope:SCOPE_DEV
; GFX1250-GISEL-NEXT: s_wait_storecnt 0x0
@@ -536,7 +535,6 @@ define amdgpu_ps float @flat_xchg_saddr_uniform_ptr_in_vgprs_rtn_immoffset(i32 %
; GFX1250-GISEL-NEXT: ds_load_b64 v[2:3], v2
; GFX1250-GISEL-NEXT: s_wait_dscnt 0x0
; GFX1250-GISEL-NEXT: v_add_co_u32 v2, vcc_lo, v2, v0
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: global_wb scope:SCOPE_DEV
; GFX1250-GISEL-NEXT: s_wait_storecnt 0x0
@@ -613,7 +611,6 @@ define amdgpu_ps void @flat_xchg_saddr_uniform_ptr_in_vgprs_nortn(i32 %voffset,
; GFX1250-GISEL-NEXT: ds_load_b64 v[2:3], v2
; GFX1250-GISEL-NEXT: s_wait_dscnt 0x0
; GFX1250-GISEL-NEXT: v_add_co_u32 v2, vcc_lo, v2, v0
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: global_wb scope:SCOPE_DEV
; GFX1250-GISEL-NEXT: s_wait_storecnt 0x0
@@ -688,7 +685,6 @@ define amdgpu_ps void @flat_xchg_saddr_uniform_ptr_in_vgprs_nortn_immoffset(i32
; GFX1250-GISEL-NEXT: ds_load_b64 v[2:3], v2
; GFX1250-GISEL-NEXT: s_wait_dscnt 0x0
; GFX1250-GISEL-NEXT: v_add_co_u32 v2, vcc_lo, v2, v0
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: global_wb scope:SCOPE_DEV
; GFX1250-GISEL-NEXT: s_wait_storecnt 0x0
@@ -799,7 +795,7 @@ define amdgpu_ps <2 x float> @flat_xchg_saddr_i64_rtn(ptr inreg %sbase, i32 %vof
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_dual_mov_b32 v4, v1 :: v_dual_mov_b32 v5, v2
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v2, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
@@ -989,10 +985,10 @@ define amdgpu_ps <2 x float> @flat_xchg_saddr_i64_rtn_neg128(ptr inreg %sbase, i
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_dual_mov_b32 v4, v1 :: v_dual_mov_b32 v5, v2
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v6, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v7, null, 0, v3, vcc_lo
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v2, vcc_lo, 0xffffff80, v6
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v3, null, -1, v7, vcc_lo
; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
@@ -1142,9 +1138,10 @@ define amdgpu_ps void @flat_xchg_saddr_i64_nortn(ptr inreg %sbase, i32 %voffset,
; GFX1250-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1250-SDAG-NEXT: v_add_nc_u64_e32 v[0:1], s[2:3], v[0:1]
; GFX1250-SDAG-NEXT: v_xor_b32_e32 v4, src_flat_scratch_base_hi, v1
-; GFX1250-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1250-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1250-SDAG-NEXT: v_cmpx_lt_u32_e32 0x3ffffff, v4
; GFX1250-SDAG-NEXT: s_xor_b32 s0, exec_lo, s0
+; GFX1250-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1250-SDAG-NEXT: s_cbranch_execnz .LBB12_3
; GFX1250-SDAG-NEXT: ; %bb.1: ; %Flow
; GFX1250-SDAG-NEXT: s_and_not1_saveexec_b32 s0, s0
@@ -1179,12 +1176,13 @@ define amdgpu_ps void @flat_xchg_saddr_i64_nortn(ptr inreg %sbase, i32 %voffset,
; GFX1250-GISEL-NEXT: v_dual_mov_b32 v4, v1 :: v_dual_mov_b32 v5, v2
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
; GFX1250-GISEL-NEXT: s_mov_b32 s0, exec_lo
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1250-GISEL-NEXT: v_xor_b32_e32 v2, src_flat_scratch_base_hi, v1
; GFX1250-GISEL-NEXT: v_cmpx_le_u32_e32 0x4000000, v2
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1250-GISEL-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX1250-GISEL-NEXT: s_cbranch_execnz .LBB12_3
; GFX1250-GISEL-NEXT: ; %bb.1: ; %Flow
@@ -1300,6 +1298,7 @@ define amdgpu_ps void @flat_xchg_saddr_i64_nortn_neg128(ptr inreg %sbase, i32 %v
; GFX1250-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1250-SDAG-NEXT: v_xor_b32_e32 v4, src_flat_scratch_base_hi, v1
; GFX1250-SDAG-NEXT: v_cmpx_lt_u32_e32 0x3ffffff, v4
+; GFX1250-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1250-SDAG-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX1250-SDAG-NEXT: s_cbranch_execnz .LBB13_3
; GFX1250-SDAG-NEXT: ; %bb.1: ; %Flow
@@ -1335,15 +1334,16 @@ define amdgpu_ps void @flat_xchg_saddr_i64_nortn_neg128(ptr inreg %sbase, i32 %v
; GFX1250-GISEL-NEXT: v_dual_mov_b32 v4, v1 :: v_dual_mov_b32 v5, v2
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
; GFX1250-GISEL-NEXT: s_mov_b32 s0, exec_lo
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v2, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, 0xffffff80, v2
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, -1, v3, vcc_lo
; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1250-GISEL-NEXT: v_xor_b32_e32 v6, src_flat_scratch_base_hi, v1
; GFX1250-GISEL-NEXT: v_cmpx_le_u32_e32 0x4000000, v6
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1250-GISEL-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX1250-GISEL-NEXT: s_cbranch_execnz .LBB13_3
; GFX1250-GISEL-NEXT: ; %bb.1: ; %Flow
@@ -1479,7 +1479,7 @@ define amdgpu_ps float @flat_add_saddr_i32_rtn(ptr inreg %sbase, i32 %voffset, i
; GFX1250-GISEL-NEXT: v_nop
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v2, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: global_wb scope:SCOPE_DEV
@@ -1546,7 +1546,7 @@ define amdgpu_ps float @flat_add_saddr_i32_rtn_neg128(ptr inreg %sbase, i32 %vof
; GFX1250-GISEL-NEXT: v_nop
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v2, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: global_wb scope:SCOPE_DEV
@@ -1620,7 +1620,7 @@ define amdgpu_ps void @flat_add_saddr_i32_nortn(ptr inreg %sbase, i32 %voffset,
; GFX1250-GISEL-NEXT: v_nop
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v2, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: global_wb scope:SCOPE_DEV
@@ -1686,7 +1686,7 @@ define amdgpu_ps void @flat_add_saddr_i32_nortn_neg128(ptr inreg %sbase, i32 %vo
; GFX1250-GISEL-NEXT: v_nop
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v2, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: global_wb scope:SCOPE_DEV
@@ -1791,7 +1791,7 @@ define amdgpu_ps <2 x float> @flat_add_saddr_i64_rtn(ptr inreg %sbase, i32 %voff
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_dual_mov_b32 v4, v1 :: v_dual_mov_b32 v5, v2
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v2, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
@@ -1985,10 +1985,10 @@ define amdgpu_ps <2 x float> @flat_add_saddr_i64_rtn_neg128(ptr inreg %sbase, i3
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_dual_mov_b32 v4, v1 :: v_dual_mov_b32 v5, v2
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v6, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v7, null, 0, v3, vcc_lo
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v2, vcc_lo, 0xffffff80, v6
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v3, null, -1, v7, vcc_lo
; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
@@ -2142,9 +2142,10 @@ define amdgpu_ps void @flat_add_saddr_i64_nortn(ptr inreg %sbase, i32 %voffset,
; GFX1250-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1250-SDAG-NEXT: v_add_nc_u64_e32 v[0:1], s[2:3], v[0:1]
; GFX1250-SDAG-NEXT: v_xor_b32_e32 v4, src_flat_scratch_base_hi, v1
-; GFX1250-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1250-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1250-SDAG-NEXT: v_cmpx_lt_u32_e32 0x3ffffff, v4
; GFX1250-SDAG-NEXT: s_xor_b32 s0, exec_lo, s0
+; GFX1250-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1250-SDAG-NEXT: s_cbranch_execnz .LBB20_3
; GFX1250-SDAG-NEXT: ; %bb.1: ; %Flow
; GFX1250-SDAG-NEXT: s_and_not1_saveexec_b32 s0, s0
@@ -2182,12 +2183,13 @@ define amdgpu_ps void @flat_add_saddr_i64_nortn(ptr inreg %sbase, i32 %voffset,
; GFX1250-GISEL-NEXT: v_dual_mov_b32 v4, v1 :: v_dual_mov_b32 v5, v2
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
; GFX1250-GISEL-NEXT: s_mov_b32 s0, exec_lo
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1250-GISEL-NEXT: v_xor_b32_e32 v2, src_flat_scratch_base_hi, v1
; GFX1250-GISEL-NEXT: v_cmpx_le_u32_e32 0x4000000, v2
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1250-GISEL-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX1250-GISEL-NEXT: s_cbranch_execnz .LBB20_3
; GFX1250-GISEL-NEXT: ; %bb.1: ; %Flow
@@ -2314,6 +2316,7 @@ define amdgpu_ps void @flat_add_saddr_i64_nortn_neg128(ptr inreg %sbase, i32 %vo
; GFX1250-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1250-SDAG-NEXT: v_xor_b32_e32 v4, src_flat_scratch_base_hi, v1
; GFX1250-SDAG-NEXT: v_cmpx_lt_u32_e32 0x3ffffff, v4
+; GFX1250-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1250-SDAG-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX1250-SDAG-NEXT: s_cbranch_execnz .LBB21_3
; GFX1250-SDAG-NEXT: ; %bb.1: ; %Flow
@@ -2352,15 +2355,16 @@ define amdgpu_ps void @flat_add_saddr_i64_nortn_neg128(ptr inreg %sbase, i32 %vo
; GFX1250-GISEL-NEXT: v_dual_mov_b32 v4, v1 :: v_dual_mov_b32 v5, v2
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
; GFX1250-GISEL-NEXT: s_mov_b32 s0, exec_lo
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v2, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, 0xffffff80, v2
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, -1, v3, vcc_lo
; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1250-GISEL-NEXT: v_xor_b32_e32 v6, src_flat_scratch_base_hi, v1
; GFX1250-GISEL-NEXT: v_cmpx_le_u32_e32 0x4000000, v6
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1250-GISEL-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX1250-GISEL-NEXT: s_cbranch_execnz .LBB21_3
; GFX1250-GISEL-NEXT: ; %bb.1: ; %Flow
@@ -2507,7 +2511,7 @@ define amdgpu_ps float @flat_sub_saddr_i32_rtn(ptr inreg %sbase, i32 %voffset, i
; GFX1250-GISEL-NEXT: v_nop
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v2, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: global_wb scope:SCOPE_DEV
@@ -2574,7 +2578,7 @@ define amdgpu_ps float @flat_sub_saddr_i32_rtn_neg128(ptr inreg %sbase, i32 %vof
; GFX1250-GISEL-NEXT: v_nop
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v2, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: global_wb scope:SCOPE_DEV
@@ -2648,7 +2652,7 @@ define amdgpu_ps void @flat_sub_saddr_i32_nortn(ptr inreg %sbase, i32 %voffset,
; GFX1250-GISEL-NEXT: v_nop
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v2, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: global_wb scope:SCOPE_DEV
@@ -2714,7 +2718,7 @@ define amdgpu_ps void @flat_sub_saddr_i32_nortn_neg128(ptr inreg %sbase, i32 %vo
; GFX1250-GISEL-NEXT: v_nop
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v2, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: global_wb scope:SCOPE_DEV
@@ -2819,7 +2823,7 @@ define amdgpu_ps <2 x float> @flat_sub_saddr_i64_rtn(ptr inreg %sbase, i32 %voff
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_dual_mov_b32 v4, v1 :: v_dual_mov_b32 v5, v2
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v2, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
@@ -3015,10 +3019,10 @@ define amdgpu_ps <2 x float> @flat_sub_saddr_i64_rtn_neg128(ptr inreg %sbase, i3
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_dual_mov_b32 v4, v1 :: v_dual_mov_b32 v5, v2
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v6, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v7, null, 0, v3, vcc_lo
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v2, vcc_lo, 0xffffff80, v6
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v3, null, -1, v7, vcc_lo
; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
@@ -3174,9 +3178,10 @@ define amdgpu_ps void @flat_sub_saddr_i64_nortn(ptr inreg %sbase, i32 %voffset,
; GFX1250-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1250-SDAG-NEXT: v_add_nc_u64_e32 v[0:1], s[2:3], v[0:1]
; GFX1250-SDAG-NEXT: v_xor_b32_e32 v4, src_flat_scratch_base_hi, v1
-; GFX1250-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1250-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1250-SDAG-NEXT: v_cmpx_lt_u32_e32 0x3ffffff, v4
; GFX1250-SDAG-NEXT: s_xor_b32 s0, exec_lo, s0
+; GFX1250-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1250-SDAG-NEXT: s_cbranch_execnz .LBB28_3
; GFX1250-SDAG-NEXT: ; %bb.1: ; %Flow
; GFX1250-SDAG-NEXT: s_and_not1_saveexec_b32 s0, s0
@@ -3214,12 +3219,13 @@ define amdgpu_ps void @flat_sub_saddr_i64_nortn(ptr inreg %sbase, i32 %voffset,
; GFX1250-GISEL-NEXT: v_dual_mov_b32 v4, v1 :: v_dual_mov_b32 v5, v2
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
; GFX1250-GISEL-NEXT: s_mov_b32 s0, exec_lo
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1250-GISEL-NEXT: v_xor_b32_e32 v2, src_flat_scratch_base_hi, v1
; GFX1250-GISEL-NEXT: v_cmpx_le_u32_e32 0x4000000, v2
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1250-GISEL-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX1250-GISEL-NEXT: s_cbranch_execnz .LBB28_3
; GFX1250-GISEL-NEXT: ; %bb.1: ; %Flow
@@ -3348,6 +3354,7 @@ define amdgpu_ps void @flat_sub_saddr_i64_nortn_neg128(ptr inreg %sbase, i32 %vo
; GFX1250-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1250-SDAG-NEXT: v_xor_b32_e32 v4, src_flat_scratch_base_hi, v1
; GFX1250-SDAG-NEXT: v_cmpx_lt_u32_e32 0x3ffffff, v4
+; GFX1250-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1250-SDAG-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX1250-SDAG-NEXT: s_cbranch_execnz .LBB29_3
; GFX1250-SDAG-NEXT: ; %bb.1: ; %Flow
@@ -3386,15 +3393,16 @@ define amdgpu_ps void @flat_sub_saddr_i64_nortn_neg128(ptr inreg %sbase, i32 %vo
; GFX1250-GISEL-NEXT: v_dual_mov_b32 v4, v1 :: v_dual_mov_b32 v5, v2
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
; GFX1250-GISEL-NEXT: s_mov_b32 s0, exec_lo
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v2, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, 0xffffff80, v2
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, -1, v3, vcc_lo
; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1250-GISEL-NEXT: v_xor_b32_e32 v6, src_flat_scratch_base_hi, v1
; GFX1250-GISEL-NEXT: v_cmpx_le_u32_e32 0x4000000, v6
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1250-GISEL-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX1250-GISEL-NEXT: s_cbranch_execnz .LBB29_3
; GFX1250-GISEL-NEXT: ; %bb.1: ; %Flow
@@ -3543,7 +3551,7 @@ define amdgpu_ps float @flat_and_saddr_i32_rtn(ptr inreg %sbase, i32 %voffset, i
; GFX1250-GISEL-NEXT: v_nop
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v2, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: global_wb scope:SCOPE_DEV
@@ -3610,7 +3618,7 @@ define amdgpu_ps float @flat_and_saddr_i32_rtn_neg128(ptr inreg %sbase, i32 %vof
; GFX1250-GISEL-NEXT: v_nop
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v2, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: global_wb scope:SCOPE_DEV
@@ -3684,7 +3692,7 @@ define amdgpu_ps void @flat_and_saddr_i32_nortn(ptr inreg %sbase, i32 %voffset,
; GFX1250-GISEL-NEXT: v_nop
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v2, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: global_wb scope:SCOPE_DEV
@@ -3750,7 +3758,7 @@ define amdgpu_ps void @flat_and_saddr_i32_nortn_neg128(ptr inreg %sbase, i32 %vo
; GFX1250-GISEL-NEXT: v_nop
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v2, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: global_wb scope:SCOPE_DEV
@@ -3856,7 +3864,7 @@ define amdgpu_ps <2 x float> @flat_and_saddr_i64_rtn(ptr inreg %sbase, i32 %voff
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_dual_mov_b32 v4, v1 :: v_dual_mov_b32 v5, v2
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v2, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
@@ -4052,10 +4060,10 @@ define amdgpu_ps <2 x float> @flat_and_saddr_i64_rtn_neg128(ptr inreg %sbase, i3
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_dual_mov_b32 v4, v1 :: v_dual_mov_b32 v5, v2
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v6, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v7, null, 0, v3, vcc_lo
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v2, vcc_lo, 0xffffff80, v6
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v3, null, -1, v7, vcc_lo
; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
@@ -4210,9 +4218,10 @@ define amdgpu_ps void @flat_and_saddr_i64_nortn(ptr inreg %sbase, i32 %voffset,
; GFX1250-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1250-SDAG-NEXT: v_add_nc_u64_e32 v[0:1], s[2:3], v[0:1]
; GFX1250-SDAG-NEXT: v_xor_b32_e32 v4, src_flat_scratch_base_hi, v1
-; GFX1250-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1250-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1250-SDAG-NEXT: v_cmpx_lt_u32_e32 0x3ffffff, v4
; GFX1250-SDAG-NEXT: s_xor_b32 s0, exec_lo, s0
+; GFX1250-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1250-SDAG-NEXT: s_cbranch_execnz .LBB36_3
; GFX1250-SDAG-NEXT: ; %bb.1: ; %Flow
; GFX1250-SDAG-NEXT: s_and_not1_saveexec_b32 s0, s0
@@ -4251,12 +4260,13 @@ define amdgpu_ps void @flat_and_saddr_i64_nortn(ptr inreg %sbase, i32 %voffset,
; GFX1250-GISEL-NEXT: v_dual_mov_b32 v4, v1 :: v_dual_mov_b32 v5, v2
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
; GFX1250-GISEL-NEXT: s_mov_b32 s0, exec_lo
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1250-GISEL-NEXT: v_xor_b32_e32 v2, src_flat_scratch_base_hi, v1
; GFX1250-GISEL-NEXT: v_cmpx_le_u32_e32 0x4000000, v2
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1250-GISEL-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX1250-GISEL-NEXT: s_cbranch_execnz .LBB36_3
; GFX1250-GISEL-NEXT: ; %bb.1: ; %Flow
@@ -4384,6 +4394,7 @@ define amdgpu_ps void @flat_and_saddr_i64_nortn_neg128(ptr inreg %sbase, i32 %vo
; GFX1250-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1250-SDAG-NEXT: v_xor_b32_e32 v4, src_flat_scratch_base_hi, v1
; GFX1250-SDAG-NEXT: v_cmpx_lt_u32_e32 0x3ffffff, v4
+; GFX1250-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1250-SDAG-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX1250-SDAG-NEXT: s_cbranch_execnz .LBB37_3
; GFX1250-SDAG-NEXT: ; %bb.1: ; %Flow
@@ -4423,15 +4434,16 @@ define amdgpu_ps void @flat_and_saddr_i64_nortn_neg128(ptr inreg %sbase, i32 %vo
; GFX1250-GISEL-NEXT: v_dual_mov_b32 v4, v1 :: v_dual_mov_b32 v5, v2
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
; GFX1250-GISEL-NEXT: s_mov_b32 s0, exec_lo
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v2, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, 0xffffff80, v2
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, -1, v3, vcc_lo
; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1250-GISEL-NEXT: v_xor_b32_e32 v6, src_flat_scratch_base_hi, v1
; GFX1250-GISEL-NEXT: v_cmpx_le_u32_e32 0x4000000, v6
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1250-GISEL-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX1250-GISEL-NEXT: s_cbranch_execnz .LBB37_3
; GFX1250-GISEL-NEXT: ; %bb.1: ; %Flow
@@ -4579,7 +4591,7 @@ define amdgpu_ps float @flat_or_saddr_i32_rtn(ptr inreg %sbase, i32 %voffset, i3
; GFX1250-GISEL-NEXT: v_nop
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v2, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: global_wb scope:SCOPE_DEV
@@ -4646,7 +4658,7 @@ define amdgpu_ps float @flat_or_saddr_i32_rtn_neg128(ptr inreg %sbase, i32 %voff
; GFX1250-GISEL-NEXT: v_nop
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v2, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: global_wb scope:SCOPE_DEV
@@ -4720,7 +4732,7 @@ define amdgpu_ps void @flat_or_saddr_i32_nortn(ptr inreg %sbase, i32 %voffset, i
; GFX1250-GISEL-NEXT: v_nop
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v2, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: global_wb scope:SCOPE_DEV
@@ -4786,7 +4798,7 @@ define amdgpu_ps void @flat_or_saddr_i32_nortn_neg128(ptr inreg %sbase, i32 %vof
; GFX1250-GISEL-NEXT: v_nop
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v2, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: global_wb scope:SCOPE_DEV
@@ -4892,7 +4904,7 @@ define amdgpu_ps <2 x float> @flat_or_saddr_i64_rtn(ptr inreg %sbase, i32 %voffs
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_dual_mov_b32 v4, v1 :: v_dual_mov_b32 v5, v2
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v2, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
@@ -5088,10 +5100,10 @@ define amdgpu_ps <2 x float> @flat_or_saddr_i64_rtn_neg128(ptr inreg %sbase, i32
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_dual_mov_b32 v4, v1 :: v_dual_mov_b32 v5, v2
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v6, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v7, null, 0, v3, vcc_lo
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v2, vcc_lo, 0xffffff80, v6
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v3, null, -1, v7, vcc_lo
; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
@@ -5246,9 +5258,10 @@ define amdgpu_ps void @flat_or_saddr_i64_nortn(ptr inreg %sbase, i32 %voffset, i
; GFX1250-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1250-SDAG-NEXT: v_add_nc_u64_e32 v[0:1], s[2:3], v[0:1]
; GFX1250-SDAG-NEXT: v_xor_b32_e32 v4, src_flat_scratch_base_hi, v1
-; GFX1250-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1250-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1250-SDAG-NEXT: v_cmpx_lt_u32_e32 0x3ffffff, v4
; GFX1250-SDAG-NEXT: s_xor_b32 s0, exec_lo, s0
+; GFX1250-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1250-SDAG-NEXT: s_cbranch_execnz .LBB44_3
; GFX1250-SDAG-NEXT: ; %bb.1: ; %Flow
; GFX1250-SDAG-NEXT: s_and_not1_saveexec_b32 s0, s0
@@ -5287,12 +5300,13 @@ define amdgpu_ps void @flat_or_saddr_i64_nortn(ptr inreg %sbase, i32 %voffset, i
; GFX1250-GISEL-NEXT: v_dual_mov_b32 v4, v1 :: v_dual_mov_b32 v5, v2
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
; GFX1250-GISEL-NEXT: s_mov_b32 s0, exec_lo
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1250-GISEL-NEXT: v_xor_b32_e32 v2, src_flat_scratch_base_hi, v1
; GFX1250-GISEL-NEXT: v_cmpx_le_u32_e32 0x4000000, v2
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1250-GISEL-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX1250-GISEL-NEXT: s_cbranch_execnz .LBB44_3
; GFX1250-GISEL-NEXT: ; %bb.1: ; %Flow
@@ -5420,6 +5434,7 @@ define amdgpu_ps void @flat_or_saddr_i64_nortn_neg128(ptr inreg %sbase, i32 %vof
; GFX1250-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1250-SDAG-NEXT: v_xor_b32_e32 v4, src_flat_scratch_base_hi, v1
; GFX1250-SDAG-NEXT: v_cmpx_lt_u32_e32 0x3ffffff, v4
+; GFX1250-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1250-SDAG-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX1250-SDAG-NEXT: s_cbranch_execnz .LBB45_3
; GFX1250-SDAG-NEXT: ; %bb.1: ; %Flow
@@ -5459,15 +5474,16 @@ define amdgpu_ps void @flat_or_saddr_i64_nortn_neg128(ptr inreg %sbase, i32 %vof
; GFX1250-GISEL-NEXT: v_dual_mov_b32 v4, v1 :: v_dual_mov_b32 v5, v2
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
; GFX1250-GISEL-NEXT: s_mov_b32 s0, exec_lo
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v2, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, 0xffffff80, v2
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, -1, v3, vcc_lo
; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1250-GISEL-NEXT: v_xor_b32_e32 v6, src_flat_scratch_base_hi, v1
; GFX1250-GISEL-NEXT: v_cmpx_le_u32_e32 0x4000000, v6
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1250-GISEL-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX1250-GISEL-NEXT: s_cbranch_execnz .LBB45_3
; GFX1250-GISEL-NEXT: ; %bb.1: ; %Flow
@@ -5615,7 +5631,7 @@ define amdgpu_ps float @flat_xor_saddr_i32_rtn(ptr inreg %sbase, i32 %voffset, i
; GFX1250-GISEL-NEXT: v_nop
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v2, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: global_wb scope:SCOPE_DEV
@@ -5682,7 +5698,7 @@ define amdgpu_ps float @flat_xor_saddr_i32_rtn_neg128(ptr inreg %sbase, i32 %vof
; GFX1250-GISEL-NEXT: v_nop
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v2, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: global_wb scope:SCOPE_DEV
@@ -5756,7 +5772,7 @@ define amdgpu_ps void @flat_xor_saddr_i32_nortn(ptr inreg %sbase, i32 %voffset,
; GFX1250-GISEL-NEXT: v_nop
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v2, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: global_wb scope:SCOPE_DEV
@@ -5822,7 +5838,7 @@ define amdgpu_ps void @flat_xor_saddr_i32_nortn_neg128(ptr inreg %sbase, i32 %vo
; GFX1250-GISEL-NEXT: v_nop
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v2, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: global_wb scope:SCOPE_DEV
@@ -5928,7 +5944,7 @@ define amdgpu_ps <2 x float> @flat_xor_saddr_i64_rtn(ptr inreg %sbase, i32 %voff
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_dual_mov_b32 v4, v1 :: v_dual_mov_b32 v5, v2
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v2, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
@@ -6124,10 +6140,10 @@ define amdgpu_ps <2 x float> @flat_xor_saddr_i64_rtn_neg128(ptr inreg %sbase, i3
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_dual_mov_b32 v4, v1 :: v_dual_mov_b32 v5, v2
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v6, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v7, null, 0, v3, vcc_lo
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v2, vcc_lo, 0xffffff80, v6
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v3, null, -1, v7, vcc_lo
; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
@@ -6282,9 +6298,10 @@ define amdgpu_ps void @flat_xor_saddr_i64_nortn(ptr inreg %sbase, i32 %voffset,
; GFX1250-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1250-SDAG-NEXT: v_add_nc_u64_e32 v[0:1], s[2:3], v[0:1]
; GFX1250-SDAG-NEXT: v_xor_b32_e32 v4, src_flat_scratch_base_hi, v1
-; GFX1250-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1250-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1250-SDAG-NEXT: v_cmpx_lt_u32_e32 0x3ffffff, v4
; GFX1250-SDAG-NEXT: s_xor_b32 s0, exec_lo, s0
+; GFX1250-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1250-SDAG-NEXT: s_cbranch_execnz .LBB52_3
; GFX1250-SDAG-NEXT: ; %bb.1: ; %Flow
; GFX1250-SDAG-NEXT: s_and_not1_saveexec_b32 s0, s0
@@ -6323,12 +6340,13 @@ define amdgpu_ps void @flat_xor_saddr_i64_nortn(ptr inreg %sbase, i32 %voffset,
; GFX1250-GISEL-NEXT: v_dual_mov_b32 v4, v1 :: v_dual_mov_b32 v5, v2
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
; GFX1250-GISEL-NEXT: s_mov_b32 s0, exec_lo
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1250-GISEL-NEXT: v_xor_b32_e32 v2, src_flat_scratch_base_hi, v1
; GFX1250-GISEL-NEXT: v_cmpx_le_u32_e32 0x4000000, v2
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1250-GISEL-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX1250-GISEL-NEXT: s_cbranch_execnz .LBB52_3
; GFX1250-GISEL-NEXT: ; %bb.1: ; %Flow
@@ -6456,6 +6474,7 @@ define amdgpu_ps void @flat_xor_saddr_i64_nortn_neg128(ptr inreg %sbase, i32 %vo
; GFX1250-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1250-SDAG-NEXT: v_xor_b32_e32 v4, src_flat_scratch_base_hi, v1
; GFX1250-SDAG-NEXT: v_cmpx_lt_u32_e32 0x3ffffff, v4
+; GFX1250-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1250-SDAG-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX1250-SDAG-NEXT: s_cbranch_execnz .LBB53_3
; GFX1250-SDAG-NEXT: ; %bb.1: ; %Flow
@@ -6495,15 +6514,16 @@ define amdgpu_ps void @flat_xor_saddr_i64_nortn_neg128(ptr inreg %sbase, i32 %vo
; GFX1250-GISEL-NEXT: v_dual_mov_b32 v4, v1 :: v_dual_mov_b32 v5, v2
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
; GFX1250-GISEL-NEXT: s_mov_b32 s0, exec_lo
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v2, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, 0xffffff80, v2
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, -1, v3, vcc_lo
; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1250-GISEL-NEXT: v_xor_b32_e32 v6, src_flat_scratch_base_hi, v1
; GFX1250-GISEL-NEXT: v_cmpx_le_u32_e32 0x4000000, v6
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1250-GISEL-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX1250-GISEL-NEXT: s_cbranch_execnz .LBB53_3
; GFX1250-GISEL-NEXT: ; %bb.1: ; %Flow
@@ -6647,7 +6667,7 @@ define amdgpu_ps float @flat_max_saddr_i32_rtn(ptr inreg %sbase, i32 %voffset, i
; GFX1250-GISEL-NEXT: v_nop
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v2, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: flat_atomic_max_i32 v0, v[2:3], v1 th:TH_ATOMIC_RETURN
@@ -6700,7 +6720,7 @@ define amdgpu_ps float @flat_max_saddr_i32_rtn_neg128(ptr inreg %sbase, i32 %vof
; GFX1250-GISEL-NEXT: v_nop
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v2, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: flat_atomic_max_i32 v0, v[2:3], v1 offset:-128 th:TH_ATOMIC_RETURN
@@ -6760,7 +6780,7 @@ define amdgpu_ps void @flat_max_saddr_i32_nortn(ptr inreg %sbase, i32 %voffset,
; GFX1250-GISEL-NEXT: v_nop
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v2, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: flat_atomic_max_i32 v[2:3], v1
@@ -6812,7 +6832,7 @@ define amdgpu_ps void @flat_max_saddr_i32_nortn_neg128(ptr inreg %sbase, i32 %vo
; GFX1250-GISEL-NEXT: v_nop
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v2, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: flat_atomic_max_i32 v[2:3], v1 offset:-128
@@ -6903,7 +6923,7 @@ define amdgpu_ps <2 x float> @flat_max_saddr_i64_rtn(ptr inreg %sbase, i32 %voff
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_dual_mov_b32 v4, v1 :: v_dual_mov_b32 v5, v2
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v2, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
@@ -7089,10 +7109,10 @@ define amdgpu_ps <2 x float> @flat_max_saddr_i64_rtn_neg128(ptr inreg %sbase, i3
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_dual_mov_b32 v4, v1 :: v_dual_mov_b32 v5, v2
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v6, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v7, null, 0, v3, vcc_lo
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v2, vcc_lo, 0xffffff80, v6
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v3, null, -1, v7, vcc_lo
; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
@@ -7242,9 +7262,10 @@ define amdgpu_ps void @flat_max_saddr_i64_nortn(ptr inreg %sbase, i32 %voffset,
; GFX1250-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1250-SDAG-NEXT: v_add_nc_u64_e32 v[0:1], s[2:3], v[0:1]
; GFX1250-SDAG-NEXT: v_xor_b32_e32 v4, src_flat_scratch_base_hi, v1
-; GFX1250-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1250-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1250-SDAG-NEXT: v_cmpx_lt_u32_e32 0x3ffffff, v4
; GFX1250-SDAG-NEXT: s_xor_b32 s0, exec_lo, s0
+; GFX1250-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1250-SDAG-NEXT: s_cbranch_execnz .LBB60_3
; GFX1250-SDAG-NEXT: ; %bb.1: ; %Flow
; GFX1250-SDAG-NEXT: s_and_not1_saveexec_b32 s0, s0
@@ -7279,12 +7300,13 @@ define amdgpu_ps void @flat_max_saddr_i64_nortn(ptr inreg %sbase, i32 %voffset,
; GFX1250-GISEL-NEXT: v_dual_mov_b32 v4, v1 :: v_dual_mov_b32 v5, v2
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
; GFX1250-GISEL-NEXT: s_mov_b32 s0, exec_lo
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1250-GISEL-NEXT: v_xor_b32_e32 v2, src_flat_scratch_base_hi, v1
; GFX1250-GISEL-NEXT: v_cmpx_le_u32_e32 0x4000000, v2
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1250-GISEL-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX1250-GISEL-NEXT: s_cbranch_execnz .LBB60_3
; GFX1250-GISEL-NEXT: ; %bb.1: ; %Flow
@@ -7406,6 +7428,7 @@ define amdgpu_ps void @flat_max_saddr_i64_nortn_neg128(ptr inreg %sbase, i32 %vo
; GFX1250-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1250-SDAG-NEXT: v_xor_b32_e32 v4, src_flat_scratch_base_hi, v1
; GFX1250-SDAG-NEXT: v_cmpx_lt_u32_e32 0x3ffffff, v4
+; GFX1250-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1250-SDAG-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX1250-SDAG-NEXT: s_cbranch_execnz .LBB61_3
; GFX1250-SDAG-NEXT: ; %bb.1: ; %Flow
@@ -7441,15 +7464,16 @@ define amdgpu_ps void @flat_max_saddr_i64_nortn_neg128(ptr inreg %sbase, i32 %vo
; GFX1250-GISEL-NEXT: v_dual_mov_b32 v4, v1 :: v_dual_mov_b32 v5, v2
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
; GFX1250-GISEL-NEXT: s_mov_b32 s0, exec_lo
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v2, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, 0xffffff80, v2
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, -1, v3, vcc_lo
; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1250-GISEL-NEXT: v_xor_b32_e32 v6, src_flat_scratch_base_hi, v1
; GFX1250-GISEL-NEXT: v_cmpx_le_u32_e32 0x4000000, v6
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1250-GISEL-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX1250-GISEL-NEXT: s_cbranch_execnz .LBB61_3
; GFX1250-GISEL-NEXT: ; %bb.1: ; %Flow
@@ -7587,7 +7611,7 @@ define amdgpu_ps float @flat_min_saddr_i32_rtn(ptr inreg %sbase, i32 %voffset, i
; GFX1250-GISEL-NEXT: v_nop
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v2, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: flat_atomic_min_i32 v0, v[2:3], v1 th:TH_ATOMIC_RETURN
@@ -7640,7 +7664,7 @@ define amdgpu_ps float @flat_min_saddr_i32_rtn_neg128(ptr inreg %sbase, i32 %vof
; GFX1250-GISEL-NEXT: v_nop
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v2, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: flat_atomic_min_i32 v0, v[2:3], v1 offset:-128 th:TH_ATOMIC_RETURN
@@ -7700,7 +7724,7 @@ define amdgpu_ps void @flat_min_saddr_i32_nortn(ptr inreg %sbase, i32 %voffset,
; GFX1250-GISEL-NEXT: v_nop
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v2, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: flat_atomic_min_i32 v[2:3], v1
@@ -7752,7 +7776,7 @@ define amdgpu_ps void @flat_min_saddr_i32_nortn_neg128(ptr inreg %sbase, i32 %vo
; GFX1250-GISEL-NEXT: v_nop
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v2, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: flat_atomic_min_i32 v[2:3], v1 offset:-128
@@ -7843,7 +7867,7 @@ define amdgpu_ps <2 x float> @flat_min_saddr_i64_rtn(ptr inreg %sbase, i32 %voff
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_dual_mov_b32 v4, v1 :: v_dual_mov_b32 v5, v2
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v2, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
@@ -8029,10 +8053,10 @@ define amdgpu_ps <2 x float> @flat_min_saddr_i64_rtn_neg128(ptr inreg %sbase, i3
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_dual_mov_b32 v4, v1 :: v_dual_mov_b32 v5, v2
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v6, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v7, null, 0, v3, vcc_lo
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v2, vcc_lo, 0xffffff80, v6
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v3, null, -1, v7, vcc_lo
; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
@@ -8182,9 +8206,10 @@ define amdgpu_ps void @flat_min_saddr_i64_nortn(ptr inreg %sbase, i32 %voffset,
; GFX1250-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1250-SDAG-NEXT: v_add_nc_u64_e32 v[0:1], s[2:3], v[0:1]
; GFX1250-SDAG-NEXT: v_xor_b32_e32 v4, src_flat_scratch_base_hi, v1
-; GFX1250-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1250-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1250-SDAG-NEXT: v_cmpx_lt_u32_e32 0x3ffffff, v4
; GFX1250-SDAG-NEXT: s_xor_b32 s0, exec_lo, s0
+; GFX1250-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1250-SDAG-NEXT: s_cbranch_execnz .LBB68_3
; GFX1250-SDAG-NEXT: ; %bb.1: ; %Flow
; GFX1250-SDAG-NEXT: s_and_not1_saveexec_b32 s0, s0
@@ -8219,12 +8244,13 @@ define amdgpu_ps void @flat_min_saddr_i64_nortn(ptr inreg %sbase, i32 %voffset,
; GFX1250-GISEL-NEXT: v_dual_mov_b32 v4, v1 :: v_dual_mov_b32 v5, v2
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
; GFX1250-GISEL-NEXT: s_mov_b32 s0, exec_lo
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1250-GISEL-NEXT: v_xor_b32_e32 v2, src_flat_scratch_base_hi, v1
; GFX1250-GISEL-NEXT: v_cmpx_le_u32_e32 0x4000000, v2
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1250-GISEL-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX1250-GISEL-NEXT: s_cbranch_execnz .LBB68_3
; GFX1250-GISEL-NEXT: ; %bb.1: ; %Flow
@@ -8346,6 +8372,7 @@ define amdgpu_ps void @flat_min_saddr_i64_nortn_neg128(ptr inreg %sbase, i32 %vo
; GFX1250-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1250-SDAG-NEXT: v_xor_b32_e32 v4, src_flat_scratch_base_hi, v1
; GFX1250-SDAG-NEXT: v_cmpx_lt_u32_e32 0x3ffffff, v4
+; GFX1250-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1250-SDAG-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX1250-SDAG-NEXT: s_cbranch_execnz .LBB69_3
; GFX1250-SDAG-NEXT: ; %bb.1: ; %Flow
@@ -8381,15 +8408,16 @@ define amdgpu_ps void @flat_min_saddr_i64_nortn_neg128(ptr inreg %sbase, i32 %vo
; GFX1250-GISEL-NEXT: v_dual_mov_b32 v4, v1 :: v_dual_mov_b32 v5, v2
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
; GFX1250-GISEL-NEXT: s_mov_b32 s0, exec_lo
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v2, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, 0xffffff80, v2
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, -1, v3, vcc_lo
; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1250-GISEL-NEXT: v_xor_b32_e32 v6, src_flat_scratch_base_hi, v1
; GFX1250-GISEL-NEXT: v_cmpx_le_u32_e32 0x4000000, v6
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1250-GISEL-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX1250-GISEL-NEXT: s_cbranch_execnz .LBB69_3
; GFX1250-GISEL-NEXT: ; %bb.1: ; %Flow
@@ -8527,7 +8555,7 @@ define amdgpu_ps float @flat_umax_saddr_i32_rtn(ptr inreg %sbase, i32 %voffset,
; GFX1250-GISEL-NEXT: v_nop
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v2, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: flat_atomic_max_u32 v0, v[2:3], v1 th:TH_ATOMIC_RETURN
@@ -8580,7 +8608,7 @@ define amdgpu_ps float @flat_umax_saddr_i32_rtn_neg128(ptr inreg %sbase, i32 %vo
; GFX1250-GISEL-NEXT: v_nop
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v2, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: flat_atomic_max_u32 v0, v[2:3], v1 offset:-128 th:TH_ATOMIC_RETURN
@@ -8640,7 +8668,7 @@ define amdgpu_ps void @flat_umax_saddr_i32_nortn(ptr inreg %sbase, i32 %voffset,
; GFX1250-GISEL-NEXT: v_nop
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v2, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: flat_atomic_max_u32 v[2:3], v1
@@ -8692,7 +8720,7 @@ define amdgpu_ps void @flat_umax_saddr_i32_nortn_neg128(ptr inreg %sbase, i32 %v
; GFX1250-GISEL-NEXT: v_nop
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v2, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: flat_atomic_max_u32 v[2:3], v1 offset:-128
@@ -8783,7 +8811,7 @@ define amdgpu_ps <2 x float> @flat_umax_saddr_i64_rtn(ptr inreg %sbase, i32 %vof
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_dual_mov_b32 v4, v1 :: v_dual_mov_b32 v5, v2
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v2, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
@@ -8969,10 +8997,10 @@ define amdgpu_ps <2 x float> @flat_umax_saddr_i64_rtn_neg128(ptr inreg %sbase, i
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_dual_mov_b32 v4, v1 :: v_dual_mov_b32 v5, v2
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v6, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v7, null, 0, v3, vcc_lo
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v2, vcc_lo, 0xffffff80, v6
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v3, null, -1, v7, vcc_lo
; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
@@ -9122,9 +9150,10 @@ define amdgpu_ps void @flat_umax_saddr_i64_nortn(ptr inreg %sbase, i32 %voffset,
; GFX1250-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1250-SDAG-NEXT: v_add_nc_u64_e32 v[0:1], s[2:3], v[0:1]
; GFX1250-SDAG-NEXT: v_xor_b32_e32 v4, src_flat_scratch_base_hi, v1
-; GFX1250-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1250-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1250-SDAG-NEXT: v_cmpx_lt_u32_e32 0x3ffffff, v4
; GFX1250-SDAG-NEXT: s_xor_b32 s0, exec_lo, s0
+; GFX1250-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1250-SDAG-NEXT: s_cbranch_execnz .LBB76_3
; GFX1250-SDAG-NEXT: ; %bb.1: ; %Flow
; GFX1250-SDAG-NEXT: s_and_not1_saveexec_b32 s0, s0
@@ -9159,12 +9188,13 @@ define amdgpu_ps void @flat_umax_saddr_i64_nortn(ptr inreg %sbase, i32 %voffset,
; GFX1250-GISEL-NEXT: v_dual_mov_b32 v4, v1 :: v_dual_mov_b32 v5, v2
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
; GFX1250-GISEL-NEXT: s_mov_b32 s0, exec_lo
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1250-GISEL-NEXT: v_xor_b32_e32 v2, src_flat_scratch_base_hi, v1
; GFX1250-GISEL-NEXT: v_cmpx_le_u32_e32 0x4000000, v2
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1250-GISEL-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX1250-GISEL-NEXT: s_cbranch_execnz .LBB76_3
; GFX1250-GISEL-NEXT: ; %bb.1: ; %Flow
@@ -9286,6 +9316,7 @@ define amdgpu_ps void @flat_umax_saddr_i64_nortn_neg128(ptr inreg %sbase, i32 %v
; GFX1250-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1250-SDAG-NEXT: v_xor_b32_e32 v4, src_flat_scratch_base_hi, v1
; GFX1250-SDAG-NEXT: v_cmpx_lt_u32_e32 0x3ffffff, v4
+; GFX1250-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1250-SDAG-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX1250-SDAG-NEXT: s_cbranch_execnz .LBB77_3
; GFX1250-SDAG-NEXT: ; %bb.1: ; %Flow
@@ -9321,15 +9352,16 @@ define amdgpu_ps void @flat_umax_saddr_i64_nortn_neg128(ptr inreg %sbase, i32 %v
; GFX1250-GISEL-NEXT: v_dual_mov_b32 v4, v1 :: v_dual_mov_b32 v5, v2
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
; GFX1250-GISEL-NEXT: s_mov_b32 s0, exec_lo
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v2, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, 0xffffff80, v2
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, -1, v3, vcc_lo
; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1250-GISEL-NEXT: v_xor_b32_e32 v6, src_flat_scratch_base_hi, v1
; GFX1250-GISEL-NEXT: v_cmpx_le_u32_e32 0x4000000, v6
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1250-GISEL-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX1250-GISEL-NEXT: s_cbranch_execnz .LBB77_3
; GFX1250-GISEL-NEXT: ; %bb.1: ; %Flow
@@ -9467,7 +9499,7 @@ define amdgpu_ps float @flat_umin_saddr_i32_rtn(ptr inreg %sbase, i32 %voffset,
; GFX1250-GISEL-NEXT: v_nop
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v2, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: flat_atomic_min_u32 v0, v[2:3], v1 th:TH_ATOMIC_RETURN
@@ -9520,7 +9552,7 @@ define amdgpu_ps float @flat_umin_saddr_i32_rtn_neg128(ptr inreg %sbase, i32 %vo
; GFX1250-GISEL-NEXT: v_nop
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v2, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: flat_atomic_min_u32 v0, v[2:3], v1 offset:-128 th:TH_ATOMIC_RETURN
@@ -9580,7 +9612,7 @@ define amdgpu_ps void @flat_umin_saddr_i32_nortn(ptr inreg %sbase, i32 %voffset,
; GFX1250-GISEL-NEXT: v_nop
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v2, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: flat_atomic_min_u32 v[2:3], v1
@@ -9632,7 +9664,7 @@ define amdgpu_ps void @flat_umin_saddr_i32_nortn_neg128(ptr inreg %sbase, i32 %v
; GFX1250-GISEL-NEXT: v_nop
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v2, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: flat_atomic_min_u32 v[2:3], v1 offset:-128
@@ -9723,7 +9755,7 @@ define amdgpu_ps <2 x float> @flat_umin_saddr_i64_rtn(ptr inreg %sbase, i32 %vof
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_dual_mov_b32 v4, v1 :: v_dual_mov_b32 v5, v2
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v2, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
@@ -9909,10 +9941,10 @@ define amdgpu_ps <2 x float> @flat_umin_saddr_i64_rtn_neg128(ptr inreg %sbase, i
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_dual_mov_b32 v4, v1 :: v_dual_mov_b32 v5, v2
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v6, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v7, null, 0, v3, vcc_lo
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v2, vcc_lo, 0xffffff80, v6
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v3, null, -1, v7, vcc_lo
; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
@@ -10062,9 +10094,10 @@ define amdgpu_ps void @flat_umin_saddr_i64_nortn(ptr inreg %sbase, i32 %voffset,
; GFX1250-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1250-SDAG-NEXT: v_add_nc_u64_e32 v[0:1], s[2:3], v[0:1]
; GFX1250-SDAG-NEXT: v_xor_b32_e32 v4, src_flat_scratch_base_hi, v1
-; GFX1250-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1250-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1250-SDAG-NEXT: v_cmpx_lt_u32_e32 0x3ffffff, v4
; GFX1250-SDAG-NEXT: s_xor_b32 s0, exec_lo, s0
+; GFX1250-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1250-SDAG-NEXT: s_cbranch_execnz .LBB84_3
; GFX1250-SDAG-NEXT: ; %bb.1: ; %Flow
; GFX1250-SDAG-NEXT: s_and_not1_saveexec_b32 s0, s0
@@ -10099,12 +10132,13 @@ define amdgpu_ps void @flat_umin_saddr_i64_nortn(ptr inreg %sbase, i32 %voffset,
; GFX1250-GISEL-NEXT: v_dual_mov_b32 v4, v1 :: v_dual_mov_b32 v5, v2
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
; GFX1250-GISEL-NEXT: s_mov_b32 s0, exec_lo
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1250-GISEL-NEXT: v_xor_b32_e32 v2, src_flat_scratch_base_hi, v1
; GFX1250-GISEL-NEXT: v_cmpx_le_u32_e32 0x4000000, v2
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1250-GISEL-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX1250-GISEL-NEXT: s_cbranch_execnz .LBB84_3
; GFX1250-GISEL-NEXT: ; %bb.1: ; %Flow
@@ -10226,6 +10260,7 @@ define amdgpu_ps void @flat_umin_saddr_i64_nortn_neg128(ptr inreg %sbase, i32 %v
; GFX1250-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1250-SDAG-NEXT: v_xor_b32_e32 v4, src_flat_scratch_base_hi, v1
; GFX1250-SDAG-NEXT: v_cmpx_lt_u32_e32 0x3ffffff, v4
+; GFX1250-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1250-SDAG-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX1250-SDAG-NEXT: s_cbranch_execnz .LBB85_3
; GFX1250-SDAG-NEXT: ; %bb.1: ; %Flow
@@ -10261,15 +10296,16 @@ define amdgpu_ps void @flat_umin_saddr_i64_nortn_neg128(ptr inreg %sbase, i32 %v
; GFX1250-GISEL-NEXT: v_dual_mov_b32 v4, v1 :: v_dual_mov_b32 v5, v2
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
; GFX1250-GISEL-NEXT: s_mov_b32 s0, exec_lo
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v2, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, 0xffffff80, v2
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, -1, v3, vcc_lo
; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1250-GISEL-NEXT: v_xor_b32_e32 v6, src_flat_scratch_base_hi, v1
; GFX1250-GISEL-NEXT: v_cmpx_le_u32_e32 0x4000000, v6
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1250-GISEL-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX1250-GISEL-NEXT: s_cbranch_execnz .LBB85_3
; GFX1250-GISEL-NEXT: ; %bb.1: ; %Flow
@@ -10412,7 +10448,7 @@ define amdgpu_ps float @flat_cmpxchg_saddr_i32_rtn(ptr inreg %sbase, i32 %voffse
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[4:5], s[2:3]
; GFX1250-GISEL-NEXT: v_mov_b32_e32 v3, v1
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_3)
; GFX1250-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, v4, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v5, vcc_lo
; GFX1250-GISEL-NEXT: global_wb scope:SCOPE_SYS
@@ -10482,7 +10518,7 @@ define amdgpu_ps float @flat_cmpxchg_saddr_i32_rtn_neg128(ptr inreg %sbase, i32
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[4:5], s[2:3]
; GFX1250-GISEL-NEXT: v_mov_b32_e32 v3, v1
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_3)
; GFX1250-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, v4, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v5, vcc_lo
; GFX1250-GISEL-NEXT: global_wb scope:SCOPE_SYS
@@ -10559,7 +10595,7 @@ define amdgpu_ps void @flat_cmpxchg_saddr_i32_nortn(ptr inreg %sbase, i32 %voffs
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[4:5], s[2:3]
; GFX1250-GISEL-NEXT: v_mov_b32_e32 v3, v1
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_3)
; GFX1250-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, v4, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v5, vcc_lo
; GFX1250-GISEL-NEXT: global_wb scope:SCOPE_SYS
@@ -10627,7 +10663,7 @@ define amdgpu_ps void @flat_cmpxchg_saddr_i32_nortn_neg128(ptr inreg %sbase, i32
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[4:5], s[2:3]
; GFX1250-GISEL-NEXT: v_mov_b32_e32 v3, v1
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_3)
; GFX1250-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, v4, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v5, vcc_lo
; GFX1250-GISEL-NEXT: global_wb scope:SCOPE_SYS
@@ -10736,7 +10772,7 @@ define amdgpu_ps <2 x float> @flat_cmpxchg_saddr_i64_rtn(ptr inreg %sbase, i32 %
; GFX1250-GISEL-NEXT: v_dual_mov_b32 v8, v1 :: v_dual_mov_b32 v9, v2
; GFX1250-GISEL-NEXT: v_dual_mov_b32 v6, v3 :: v_dual_mov_b32 v7, v4
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v2, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
@@ -10943,10 +10979,10 @@ define amdgpu_ps <2 x float> @flat_cmpxchg_saddr_i64_rtn_neg128(ptr inreg %sbase
; GFX1250-GISEL-NEXT: v_dual_mov_b32 v8, v1 :: v_dual_mov_b32 v9, v2
; GFX1250-GISEL-NEXT: v_dual_mov_b32 v6, v3 :: v_dual_mov_b32 v7, v4
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v4, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v5, null, 0, v3, vcc_lo
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v2, vcc_lo, 0xffffff80, v4
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v3, null, -1, v5, vcc_lo
; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
@@ -11111,9 +11147,10 @@ define amdgpu_ps void @flat_cmpxchg_saddr_i64_nortn(ptr inreg %sbase, i32 %voffs
; GFX1250-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1250-SDAG-NEXT: v_add_nc_u64_e32 v[0:1], s[2:3], v[0:1]
; GFX1250-SDAG-NEXT: v_xor_b32_e32 v2, src_flat_scratch_base_hi, v1
-; GFX1250-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1250-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1250-SDAG-NEXT: v_cmpx_lt_u32_e32 0x3ffffff, v2
; GFX1250-SDAG-NEXT: s_xor_b32 s0, exec_lo, s0
+; GFX1250-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1250-SDAG-NEXT: s_cbranch_execnz .LBB92_3
; GFX1250-SDAG-NEXT: ; %bb.1: ; %Flow
; GFX1250-SDAG-NEXT: s_and_not1_saveexec_b32 s0, s0
@@ -11153,12 +11190,13 @@ define amdgpu_ps void @flat_cmpxchg_saddr_i64_nortn(ptr inreg %sbase, i32 %voffs
; GFX1250-GISEL-NEXT: v_dual_mov_b32 v6, v3 :: v_dual_mov_b32 v7, v4
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
; GFX1250-GISEL-NEXT: s_mov_b32 s0, exec_lo
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1250-GISEL-NEXT: v_xor_b32_e32 v2, src_flat_scratch_base_hi, v1
; GFX1250-GISEL-NEXT: v_cmpx_le_u32_e32 0x4000000, v2
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1250-GISEL-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX1250-GISEL-NEXT: s_cbranch_execnz .LBB92_3
; GFX1250-GISEL-NEXT: ; %bb.1: ; %Flow
@@ -11295,6 +11333,7 @@ define amdgpu_ps void @flat_cmpxchg_saddr_i64_nortn_neg128(ptr inreg %sbase, i32
; GFX1250-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1250-SDAG-NEXT: v_xor_b32_e32 v2, src_flat_scratch_base_hi, v1
; GFX1250-SDAG-NEXT: v_cmpx_lt_u32_e32 0x3ffffff, v2
+; GFX1250-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1250-SDAG-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX1250-SDAG-NEXT: s_cbranch_execnz .LBB93_3
; GFX1250-SDAG-NEXT: ; %bb.1: ; %Flow
@@ -11335,15 +11374,16 @@ define amdgpu_ps void @flat_cmpxchg_saddr_i64_nortn_neg128(ptr inreg %sbase, i32
; GFX1250-GISEL-NEXT: v_dual_mov_b32 v6, v3 :: v_dual_mov_b32 v7, v4
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
; GFX1250-GISEL-NEXT: s_mov_b32 s0, exec_lo
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v2, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, 0xffffff80, v2
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, -1, v3, vcc_lo
; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1250-GISEL-NEXT: v_xor_b32_e32 v4, src_flat_scratch_base_hi, v1
; GFX1250-GISEL-NEXT: v_cmpx_le_u32_e32 0x4000000, v4
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1250-GISEL-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX1250-GISEL-NEXT: s_cbranch_execnz .LBB93_3
; GFX1250-GISEL-NEXT: ; %bb.1: ; %Flow
@@ -11495,7 +11535,7 @@ define amdgpu_ps float @flat_inc_saddr_i32_rtn(ptr inreg %sbase, i32 %voffset, i
; GFX1250-GISEL-NEXT: v_nop
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v2, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: flat_atomic_inc_u32 v0, v[2:3], v1 th:TH_ATOMIC_RETURN scope:SCOPE_DEV
@@ -11548,7 +11588,7 @@ define amdgpu_ps float @flat_inc_saddr_i32_rtn_neg128(ptr inreg %sbase, i32 %vof
; GFX1250-GISEL-NEXT: v_nop
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v2, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: flat_atomic_inc_u32 v0, v[2:3], v1 offset:-128 th:TH_ATOMIC_RETURN scope:SCOPE_DEV
@@ -11607,7 +11647,7 @@ define amdgpu_ps void @flat_inc_saddr_i32_nortn(ptr inreg %sbase, i32 %voffset,
; GFX1250-GISEL-NEXT: v_nop
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v2, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: flat_atomic_inc_u32 v[2:3], v1 scope:SCOPE_DEV
@@ -11655,7 +11695,7 @@ define amdgpu_ps void @flat_inc_saddr_i32_nortn_neg128(ptr inreg %sbase, i32 %vo
; GFX1250-GISEL-NEXT: v_nop
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v2, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: flat_atomic_inc_u32 v[2:3], v1 offset:-128 scope:SCOPE_DEV
@@ -11747,7 +11787,7 @@ define amdgpu_ps <2 x float> @flat_inc_saddr_i64_rtn(ptr inreg %sbase, i32 %voff
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_dual_mov_b32 v4, v1 :: v_dual_mov_b32 v5, v2
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v2, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
@@ -11947,10 +11987,10 @@ define amdgpu_ps <2 x float> @flat_inc_saddr_i64_rtn_neg128(ptr inreg %sbase, i3
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_dual_mov_b32 v4, v1 :: v_dual_mov_b32 v5, v2
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v6, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v7, null, 0, v3, vcc_lo
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v2, vcc_lo, 0xffffff80, v6
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v3, null, -1, v7, vcc_lo
; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
@@ -12110,9 +12150,10 @@ define amdgpu_ps void @flat_inc_saddr_i64_nortn(ptr inreg %sbase, i32 %voffset,
; GFX1250-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1250-SDAG-NEXT: v_add_nc_u64_e32 v[0:1], s[2:3], v[0:1]
; GFX1250-SDAG-NEXT: v_xor_b32_e32 v4, src_flat_scratch_base_hi, v1
-; GFX1250-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1250-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1250-SDAG-NEXT: v_cmpx_lt_u32_e32 0x3ffffff, v4
; GFX1250-SDAG-NEXT: s_xor_b32 s0, exec_lo, s0
+; GFX1250-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1250-SDAG-NEXT: s_cbranch_execnz .LBB100_3
; GFX1250-SDAG-NEXT: ; %bb.1: ; %Flow
; GFX1250-SDAG-NEXT: s_and_not1_saveexec_b32 s0, s0
@@ -12148,12 +12189,13 @@ define amdgpu_ps void @flat_inc_saddr_i64_nortn(ptr inreg %sbase, i32 %voffset,
; GFX1250-GISEL-NEXT: v_dual_mov_b32 v4, v1 :: v_dual_mov_b32 v5, v2
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
; GFX1250-GISEL-NEXT: s_mov_b32 s0, exec_lo
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1250-GISEL-NEXT: v_xor_b32_e32 v2, src_flat_scratch_base_hi, v1
; GFX1250-GISEL-NEXT: v_cmpx_le_u32_e32 0x4000000, v2
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1250-GISEL-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX1250-GISEL-NEXT: s_cbranch_execnz .LBB100_3
; GFX1250-GISEL-NEXT: ; %bb.1: ; %Flow
@@ -12280,6 +12322,7 @@ define amdgpu_ps void @flat_inc_saddr_i64_nortn_neg128(ptr inreg %sbase, i32 %vo
; GFX1250-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1250-SDAG-NEXT: v_xor_b32_e32 v4, src_flat_scratch_base_hi, v1
; GFX1250-SDAG-NEXT: v_cmpx_lt_u32_e32 0x3ffffff, v4
+; GFX1250-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1250-SDAG-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX1250-SDAG-NEXT: s_cbranch_execnz .LBB101_3
; GFX1250-SDAG-NEXT: ; %bb.1: ; %Flow
@@ -12316,15 +12359,16 @@ define amdgpu_ps void @flat_inc_saddr_i64_nortn_neg128(ptr inreg %sbase, i32 %vo
; GFX1250-GISEL-NEXT: v_dual_mov_b32 v4, v1 :: v_dual_mov_b32 v5, v2
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
; GFX1250-GISEL-NEXT: s_mov_b32 s0, exec_lo
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v2, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, 0xffffff80, v2
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, -1, v3, vcc_lo
; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1250-GISEL-NEXT: v_xor_b32_e32 v6, src_flat_scratch_base_hi, v1
; GFX1250-GISEL-NEXT: v_cmpx_le_u32_e32 0x4000000, v6
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1250-GISEL-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX1250-GISEL-NEXT: s_cbranch_execnz .LBB101_3
; GFX1250-GISEL-NEXT: ; %bb.1: ; %Flow
@@ -12468,7 +12512,7 @@ define amdgpu_ps float @flat_dec_saddr_i32_rtn(ptr inreg %sbase, i32 %voffset, i
; GFX1250-GISEL-NEXT: v_nop
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v2, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: flat_atomic_dec_u32 v0, v[2:3], v1 th:TH_ATOMIC_RETURN scope:SCOPE_DEV
@@ -12521,7 +12565,7 @@ define amdgpu_ps float @flat_dec_saddr_i32_rtn_neg128(ptr inreg %sbase, i32 %vof
; GFX1250-GISEL-NEXT: v_nop
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v2, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: flat_atomic_dec_u32 v0, v[2:3], v1 offset:-128 th:TH_ATOMIC_RETURN scope:SCOPE_DEV
@@ -12580,7 +12624,7 @@ define amdgpu_ps void @flat_dec_saddr_i32_nortn(ptr inreg %sbase, i32 %voffset,
; GFX1250-GISEL-NEXT: v_nop
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v2, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: flat_atomic_dec_u32 v[2:3], v1 scope:SCOPE_DEV
@@ -12628,7 +12672,7 @@ define amdgpu_ps void @flat_dec_saddr_i32_nortn_neg128(ptr inreg %sbase, i32 %vo
; GFX1250-GISEL-NEXT: v_nop
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v2, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: flat_atomic_dec_u32 v[2:3], v1 offset:-128 scope:SCOPE_DEV
@@ -12699,7 +12743,7 @@ define amdgpu_ps <2 x float> @flat_dec_saddr_i64_rtn(ptr inreg %sbase, i32 %voff
; GFX1250-SDAG-NEXT: v_cmp_ne_u64_e32 vcc_lo, 0, v[4:5]
; GFX1250-SDAG-NEXT: s_wait_loadcnt_dscnt 0x0
; GFX1250-SDAG-NEXT: v_subrev_nc_u32_e32 v0, src_flat_scratch_base_lo, v4
-; GFX1250-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_4) | instid1(VALU_DEP_1)
+; GFX1250-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1250-SDAG-NEXT: v_cndmask_b32_e32 v4, -1, v0, vcc_lo
; GFX1250-SDAG-NEXT: scratch_load_b64 v[0:1], v4, off
; GFX1250-SDAG-NEXT: s_wait_loadcnt 0x0
@@ -12707,6 +12751,7 @@ define amdgpu_ps <2 x float> @flat_dec_saddr_i64_rtn(ptr inreg %sbase, i32 %voff
; GFX1250-SDAG-NEXT: v_sub_co_u32 v5, s0, v0, 1
; GFX1250-SDAG-NEXT: v_subrev_co_ci_u32_e64 v6, s0, 0, v1, s0
; GFX1250-SDAG-NEXT: s_or_b32 vcc_lo, s0, vcc_lo
+; GFX1250-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1250-SDAG-NEXT: v_dual_cndmask_b32 v3, v6, v3 :: v_dual_cndmask_b32 v2, v5, v2
; GFX1250-SDAG-NEXT: scratch_store_b64 v4, v[2:3], off
; GFX1250-SDAG-NEXT: s_wait_xcnt 0x0
@@ -12722,7 +12767,7 @@ define amdgpu_ps <2 x float> @flat_dec_saddr_i64_rtn(ptr inreg %sbase, i32 %voff
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_dual_mov_b32 v4, v1 :: v_dual_mov_b32 v5, v2
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v2, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
@@ -12751,7 +12796,7 @@ define amdgpu_ps <2 x float> @flat_dec_saddr_i64_rtn(ptr inreg %sbase, i32 %voff
; GFX1250-GISEL-NEXT: v_cmp_ne_u64_e32 vcc_lo, 0, v[2:3]
; GFX1250-GISEL-NEXT: s_wait_loadcnt_dscnt 0x0
; GFX1250-GISEL-NEXT: v_subrev_nc_u32_e32 v0, src_flat_scratch_base_lo, v2
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_4) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1250-GISEL-NEXT: v_cndmask_b32_e32 v6, -1, v0, vcc_lo
; GFX1250-GISEL-NEXT: scratch_load_b64 v[0:1], v6, off
; GFX1250-GISEL-NEXT: s_wait_loadcnt 0x0
@@ -12759,6 +12804,7 @@ define amdgpu_ps <2 x float> @flat_dec_saddr_i64_rtn(ptr inreg %sbase, i32 %voff
; GFX1250-GISEL-NEXT: v_sub_co_u32 v2, s0, v0, 1
; GFX1250-GISEL-NEXT: v_subrev_co_ci_u32_e64 v3, s0, 0, v1, s0
; GFX1250-GISEL-NEXT: s_or_b32 vcc_lo, s0, vcc_lo
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1250-GISEL-NEXT: v_dual_cndmask_b32 v2, v2, v4 :: v_dual_cndmask_b32 v3, v3, v5
; GFX1250-GISEL-NEXT: scratch_store_b64 v6, v[2:3], off
; GFX1250-GISEL-NEXT: s_wait_xcnt 0x0
@@ -12905,7 +12951,7 @@ define amdgpu_ps <2 x float> @flat_dec_saddr_i64_rtn_neg128(ptr inreg %sbase, i3
; GFX1250-SDAG-NEXT: v_cmp_ne_u64_e32 vcc_lo, 0, v[4:5]
; GFX1250-SDAG-NEXT: s_wait_loadcnt_dscnt 0x0
; GFX1250-SDAG-NEXT: v_subrev_nc_u32_e32 v0, src_flat_scratch_base_lo, v4
-; GFX1250-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_4) | instid1(VALU_DEP_1)
+; GFX1250-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1250-SDAG-NEXT: v_cndmask_b32_e32 v4, -1, v0, vcc_lo
; GFX1250-SDAG-NEXT: scratch_load_b64 v[0:1], v4, off
; GFX1250-SDAG-NEXT: s_wait_loadcnt 0x0
@@ -12913,6 +12959,7 @@ define amdgpu_ps <2 x float> @flat_dec_saddr_i64_rtn_neg128(ptr inreg %sbase, i3
; GFX1250-SDAG-NEXT: v_sub_co_u32 v5, s0, v0, 1
; GFX1250-SDAG-NEXT: v_subrev_co_ci_u32_e64 v6, s0, 0, v1, s0
; GFX1250-SDAG-NEXT: s_or_b32 vcc_lo, s0, vcc_lo
+; GFX1250-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1250-SDAG-NEXT: v_dual_cndmask_b32 v3, v6, v3 :: v_dual_cndmask_b32 v2, v5, v2
; GFX1250-SDAG-NEXT: scratch_store_b64 v4, v[2:3], off
; GFX1250-SDAG-NEXT: s_wait_xcnt 0x0
@@ -12928,10 +12975,10 @@ define amdgpu_ps <2 x float> @flat_dec_saddr_i64_rtn_neg128(ptr inreg %sbase, i3
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_dual_mov_b32 v4, v1 :: v_dual_mov_b32 v5, v2
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v6, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v7, null, 0, v3, vcc_lo
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v2, vcc_lo, 0xffffff80, v6
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v3, null, -1, v7, vcc_lo
; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
@@ -12960,7 +13007,7 @@ define amdgpu_ps <2 x float> @flat_dec_saddr_i64_rtn_neg128(ptr inreg %sbase, i3
; GFX1250-GISEL-NEXT: v_cmp_ne_u64_e32 vcc_lo, 0, v[2:3]
; GFX1250-GISEL-NEXT: s_wait_loadcnt_dscnt 0x0
; GFX1250-GISEL-NEXT: v_subrev_nc_u32_e32 v0, src_flat_scratch_base_lo, v2
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_4) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1250-GISEL-NEXT: v_cndmask_b32_e32 v6, -1, v0, vcc_lo
; GFX1250-GISEL-NEXT: scratch_load_b64 v[0:1], v6, off
; GFX1250-GISEL-NEXT: s_wait_loadcnt 0x0
@@ -12968,6 +13015,7 @@ define amdgpu_ps <2 x float> @flat_dec_saddr_i64_rtn_neg128(ptr inreg %sbase, i3
; GFX1250-GISEL-NEXT: v_sub_co_u32 v2, s0, v0, 1
; GFX1250-GISEL-NEXT: v_subrev_co_ci_u32_e64 v3, s0, 0, v1, s0
; GFX1250-GISEL-NEXT: s_or_b32 vcc_lo, s0, vcc_lo
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1250-GISEL-NEXT: v_dual_cndmask_b32 v2, v2, v4 :: v_dual_cndmask_b32 v3, v3, v5
; GFX1250-GISEL-NEXT: scratch_store_b64 v6, v[2:3], off
; GFX1250-GISEL-NEXT: s_wait_xcnt 0x0
@@ -13095,9 +13143,10 @@ define amdgpu_ps void @flat_dec_saddr_i64_nortn(ptr inreg %sbase, i32 %voffset,
; GFX1250-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1250-SDAG-NEXT: v_add_nc_u64_e32 v[0:1], s[2:3], v[0:1]
; GFX1250-SDAG-NEXT: v_xor_b32_e32 v4, src_flat_scratch_base_hi, v1
-; GFX1250-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1250-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1250-SDAG-NEXT: v_cmpx_lt_u32_e32 0x3ffffff, v4
; GFX1250-SDAG-NEXT: s_xor_b32 s0, exec_lo, s0
+; GFX1250-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1250-SDAG-NEXT: s_cbranch_execnz .LBB108_3
; GFX1250-SDAG-NEXT: ; %bb.1: ; %Flow
; GFX1250-SDAG-NEXT: s_and_not1_saveexec_b32 s0, s0
@@ -13114,7 +13163,7 @@ define amdgpu_ps void @flat_dec_saddr_i64_nortn(ptr inreg %sbase, i32 %voffset,
; GFX1250-SDAG-NEXT: .LBB108_4: ; %atomicrmw.private
; GFX1250-SDAG-NEXT: v_cmp_ne_u64_e32 vcc_lo, 0, v[0:1]
; GFX1250-SDAG-NEXT: v_subrev_nc_u32_e32 v4, src_flat_scratch_base_lo, v0
-; GFX1250-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_4) | instid1(VALU_DEP_1)
+; GFX1250-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1250-SDAG-NEXT: v_cndmask_b32_e32 v4, -1, v4, vcc_lo
; GFX1250-SDAG-NEXT: scratch_load_b64 v[0:1], v4, off
; GFX1250-SDAG-NEXT: s_wait_loadcnt 0x0
@@ -13122,6 +13171,7 @@ define amdgpu_ps void @flat_dec_saddr_i64_nortn(ptr inreg %sbase, i32 %voffset,
; GFX1250-SDAG-NEXT: v_sub_co_u32 v0, s0, v0, 1
; GFX1250-SDAG-NEXT: v_subrev_co_ci_u32_e64 v1, s0, 0, v1, s0
; GFX1250-SDAG-NEXT: s_or_b32 vcc_lo, s0, vcc_lo
+; GFX1250-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1250-SDAG-NEXT: v_dual_cndmask_b32 v1, v1, v3 :: v_dual_cndmask_b32 v0, v0, v2
; GFX1250-SDAG-NEXT: scratch_store_b64 v4, v[0:1], off
; GFX1250-SDAG-NEXT: s_endpgm
@@ -13135,12 +13185,13 @@ define amdgpu_ps void @flat_dec_saddr_i64_nortn(ptr inreg %sbase, i32 %voffset,
; GFX1250-GISEL-NEXT: v_dual_mov_b32 v4, v1 :: v_dual_mov_b32 v5, v2
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
; GFX1250-GISEL-NEXT: s_mov_b32 s0, exec_lo
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1250-GISEL-NEXT: v_xor_b32_e32 v2, src_flat_scratch_base_hi, v1
; GFX1250-GISEL-NEXT: v_cmpx_le_u32_e32 0x4000000, v2
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1250-GISEL-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX1250-GISEL-NEXT: s_cbranch_execnz .LBB108_3
; GFX1250-GISEL-NEXT: ; %bb.1: ; %Flow
@@ -13158,7 +13209,7 @@ define amdgpu_ps void @flat_dec_saddr_i64_nortn(ptr inreg %sbase, i32 %voffset,
; GFX1250-GISEL-NEXT: .LBB108_4: ; %atomicrmw.private
; GFX1250-GISEL-NEXT: v_cmp_ne_u64_e32 vcc_lo, 0, v[0:1]
; GFX1250-GISEL-NEXT: v_subrev_nc_u32_e32 v2, src_flat_scratch_base_lo, v0
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_4) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1250-GISEL-NEXT: v_cndmask_b32_e32 v2, -1, v2, vcc_lo
; GFX1250-GISEL-NEXT: scratch_load_b64 v[0:1], v2, off
; GFX1250-GISEL-NEXT: s_wait_loadcnt 0x0
@@ -13166,6 +13217,7 @@ define amdgpu_ps void @flat_dec_saddr_i64_nortn(ptr inreg %sbase, i32 %voffset,
; GFX1250-GISEL-NEXT: v_sub_co_u32 v0, s0, v0, 1
; GFX1250-GISEL-NEXT: v_subrev_co_ci_u32_e64 v1, s0, 0, v1, s0
; GFX1250-GISEL-NEXT: s_or_b32 vcc_lo, s0, vcc_lo
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1250-GISEL-NEXT: v_dual_cndmask_b32 v0, v0, v4 :: v_dual_cndmask_b32 v1, v1, v5
; GFX1250-GISEL-NEXT: scratch_store_b64 v2, v[0:1], off
; GFX1250-GISEL-NEXT: s_endpgm
@@ -13271,6 +13323,7 @@ define amdgpu_ps void @flat_dec_saddr_i64_nortn_neg128(ptr inreg %sbase, i32 %vo
; GFX1250-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1250-SDAG-NEXT: v_xor_b32_e32 v4, src_flat_scratch_base_hi, v1
; GFX1250-SDAG-NEXT: v_cmpx_lt_u32_e32 0x3ffffff, v4
+; GFX1250-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1250-SDAG-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX1250-SDAG-NEXT: s_cbranch_execnz .LBB109_3
; GFX1250-SDAG-NEXT: ; %bb.1: ; %Flow
@@ -13288,7 +13341,7 @@ define amdgpu_ps void @flat_dec_saddr_i64_nortn_neg128(ptr inreg %sbase, i32 %vo
; GFX1250-SDAG-NEXT: .LBB109_4: ; %atomicrmw.private
; GFX1250-SDAG-NEXT: v_cmp_ne_u64_e32 vcc_lo, 0, v[0:1]
; GFX1250-SDAG-NEXT: v_subrev_nc_u32_e32 v4, src_flat_scratch_base_lo, v0
-; GFX1250-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_4) | instid1(VALU_DEP_1)
+; GFX1250-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1250-SDAG-NEXT: v_cndmask_b32_e32 v4, -1, v4, vcc_lo
; GFX1250-SDAG-NEXT: scratch_load_b64 v[0:1], v4, off
; GFX1250-SDAG-NEXT: s_wait_loadcnt 0x0
@@ -13296,6 +13349,7 @@ define amdgpu_ps void @flat_dec_saddr_i64_nortn_neg128(ptr inreg %sbase, i32 %vo
; GFX1250-SDAG-NEXT: v_sub_co_u32 v0, s0, v0, 1
; GFX1250-SDAG-NEXT: v_subrev_co_ci_u32_e64 v1, s0, 0, v1, s0
; GFX1250-SDAG-NEXT: s_or_b32 vcc_lo, s0, vcc_lo
+; GFX1250-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1250-SDAG-NEXT: v_dual_cndmask_b32 v1, v1, v3 :: v_dual_cndmask_b32 v0, v0, v2
; GFX1250-SDAG-NEXT: scratch_store_b64 v4, v[0:1], off
; GFX1250-SDAG-NEXT: s_endpgm
@@ -13309,15 +13363,16 @@ define amdgpu_ps void @flat_dec_saddr_i64_nortn_neg128(ptr inreg %sbase, i32 %vo
; GFX1250-GISEL-NEXT: v_dual_mov_b32 v4, v1 :: v_dual_mov_b32 v5, v2
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
; GFX1250-GISEL-NEXT: s_mov_b32 s0, exec_lo
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v2, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, 0xffffff80, v2
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, -1, v3, vcc_lo
; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1250-GISEL-NEXT: v_xor_b32_e32 v6, src_flat_scratch_base_hi, v1
; GFX1250-GISEL-NEXT: v_cmpx_le_u32_e32 0x4000000, v6
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1250-GISEL-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX1250-GISEL-NEXT: s_cbranch_execnz .LBB109_3
; GFX1250-GISEL-NEXT: ; %bb.1: ; %Flow
@@ -13335,7 +13390,7 @@ define amdgpu_ps void @flat_dec_saddr_i64_nortn_neg128(ptr inreg %sbase, i32 %vo
; GFX1250-GISEL-NEXT: .LBB109_4: ; %atomicrmw.private
; GFX1250-GISEL-NEXT: v_cmp_ne_u64_e32 vcc_lo, 0, v[0:1]
; GFX1250-GISEL-NEXT: v_subrev_nc_u32_e32 v2, src_flat_scratch_base_lo, v0
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_4) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1250-GISEL-NEXT: v_cndmask_b32_e32 v2, -1, v2, vcc_lo
; GFX1250-GISEL-NEXT: scratch_load_b64 v[0:1], v2, off
; GFX1250-GISEL-NEXT: s_wait_loadcnt 0x0
@@ -13343,6 +13398,7 @@ define amdgpu_ps void @flat_dec_saddr_i64_nortn_neg128(ptr inreg %sbase, i32 %vo
; GFX1250-GISEL-NEXT: v_sub_co_u32 v0, s0, v0, 1
; GFX1250-GISEL-NEXT: v_subrev_co_ci_u32_e64 v1, s0, 0, v1, s0
; GFX1250-GISEL-NEXT: s_or_b32 vcc_lo, s0, vcc_lo
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1250-GISEL-NEXT: v_dual_cndmask_b32 v0, v0, v4 :: v_dual_cndmask_b32 v1, v1, v5
; GFX1250-GISEL-NEXT: scratch_store_b64 v2, v[0:1], off
; GFX1250-GISEL-NEXT: s_endpgm
@@ -13446,16 +13502,18 @@ define double @flat_atomic_fadd_f64_saddr_rtn(ptr inreg %ptr, double %data) {
; GFX1250-SDAG-NEXT: s_wait_kmcnt 0x0
; GFX1250-SDAG-NEXT: s_mov_b64 s[2:3], src_shared_base
; GFX1250-SDAG-NEXT: s_add_nc_u64 s[0:1], s[0:1], 0x50
-; GFX1250-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
+; GFX1250-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1250-SDAG-NEXT: s_cmp_eq_u32 s1, s3
; GFX1250-SDAG-NEXT: s_cselect_b32 s2, 1, 0
+; GFX1250-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1250-SDAG-NEXT: s_cmp_lg_u32 s2, 1
; GFX1250-SDAG-NEXT: s_cbranch_scc0 .LBB110_3
; GFX1250-SDAG-NEXT: ; %bb.1: ; %atomicrmw.check.private
; GFX1250-SDAG-NEXT: s_xor_b32 s2, s1, src_flat_scratch_base_hi
-; GFX1250-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
+; GFX1250-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1250-SDAG-NEXT: s_cmp_lt_u32 s2, 0x4000000
; GFX1250-SDAG-NEXT: s_cselect_b32 s2, 1, 0
+; GFX1250-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1250-SDAG-NEXT: s_cmp_lg_u32 s2, 1
; GFX1250-SDAG-NEXT: s_cbranch_scc0 .LBB110_4
; GFX1250-SDAG-NEXT: ; %bb.2: ; %atomicrmw.global
@@ -13477,10 +13535,12 @@ define double @flat_atomic_fadd_f64_saddr_rtn(ptr inreg %ptr, double %data) {
; GFX1250-SDAG-NEXT: s_and_b32 s2, s2, exec_lo
; GFX1250-SDAG-NEXT: s_cselect_b32 s2, 1, 0
; GFX1250-SDAG-NEXT: s_cmp_lg_u32 s2, 1
+; GFX1250-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1250-SDAG-NEXT: s_cbranch_scc1 .LBB110_7
; GFX1250-SDAG-NEXT: ; %bb.6: ; %atomicrmw.private
; GFX1250-SDAG-NEXT: s_sub_co_i32 s2, s0, src_flat_scratch_base_lo
; GFX1250-SDAG-NEXT: s_cmp_lg_u64 s[0:1], 0
+; GFX1250-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1250-SDAG-NEXT: s_cselect_b32 s2, s2, -1
; GFX1250-SDAG-NEXT: scratch_load_b64 v[2:3], off, s2
; GFX1250-SDAG-NEXT: s_wait_loadcnt 0x0
@@ -13494,11 +13554,12 @@ define double @flat_atomic_fadd_f64_saddr_rtn(ptr inreg %ptr, double %data) {
; GFX1250-SDAG-NEXT: s_and_b32 s2, s2, exec_lo
; GFX1250-SDAG-NEXT: s_cselect_b32 s2, 1, 0
; GFX1250-SDAG-NEXT: s_cmp_lg_u32 s2, 1
+; GFX1250-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1250-SDAG-NEXT: s_cbranch_scc1 .LBB110_10
; GFX1250-SDAG-NEXT: ; %bb.9: ; %atomicrmw.shared
; GFX1250-SDAG-NEXT: s_cmp_lg_u64 s[0:1], 0
+; GFX1250-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1250-SDAG-NEXT: s_cselect_b32 s0, s0, -1
-; GFX1250-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1250-SDAG-NEXT: v_mov_b32_e32 v2, s0
; GFX1250-SDAG-NEXT: s_wait_storecnt 0x0
; GFX1250-SDAG-NEXT: ds_add_rtn_f64 v[2:3], v2, v[0:1]
@@ -13517,6 +13578,7 @@ define double @flat_atomic_fadd_f64_saddr_rtn(ptr inreg %ptr, double %data) {
; GFX1250-GISEL-NEXT: s_mov_b32 s2, 1
; GFX1250-GISEL-NEXT: s_cmp_lg_u32 s1, s3
; GFX1250-GISEL-NEXT: ; implicit-def: $vgpr2_vgpr3
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1250-GISEL-NEXT: s_cbranch_scc0 .LBB110_6
; GFX1250-GISEL-NEXT: ; %bb.1: ; %atomicrmw.check.private
; GFX1250-GISEL-NEXT: s_xor_b32 s2, s1, src_flat_scratch_base_hi
@@ -13533,12 +13595,13 @@ define double @flat_atomic_fadd_f64_saddr_rtn(ptr inreg %ptr, double %data) {
; GFX1250-GISEL-NEXT: s_wait_loadcnt 0x0
; GFX1250-GISEL-NEXT: .LBB110_3: ; %Flow
; GFX1250-GISEL-NEXT: s_xor_b32 s2, s2, 1
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1250-GISEL-NEXT: s_cmp_lg_u32 s2, 0
; GFX1250-GISEL-NEXT: s_cbranch_scc1 .LBB110_5
; GFX1250-GISEL-NEXT: ; %bb.4: ; %atomicrmw.private
; GFX1250-GISEL-NEXT: s_sub_co_i32 s2, s0, src_flat_scratch_base_lo
; GFX1250-GISEL-NEXT: s_cmp_lg_u64 s[0:1], 0
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1250-GISEL-NEXT: s_cselect_b32 s2, s2, -1
; GFX1250-GISEL-NEXT: scratch_load_b64 v[2:3], off, s2
; GFX1250-GISEL-NEXT: s_wait_loadcnt 0x0
@@ -13551,11 +13614,12 @@ define double @flat_atomic_fadd_f64_saddr_rtn(ptr inreg %ptr, double %data) {
; GFX1250-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1250-GISEL-NEXT: s_xor_b32 s2, s2, 1
; GFX1250-GISEL-NEXT: s_cmp_lg_u32 s2, 0
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1250-GISEL-NEXT: s_cbranch_scc1 .LBB110_8
; GFX1250-GISEL-NEXT: ; %bb.7: ; %atomicrmw.shared
; GFX1250-GISEL-NEXT: s_cmp_lg_u64 s[0:1], 0
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1250-GISEL-NEXT: s_cselect_b32 s0, s0, -1
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1250-GISEL-NEXT: v_mov_b32_e32 v2, s0
; GFX1250-GISEL-NEXT: s_wait_storecnt 0x0
; GFX1250-GISEL-NEXT: ds_add_rtn_f64 v[2:3], v2, v[0:1]
@@ -13685,17 +13749,19 @@ define void @flat_atomic_fadd_f64_saddr_nortn(ptr inreg %ptr, double %data) {
; GFX1250-SDAG-NEXT: s_wait_kmcnt 0x0
; GFX1250-SDAG-NEXT: s_mov_b64 s[2:3], src_shared_base
; GFX1250-SDAG-NEXT: s_add_nc_u64 s[0:1], s[0:1], 0x50
-; GFX1250-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
+; GFX1250-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1250-SDAG-NEXT: s_cmp_eq_u32 s1, s3
; GFX1250-SDAG-NEXT: s_cselect_b32 s2, 1, 0
+; GFX1250-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1250-SDAG-NEXT: s_cmp_lg_u32 s2, 1
; GFX1250-SDAG-NEXT: s_mov_b32 s2, -1
; GFX1250-SDAG-NEXT: s_cbranch_scc0 .LBB111_6
; GFX1250-SDAG-NEXT: ; %bb.1: ; %atomicrmw.check.private
; GFX1250-SDAG-NEXT: s_xor_b32 s2, s1, src_flat_scratch_base_hi
-; GFX1250-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
+; GFX1250-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1250-SDAG-NEXT: s_cmp_lt_u32 s2, 0x4000000
; GFX1250-SDAG-NEXT: s_cselect_b32 s2, 1, 0
+; GFX1250-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1250-SDAG-NEXT: s_cmp_lg_u32 s2, 1
; GFX1250-SDAG-NEXT: s_mov_b32 s2, -1
; GFX1250-SDAG-NEXT: s_cbranch_scc0 .LBB111_3
@@ -13708,12 +13774,13 @@ define void @flat_atomic_fadd_f64_saddr_nortn(ptr inreg %ptr, double %data) {
; GFX1250-SDAG-NEXT: .LBB111_3: ; %Flow
; GFX1250-SDAG-NEXT: s_and_b32 s2, s2, exec_lo
; GFX1250-SDAG-NEXT: s_cselect_b32 s2, 1, 0
-; GFX1250-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1250-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1250-SDAG-NEXT: s_cmp_lg_u32 s2, 1
; GFX1250-SDAG-NEXT: s_cbranch_scc1 .LBB111_5
; GFX1250-SDAG-NEXT: ; %bb.4: ; %atomicrmw.private
; GFX1250-SDAG-NEXT: s_sub_co_i32 s2, s0, src_flat_scratch_base_lo
; GFX1250-SDAG-NEXT: s_cmp_lg_u64 s[0:1], 0
+; GFX1250-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1250-SDAG-NEXT: s_cselect_b32 s2, s2, -1
; GFX1250-SDAG-NEXT: scratch_load_b64 v[2:3], off, s2
; GFX1250-SDAG-NEXT: s_wait_loadcnt 0x0
@@ -13727,11 +13794,12 @@ define void @flat_atomic_fadd_f64_saddr_nortn(ptr inreg %ptr, double %data) {
; GFX1250-SDAG-NEXT: s_and_b32 s2, s2, exec_lo
; GFX1250-SDAG-NEXT: s_cselect_b32 s2, 1, 0
; GFX1250-SDAG-NEXT: s_cmp_lg_u32 s2, 1
+; GFX1250-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1250-SDAG-NEXT: s_cbranch_scc1 .LBB111_8
; GFX1250-SDAG-NEXT: ; %bb.7: ; %atomicrmw.shared
; GFX1250-SDAG-NEXT: s_cmp_lg_u64 s[0:1], 0
+; GFX1250-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1250-SDAG-NEXT: s_cselect_b32 s0, s0, -1
-; GFX1250-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1250-SDAG-NEXT: v_mov_b32_e32 v2, s0
; GFX1250-SDAG-NEXT: s_wait_storecnt 0x0
; GFX1250-SDAG-NEXT: ds_add_f64 v2, v[0:1]
@@ -13748,6 +13816,7 @@ define void @flat_atomic_fadd_f64_saddr_nortn(ptr inreg %ptr, double %data) {
; GFX1250-GISEL-NEXT: s_add_co_ci_u32 s1, s1, 0
; GFX1250-GISEL-NEXT: s_mov_b32 s2, 1
; GFX1250-GISEL-NEXT: s_cmp_lg_u32 s1, s3
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1250-GISEL-NEXT: s_cbranch_scc0 .LBB111_6
; GFX1250-GISEL-NEXT: ; %bb.1: ; %atomicrmw.check.private
; GFX1250-GISEL-NEXT: s_xor_b32 s2, s1, src_flat_scratch_base_hi
@@ -13763,12 +13832,13 @@ define void @flat_atomic_fadd_f64_saddr_nortn(ptr inreg %ptr, double %data) {
; GFX1250-GISEL-NEXT: s_wait_storecnt 0x0
; GFX1250-GISEL-NEXT: .LBB111_3: ; %Flow
; GFX1250-GISEL-NEXT: s_xor_b32 s2, s2, 1
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1250-GISEL-NEXT: s_cmp_lg_u32 s2, 0
; GFX1250-GISEL-NEXT: s_cbranch_scc1 .LBB111_5
; GFX1250-GISEL-NEXT: ; %bb.4: ; %atomicrmw.private
; GFX1250-GISEL-NEXT: s_sub_co_i32 s2, s0, src_flat_scratch_base_lo
; GFX1250-GISEL-NEXT: s_cmp_lg_u64 s[0:1], 0
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1250-GISEL-NEXT: s_cselect_b32 s2, s2, -1
; GFX1250-GISEL-NEXT: scratch_load_b64 v[2:3], off, s2
; GFX1250-GISEL-NEXT: s_wait_loadcnt 0x0
@@ -13781,11 +13851,12 @@ define void @flat_atomic_fadd_f64_saddr_nortn(ptr inreg %ptr, double %data) {
; GFX1250-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1250-GISEL-NEXT: s_xor_b32 s2, s2, 1
; GFX1250-GISEL-NEXT: s_cmp_lg_u32 s2, 0
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1250-GISEL-NEXT: s_cbranch_scc1 .LBB111_8
; GFX1250-GISEL-NEXT: ; %bb.7: ; %atomicrmw.shared
; GFX1250-GISEL-NEXT: s_cmp_lg_u64 s[0:1], 0
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1250-GISEL-NEXT: s_cselect_b32 s0, s0, -1
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1250-GISEL-NEXT: v_mov_b32_e32 v2, s0
; GFX1250-GISEL-NEXT: s_wait_storecnt 0x0
; GFX1250-GISEL-NEXT: ds_add_f64 v2, v[0:1]
@@ -13902,9 +13973,10 @@ define double @flat_atomic_fmax_f64_saddr_rtn(ptr inreg %ptr, double %data) {
; GFX1250-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1250-SDAG-NEXT: s_xor_b32 s2, s1, src_flat_scratch_base_hi
; GFX1250-SDAG-NEXT: s_cmp_lt_u32 s2, 0x4000000
+; GFX1250-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1250-SDAG-NEXT: s_cselect_b32 s2, 1, 0
-; GFX1250-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1250-SDAG-NEXT: s_cmp_lg_u32 s2, 1
+; GFX1250-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1250-SDAG-NEXT: s_cbranch_scc0 .LBB112_2
; GFX1250-SDAG-NEXT: ; %bb.1: ; %atomicrmw.global
; GFX1250-SDAG-NEXT: v_mov_b32_e32 v2, 0
@@ -13921,6 +13993,7 @@ define double @flat_atomic_fmax_f64_saddr_rtn(ptr inreg %ptr, double %data) {
; GFX1250-SDAG-NEXT: s_and_b32 s2, s2, exec_lo
; GFX1250-SDAG-NEXT: s_cselect_b32 s2, 1, 0
; GFX1250-SDAG-NEXT: s_cmp_lg_u32 s2, 1
+; GFX1250-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1250-SDAG-NEXT: s_cbranch_scc1 .LBB112_5
; GFX1250-SDAG-NEXT: ; %bb.4: ; %atomicrmw.private
; GFX1250-SDAG-NEXT: s_sub_co_i32 s2, s0, src_flat_scratch_base_lo
@@ -13958,7 +14031,7 @@ define double @flat_atomic_fmax_f64_saddr_rtn(ptr inreg %ptr, double %data) {
; GFX1250-GISEL-NEXT: s_wait_loadcnt_dscnt 0x0
; GFX1250-GISEL-NEXT: .LBB112_2: ; %Flow
; GFX1250-GISEL-NEXT: s_xor_b32 s0, s4, 1
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1250-GISEL-NEXT: s_cmp_lg_u32 s0, 0
; GFX1250-GISEL-NEXT: s_cbranch_scc1 .LBB112_4
; GFX1250-GISEL-NEXT: ; %bb.3: ; %atomicrmw.private
@@ -14063,8 +14136,8 @@ define void @flat_atomic_fmax_f64_saddr_nortn(ptr inreg %ptr, double %data) {
; GFX1250-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1250-SDAG-NEXT: s_xor_b32 s2, s1, src_flat_scratch_base_hi
; GFX1250-SDAG-NEXT: s_cmp_lt_u32 s2, 0x4000000
+; GFX1250-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1250-SDAG-NEXT: s_cselect_b32 s2, 1, 0
-; GFX1250-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1250-SDAG-NEXT: s_cmp_lg_u32 s2, 1
; GFX1250-SDAG-NEXT: s_mov_b32 s2, -1
; GFX1250-SDAG-NEXT: s_cbranch_scc0 .LBB113_2
@@ -14077,7 +14150,7 @@ define void @flat_atomic_fmax_f64_saddr_nortn(ptr inreg %ptr, double %data) {
; GFX1250-SDAG-NEXT: .LBB113_2: ; %Flow
; GFX1250-SDAG-NEXT: s_and_b32 s2, s2, exec_lo
; GFX1250-SDAG-NEXT: s_cselect_b32 s2, 1, 0
-; GFX1250-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1250-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1250-SDAG-NEXT: s_cmp_lg_u32 s2, 1
; GFX1250-SDAG-NEXT: s_cbranch_scc1 .LBB113_4
; GFX1250-SDAG-NEXT: ; %bb.3: ; %atomicrmw.private
@@ -14114,7 +14187,7 @@ define void @flat_atomic_fmax_f64_saddr_nortn(ptr inreg %ptr, double %data) {
; GFX1250-GISEL-NEXT: .LBB113_2: ; %Flow
; GFX1250-GISEL-NEXT: s_wait_xcnt 0x0
; GFX1250-GISEL-NEXT: s_xor_b32 s0, s4, 1
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1250-GISEL-NEXT: s_cmp_lg_u32 s0, 0
; GFX1250-GISEL-NEXT: s_cbranch_scc1 .LBB113_4
; GFX1250-GISEL-NEXT: ; %bb.3: ; %atomicrmw.private
@@ -14209,9 +14282,10 @@ define double @flat_atomic_fmin_f64_saddr_rtn(ptr inreg %ptr, double %data) {
; GFX1250-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1250-SDAG-NEXT: s_xor_b32 s2, s1, src_flat_scratch_base_hi
; GFX1250-SDAG-NEXT: s_cmp_lt_u32 s2, 0x4000000
+; GFX1250-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1250-SDAG-NEXT: s_cselect_b32 s2, 1, 0
-; GFX1250-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1250-SDAG-NEXT: s_cmp_lg_u32 s2, 1
+; GFX1250-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1250-SDAG-NEXT: s_cbranch_scc0 .LBB114_2
; GFX1250-SDAG-NEXT: ; %bb.1: ; %atomicrmw.global
; GFX1250-SDAG-NEXT: v_mov_b32_e32 v2, 0
@@ -14228,6 +14302,7 @@ define double @flat_atomic_fmin_f64_saddr_rtn(ptr inreg %ptr, double %data) {
; GFX1250-SDAG-NEXT: s_and_b32 s2, s2, exec_lo
; GFX1250-SDAG-NEXT: s_cselect_b32 s2, 1, 0
; GFX1250-SDAG-NEXT: s_cmp_lg_u32 s2, 1
+; GFX1250-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1250-SDAG-NEXT: s_cbranch_scc1 .LBB114_5
; GFX1250-SDAG-NEXT: ; %bb.4: ; %atomicrmw.private
; GFX1250-SDAG-NEXT: s_sub_co_i32 s2, s0, src_flat_scratch_base_lo
@@ -14265,7 +14340,7 @@ define double @flat_atomic_fmin_f64_saddr_rtn(ptr inreg %ptr, double %data) {
; GFX1250-GISEL-NEXT: s_wait_loadcnt_dscnt 0x0
; GFX1250-GISEL-NEXT: .LBB114_2: ; %Flow
; GFX1250-GISEL-NEXT: s_xor_b32 s0, s4, 1
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1250-GISEL-NEXT: s_cmp_lg_u32 s0, 0
; GFX1250-GISEL-NEXT: s_cbranch_scc1 .LBB114_4
; GFX1250-GISEL-NEXT: ; %bb.3: ; %atomicrmw.private
@@ -14370,8 +14445,8 @@ define void @flat_atomic_fmin_f64_saddr_nortn(ptr inreg %ptr, double %data) {
; GFX1250-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1250-SDAG-NEXT: s_xor_b32 s2, s1, src_flat_scratch_base_hi
; GFX1250-SDAG-NEXT: s_cmp_lt_u32 s2, 0x4000000
+; GFX1250-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1250-SDAG-NEXT: s_cselect_b32 s2, 1, 0
-; GFX1250-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1250-SDAG-NEXT: s_cmp_lg_u32 s2, 1
; GFX1250-SDAG-NEXT: s_mov_b32 s2, -1
; GFX1250-SDAG-NEXT: s_cbranch_scc0 .LBB115_2
@@ -14384,7 +14459,7 @@ define void @flat_atomic_fmin_f64_saddr_nortn(ptr inreg %ptr, double %data) {
; GFX1250-SDAG-NEXT: .LBB115_2: ; %Flow
; GFX1250-SDAG-NEXT: s_and_b32 s2, s2, exec_lo
; GFX1250-SDAG-NEXT: s_cselect_b32 s2, 1, 0
-; GFX1250-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1250-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1250-SDAG-NEXT: s_cmp_lg_u32 s2, 1
; GFX1250-SDAG-NEXT: s_cbranch_scc1 .LBB115_4
; GFX1250-SDAG-NEXT: ; %bb.3: ; %atomicrmw.private
@@ -14421,7 +14496,7 @@ define void @flat_atomic_fmin_f64_saddr_nortn(ptr inreg %ptr, double %data) {
; GFX1250-GISEL-NEXT: .LBB115_2: ; %Flow
; GFX1250-GISEL-NEXT: s_wait_xcnt 0x0
; GFX1250-GISEL-NEXT: s_xor_b32 s0, s4, 1
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1250-GISEL-NEXT: s_cmp_lg_u32 s0, 0
; GFX1250-GISEL-NEXT: s_cbranch_scc1 .LBB115_4
; GFX1250-GISEL-NEXT: ; %bb.3: ; %atomicrmw.private
@@ -14810,7 +14885,7 @@ define <2 x half> @flat_atomic_fmax_v2f16_saddr_rtn(ptr inreg %ptr, <2 x half> %
; GFX1250-NEXT: s_wait_loadcnt_dscnt 0x0
; GFX1250-NEXT: v_cmp_eq_u32_e32 vcc_lo, v0, v5
; GFX1250-NEXT: s_or_b32 s2, vcc_lo, s2
-; GFX1250-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1250-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1250-NEXT: s_and_not1_b32 exec_lo, exec_lo, s2
; GFX1250-NEXT: s_cbranch_execnz .LBB124_1
; GFX1250-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -14868,7 +14943,7 @@ define void @flat_atomic_fmax_v2f16_saddr_nortn(ptr inreg %ptr, <2 x half> %data
; GFX1250-NEXT: v_cmp_eq_u32_e32 vcc_lo, v0, v1
; GFX1250-NEXT: v_mov_b32_e32 v1, v0
; GFX1250-NEXT: s_or_b32 s2, vcc_lo, s2
-; GFX1250-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1250-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1250-NEXT: s_and_not1_b32 exec_lo, exec_lo, s2
; GFX1250-NEXT: s_cbranch_execnz .LBB125_1
; GFX1250-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -14927,7 +15002,7 @@ define <2 x half> @flat_atomic_fmin_v2f16_saddr_rtn(ptr inreg %ptr, <2 x half> %
; GFX1250-NEXT: s_wait_loadcnt_dscnt 0x0
; GFX1250-NEXT: v_cmp_eq_u32_e32 vcc_lo, v0, v5
; GFX1250-NEXT: s_or_b32 s2, vcc_lo, s2
-; GFX1250-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1250-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1250-NEXT: s_and_not1_b32 exec_lo, exec_lo, s2
; GFX1250-NEXT: s_cbranch_execnz .LBB126_1
; GFX1250-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -14985,7 +15060,7 @@ define void @flat_atomic_fmin_v2f16_saddr_nortn(ptr inreg %ptr, <2 x half> %data
; GFX1250-NEXT: v_cmp_eq_u32_e32 vcc_lo, v0, v1
; GFX1250-NEXT: v_mov_b32_e32 v1, v0
; GFX1250-NEXT: s_or_b32 s2, vcc_lo, s2
-; GFX1250-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1250-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1250-NEXT: s_and_not1_b32 exec_lo, exec_lo, s2
; GFX1250-NEXT: s_cbranch_execnz .LBB127_1
; GFX1250-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -15105,11 +15180,12 @@ define <2 x bfloat> @flat_atomic_fmax_v2bf16_saddr_rtn(ptr inreg %ptr, <2 x bflo
; GFX1250-NEXT: s_wait_loadcnt_dscnt 0x0
; GFX1250-NEXT: v_cmp_eq_u32_e32 vcc_lo, v1, v5
; GFX1250-NEXT: s_or_b32 s2, vcc_lo, s2
-; GFX1250-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1250-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1250-NEXT: s_and_not1_b32 exec_lo, exec_lo, s2
; GFX1250-NEXT: s_cbranch_execnz .LBB130_1
; GFX1250-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX1250-NEXT: s_or_b32 exec_lo, exec_lo, s2
+; GFX1250-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1250-NEXT: v_mov_b32_e32 v0, v1
; GFX1250-NEXT: s_set_pc_i64 s[30:31]
;
@@ -15164,7 +15240,7 @@ define void @flat_atomic_fmax_v2bf16_saddr_nortn(ptr inreg %ptr, <2 x bfloat> %d
; GFX1250-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v3
; GFX1250-NEXT: v_mov_b32_e32 v3, v2
; GFX1250-NEXT: s_or_b32 s2, vcc_lo, s2
-; GFX1250-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1250-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1250-NEXT: s_and_not1_b32 exec_lo, exec_lo, s2
; GFX1250-NEXT: s_cbranch_execnz .LBB131_1
; GFX1250-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -15223,11 +15299,12 @@ define <2 x bfloat> @flat_atomic_fmin_v2bf16_saddr_rtn(ptr inreg %ptr, <2 x bflo
; GFX1250-NEXT: s_wait_loadcnt_dscnt 0x0
; GFX1250-NEXT: v_cmp_eq_u32_e32 vcc_lo, v1, v5
; GFX1250-NEXT: s_or_b32 s2, vcc_lo, s2
-; GFX1250-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1250-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1250-NEXT: s_and_not1_b32 exec_lo, exec_lo, s2
; GFX1250-NEXT: s_cbranch_execnz .LBB132_1
; GFX1250-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX1250-NEXT: s_or_b32 exec_lo, exec_lo, s2
+; GFX1250-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1250-NEXT: v_mov_b32_e32 v0, v1
; GFX1250-NEXT: s_set_pc_i64 s[30:31]
;
@@ -15282,7 +15359,7 @@ define void @flat_atomic_fmin_v2bf16_saddr_nortn(ptr inreg %ptr, <2 x bfloat> %d
; GFX1250-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v3
; GFX1250-NEXT: v_mov_b32_e32 v3, v2
; GFX1250-NEXT: s_or_b32 s2, vcc_lo, s2
-; GFX1250-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1250-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1250-NEXT: s_and_not1_b32 exec_lo, exec_lo, s2
; GFX1250-NEXT: s_cbranch_execnz .LBB133_1
; GFX1250-NEXT: ; %bb.2: ; %atomicrmw.end
diff --git a/llvm/test/CodeGen/AMDGPU/flat-saddr-load.ll b/llvm/test/CodeGen/AMDGPU/flat-saddr-load.ll
index 0aada3a1c29767..a3eaa35b246d55 100644
--- a/llvm/test/CodeGen/AMDGPU/flat-saddr-load.ll
+++ b/llvm/test/CodeGen/AMDGPU/flat-saddr-load.ll
@@ -97,7 +97,6 @@ define amdgpu_ps float @flat_load_saddr_i8_offset_neg8388609(ptr inreg %sbase) {
; GFX1250-SDAG-NEXT: v_nop
; GFX1250-SDAG-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-SDAG-NEXT: v_add_co_u32 v0, s0, 0xff800000, s2
-; GFX1250-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1250-SDAG-NEXT: v_add_co_ci_u32_e64 v1, null, -1, s3, s0
; GFX1250-SDAG-NEXT: flat_load_u8 v0, v[0:1] offset:-1
; GFX1250-SDAG-NEXT: s_wait_loadcnt_dscnt 0x0
@@ -124,7 +123,6 @@ define amdgpu_ps float @flat_load_saddr_i8_offset_neg8388609(ptr inreg %sbase) {
; GFX1250-NOECC-NEXT: v_nop
; GFX1250-NOECC-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-NOECC-NEXT: v_add_co_u32 v0, s0, 0xff800000, s2
-; GFX1250-NOECC-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1250-NOECC-NEXT: v_add_co_ci_u32_e64 v1, null, -1, s3, s0
; GFX1250-NOECC-NEXT: flat_load_u8 v0, v[0:1] offset:-1
; GFX1250-NOECC-NEXT: s_wait_loadcnt_dscnt 0x0
@@ -144,7 +142,6 @@ define amdgpu_ps float @flat_load_saddr_i8_offset_0xFFFFFFFF(ptr inreg %sbase) {
; GFX1250-SDAG-NEXT: v_nop
; GFX1250-SDAG-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-SDAG-NEXT: v_add_co_u32 v0, s0, 0xff800000, s2
-; GFX1250-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1250-SDAG-NEXT: v_add_co_ci_u32_e64 v1, null, 0, s3, s0
; GFX1250-SDAG-NEXT: flat_load_u8 v0, v[0:1] offset:8388607
; GFX1250-SDAG-NEXT: s_wait_loadcnt_dscnt 0x0
@@ -171,7 +168,6 @@ define amdgpu_ps float @flat_load_saddr_i8_offset_0xFFFFFFFF(ptr inreg %sbase) {
; GFX1250-NOECC-NEXT: v_nop
; GFX1250-NOECC-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-NOECC-NEXT: v_add_co_u32 v0, s0, 0xff800000, s2
-; GFX1250-NOECC-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1250-NOECC-NEXT: v_add_co_ci_u32_e64 v1, null, 0, s3, s0
; GFX1250-NOECC-NEXT: flat_load_u8 v0, v[0:1] offset:8388607
; GFX1250-NOECC-NEXT: s_wait_loadcnt_dscnt 0x0
@@ -236,7 +232,6 @@ define amdgpu_ps float @flat_load_saddr_i8_offset_0x100000001(ptr inreg %sbase)
; GFX1250-SDAG-NEXT: v_nop
; GFX1250-SDAG-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-SDAG-NEXT: v_add_co_u32 v0, s0, 0, s2
-; GFX1250-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1250-SDAG-NEXT: v_add_co_ci_u32_e64 v1, null, 1, s3, s0
; GFX1250-SDAG-NEXT: flat_load_u8 v0, v[0:1] offset:1
; GFX1250-SDAG-NEXT: s_wait_loadcnt_dscnt 0x0
@@ -263,7 +258,6 @@ define amdgpu_ps float @flat_load_saddr_i8_offset_0x100000001(ptr inreg %sbase)
; GFX1250-NOECC-NEXT: v_nop
; GFX1250-NOECC-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-NOECC-NEXT: v_add_co_u32 v0, s0, 0, s2
-; GFX1250-NOECC-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1250-NOECC-NEXT: v_add_co_ci_u32_e64 v1, null, 1, s3, s0
; GFX1250-NOECC-NEXT: flat_load_u8 v0, v[0:1] offset:1
; GFX1250-NOECC-NEXT: s_wait_loadcnt_dscnt 0x0
@@ -283,7 +277,6 @@ define amdgpu_ps float @flat_load_saddr_i8_offset_0x100000FFF(ptr inreg %sbase)
; GFX1250-SDAG-NEXT: v_nop
; GFX1250-SDAG-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-SDAG-NEXT: v_add_co_u32 v0, s0, 0, s2
-; GFX1250-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1250-SDAG-NEXT: v_add_co_ci_u32_e64 v1, null, 1, s3, s0
; GFX1250-SDAG-NEXT: flat_load_u8 v0, v[0:1] offset:4095
; GFX1250-SDAG-NEXT: s_wait_loadcnt_dscnt 0x0
@@ -310,7 +303,6 @@ define amdgpu_ps float @flat_load_saddr_i8_offset_0x100000FFF(ptr inreg %sbase)
; GFX1250-NOECC-NEXT: v_nop
; GFX1250-NOECC-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-NOECC-NEXT: v_add_co_u32 v0, s0, 0, s2
-; GFX1250-NOECC-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1250-NOECC-NEXT: v_add_co_ci_u32_e64 v1, null, 1, s3, s0
; GFX1250-NOECC-NEXT: flat_load_u8 v0, v[0:1] offset:4095
; GFX1250-NOECC-NEXT: s_wait_loadcnt_dscnt 0x0
@@ -330,7 +322,6 @@ define amdgpu_ps float @flat_load_saddr_i8_offset_0x100001000(ptr inreg %sbase)
; GFX1250-SDAG-NEXT: v_nop
; GFX1250-SDAG-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-SDAG-NEXT: v_add_co_u32 v0, s0, 0, s2
-; GFX1250-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1250-SDAG-NEXT: v_add_co_ci_u32_e64 v1, null, 1, s3, s0
; GFX1250-SDAG-NEXT: flat_load_u8 v0, v[0:1] offset:4096
; GFX1250-SDAG-NEXT: s_wait_loadcnt_dscnt 0x0
@@ -357,7 +348,6 @@ define amdgpu_ps float @flat_load_saddr_i8_offset_0x100001000(ptr inreg %sbase)
; GFX1250-NOECC-NEXT: v_nop
; GFX1250-NOECC-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-NOECC-NEXT: v_add_co_u32 v0, s0, 0, s2
-; GFX1250-NOECC-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1250-NOECC-NEXT: v_add_co_ci_u32_e64 v1, null, 1, s3, s0
; GFX1250-NOECC-NEXT: flat_load_u8 v0, v[0:1] offset:4096
; GFX1250-NOECC-NEXT: s_wait_loadcnt_dscnt 0x0
@@ -377,7 +367,6 @@ define amdgpu_ps float @flat_load_saddr_i8_offset_neg0xFFFFFFFF(ptr inreg %sbase
; GFX1250-SDAG-NEXT: v_nop
; GFX1250-SDAG-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-SDAG-NEXT: v_add_co_u32 v0, s0, 0x800000, s2
-; GFX1250-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1250-SDAG-NEXT: v_add_co_ci_u32_e64 v1, null, -1, s3, s0
; GFX1250-SDAG-NEXT: flat_load_u8 v0, v[0:1] offset:-8388607
; GFX1250-SDAG-NEXT: s_wait_loadcnt_dscnt 0x0
@@ -404,7 +393,6 @@ define amdgpu_ps float @flat_load_saddr_i8_offset_neg0xFFFFFFFF(ptr inreg %sbase
; GFX1250-NOECC-NEXT: v_nop
; GFX1250-NOECC-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-NOECC-NEXT: v_add_co_u32 v0, s0, 0x800000, s2
-; GFX1250-NOECC-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1250-NOECC-NEXT: v_add_co_ci_u32_e64 v1, null, -1, s3, s0
; GFX1250-NOECC-NEXT: flat_load_u8 v0, v[0:1] offset:-8388607
; GFX1250-NOECC-NEXT: s_wait_loadcnt_dscnt 0x0
@@ -469,7 +457,6 @@ define amdgpu_ps float @flat_load_saddr_i8_offset_neg0x100000001(ptr inreg %sbas
; GFX1250-SDAG-NEXT: v_nop
; GFX1250-SDAG-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-SDAG-NEXT: v_add_co_u32 v0, s0, 0, s2
-; GFX1250-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1250-SDAG-NEXT: v_add_co_ci_u32_e64 v1, null, -1, s3, s0
; GFX1250-SDAG-NEXT: flat_load_u8 v0, v[0:1] offset:-1
; GFX1250-SDAG-NEXT: s_wait_loadcnt_dscnt 0x0
@@ -496,7 +483,6 @@ define amdgpu_ps float @flat_load_saddr_i8_offset_neg0x100000001(ptr inreg %sbas
; GFX1250-NOECC-NEXT: v_nop
; GFX1250-NOECC-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-NOECC-NEXT: v_add_co_u32 v0, s0, 0, s2
-; GFX1250-NOECC-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1250-NOECC-NEXT: v_add_co_ci_u32_e64 v1, null, -1, s3, s0
; GFX1250-NOECC-NEXT: flat_load_u8 v0, v[0:1] offset:-1
; GFX1250-NOECC-NEXT: s_wait_loadcnt_dscnt 0x0
@@ -534,7 +520,7 @@ define amdgpu_ps float @flat_load_saddr_i8_zext_vgpr(ptr inreg %sbase, i32 %voff
; GFX1250-GISEL-NEXT: v_nop
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: flat_load_u8 v0, v[0:1]
@@ -583,7 +569,7 @@ define amdgpu_ps float @flat_load_saddr_i8_zext_vgpr_offset_8388607(ptr inreg %s
; GFX1250-GISEL-NEXT: v_nop
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: flat_load_u8 v0, v[0:1] offset:8388607
@@ -623,7 +609,7 @@ define amdgpu_ps float @flat_load_saddr_i8_zext_vgpr_offset_8388608(ptr inreg %s
; GFX1250-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1250-SDAG-NEXT: v_add_nc_u64_e32 v[0:1], s[2:3], v[0:1]
; GFX1250-SDAG-NEXT: v_add_co_u32 v0, vcc_lo, 0x800000, v0
-; GFX1250-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1250-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX1250-SDAG-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1250-SDAG-NEXT: flat_load_u8 v0, v[0:1]
; GFX1250-SDAG-NEXT: s_wait_loadcnt_dscnt 0x0
@@ -636,10 +622,10 @@ define amdgpu_ps float @flat_load_saddr_i8_zext_vgpr_offset_8388608(ptr inreg %s
; GFX1250-GISEL-NEXT: v_nop
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v3, vcc_lo
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, 0x800000, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1250-GISEL-NEXT: flat_load_u8 v0, v[0:1]
@@ -656,7 +642,7 @@ define amdgpu_ps float @flat_load_saddr_i8_zext_vgpr_offset_8388608(ptr inreg %s
; GFX1250-NOECC-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1250-NOECC-NEXT: v_add_nc_u64_e32 v[0:1], s[2:3], v[0:1]
; GFX1250-NOECC-NEXT: v_add_co_u32 v0, vcc_lo, 0x800000, v0
-; GFX1250-NOECC-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1250-NOECC-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX1250-NOECC-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1250-NOECC-NEXT: flat_load_u8 v0, v[0:1]
; GFX1250-NOECC-NEXT: s_wait_loadcnt_dscnt 0x0
@@ -692,7 +678,7 @@ define amdgpu_ps float @flat_load_saddr_i8_zext_vgpr_offset_neg8388608(ptr inreg
; GFX1250-GISEL-NEXT: v_nop
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: flat_load_u8 v0, v[0:1] offset:-8388608
@@ -742,7 +728,7 @@ define amdgpu_ps float @flat_load_saddr_i8_zext_vgpr_offset_neg8388607(ptr inreg
; GFX1250-GISEL-NEXT: v_nop
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: flat_load_u8 v0, v[0:1] offset:-8388607
@@ -791,7 +777,7 @@ define amdgpu_ps float @flat_load_saddr_i8_zext_vgpr_offset_8388607_gep_order(pt
; GFX1250-GISEL-NEXT: v_nop
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: flat_load_u8 v0, v[0:1] offset:8388607
@@ -841,7 +827,7 @@ define amdgpu_ps float @flat_load_saddr_i8_zext_vgpr_ptrtoint(ptr inreg %sbase,
; GFX1250-GISEL-NEXT: v_nop
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: flat_load_u8 v0, v[0:1]
@@ -892,7 +878,7 @@ define amdgpu_ps float @flat_load_saddr_i8_zext_vgpr_ptrtoint_commute_add(ptr in
; GFX1250-GISEL-NEXT: v_nop
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: flat_load_u8 v0, v[0:1]
@@ -944,10 +930,10 @@ define amdgpu_ps float @flat_load_saddr_i8_zext_vgpr_ptrtoint_commute_add_imm_of
; GFX1250-GISEL-NEXT: v_nop
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v3, vcc_lo
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, 0x80, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1250-GISEL-NEXT: flat_load_u8 v0, v[0:1]
@@ -1001,10 +987,10 @@ define amdgpu_ps float @flat_load_saddr_i8_zext_vgpr_ptrtoint_commute_add_imm_of
; GFX1250-GISEL-NEXT: v_nop
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v3, vcc_lo
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, 0x80, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1250-GISEL-NEXT: flat_load_u8 v0, v[0:1]
@@ -1067,7 +1053,6 @@ define amdgpu_ps float @flat_load_saddr_uniform_ptr_in_vgprs(i32 %voffset) {
; GFX1250-GISEL-NEXT: ds_load_b64 v[2:3], v1
; GFX1250-GISEL-NEXT: s_wait_dscnt 0x0
; GFX1250-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, v2, v0
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: flat_load_u8 v0, v[0:1]
; GFX1250-GISEL-NEXT: s_wait_loadcnt_dscnt 0x0
@@ -1121,7 +1106,6 @@ define amdgpu_ps float @flat_load_saddr_uniform_ptr_in_vgprs_immoffset(i32 %voff
; GFX1250-GISEL-NEXT: ds_load_b64 v[2:3], v1
; GFX1250-GISEL-NEXT: s_wait_dscnt 0x0
; GFX1250-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, v2, v0
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: flat_load_u8 v0, v[0:1] offset:42
; GFX1250-GISEL-NEXT: s_wait_loadcnt_dscnt 0x0
@@ -1381,7 +1365,7 @@ define amdgpu_ps float @flat_load_i8_vgpr64_sgpr32(ptr %vbase, i32 inreg %soffse
; GFX1250-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
; GFX1250-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, v1, v3, vcc_lo
; GFX1250-GISEL-NEXT: flat_load_u8 v0, v[0:1]
; GFX1250-GISEL-NEXT: s_wait_loadcnt_dscnt 0x0
@@ -1432,7 +1416,7 @@ define amdgpu_ps float @flat_load_i8_vgpr64_sgpr32_offset_8388607(ptr %vbase, i3
; GFX1250-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
; GFX1250-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, v1, v3, vcc_lo
; GFX1250-GISEL-NEXT: flat_load_u8 v0, v[0:1] offset:8388607
; GFX1250-GISEL-NEXT: s_wait_loadcnt_dscnt 0x0
@@ -1515,7 +1499,7 @@ define amdgpu_ps float @flat_load_saddr_f32_natural_addressing_immoffset(ptr inr
; GFX1250-GISEL-NEXT: s_wait_xcnt 0x0
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[0:1], s[2:3]
; GFX1250-GISEL-NEXT: s_wait_loadcnt_dscnt 0x0
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1250-GISEL-NEXT: flat_load_b32 v0, v[0:1] offset:128
@@ -1573,7 +1557,7 @@ define amdgpu_ps float @flat_load_f32_saddr_zext_vgpr_range(ptr inreg %sbase, pt
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[0:1], s[2:3]
; GFX1250-GISEL-NEXT: s_wait_loadcnt_dscnt 0x0
; GFX1250-GISEL-NEXT: v_lshlrev_b32_e32 v2, 2, v2
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_3)
; GFX1250-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1250-GISEL-NEXT: flat_load_b32 v0, v[0:1]
@@ -1629,7 +1613,7 @@ define amdgpu_ps float @flat_load_f32_saddr_zext_vgpr_range_imm_offset(ptr inreg
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[0:1], s[2:3]
; GFX1250-GISEL-NEXT: s_wait_loadcnt_dscnt 0x0
; GFX1250-GISEL-NEXT: v_lshlrev_b32_e32 v2, 2, v2
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_3)
; GFX1250-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1250-GISEL-NEXT: flat_load_b32 v0, v[0:1] offset:400
@@ -1703,7 +1687,7 @@ define amdgpu_ps half @flat_load_saddr_i16(ptr inreg %sbase, i32 %voffset) {
; GFX1250-GISEL-NEXT: v_nop
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: flat_load_u16 v0, v[0:1]
@@ -1763,7 +1747,7 @@ define amdgpu_ps half @flat_load_saddr_i16_immneg128(ptr inreg %sbase, i32 %voff
; GFX1250-GISEL-NEXT: v_nop
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: flat_load_u16 v0, v[0:1] offset:-128
@@ -1824,7 +1808,7 @@ define amdgpu_ps half @flat_load_saddr_f16(ptr inreg %sbase, i32 %voffset) {
; GFX1250-GISEL-NEXT: v_nop
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: flat_load_u16 v0, v[0:1]
@@ -1883,7 +1867,7 @@ define amdgpu_ps half @flat_load_saddr_f16_immneg128(ptr inreg %sbase, i32 %voff
; GFX1250-GISEL-NEXT: v_nop
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: flat_load_u16 v0, v[0:1] offset:-128
@@ -1943,7 +1927,7 @@ define amdgpu_ps float @flat_load_saddr_i32(ptr inreg %sbase, i32 %voffset) {
; GFX1250-GISEL-NEXT: v_nop
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: flat_load_b32 v0, v[0:1]
@@ -1990,7 +1974,7 @@ define amdgpu_ps float @flat_load_saddr_i32_immneg128(ptr inreg %sbase, i32 %vof
; GFX1250-GISEL-NEXT: v_nop
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: flat_load_b32 v0, v[0:1] offset:-128
@@ -2038,7 +2022,7 @@ define amdgpu_ps float @flat_load_saddr_f32(ptr inreg %sbase, i32 %voffset) {
; GFX1250-GISEL-NEXT: v_nop
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: flat_load_b32 v0, v[0:1]
@@ -2084,7 +2068,7 @@ define amdgpu_ps float @flat_load_saddr_f32_immneg128(ptr inreg %sbase, i32 %vof
; GFX1250-GISEL-NEXT: v_nop
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: flat_load_b32 v0, v[0:1] offset:-128
@@ -2131,7 +2115,7 @@ define amdgpu_ps <2 x half> @flat_load_saddr_v2i16(ptr inreg %sbase, i32 %voffse
; GFX1250-GISEL-NEXT: v_nop
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: flat_load_b32 v0, v[0:1]
@@ -2178,7 +2162,7 @@ define amdgpu_ps <2 x half> @flat_load_saddr_v2i16_immneg128(ptr inreg %sbase, i
; GFX1250-GISEL-NEXT: v_nop
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: flat_load_b32 v0, v[0:1] offset:-128
@@ -2226,7 +2210,7 @@ define amdgpu_ps <2 x half> @flat_load_saddr_v2f16(ptr inreg %sbase, i32 %voffse
; GFX1250-GISEL-NEXT: v_nop
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: flat_load_b32 v0, v[0:1]
@@ -2272,7 +2256,7 @@ define amdgpu_ps <2 x half> @flat_load_saddr_v2f16_immneg128(ptr inreg %sbase, i
; GFX1250-GISEL-NEXT: v_nop
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: flat_load_b32 v0, v[0:1] offset:-128
@@ -2319,7 +2303,7 @@ define amdgpu_ps <2 x half> @flat_load_saddr_p3(ptr inreg %sbase, i32 %voffset)
; GFX1250-GISEL-NEXT: v_nop
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: flat_load_b32 v0, v[0:1]
@@ -2367,7 +2351,7 @@ define amdgpu_ps <2 x half> @flat_load_saddr_p3_immneg128(ptr inreg %sbase, i32
; GFX1250-GISEL-NEXT: v_nop
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: flat_load_b32 v0, v[0:1] offset:-128
@@ -2416,7 +2400,7 @@ define amdgpu_ps <2 x float> @flat_load_saddr_f64(ptr inreg %sbase, i32 %voffset
; GFX1250-GISEL-NEXT: v_nop
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: flat_load_b64 v[0:1], v[0:1]
@@ -2463,7 +2447,7 @@ define amdgpu_ps <2 x float> @flat_load_saddr_f64_immneg128(ptr inreg %sbase, i3
; GFX1250-GISEL-NEXT: v_nop
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: flat_load_b64 v[0:1], v[0:1] offset:-128
@@ -2511,7 +2495,7 @@ define amdgpu_ps <2 x float> @flat_load_saddr_i64(ptr inreg %sbase, i32 %voffset
; GFX1250-GISEL-NEXT: v_nop
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: flat_load_b64 v[0:1], v[0:1]
@@ -2558,7 +2542,7 @@ define amdgpu_ps <2 x float> @flat_load_saddr_i64_immneg128(ptr inreg %sbase, i3
; GFX1250-GISEL-NEXT: v_nop
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: flat_load_b64 v[0:1], v[0:1] offset:-128
@@ -2606,7 +2590,7 @@ define amdgpu_ps <2 x float> @flat_load_saddr_v2f32(ptr inreg %sbase, i32 %voffs
; GFX1250-GISEL-NEXT: v_nop
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: flat_load_b64 v[0:1], v[0:1]
@@ -2652,7 +2636,7 @@ define amdgpu_ps <2 x float> @flat_load_saddr_v2f32_immneg128(ptr inreg %sbase,
; GFX1250-GISEL-NEXT: v_nop
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: flat_load_b64 v[0:1], v[0:1] offset:-128
@@ -2699,7 +2683,7 @@ define amdgpu_ps <2 x float> @flat_load_saddr_v2i32(ptr inreg %sbase, i32 %voffs
; GFX1250-GISEL-NEXT: v_nop
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: flat_load_b64 v[0:1], v[0:1]
@@ -2746,7 +2730,7 @@ define amdgpu_ps <2 x float> @flat_load_saddr_v2i32_immneg128(ptr inreg %sbase,
; GFX1250-GISEL-NEXT: v_nop
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: flat_load_b64 v[0:1], v[0:1] offset:-128
@@ -2794,7 +2778,7 @@ define amdgpu_ps <2 x float> @flat_load_saddr_v4i16(ptr inreg %sbase, i32 %voffs
; GFX1250-GISEL-NEXT: v_nop
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: flat_load_b64 v[0:1], v[0:1]
@@ -2841,7 +2825,7 @@ define amdgpu_ps <2 x float> @flat_load_saddr_v4i16_immneg128(ptr inreg %sbase,
; GFX1250-GISEL-NEXT: v_nop
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: flat_load_b64 v[0:1], v[0:1] offset:-128
@@ -2889,7 +2873,7 @@ define amdgpu_ps <2 x float> @flat_load_saddr_v4f16(ptr inreg %sbase, i32 %voffs
; GFX1250-GISEL-NEXT: v_nop
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: flat_load_b64 v[0:1], v[0:1]
@@ -2936,7 +2920,7 @@ define amdgpu_ps <2 x float> @flat_load_saddr_v4f16_immneg128(ptr inreg %sbase,
; GFX1250-GISEL-NEXT: v_nop
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: flat_load_b64 v[0:1], v[0:1] offset:-128
@@ -2984,7 +2968,7 @@ define amdgpu_ps <2 x float> @flat_load_saddr_p1(ptr inreg %sbase, i32 %voffset)
; GFX1250-GISEL-NEXT: v_nop
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: flat_load_b64 v[0:1], v[0:1]
@@ -3032,7 +3016,7 @@ define amdgpu_ps <2 x float> @flat_load_saddr_p1_immneg128(ptr inreg %sbase, i32
; GFX1250-GISEL-NEXT: v_nop
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: flat_load_b64 v[0:1], v[0:1] offset:-128
@@ -3081,7 +3065,7 @@ define amdgpu_ps <3 x float> @flat_load_saddr_v3f32(ptr inreg %sbase, i32 %voffs
; GFX1250-GISEL-NEXT: v_nop
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: flat_load_b96 v[0:2], v[0:1]
@@ -3127,7 +3111,7 @@ define amdgpu_ps <3 x float> @flat_load_saddr_v3f32_immneg128(ptr inreg %sbase,
; GFX1250-GISEL-NEXT: v_nop
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: flat_load_b96 v[0:2], v[0:1] offset:-128
@@ -3174,7 +3158,7 @@ define amdgpu_ps <3 x float> @flat_load_saddr_v3i32(ptr inreg %sbase, i32 %voffs
; GFX1250-GISEL-NEXT: v_nop
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: flat_load_b96 v[0:2], v[0:1]
@@ -3221,7 +3205,7 @@ define amdgpu_ps <3 x float> @flat_load_saddr_v3i32_immneg128(ptr inreg %sbase,
; GFX1250-GISEL-NEXT: v_nop
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: flat_load_b96 v[0:2], v[0:1] offset:-128
@@ -3269,7 +3253,7 @@ define amdgpu_ps <6 x half> @flat_load_saddr_v6f16(ptr inreg %sbase, i32 %voffse
; GFX1250-GISEL-NEXT: v_nop
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: flat_load_b96 v[0:2], v[0:1]
@@ -3315,7 +3299,7 @@ define amdgpu_ps <6 x half> @flat_load_saddr_v6f16_immneg128(ptr inreg %sbase, i
; GFX1250-GISEL-NEXT: v_nop
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: flat_load_b96 v[0:2], v[0:1] offset:-128
@@ -3362,7 +3346,7 @@ define amdgpu_ps <4 x float> @flat_load_saddr_v4f32(ptr inreg %sbase, i32 %voffs
; GFX1250-GISEL-NEXT: v_nop
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: flat_load_b128 v[0:3], v[0:1]
@@ -3408,7 +3392,7 @@ define amdgpu_ps <4 x float> @flat_load_saddr_v4f32_immneg128(ptr inreg %sbase,
; GFX1250-GISEL-NEXT: v_nop
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: flat_load_b128 v[0:3], v[0:1] offset:-128
@@ -3455,7 +3439,7 @@ define amdgpu_ps <4 x float> @flat_load_saddr_v4i32(ptr inreg %sbase, i32 %voffs
; GFX1250-GISEL-NEXT: v_nop
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: flat_load_b128 v[0:3], v[0:1]
@@ -3502,7 +3486,7 @@ define amdgpu_ps <4 x float> @flat_load_saddr_v4i32_immneg128(ptr inreg %sbase,
; GFX1250-GISEL-NEXT: v_nop
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: flat_load_b128 v[0:3], v[0:1] offset:-128
@@ -3550,7 +3534,7 @@ define amdgpu_ps <4 x float> @flat_load_saddr_v2i64(ptr inreg %sbase, i32 %voffs
; GFX1250-GISEL-NEXT: v_nop
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: flat_load_b128 v[0:3], v[0:1]
@@ -3597,7 +3581,7 @@ define amdgpu_ps <4 x float> @flat_load_saddr_v2i64_immneg128(ptr inreg %sbase,
; GFX1250-GISEL-NEXT: v_nop
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: flat_load_b128 v[0:3], v[0:1] offset:-128
@@ -3645,7 +3629,7 @@ define amdgpu_ps <4 x float> @flat_load_saddr_i128(ptr inreg %sbase, i32 %voffse
; GFX1250-GISEL-NEXT: v_nop
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: flat_load_b128 v[0:3], v[0:1]
@@ -3692,7 +3676,7 @@ define amdgpu_ps <4 x float> @flat_load_saddr_i128_immneg128(ptr inreg %sbase, i
; GFX1250-GISEL-NEXT: v_nop
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: flat_load_b128 v[0:3], v[0:1] offset:-128
@@ -3740,7 +3724,7 @@ define amdgpu_ps <4 x float> @flat_load_saddr_v2p1(ptr inreg %sbase, i32 %voffse
; GFX1250-GISEL-NEXT: v_nop
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: flat_load_b128 v[0:3], v[0:1]
@@ -3788,7 +3772,7 @@ define amdgpu_ps <4 x float> @flat_load_saddr_v2p1_immneg128(ptr inreg %sbase, i
; GFX1250-GISEL-NEXT: v_nop
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: flat_load_b128 v[0:3], v[0:1] offset:-128
@@ -3837,7 +3821,7 @@ define amdgpu_ps <4 x float> @flat_load_saddr_v4p3(ptr inreg %sbase, i32 %voffse
; GFX1250-GISEL-NEXT: v_nop
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: flat_load_b128 v[0:3], v[0:1]
@@ -3885,7 +3869,7 @@ define amdgpu_ps <4 x float> @flat_load_saddr_v4p3_immneg128(ptr inreg %sbase, i
; GFX1250-GISEL-NEXT: v_nop
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: flat_load_b128 v[0:3], v[0:1] offset:-128
@@ -3938,7 +3922,7 @@ define amdgpu_ps float @flat_sextload_saddr_i8(ptr inreg %sbase, i32 %voffset) {
; GFX1250-GISEL-NEXT: v_nop
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: flat_load_i8 v0, v[0:1]
@@ -3986,7 +3970,7 @@ define amdgpu_ps float @flat_sextload_saddr_i8_immneg128(ptr inreg %sbase, i32 %
; GFX1250-GISEL-NEXT: v_nop
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: flat_load_i8 v0, v[0:1] offset:-128
@@ -4035,7 +4019,7 @@ define amdgpu_ps float @flat_sextload_saddr_i16(ptr inreg %sbase, i32 %voffset)
; GFX1250-GISEL-NEXT: v_nop
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: flat_load_i16 v0, v[0:1]
@@ -4083,7 +4067,7 @@ define amdgpu_ps float @flat_sextload_saddr_i16_immneg128(ptr inreg %sbase, i32
; GFX1250-GISEL-NEXT: v_nop
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: flat_load_i16 v0, v[0:1] offset:-128
@@ -4132,7 +4116,7 @@ define amdgpu_ps float @flat_zextload_saddr_i8(ptr inreg %sbase, i32 %voffset) {
; GFX1250-GISEL-NEXT: v_nop
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: flat_load_u8 v0, v[0:1]
@@ -4180,7 +4164,7 @@ define amdgpu_ps float @flat_zextload_saddr_i8_immneg128(ptr inreg %sbase, i32 %
; GFX1250-GISEL-NEXT: v_nop
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: flat_load_u8 v0, v[0:1] offset:-128
@@ -4229,7 +4213,7 @@ define amdgpu_ps float @flat_zextload_saddr_i16(ptr inreg %sbase, i32 %voffset)
; GFX1250-GISEL-NEXT: v_nop
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: flat_load_u16 v0, v[0:1]
@@ -4277,7 +4261,7 @@ define amdgpu_ps float @flat_zextload_saddr_i16_immneg128(ptr inreg %sbase, i32
; GFX1250-GISEL-NEXT: v_nop
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: flat_load_u16 v0, v[0:1] offset:-128
@@ -4332,7 +4316,7 @@ define amdgpu_ps float @atomic_flat_load_saddr_i32(ptr inreg %sbase, i32 %voffse
; GFX1250-GISEL-NEXT: v_nop
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: flat_load_b32 v0, v[0:1] scope:SCOPE_SYS
@@ -4385,7 +4369,7 @@ define amdgpu_ps float @atomic_flat_load_saddr_i32_immneg128(ptr inreg %sbase, i
; GFX1250-GISEL-NEXT: v_nop
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: flat_load_b32 v0, v[0:1] offset:-128 scope:SCOPE_SYS
@@ -4439,7 +4423,7 @@ define amdgpu_ps <2 x float> @atomic_flat_load_saddr_i64(ptr inreg %sbase, i32 %
; GFX1250-GISEL-NEXT: v_nop
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: flat_load_b64 v[0:1], v[0:1] scope:SCOPE_SYS
@@ -4492,7 +4476,7 @@ define amdgpu_ps <2 x float> @atomic_flat_load_saddr_i64_immneg128(ptr inreg %sb
; GFX1250-GISEL-NEXT: v_nop
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: flat_load_b64 v[0:1], v[0:1] offset:-128 scope:SCOPE_SYS
@@ -4548,7 +4532,7 @@ define amdgpu_ps <2 x half> @flat_load_saddr_i16_d16lo_undef_hi(ptr inreg %sbase
; GFX1250-GISEL-FAKE16-NEXT: v_nop
; GFX1250-GISEL-FAKE16-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-FAKE16-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-FAKE16-NEXT: v_add_co_u32 v0, vcc_lo, v2, v0
; GFX1250-GISEL-FAKE16-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v3, vcc_lo
; GFX1250-GISEL-FAKE16-NEXT: flat_load_u16 v0, v[0:1]
@@ -4565,7 +4549,7 @@ define amdgpu_ps <2 x half> @flat_load_saddr_i16_d16lo_undef_hi(ptr inreg %sbase
; GFX1250-GISEL-TRUE16-NEXT: v_nop
; GFX1250-GISEL-TRUE16-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-TRUE16-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-TRUE16-NEXT: v_add_co_u32 v0, vcc_lo, v2, v0
; GFX1250-GISEL-TRUE16-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v3, vcc_lo
; GFX1250-GISEL-TRUE16-NEXT: flat_load_u16 v0, v[0:1]
@@ -4613,7 +4597,7 @@ define amdgpu_ps <2 x half> @flat_load_saddr_i16_d16lo_undef_hi_immneg128(ptr in
; GFX1250-GISEL-FAKE16-NEXT: v_nop
; GFX1250-GISEL-FAKE16-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-FAKE16-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-FAKE16-NEXT: v_add_co_u32 v0, vcc_lo, v2, v0
; GFX1250-GISEL-FAKE16-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v3, vcc_lo
; GFX1250-GISEL-FAKE16-NEXT: flat_load_u16 v0, v[0:1] offset:-128
@@ -4630,7 +4614,7 @@ define amdgpu_ps <2 x half> @flat_load_saddr_i16_d16lo_undef_hi_immneg128(ptr in
; GFX1250-GISEL-TRUE16-NEXT: v_nop
; GFX1250-GISEL-TRUE16-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-TRUE16-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-TRUE16-NEXT: v_add_co_u32 v0, vcc_lo, v2, v0
; GFX1250-GISEL-TRUE16-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v3, vcc_lo
; GFX1250-GISEL-TRUE16-NEXT: flat_load_u16 v0, v[0:1] offset:-128
@@ -4680,7 +4664,7 @@ define amdgpu_ps <2 x half> @flat_load_saddr_i16_d16lo_zero_hi(ptr inreg %sbase,
; GFX1250-GISEL-FAKE16-NEXT: v_nop
; GFX1250-GISEL-FAKE16-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-FAKE16-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-FAKE16-NEXT: v_add_co_u32 v0, vcc_lo, v2, v0
; GFX1250-GISEL-FAKE16-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v3, vcc_lo
; GFX1250-GISEL-FAKE16-NEXT: flat_load_u16 v0, v[0:1]
@@ -4709,7 +4693,7 @@ define amdgpu_ps <2 x half> @flat_load_saddr_i16_d16lo_zero_hi(ptr inreg %sbase,
; GFX1250-GISEL-TRUE16-NEXT: v_nop
; GFX1250-GISEL-TRUE16-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-TRUE16-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-TRUE16-NEXT: v_add_co_u32 v0, vcc_lo, v2, v0
; GFX1250-GISEL-TRUE16-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v3, vcc_lo
; GFX1250-GISEL-TRUE16-NEXT: flat_load_u16 v0, v[0:1]
@@ -4760,7 +4744,7 @@ define amdgpu_ps <2 x half> @flat_load_saddr_i16_d16lo_zero_hi_immneg128(ptr inr
; GFX1250-GISEL-FAKE16-NEXT: v_nop
; GFX1250-GISEL-FAKE16-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-FAKE16-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-FAKE16-NEXT: v_add_co_u32 v0, vcc_lo, v2, v0
; GFX1250-GISEL-FAKE16-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v3, vcc_lo
; GFX1250-GISEL-FAKE16-NEXT: flat_load_u16 v0, v[0:1] offset:-128
@@ -4789,7 +4773,7 @@ define amdgpu_ps <2 x half> @flat_load_saddr_i16_d16lo_zero_hi_immneg128(ptr inr
; GFX1250-GISEL-TRUE16-NEXT: v_nop
; GFX1250-GISEL-TRUE16-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-TRUE16-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-TRUE16-NEXT: v_add_co_u32 v0, vcc_lo, v2, v0
; GFX1250-GISEL-TRUE16-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v3, vcc_lo
; GFX1250-GISEL-TRUE16-NEXT: flat_load_u16 v0, v[0:1] offset:-128
@@ -4841,7 +4825,7 @@ define amdgpu_ps <2 x half> @flat_load_saddr_i16_d16lo_reg_hi(ptr inreg %sbase,
; GFX1250-GISEL-NEXT: v_nop
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v2, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: flat_load_u16 v0, v[2:3]
@@ -4908,7 +4892,7 @@ define amdgpu_ps <2 x half> @flat_load_saddr_i16_d16lo_reg_hi_immneg128(ptr inre
; GFX1250-GISEL-NEXT: v_nop
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v2, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: flat_load_u16 v0, v[2:3] offset:-128
@@ -4976,7 +4960,7 @@ define amdgpu_ps <2 x half> @flat_load_saddr_i16_d16lo_zexti8_reg_hi(ptr inreg %
; GFX1250-GISEL-NEXT: v_nop
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v2, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: flat_load_u8 v0, v[2:3]
@@ -5044,7 +5028,7 @@ define amdgpu_ps <2 x half> @flat_load_saddr_i16_d16lo_zexti8_reg_hi_immneg128(p
; GFX1250-GISEL-NEXT: v_nop
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v2, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: flat_load_u8 v0, v[2:3] offset:-128
@@ -5113,7 +5097,7 @@ define amdgpu_ps <2 x half> @flat_load_saddr_i16_d16lo_sexti8_reg_hi(ptr inreg %
; GFX1250-GISEL-FAKE16-NEXT: v_nop
; GFX1250-GISEL-FAKE16-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-FAKE16-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-FAKE16-NEXT: v_add_co_u32 v2, vcc_lo, v2, v0
; GFX1250-GISEL-FAKE16-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX1250-GISEL-FAKE16-NEXT: flat_load_i8 v0, v[2:3]
@@ -5146,7 +5130,7 @@ define amdgpu_ps <2 x half> @flat_load_saddr_i16_d16lo_sexti8_reg_hi(ptr inreg %
; GFX1250-GISEL-TRUE16-NEXT: v_nop
; GFX1250-GISEL-TRUE16-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-TRUE16-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-TRUE16-NEXT: v_add_co_u32 v2, vcc_lo, v2, v0
; GFX1250-GISEL-TRUE16-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX1250-GISEL-TRUE16-NEXT: flat_load_i8 v0, v[2:3]
@@ -5200,7 +5184,7 @@ define amdgpu_ps <2 x half> @flat_load_saddr_i16_d16lo_sexti8_reg_hi_immneg128(p
; GFX1250-GISEL-FAKE16-NEXT: v_nop
; GFX1250-GISEL-FAKE16-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-FAKE16-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-FAKE16-NEXT: v_add_co_u32 v2, vcc_lo, v2, v0
; GFX1250-GISEL-FAKE16-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX1250-GISEL-FAKE16-NEXT: flat_load_i8 v0, v[2:3] offset:-128
@@ -5233,7 +5217,7 @@ define amdgpu_ps <2 x half> @flat_load_saddr_i16_d16lo_sexti8_reg_hi_immneg128(p
; GFX1250-GISEL-TRUE16-NEXT: v_nop
; GFX1250-GISEL-TRUE16-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-TRUE16-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-TRUE16-NEXT: v_add_co_u32 v2, vcc_lo, v2, v0
; GFX1250-GISEL-TRUE16-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX1250-GISEL-TRUE16-NEXT: flat_load_i8 v0, v[2:3] offset:-128
@@ -5292,7 +5276,7 @@ define amdgpu_ps <2 x half> @flat_load_saddr_i16_d16hi_undef_hi(ptr inreg %sbase
; GFX1250-GISEL-FAKE16-NEXT: v_nop
; GFX1250-GISEL-FAKE16-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-FAKE16-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-FAKE16-NEXT: v_add_co_u32 v0, vcc_lo, v2, v0
; GFX1250-GISEL-FAKE16-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v3, vcc_lo
; GFX1250-GISEL-FAKE16-NEXT: flat_load_u16 v0, v[0:1]
@@ -5324,7 +5308,7 @@ define amdgpu_ps <2 x half> @flat_load_saddr_i16_d16hi_undef_hi(ptr inreg %sbase
; GFX1250-GISEL-TRUE16-NEXT: v_nop
; GFX1250-GISEL-TRUE16-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-TRUE16-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-TRUE16-NEXT: v_add_co_u32 v0, vcc_lo, v2, v0
; GFX1250-GISEL-TRUE16-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v3, vcc_lo
; GFX1250-GISEL-TRUE16-NEXT: flat_load_u16 v0, v[0:1]
@@ -5374,7 +5358,7 @@ define amdgpu_ps <2 x half> @flat_load_saddr_i16_d16hi_undef_hi_immneg128(ptr in
; GFX1250-GISEL-FAKE16-NEXT: v_nop
; GFX1250-GISEL-FAKE16-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-FAKE16-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-FAKE16-NEXT: v_add_co_u32 v0, vcc_lo, v2, v0
; GFX1250-GISEL-FAKE16-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v3, vcc_lo
; GFX1250-GISEL-FAKE16-NEXT: flat_load_u16 v0, v[0:1] offset:-128
@@ -5406,7 +5390,7 @@ define amdgpu_ps <2 x half> @flat_load_saddr_i16_d16hi_undef_hi_immneg128(ptr in
; GFX1250-GISEL-TRUE16-NEXT: v_nop
; GFX1250-GISEL-TRUE16-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-TRUE16-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-TRUE16-NEXT: v_add_co_u32 v0, vcc_lo, v2, v0
; GFX1250-GISEL-TRUE16-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v3, vcc_lo
; GFX1250-GISEL-TRUE16-NEXT: flat_load_u16 v0, v[0:1] offset:-128
@@ -5457,7 +5441,7 @@ define amdgpu_ps <2 x half> @flat_load_saddr_i16_d16hi_zero_hi(ptr inreg %sbase,
; GFX1250-GISEL-NEXT: v_nop
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: flat_load_u16 v0, v[0:1]
@@ -5538,7 +5522,7 @@ define amdgpu_ps <2 x half> @flat_load_saddr_i16_d16hi_zero_hi_immneg128(ptr inr
; GFX1250-GISEL-NEXT: v_nop
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: flat_load_u16 v0, v[0:1] offset:-128
@@ -5620,7 +5604,7 @@ define amdgpu_ps <2 x half> @flat_load_saddr_i16_d16hi_reg_hi(ptr inreg %sbase,
; GFX1250-GISEL-NEXT: v_nop
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v2, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: flat_load_u16 v0, v[2:3]
@@ -5688,7 +5672,7 @@ define amdgpu_ps <2 x half> @flat_load_saddr_i16_d16hi_reg_hi_immneg128(ptr inre
; GFX1250-GISEL-NEXT: v_nop
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v2, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: flat_load_u16 v0, v[2:3] offset:-128
@@ -5757,7 +5741,7 @@ define amdgpu_ps <2 x half> @flat_load_saddr_i16_d16hi_zexti8_reg_hi(ptr inreg %
; GFX1250-GISEL-NEXT: v_nop
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v2, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: flat_load_u8 v0, v[2:3]
@@ -5826,7 +5810,7 @@ define amdgpu_ps <2 x half> @flat_load_saddr_i16_d16hi_zexti8_reg_hi_immneg128(p
; GFX1250-GISEL-NEXT: v_nop
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v2, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: flat_load_u8 v0, v[2:3] offset:-128
@@ -5896,7 +5880,7 @@ define amdgpu_ps <2 x half> @flat_load_saddr_i16_d16hi_sexti8_reg_hi(ptr inreg %
; GFX1250-GISEL-FAKE16-NEXT: v_nop
; GFX1250-GISEL-FAKE16-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-FAKE16-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-FAKE16-NEXT: v_add_co_u32 v2, vcc_lo, v2, v0
; GFX1250-GISEL-FAKE16-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX1250-GISEL-FAKE16-NEXT: flat_load_i8 v0, v[2:3]
@@ -5929,7 +5913,7 @@ define amdgpu_ps <2 x half> @flat_load_saddr_i16_d16hi_sexti8_reg_hi(ptr inreg %
; GFX1250-GISEL-TRUE16-NEXT: v_nop
; GFX1250-GISEL-TRUE16-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-TRUE16-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-TRUE16-NEXT: v_add_co_u32 v2, vcc_lo, v2, v0
; GFX1250-GISEL-TRUE16-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX1250-GISEL-TRUE16-NEXT: flat_load_i8 v0, v[2:3]
@@ -5984,7 +5968,7 @@ define amdgpu_ps <2 x half> @flat_load_saddr_i16_d16hi_sexti8_reg_hi_immneg128(p
; GFX1250-GISEL-FAKE16-NEXT: v_nop
; GFX1250-GISEL-FAKE16-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-FAKE16-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-FAKE16-NEXT: v_add_co_u32 v2, vcc_lo, v2, v0
; GFX1250-GISEL-FAKE16-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX1250-GISEL-FAKE16-NEXT: flat_load_i8 v0, v[2:3] offset:-128
@@ -6017,7 +6001,7 @@ define amdgpu_ps <2 x half> @flat_load_saddr_i16_d16hi_sexti8_reg_hi_immneg128(p
; GFX1250-GISEL-TRUE16-NEXT: v_nop
; GFX1250-GISEL-TRUE16-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-TRUE16-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-TRUE16-NEXT: v_add_co_u32 v2, vcc_lo, v2, v0
; GFX1250-GISEL-TRUE16-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX1250-GISEL-TRUE16-NEXT: flat_load_i8 v0, v[2:3] offset:-128
@@ -6118,6 +6102,7 @@ define amdgpu_ps void @flat_addr_64bit_lsr_iv(ptr inreg %arg) {
; GFX1250-SDAG-NEXT: s_add_co_i32 s0, s0, -1
; GFX1250-SDAG-NEXT: s_add_nc_u64 s[2:3], s[2:3], 4
; GFX1250-SDAG-NEXT: s_cmp_eq_u32 s0, 0
+; GFX1250-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1250-SDAG-NEXT: s_cbranch_scc0 .LBB116_1
; GFX1250-SDAG-NEXT: ; %bb.2: ; %bb2
; GFX1250-SDAG-NEXT: s_endpgm
@@ -6139,6 +6124,7 @@ define amdgpu_ps void @flat_addr_64bit_lsr_iv(ptr inreg %arg) {
; GFX1250-GISEL-NEXT: s_add_co_u32 s2, s2, 4
; GFX1250-GISEL-NEXT: s_add_co_ci_u32 s3, s3, 0
; GFX1250-GISEL-NEXT: s_cmp_eq_u32 s0, 0
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1250-GISEL-NEXT: s_cbranch_scc0 .LBB116_1
; GFX1250-GISEL-NEXT: ; %bb.2: ; %bb2
; GFX1250-GISEL-NEXT: s_endpgm
@@ -6159,6 +6145,7 @@ define amdgpu_ps void @flat_addr_64bit_lsr_iv(ptr inreg %arg) {
; GFX1250-NOECC-NEXT: s_add_co_i32 s0, s0, -1
; GFX1250-NOECC-NEXT: s_add_nc_u64 s[2:3], s[2:3], 4
; GFX1250-NOECC-NEXT: s_cmp_eq_u32 s0, 0
+; GFX1250-NOECC-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1250-NOECC-NEXT: s_cbranch_scc0 .LBB116_1
; GFX1250-NOECC-NEXT: ; %bb.2: ; %bb2
; GFX1250-NOECC-NEXT: s_endpgm
@@ -6199,6 +6186,7 @@ define amdgpu_ps void @flat_addr_64bit_lsr_iv_multiload(ptr inreg %arg, ptr inre
; GFX1250-SDAG-NEXT: s_add_co_i32 s0, s0, -1
; GFX1250-SDAG-NEXT: s_add_nc_u64 s[2:3], s[2:3], 4
; GFX1250-SDAG-NEXT: s_cmp_eq_u32 s0, 0
+; GFX1250-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1250-SDAG-NEXT: s_cbranch_scc0 .LBB117_1
; GFX1250-SDAG-NEXT: ; %bb.2: ; %bb2
; GFX1250-SDAG-NEXT: s_endpgm
@@ -6222,6 +6210,7 @@ define amdgpu_ps void @flat_addr_64bit_lsr_iv_multiload(ptr inreg %arg, ptr inre
; GFX1250-GISEL-NEXT: s_add_co_u32 s2, s2, 4
; GFX1250-GISEL-NEXT: s_add_co_ci_u32 s3, s3, 0
; GFX1250-GISEL-NEXT: s_cmp_eq_u32 s0, 0
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1250-GISEL-NEXT: s_cbranch_scc0 .LBB117_1
; GFX1250-GISEL-NEXT: ; %bb.2: ; %bb2
; GFX1250-GISEL-NEXT: s_endpgm
@@ -6244,6 +6233,7 @@ define amdgpu_ps void @flat_addr_64bit_lsr_iv_multiload(ptr inreg %arg, ptr inre
; GFX1250-NOECC-NEXT: s_add_co_i32 s0, s0, -1
; GFX1250-NOECC-NEXT: s_add_nc_u64 s[2:3], s[2:3], 4
; GFX1250-NOECC-NEXT: s_cmp_eq_u32 s0, 0
+; GFX1250-NOECC-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1250-NOECC-NEXT: s_cbranch_scc0 .LBB117_1
; GFX1250-NOECC-NEXT: ; %bb.2: ; %bb2
; GFX1250-NOECC-NEXT: s_endpgm
diff --git a/llvm/test/CodeGen/AMDGPU/flat-saddr-store.ll b/llvm/test/CodeGen/AMDGPU/flat-saddr-store.ll
index 477c58bef75427..1b23f3d6a7afe5 100644
--- a/llvm/test/CodeGen/AMDGPU/flat-saddr-store.ll
+++ b/llvm/test/CodeGen/AMDGPU/flat-saddr-store.ll
@@ -32,7 +32,7 @@ define amdgpu_ps void @flat_store_saddr_i8_zext_vgpr(ptr inreg %sbase, ptr %voff
; GFX1250-GISEL-NEXT: s_wait_xcnt 0x0
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[0:1], s[2:3]
; GFX1250-GISEL-NEXT: s_wait_loadcnt_dscnt 0x0
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, v0, v3
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1250-GISEL-NEXT: flat_store_b8 v[0:1], v2
@@ -71,7 +71,7 @@ define amdgpu_ps void @flat_store_saddr_i8_zext_vgpr_offset_2047(ptr inreg %sbas
; GFX1250-GISEL-NEXT: s_wait_xcnt 0x0
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[0:1], s[2:3]
; GFX1250-GISEL-NEXT: s_wait_loadcnt_dscnt 0x0
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, v0, v3
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1250-GISEL-NEXT: flat_store_b8 v[0:1], v2 offset:2047
@@ -111,7 +111,7 @@ define amdgpu_ps void @flat_store_saddr_i8_zext_vgpr_offset_neg2048(ptr inreg %s
; GFX1250-GISEL-NEXT: s_wait_xcnt 0x0
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[0:1], s[2:3]
; GFX1250-GISEL-NEXT: s_wait_loadcnt_dscnt 0x0
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, v0, v3
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1250-GISEL-NEXT: flat_store_b8 v[0:1], v2 offset:-2048
@@ -155,7 +155,6 @@ define amdgpu_ps void @flat_store_saddr_uniform_ptr_in_vgprs(i32 %voffset, i8 %d
; GFX1250-GISEL-NEXT: ds_load_b64 v[2:3], v2
; GFX1250-GISEL-NEXT: s_wait_dscnt 0x0
; GFX1250-GISEL-NEXT: v_add_co_u32 v2, vcc_lo, v2, v0
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: flat_store_b8 v[2:3], v1
; GFX1250-GISEL-NEXT: s_endpgm
@@ -191,7 +190,6 @@ define amdgpu_ps void @flat_store_saddr_uniform_ptr_in_vgprs_immoffset(i32 %voff
; GFX1250-GISEL-NEXT: ds_load_b64 v[2:3], v2
; GFX1250-GISEL-NEXT: s_wait_dscnt 0x0
; GFX1250-GISEL-NEXT: v_add_co_u32 v2, vcc_lo, v2, v0
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: flat_store_b8 v[2:3], v1 offset:-120
; GFX1250-GISEL-NEXT: s_endpgm
@@ -227,7 +225,7 @@ define amdgpu_ps void @flat_store_saddr_i16_zext_vgpr(ptr inreg %sbase, i32 %vof
; GFX1250-GISEL-NEXT: v_nop
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v2, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: flat_store_b16 v[2:3], v1
@@ -258,7 +256,7 @@ define amdgpu_ps void @flat_store_saddr_i16_zext_vgpr_offset_neg128(ptr inreg %s
; GFX1250-GISEL-NEXT: v_nop
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v2, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: flat_store_b16 v[2:3], v1 offset:-128
@@ -290,7 +288,7 @@ define amdgpu_ps void @flat_store_saddr_f16_zext_vgpr(ptr inreg %sbase, i32 %vof
; GFX1250-GISEL-NEXT: v_nop
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v2, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: flat_store_b16 v[2:3], v1
@@ -321,7 +319,7 @@ define amdgpu_ps void @flat_store_saddr_f16_zext_vgpr_offset_neg128(ptr inreg %s
; GFX1250-GISEL-NEXT: v_nop
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v2, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: flat_store_b16 v[2:3], v1 offset:-128
@@ -353,7 +351,7 @@ define amdgpu_ps void @flat_store_saddr_i32_zext_vgpr(ptr inreg %sbase, i32 %vof
; GFX1250-GISEL-NEXT: v_nop
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v2, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: flat_store_b32 v[2:3], v1
@@ -384,7 +382,7 @@ define amdgpu_ps void @flat_store_saddr_i32_zext_vgpr_offset_neg128(ptr inreg %s
; GFX1250-GISEL-NEXT: v_nop
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v2, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: flat_store_b32 v[2:3], v1 offset:-128
@@ -416,7 +414,7 @@ define amdgpu_ps void @flat_store_saddr_f32_zext_vgpr(ptr inreg %sbase, i32 %vof
; GFX1250-GISEL-NEXT: v_nop
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v2, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: flat_store_b32 v[2:3], v1
@@ -447,7 +445,7 @@ define amdgpu_ps void @flat_store_saddr_f32_zext_vgpr_offset_neg128(ptr inreg %s
; GFX1250-GISEL-NEXT: v_nop
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v2, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: flat_store_b32 v[2:3], v1 offset:-128
@@ -479,7 +477,7 @@ define amdgpu_ps void @flat_store_saddr_p3_zext_vgpr(ptr inreg %sbase, i32 %voff
; GFX1250-GISEL-NEXT: v_nop
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v2, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: flat_store_b32 v[2:3], v1
@@ -510,7 +508,7 @@ define amdgpu_ps void @flat_store_saddr_p3_zext_vgpr_offset_neg128(ptr inreg %sb
; GFX1250-GISEL-NEXT: v_nop
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v2, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: flat_store_b32 v[2:3], v1 offset:-128
@@ -544,7 +542,7 @@ define amdgpu_ps void @flat_store_saddr_i64_zext_vgpr(ptr inreg %sbase, i32 %vof
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_dual_mov_b32 v4, v1 :: v_dual_mov_b32 v5, v2
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: flat_store_b64 v[0:1], v[4:5]
@@ -577,7 +575,7 @@ define amdgpu_ps void @flat_store_saddr_i64_zext_vgpr_offset_neg128(ptr inreg %s
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_dual_mov_b32 v4, v1 :: v_dual_mov_b32 v5, v2
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: flat_store_b64 v[0:1], v[4:5] offset:-128
@@ -611,7 +609,7 @@ define amdgpu_ps void @flat_store_saddr_f64_zext_vgpr(ptr inreg %sbase, i32 %vof
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_dual_mov_b32 v4, v1 :: v_dual_mov_b32 v5, v2
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: flat_store_b64 v[0:1], v[4:5]
@@ -644,7 +642,7 @@ define amdgpu_ps void @flat_store_saddr_f64_zext_vgpr_offset_neg128(ptr inreg %s
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_dual_mov_b32 v4, v1 :: v_dual_mov_b32 v5, v2
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: flat_store_b64 v[0:1], v[4:5] offset:-128
@@ -678,7 +676,7 @@ define amdgpu_ps void @flat_store_saddr_v2i32_zext_vgpr(ptr inreg %sbase, i32 %v
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_dual_mov_b32 v4, v1 :: v_dual_mov_b32 v5, v2
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: flat_store_b64 v[0:1], v[4:5]
@@ -711,7 +709,7 @@ define amdgpu_ps void @flat_store_saddr_v2i32_zext_vgpr_offset_neg128(ptr inreg
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_dual_mov_b32 v4, v1 :: v_dual_mov_b32 v5, v2
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: flat_store_b64 v[0:1], v[4:5] offset:-128
@@ -745,7 +743,7 @@ define amdgpu_ps void @flat_store_saddr_v2f32_zext_vgpr(ptr inreg %sbase, i32 %v
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_dual_mov_b32 v4, v1 :: v_dual_mov_b32 v5, v2
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: flat_store_b64 v[0:1], v[4:5]
@@ -778,7 +776,7 @@ define amdgpu_ps void @flat_store_saddr_v2f32_zext_vgpr_offset_neg128(ptr inreg
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_dual_mov_b32 v4, v1 :: v_dual_mov_b32 v5, v2
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: flat_store_b64 v[0:1], v[4:5] offset:-128
@@ -812,7 +810,7 @@ define amdgpu_ps void @flat_store_saddr_v4i16_zext_vgpr(ptr inreg %sbase, i32 %v
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_dual_mov_b32 v4, v1 :: v_dual_mov_b32 v5, v2
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: flat_store_b64 v[0:1], v[4:5]
@@ -845,7 +843,7 @@ define amdgpu_ps void @flat_store_saddr_v4i16_zext_vgpr_offset_neg128(ptr inreg
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_dual_mov_b32 v4, v1 :: v_dual_mov_b32 v5, v2
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: flat_store_b64 v[0:1], v[4:5] offset:-128
@@ -879,7 +877,7 @@ define amdgpu_ps void @flat_store_saddr_v4f16_zext_vgpr(ptr inreg %sbase, i32 %v
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_dual_mov_b32 v4, v1 :: v_dual_mov_b32 v5, v2
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: flat_store_b64 v[0:1], v[4:5]
@@ -912,7 +910,7 @@ define amdgpu_ps void @flat_store_saddr_v4f16_zext_vgpr_offset_neg128(ptr inreg
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_dual_mov_b32 v4, v1 :: v_dual_mov_b32 v5, v2
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: flat_store_b64 v[0:1], v[4:5] offset:-128
@@ -946,7 +944,7 @@ define amdgpu_ps void @flat_store_saddr_p1_zext_vgpr(ptr inreg %sbase, i32 %voff
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_dual_mov_b32 v4, v1 :: v_dual_mov_b32 v5, v2
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: flat_store_b64 v[0:1], v[4:5]
@@ -979,7 +977,7 @@ define amdgpu_ps void @flat_store_saddr_p1_zext_vgpr_offset_neg128(ptr inreg %sb
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_dual_mov_b32 v4, v1 :: v_dual_mov_b32 v5, v2
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: flat_store_b64 v[0:1], v[4:5] offset:-128
@@ -1014,7 +1012,7 @@ define amdgpu_ps void @flat_store_saddr_v3i32_zext_vgpr(ptr inreg %sbase, i32 %v
; GFX1250-GISEL-NEXT: v_dual_mov_b32 v4, v1 :: v_dual_mov_b32 v5, v2
; GFX1250-GISEL-NEXT: v_mov_b32_e32 v6, v3
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: flat_store_b96 v[0:1], v[4:6]
@@ -1048,7 +1046,7 @@ define amdgpu_ps void @flat_store_saddr_v3i32_zext_vgpr_offset_neg128(ptr inreg
; GFX1250-GISEL-NEXT: v_dual_mov_b32 v4, v1 :: v_dual_mov_b32 v5, v2
; GFX1250-GISEL-NEXT: v_mov_b32_e32 v6, v3
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: flat_store_b96 v[0:1], v[4:6] offset:-128
@@ -1083,7 +1081,7 @@ define amdgpu_ps void @flat_store_saddr_v3f32_zext_vgpr(ptr inreg %sbase, i32 %v
; GFX1250-GISEL-NEXT: v_dual_mov_b32 v4, v1 :: v_dual_mov_b32 v5, v2
; GFX1250-GISEL-NEXT: v_mov_b32_e32 v6, v3
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: flat_store_b96 v[0:1], v[4:6]
@@ -1117,7 +1115,7 @@ define amdgpu_ps void @flat_store_saddr_v3f32_zext_vgpr_offset_neg128(ptr inreg
; GFX1250-GISEL-NEXT: v_dual_mov_b32 v4, v1 :: v_dual_mov_b32 v5, v2
; GFX1250-GISEL-NEXT: v_mov_b32_e32 v6, v3
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: flat_store_b96 v[0:1], v[4:6] offset:-128
@@ -1152,7 +1150,7 @@ define amdgpu_ps void @flat_store_saddr_v6i16_zext_vgpr(ptr inreg %sbase, i32 %v
; GFX1250-GISEL-NEXT: v_dual_mov_b32 v4, v1 :: v_dual_mov_b32 v5, v2
; GFX1250-GISEL-NEXT: v_mov_b32_e32 v6, v3
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: flat_store_b96 v[0:1], v[4:6]
@@ -1186,7 +1184,7 @@ define amdgpu_ps void @flat_store_saddr_v6i16_zext_vgpr_offset_neg128(ptr inreg
; GFX1250-GISEL-NEXT: v_dual_mov_b32 v4, v1 :: v_dual_mov_b32 v5, v2
; GFX1250-GISEL-NEXT: v_mov_b32_e32 v6, v3
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: flat_store_b96 v[0:1], v[4:6] offset:-128
@@ -1221,7 +1219,7 @@ define amdgpu_ps void @flat_store_saddr_v6f16_zext_vgpr(ptr inreg %sbase, i32 %v
; GFX1250-GISEL-NEXT: v_dual_mov_b32 v4, v1 :: v_dual_mov_b32 v5, v2
; GFX1250-GISEL-NEXT: v_mov_b32_e32 v6, v3
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: flat_store_b96 v[0:1], v[4:6]
@@ -1255,7 +1253,7 @@ define amdgpu_ps void @flat_store_saddr_v6f16_zext_vgpr_offset_neg128(ptr inreg
; GFX1250-GISEL-NEXT: v_dual_mov_b32 v4, v1 :: v_dual_mov_b32 v5, v2
; GFX1250-GISEL-NEXT: v_mov_b32_e32 v6, v3
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: flat_store_b96 v[0:1], v[4:6] offset:-128
@@ -1291,7 +1289,7 @@ define amdgpu_ps void @flat_store_saddr_v4i32_zext_vgpr(ptr inreg %sbase, i32 %v
; GFX1250-GISEL-NEXT: v_dual_mov_b32 v6, v1 :: v_dual_mov_b32 v7, v2
; GFX1250-GISEL-NEXT: v_dual_mov_b32 v8, v3 :: v_dual_mov_b32 v9, v4
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: flat_store_b128 v[0:1], v[6:9]
@@ -1326,7 +1324,7 @@ define amdgpu_ps void @flat_store_saddr_v4i32_zext_vgpr_offset_neg128(ptr inreg
; GFX1250-GISEL-NEXT: v_dual_mov_b32 v6, v1 :: v_dual_mov_b32 v7, v2
; GFX1250-GISEL-NEXT: v_dual_mov_b32 v8, v3 :: v_dual_mov_b32 v9, v4
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: flat_store_b128 v[0:1], v[6:9] offset:-128
@@ -1362,7 +1360,7 @@ define amdgpu_ps void @flat_store_saddr_v4f32_zext_vgpr(ptr inreg %sbase, i32 %v
; GFX1250-GISEL-NEXT: v_dual_mov_b32 v6, v1 :: v_dual_mov_b32 v7, v2
; GFX1250-GISEL-NEXT: v_dual_mov_b32 v8, v3 :: v_dual_mov_b32 v9, v4
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: flat_store_b128 v[0:1], v[6:9]
@@ -1397,7 +1395,7 @@ define amdgpu_ps void @flat_store_saddr_v4f32_zext_vgpr_offset_neg128(ptr inreg
; GFX1250-GISEL-NEXT: v_dual_mov_b32 v6, v1 :: v_dual_mov_b32 v7, v2
; GFX1250-GISEL-NEXT: v_dual_mov_b32 v8, v3 :: v_dual_mov_b32 v9, v4
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: flat_store_b128 v[0:1], v[6:9] offset:-128
@@ -1433,7 +1431,7 @@ define amdgpu_ps void @flat_store_saddr_v2i64_zext_vgpr(ptr inreg %sbase, i32 %v
; GFX1250-GISEL-NEXT: v_dual_mov_b32 v6, v1 :: v_dual_mov_b32 v7, v2
; GFX1250-GISEL-NEXT: v_dual_mov_b32 v8, v3 :: v_dual_mov_b32 v9, v4
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: flat_store_b128 v[0:1], v[6:9]
@@ -1468,7 +1466,7 @@ define amdgpu_ps void @flat_store_saddr_v2i64_zext_vgpr_offset_neg128(ptr inreg
; GFX1250-GISEL-NEXT: v_dual_mov_b32 v6, v1 :: v_dual_mov_b32 v7, v2
; GFX1250-GISEL-NEXT: v_dual_mov_b32 v8, v3 :: v_dual_mov_b32 v9, v4
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: flat_store_b128 v[0:1], v[6:9] offset:-128
@@ -1504,7 +1502,7 @@ define amdgpu_ps void @flat_store_saddr_v2f64_zext_vgpr(ptr inreg %sbase, i32 %v
; GFX1250-GISEL-NEXT: v_dual_mov_b32 v6, v1 :: v_dual_mov_b32 v7, v2
; GFX1250-GISEL-NEXT: v_dual_mov_b32 v8, v3 :: v_dual_mov_b32 v9, v4
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: flat_store_b128 v[0:1], v[6:9]
@@ -1539,7 +1537,7 @@ define amdgpu_ps void @flat_store_saddr_v2f64_zext_vgpr_offset_neg128(ptr inreg
; GFX1250-GISEL-NEXT: v_dual_mov_b32 v6, v1 :: v_dual_mov_b32 v7, v2
; GFX1250-GISEL-NEXT: v_dual_mov_b32 v8, v3 :: v_dual_mov_b32 v9, v4
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: flat_store_b128 v[0:1], v[6:9] offset:-128
@@ -1575,7 +1573,7 @@ define amdgpu_ps void @flat_store_saddr_v8i16_zext_vgpr(ptr inreg %sbase, i32 %v
; GFX1250-GISEL-NEXT: v_dual_mov_b32 v6, v1 :: v_dual_mov_b32 v7, v2
; GFX1250-GISEL-NEXT: v_dual_mov_b32 v8, v3 :: v_dual_mov_b32 v9, v4
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: flat_store_b128 v[0:1], v[6:9]
@@ -1610,7 +1608,7 @@ define amdgpu_ps void @flat_store_saddr_v8i16_zext_vgpr_offset_neg128(ptr inreg
; GFX1250-GISEL-NEXT: v_dual_mov_b32 v6, v1 :: v_dual_mov_b32 v7, v2
; GFX1250-GISEL-NEXT: v_dual_mov_b32 v8, v3 :: v_dual_mov_b32 v9, v4
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: flat_store_b128 v[0:1], v[6:9] offset:-128
@@ -1646,7 +1644,7 @@ define amdgpu_ps void @flat_store_saddr_v8f16_zext_vgpr(ptr inreg %sbase, i32 %v
; GFX1250-GISEL-NEXT: v_dual_mov_b32 v6, v1 :: v_dual_mov_b32 v7, v2
; GFX1250-GISEL-NEXT: v_dual_mov_b32 v8, v3 :: v_dual_mov_b32 v9, v4
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: flat_store_b128 v[0:1], v[6:9]
@@ -1681,7 +1679,7 @@ define amdgpu_ps void @flat_store_saddr_v8f16_zext_vgpr_offset_neg128(ptr inreg
; GFX1250-GISEL-NEXT: v_dual_mov_b32 v6, v1 :: v_dual_mov_b32 v7, v2
; GFX1250-GISEL-NEXT: v_dual_mov_b32 v8, v3 :: v_dual_mov_b32 v9, v4
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: flat_store_b128 v[0:1], v[6:9] offset:-128
@@ -1717,7 +1715,7 @@ define amdgpu_ps void @flat_store_saddr_v2p1_zext_vgpr(ptr inreg %sbase, i32 %vo
; GFX1250-GISEL-NEXT: v_dual_mov_b32 v6, v1 :: v_dual_mov_b32 v7, v2
; GFX1250-GISEL-NEXT: v_dual_mov_b32 v8, v3 :: v_dual_mov_b32 v9, v4
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: flat_store_b128 v[0:1], v[6:9]
@@ -1752,7 +1750,7 @@ define amdgpu_ps void @flat_store_saddr_v2p1_zext_vgpr_offset_neg128(ptr inreg %
; GFX1250-GISEL-NEXT: v_dual_mov_b32 v6, v1 :: v_dual_mov_b32 v7, v2
; GFX1250-GISEL-NEXT: v_dual_mov_b32 v8, v3 :: v_dual_mov_b32 v9, v4
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: flat_store_b128 v[0:1], v[6:9] offset:-128
@@ -1788,7 +1786,7 @@ define amdgpu_ps void @flat_store_saddr_v4p3_zext_vgpr(ptr inreg %sbase, i32 %vo
; GFX1250-GISEL-NEXT: v_dual_mov_b32 v6, v1 :: v_dual_mov_b32 v7, v2
; GFX1250-GISEL-NEXT: v_dual_mov_b32 v8, v3 :: v_dual_mov_b32 v9, v4
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: flat_store_b128 v[0:1], v[6:9]
@@ -1823,7 +1821,7 @@ define amdgpu_ps void @flat_store_saddr_v4p3_zext_vgpr_offset_neg128(ptr inreg %
; GFX1250-GISEL-NEXT: v_dual_mov_b32 v6, v1 :: v_dual_mov_b32 v7, v2
; GFX1250-GISEL-NEXT: v_dual_mov_b32 v8, v3 :: v_dual_mov_b32 v9, v4
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: flat_store_b128 v[0:1], v[6:9] offset:-128
@@ -1861,7 +1859,7 @@ define amdgpu_ps void @atomic_flat_store_saddr_i32_zext_vgpr(ptr inreg %sbase, i
; GFX1250-GISEL-NEXT: v_nop
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v2, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: global_wb scope:SCOPE_SYS
@@ -1896,7 +1894,7 @@ define amdgpu_ps void @atomic_flat_store_saddr_i32_zext_vgpr_offset_neg128(ptr i
; GFX1250-GISEL-NEXT: v_nop
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v2, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: global_wb scope:SCOPE_SYS
@@ -1934,7 +1932,7 @@ define amdgpu_ps void @atomic_flat_store_saddr_i64_zext_vgpr(ptr inreg %sbase, i
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_dual_mov_b32 v4, v1 :: v_dual_mov_b32 v5, v2
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: global_wb scope:SCOPE_SYS
@@ -1971,7 +1969,7 @@ define amdgpu_ps void @atomic_flat_store_saddr_i64_zext_vgpr_offset_neg128(ptr i
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_dual_mov_b32 v4, v1 :: v_dual_mov_b32 v5, v2
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: global_wb scope:SCOPE_SYS
@@ -2009,7 +2007,7 @@ define amdgpu_ps void @flat_store_saddr_i16_d16hi_zext_vgpr(ptr inreg %sbase, i3
; GFX1250-GISEL-NEXT: v_nop
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v2, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: flat_store_d16_hi_b16 v[2:3], v1
@@ -2041,7 +2039,7 @@ define amdgpu_ps void @flat_store_saddr_i16_d16hi_zext_vgpr_offset_neg128(ptr in
; GFX1250-GISEL-NEXT: v_nop
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v2, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: flat_store_d16_hi_b16 v[2:3], v1 offset:-128
@@ -2074,7 +2072,7 @@ define amdgpu_ps void @flat_store_saddr_i16_d16hi_trunci8_zext_vgpr(ptr inreg %s
; GFX1250-GISEL-NEXT: v_nop
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v2, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: flat_store_d16_hi_b8 v[2:3], v1
@@ -2107,7 +2105,7 @@ define amdgpu_ps void @flat_store_saddr_i16_d16hi_trunci8_zext_vgpr_offset_neg12
; GFX1250-GISEL-NEXT: v_nop
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_u32 v2, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: flat_store_d16_hi_b8 v[2:3], v1 offset:-128
diff --git a/llvm/test/CodeGen/AMDGPU/flat_atomics.ll b/llvm/test/CodeGen/AMDGPU/flat_atomics.ll
index 232a170b7b2b3a..bd19b93a5c16b4 100644
--- a/llvm/test/CodeGen/AMDGPU/flat_atomics.ll
+++ b/llvm/test/CodeGen/AMDGPU/flat_atomics.ll
@@ -186,7 +186,6 @@ define amdgpu_kernel void @atomic_add_i32_max_offset_p1(ptr %out, i32 %in) {
; GFX11-NEXT: s_load_b32 s2, s[4:5], 0x2c
; GFX11-NEXT: s_waitcnt lgkmcnt(0)
; GFX11-NEXT: v_add_co_u32 v0, s0, 0x1000, s0
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, 0, s1, s0
; GFX11-NEXT: v_mov_b32_e32 v2, s2
; GFX11-NEXT: flat_atomic_add_u32 v[0:1], v2
@@ -9247,7 +9246,6 @@ define amdgpu_kernel void @atomic_inc_i32_max_offset_p1(ptr %out, i32 %in) {
; GFX11-NEXT: s_load_b32 s2, s[4:5], 0x2c
; GFX11-NEXT: s_waitcnt lgkmcnt(0)
; GFX11-NEXT: v_add_co_u32 v0, s0, 0x1000, s0
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, 0, s1, s0
; GFX11-NEXT: v_mov_b32_e32 v2, s2
; GFX11-NEXT: flat_atomic_inc_u32 v[0:1], v2
@@ -9987,7 +9985,6 @@ define amdgpu_kernel void @atomic_dec_i32_max_offset_p1(ptr %out, i32 %in) {
; GFX11-NEXT: s_load_b32 s2, s[4:5], 0x2c
; GFX11-NEXT: s_waitcnt lgkmcnt(0)
; GFX11-NEXT: v_add_co_u32 v0, s0, 0x1000, s0
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, 0, s1, s0
; GFX11-NEXT: v_mov_b32_e32 v2, s2
; GFX11-NEXT: flat_atomic_dec_u32 v[0:1], v2
diff --git a/llvm/test/CodeGen/AMDGPU/flat_atomics_i64.ll b/llvm/test/CodeGen/AMDGPU/flat_atomics_i64.ll
index 5559c3e84e9b73..ca7d38c9346941 100644
--- a/llvm/test/CodeGen/AMDGPU/flat_atomics_i64.ll
+++ b/llvm/test/CodeGen/AMDGPU/flat_atomics_i64.ll
@@ -111,9 +111,10 @@ define amdgpu_kernel void @atomic_add_i64_offset(ptr %out, i64 %in) {
; GFX12-NEXT: s_mov_b64 s[6:7], src_private_base
; GFX12-NEXT: s_wait_kmcnt 0x0
; GFX12-NEXT: s_add_nc_u64 s[0:1], s[0:1], 32
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_eq_u32 s1, s7
; GFX12-NEXT: s_cselect_b32 s4, 1, 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_lg_u32 s4, 1
; GFX12-NEXT: s_mov_b32 s4, -1
; GFX12-NEXT: s_cbranch_scc0 .LBB0_2
@@ -127,16 +128,16 @@ define amdgpu_kernel void @atomic_add_i64_offset(ptr %out, i64 %in) {
; GFX12-NEXT: .LBB0_2: ; %Flow
; GFX12-NEXT: s_and_b32 s4, s4, exec_lo
; GFX12-NEXT: s_cselect_b32 s4, 1, 0
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_lg_u32 s4, 1
; GFX12-NEXT: s_cbranch_scc1 .LBB0_4
; GFX12-NEXT: ; %bb.3: ; %atomicrmw.private
; GFX12-NEXT: s_cmp_lg_u64 s[0:1], 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s0, s0, -1
; GFX12-NEXT: scratch_load_b64 v[0:1], off, s0
; GFX12-NEXT: s_wait_loadcnt 0x0
; GFX12-NEXT: v_add_co_u32 v0, vcc_lo, v0, s2
-; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-NEXT: v_add_co_ci_u32_e64 v1, null, s3, v1, vcc_lo
; GFX12-NEXT: scratch_store_b64 off, v[0:1], s0
; GFX12-NEXT: .LBB0_4: ; %atomicrmw.phi
@@ -271,9 +272,10 @@ define amdgpu_kernel void @atomic_add_i64_ret_offset(ptr %out, ptr %out2, i64 %i
; GFX12-NEXT: s_mov_b64 s[6:7], src_private_base
; GFX12-NEXT: s_wait_kmcnt 0x0
; GFX12-NEXT: s_add_nc_u64 s[0:1], s[0:1], 32
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_eq_u32 s1, s7
; GFX12-NEXT: s_cselect_b32 s6, 1, 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_lg_u32 s6, 1
; GFX12-NEXT: s_cbranch_scc0 .LBB1_2
; GFX12-NEXT: ; %bb.1: ; %atomicrmw.global
@@ -292,14 +294,15 @@ define amdgpu_kernel void @atomic_add_i64_ret_offset(ptr %out, ptr %out2, i64 %i
; GFX12-NEXT: s_and_b32 s6, s6, exec_lo
; GFX12-NEXT: s_cselect_b32 s6, 1, 0
; GFX12-NEXT: s_cmp_lg_u32 s6, 1
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_scc1 .LBB1_5
; GFX12-NEXT: ; %bb.4: ; %atomicrmw.private
; GFX12-NEXT: s_cmp_lg_u64 s[0:1], 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s0, s0, -1
; GFX12-NEXT: scratch_load_b64 v[0:1], off, s0
; GFX12-NEXT: s_wait_loadcnt 0x0
; GFX12-NEXT: v_add_co_u32 v2, vcc_lo, v0, s4
-; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-NEXT: v_add_co_ci_u32_e64 v3, null, s5, v1, vcc_lo
; GFX12-NEXT: scratch_store_b64 off, v[2:3], s0
; GFX12-NEXT: .LBB1_5: ; %atomicrmw.end
@@ -435,9 +438,10 @@ define amdgpu_kernel void @atomic_add_i64_addr64_offset(ptr %out, i64 %in, i64 %
; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_add_nc_u64 s[0:1], s[0:1], s[4:5]
; GFX12-NEXT: s_add_nc_u64 s[0:1], s[0:1], 32
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_eq_u32 s1, s7
; GFX12-NEXT: s_cselect_b32 s4, 1, 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_lg_u32 s4, 1
; GFX12-NEXT: s_mov_b32 s4, -1
; GFX12-NEXT: s_cbranch_scc0 .LBB2_2
@@ -451,16 +455,16 @@ define amdgpu_kernel void @atomic_add_i64_addr64_offset(ptr %out, i64 %in, i64 %
; GFX12-NEXT: .LBB2_2: ; %Flow
; GFX12-NEXT: s_and_b32 s4, s4, exec_lo
; GFX12-NEXT: s_cselect_b32 s4, 1, 0
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_lg_u32 s4, 1
; GFX12-NEXT: s_cbranch_scc1 .LBB2_4
; GFX12-NEXT: ; %bb.3: ; %atomicrmw.private
; GFX12-NEXT: s_cmp_lg_u64 s[0:1], 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s0, s0, -1
; GFX12-NEXT: scratch_load_b64 v[0:1], off, s0
; GFX12-NEXT: s_wait_loadcnt 0x0
; GFX12-NEXT: v_add_co_u32 v0, vcc_lo, v0, s2
-; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-NEXT: v_add_co_ci_u32_e64 v1, null, s3, v1, vcc_lo
; GFX12-NEXT: scratch_store_b64 off, v[0:1], s0
; GFX12-NEXT: .LBB2_4: ; %atomicrmw.phi
@@ -601,9 +605,10 @@ define amdgpu_kernel void @atomic_add_i64_ret_addr64_offset(ptr %out, ptr %out2,
; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_add_nc_u64 s[0:1], s[0:1], s[6:7]
; GFX12-NEXT: s_add_nc_u64 s[0:1], s[0:1], 32
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_eq_u32 s1, s9
; GFX12-NEXT: s_cselect_b32 s6, 1, 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_lg_u32 s6, 1
; GFX12-NEXT: s_cbranch_scc0 .LBB3_2
; GFX12-NEXT: ; %bb.1: ; %atomicrmw.global
@@ -622,14 +627,15 @@ define amdgpu_kernel void @atomic_add_i64_ret_addr64_offset(ptr %out, ptr %out2,
; GFX12-NEXT: s_and_b32 s6, s6, exec_lo
; GFX12-NEXT: s_cselect_b32 s6, 1, 0
; GFX12-NEXT: s_cmp_lg_u32 s6, 1
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_scc1 .LBB3_5
; GFX12-NEXT: ; %bb.4: ; %atomicrmw.private
; GFX12-NEXT: s_cmp_lg_u64 s[0:1], 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s0, s0, -1
; GFX12-NEXT: scratch_load_b64 v[0:1], off, s0
; GFX12-NEXT: s_wait_loadcnt 0x0
; GFX12-NEXT: v_add_co_u32 v2, vcc_lo, v0, s4
-; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-NEXT: v_add_co_ci_u32_e64 v3, null, s5, v1, vcc_lo
; GFX12-NEXT: scratch_store_b64 off, v[2:3], s0
; GFX12-NEXT: .LBB3_5: ; %atomicrmw.end
@@ -748,8 +754,8 @@ define amdgpu_kernel void @atomic_add_i64(ptr %out, i64 %in) {
; GFX12-NEXT: s_mov_b64 s[6:7], src_private_base
; GFX12-NEXT: s_wait_kmcnt 0x0
; GFX12-NEXT: s_cmp_eq_u32 s1, s7
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s4, 1, 0
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_lg_u32 s4, 1
; GFX12-NEXT: s_mov_b32 s4, -1
; GFX12-NEXT: s_cbranch_scc0 .LBB4_2
@@ -763,16 +769,16 @@ define amdgpu_kernel void @atomic_add_i64(ptr %out, i64 %in) {
; GFX12-NEXT: .LBB4_2: ; %Flow
; GFX12-NEXT: s_and_b32 s4, s4, exec_lo
; GFX12-NEXT: s_cselect_b32 s4, 1, 0
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_lg_u32 s4, 1
; GFX12-NEXT: s_cbranch_scc1 .LBB4_4
; GFX12-NEXT: ; %bb.3: ; %atomicrmw.private
; GFX12-NEXT: s_cmp_lg_u64 s[0:1], 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s0, s0, -1
; GFX12-NEXT: scratch_load_b64 v[0:1], off, s0
; GFX12-NEXT: s_wait_loadcnt 0x0
; GFX12-NEXT: v_add_co_u32 v0, vcc_lo, v0, s2
-; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-NEXT: v_add_co_ci_u32_e64 v1, null, s3, v1, vcc_lo
; GFX12-NEXT: scratch_store_b64 off, v[0:1], s0
; GFX12-NEXT: .LBB4_4: ; %atomicrmw.phi
@@ -902,9 +908,10 @@ define amdgpu_kernel void @atomic_add_i64_ret(ptr %out, ptr %out2, i64 %in) {
; GFX12-NEXT: s_mov_b64 s[6:7], src_private_base
; GFX12-NEXT: s_wait_kmcnt 0x0
; GFX12-NEXT: s_cmp_eq_u32 s1, s7
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s6, 1, 0
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_lg_u32 s6, 1
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_scc0 .LBB5_2
; GFX12-NEXT: ; %bb.1: ; %atomicrmw.global
; GFX12-NEXT: v_dual_mov_b32 v0, s0 :: v_dual_mov_b32 v1, s1
@@ -922,14 +929,15 @@ define amdgpu_kernel void @atomic_add_i64_ret(ptr %out, ptr %out2, i64 %in) {
; GFX12-NEXT: s_and_b32 s6, s6, exec_lo
; GFX12-NEXT: s_cselect_b32 s6, 1, 0
; GFX12-NEXT: s_cmp_lg_u32 s6, 1
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_scc1 .LBB5_5
; GFX12-NEXT: ; %bb.4: ; %atomicrmw.private
; GFX12-NEXT: s_cmp_lg_u64 s[0:1], 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s0, s0, -1
; GFX12-NEXT: scratch_load_b64 v[0:1], off, s0
; GFX12-NEXT: s_wait_loadcnt 0x0
; GFX12-NEXT: v_add_co_u32 v2, vcc_lo, v0, s4
-; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-NEXT: v_add_co_ci_u32_e64 v3, null, s5, v1, vcc_lo
; GFX12-NEXT: scratch_store_b64 off, v[2:3], s0
; GFX12-NEXT: .LBB5_5: ; %atomicrmw.end
@@ -1059,8 +1067,8 @@ define amdgpu_kernel void @atomic_add_i64_addr64(ptr %out, i64 %in, i64 %index)
; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_add_nc_u64 s[0:1], s[0:1], s[4:5]
; GFX12-NEXT: s_cmp_eq_u32 s1, s7
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s4, 1, 0
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_lg_u32 s4, 1
; GFX12-NEXT: s_mov_b32 s4, -1
; GFX12-NEXT: s_cbranch_scc0 .LBB6_2
@@ -1074,16 +1082,16 @@ define amdgpu_kernel void @atomic_add_i64_addr64(ptr %out, i64 %in, i64 %index)
; GFX12-NEXT: .LBB6_2: ; %Flow
; GFX12-NEXT: s_and_b32 s4, s4, exec_lo
; GFX12-NEXT: s_cselect_b32 s4, 1, 0
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_lg_u32 s4, 1
; GFX12-NEXT: s_cbranch_scc1 .LBB6_4
; GFX12-NEXT: ; %bb.3: ; %atomicrmw.private
; GFX12-NEXT: s_cmp_lg_u64 s[0:1], 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s0, s0, -1
; GFX12-NEXT: scratch_load_b64 v[0:1], off, s0
; GFX12-NEXT: s_wait_loadcnt 0x0
; GFX12-NEXT: v_add_co_u32 v0, vcc_lo, v0, s2
-; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-NEXT: v_add_co_ci_u32_e64 v1, null, s3, v1, vcc_lo
; GFX12-NEXT: scratch_store_b64 off, v[0:1], s0
; GFX12-NEXT: .LBB6_4: ; %atomicrmw.phi
@@ -1219,9 +1227,10 @@ define amdgpu_kernel void @atomic_add_i64_ret_addr64(ptr %out, ptr %out2, i64 %i
; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_add_nc_u64 s[0:1], s[0:1], s[6:7]
; GFX12-NEXT: s_cmp_eq_u32 s1, s9
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s6, 1, 0
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_lg_u32 s6, 1
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_scc0 .LBB7_2
; GFX12-NEXT: ; %bb.1: ; %atomicrmw.global
; GFX12-NEXT: v_dual_mov_b32 v0, s0 :: v_dual_mov_b32 v1, s1
@@ -1239,14 +1248,15 @@ define amdgpu_kernel void @atomic_add_i64_ret_addr64(ptr %out, ptr %out2, i64 %i
; GFX12-NEXT: s_and_b32 s6, s6, exec_lo
; GFX12-NEXT: s_cselect_b32 s6, 1, 0
; GFX12-NEXT: s_cmp_lg_u32 s6, 1
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_scc1 .LBB7_5
; GFX12-NEXT: ; %bb.4: ; %atomicrmw.private
; GFX12-NEXT: s_cmp_lg_u64 s[0:1], 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s0, s0, -1
; GFX12-NEXT: scratch_load_b64 v[0:1], off, s0
; GFX12-NEXT: s_wait_loadcnt 0x0
; GFX12-NEXT: v_add_co_u32 v2, vcc_lo, v0, s4
-; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-NEXT: v_add_co_ci_u32_e64 v3, null, s5, v1, vcc_lo
; GFX12-NEXT: scratch_store_b64 off, v[2:3], s0
; GFX12-NEXT: .LBB7_5: ; %atomicrmw.end
@@ -1366,9 +1376,10 @@ define amdgpu_kernel void @atomic_and_i64_offset(ptr %out, i64 %in) {
; GFX12-NEXT: s_mov_b64 s[6:7], src_private_base
; GFX12-NEXT: s_wait_kmcnt 0x0
; GFX12-NEXT: s_add_nc_u64 s[0:1], s[0:1], 32
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_eq_u32 s1, s7
; GFX12-NEXT: s_cselect_b32 s4, 1, 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_lg_u32 s4, 1
; GFX12-NEXT: s_mov_b32 s4, -1
; GFX12-NEXT: s_cbranch_scc0 .LBB8_2
@@ -1382,11 +1393,12 @@ define amdgpu_kernel void @atomic_and_i64_offset(ptr %out, i64 %in) {
; GFX12-NEXT: .LBB8_2: ; %Flow
; GFX12-NEXT: s_and_b32 s4, s4, exec_lo
; GFX12-NEXT: s_cselect_b32 s4, 1, 0
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_lg_u32 s4, 1
; GFX12-NEXT: s_cbranch_scc1 .LBB8_4
; GFX12-NEXT: ; %bb.3: ; %atomicrmw.private
; GFX12-NEXT: s_cmp_lg_u64 s[0:1], 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s0, s0, -1
; GFX12-NEXT: scratch_load_b64 v[0:1], off, s0
; GFX12-NEXT: s_wait_loadcnt 0x0
@@ -1523,9 +1535,10 @@ define amdgpu_kernel void @atomic_and_i64_ret_offset(ptr %out, ptr %out2, i64 %i
; GFX12-NEXT: s_mov_b64 s[6:7], src_private_base
; GFX12-NEXT: s_wait_kmcnt 0x0
; GFX12-NEXT: s_add_nc_u64 s[0:1], s[0:1], 32
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_eq_u32 s1, s7
; GFX12-NEXT: s_cselect_b32 s6, 1, 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_lg_u32 s6, 1
; GFX12-NEXT: s_cbranch_scc0 .LBB9_2
; GFX12-NEXT: ; %bb.1: ; %atomicrmw.global
@@ -1544,9 +1557,11 @@ define amdgpu_kernel void @atomic_and_i64_ret_offset(ptr %out, ptr %out2, i64 %i
; GFX12-NEXT: s_and_b32 s6, s6, exec_lo
; GFX12-NEXT: s_cselect_b32 s6, 1, 0
; GFX12-NEXT: s_cmp_lg_u32 s6, 1
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_scc1 .LBB9_5
; GFX12-NEXT: ; %bb.4: ; %atomicrmw.private
; GFX12-NEXT: s_cmp_lg_u64 s[0:1], 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s0, s0, -1
; GFX12-NEXT: scratch_load_b64 v[0:1], off, s0
; GFX12-NEXT: s_wait_loadcnt 0x0
@@ -1684,9 +1699,10 @@ define amdgpu_kernel void @atomic_and_i64_addr64_offset(ptr %out, i64 %in, i64 %
; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_add_nc_u64 s[0:1], s[0:1], s[4:5]
; GFX12-NEXT: s_add_nc_u64 s[0:1], s[0:1], 32
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_eq_u32 s1, s7
; GFX12-NEXT: s_cselect_b32 s4, 1, 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_lg_u32 s4, 1
; GFX12-NEXT: s_mov_b32 s4, -1
; GFX12-NEXT: s_cbranch_scc0 .LBB10_2
@@ -1700,11 +1716,12 @@ define amdgpu_kernel void @atomic_and_i64_addr64_offset(ptr %out, i64 %in, i64 %
; GFX12-NEXT: .LBB10_2: ; %Flow
; GFX12-NEXT: s_and_b32 s4, s4, exec_lo
; GFX12-NEXT: s_cselect_b32 s4, 1, 0
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_lg_u32 s4, 1
; GFX12-NEXT: s_cbranch_scc1 .LBB10_4
; GFX12-NEXT: ; %bb.3: ; %atomicrmw.private
; GFX12-NEXT: s_cmp_lg_u64 s[0:1], 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s0, s0, -1
; GFX12-NEXT: scratch_load_b64 v[0:1], off, s0
; GFX12-NEXT: s_wait_loadcnt 0x0
@@ -1847,9 +1864,10 @@ define amdgpu_kernel void @atomic_and_i64_ret_addr64_offset(ptr %out, ptr %out2,
; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_add_nc_u64 s[0:1], s[0:1], s[6:7]
; GFX12-NEXT: s_add_nc_u64 s[0:1], s[0:1], 32
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_eq_u32 s1, s9
; GFX12-NEXT: s_cselect_b32 s6, 1, 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_lg_u32 s6, 1
; GFX12-NEXT: s_cbranch_scc0 .LBB11_2
; GFX12-NEXT: ; %bb.1: ; %atomicrmw.global
@@ -1868,9 +1886,11 @@ define amdgpu_kernel void @atomic_and_i64_ret_addr64_offset(ptr %out, ptr %out2,
; GFX12-NEXT: s_and_b32 s6, s6, exec_lo
; GFX12-NEXT: s_cselect_b32 s6, 1, 0
; GFX12-NEXT: s_cmp_lg_u32 s6, 1
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_scc1 .LBB11_5
; GFX12-NEXT: ; %bb.4: ; %atomicrmw.private
; GFX12-NEXT: s_cmp_lg_u64 s[0:1], 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s0, s0, -1
; GFX12-NEXT: scratch_load_b64 v[0:1], off, s0
; GFX12-NEXT: s_wait_loadcnt 0x0
@@ -1991,8 +2011,8 @@ define amdgpu_kernel void @atomic_and_i64(ptr %out, i64 %in) {
; GFX12-NEXT: s_mov_b64 s[6:7], src_private_base
; GFX12-NEXT: s_wait_kmcnt 0x0
; GFX12-NEXT: s_cmp_eq_u32 s1, s7
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s4, 1, 0
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_lg_u32 s4, 1
; GFX12-NEXT: s_mov_b32 s4, -1
; GFX12-NEXT: s_cbranch_scc0 .LBB12_2
@@ -2006,11 +2026,12 @@ define amdgpu_kernel void @atomic_and_i64(ptr %out, i64 %in) {
; GFX12-NEXT: .LBB12_2: ; %Flow
; GFX12-NEXT: s_and_b32 s4, s4, exec_lo
; GFX12-NEXT: s_cselect_b32 s4, 1, 0
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_lg_u32 s4, 1
; GFX12-NEXT: s_cbranch_scc1 .LBB12_4
; GFX12-NEXT: ; %bb.3: ; %atomicrmw.private
; GFX12-NEXT: s_cmp_lg_u64 s[0:1], 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s0, s0, -1
; GFX12-NEXT: scratch_load_b64 v[0:1], off, s0
; GFX12-NEXT: s_wait_loadcnt 0x0
@@ -2142,9 +2163,10 @@ define amdgpu_kernel void @atomic_and_i64_ret(ptr %out, ptr %out2, i64 %in) {
; GFX12-NEXT: s_mov_b64 s[6:7], src_private_base
; GFX12-NEXT: s_wait_kmcnt 0x0
; GFX12-NEXT: s_cmp_eq_u32 s1, s7
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s6, 1, 0
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_lg_u32 s6, 1
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_scc0 .LBB13_2
; GFX12-NEXT: ; %bb.1: ; %atomicrmw.global
; GFX12-NEXT: v_dual_mov_b32 v0, s0 :: v_dual_mov_b32 v1, s1
@@ -2162,9 +2184,11 @@ define amdgpu_kernel void @atomic_and_i64_ret(ptr %out, ptr %out2, i64 %in) {
; GFX12-NEXT: s_and_b32 s6, s6, exec_lo
; GFX12-NEXT: s_cselect_b32 s6, 1, 0
; GFX12-NEXT: s_cmp_lg_u32 s6, 1
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_scc1 .LBB13_5
; GFX12-NEXT: ; %bb.4: ; %atomicrmw.private
; GFX12-NEXT: s_cmp_lg_u64 s[0:1], 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s0, s0, -1
; GFX12-NEXT: scratch_load_b64 v[0:1], off, s0
; GFX12-NEXT: s_wait_loadcnt 0x0
@@ -2296,8 +2320,8 @@ define amdgpu_kernel void @atomic_and_i64_addr64(ptr %out, i64 %in, i64 %index)
; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_add_nc_u64 s[0:1], s[0:1], s[4:5]
; GFX12-NEXT: s_cmp_eq_u32 s1, s7
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s4, 1, 0
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_lg_u32 s4, 1
; GFX12-NEXT: s_mov_b32 s4, -1
; GFX12-NEXT: s_cbranch_scc0 .LBB14_2
@@ -2311,11 +2335,12 @@ define amdgpu_kernel void @atomic_and_i64_addr64(ptr %out, i64 %in, i64 %index)
; GFX12-NEXT: .LBB14_2: ; %Flow
; GFX12-NEXT: s_and_b32 s4, s4, exec_lo
; GFX12-NEXT: s_cselect_b32 s4, 1, 0
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_lg_u32 s4, 1
; GFX12-NEXT: s_cbranch_scc1 .LBB14_4
; GFX12-NEXT: ; %bb.3: ; %atomicrmw.private
; GFX12-NEXT: s_cmp_lg_u64 s[0:1], 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s0, s0, -1
; GFX12-NEXT: scratch_load_b64 v[0:1], off, s0
; GFX12-NEXT: s_wait_loadcnt 0x0
@@ -2453,9 +2478,10 @@ define amdgpu_kernel void @atomic_and_i64_ret_addr64(ptr %out, ptr %out2, i64 %i
; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_add_nc_u64 s[0:1], s[0:1], s[6:7]
; GFX12-NEXT: s_cmp_eq_u32 s1, s9
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s6, 1, 0
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_lg_u32 s6, 1
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_scc0 .LBB15_2
; GFX12-NEXT: ; %bb.1: ; %atomicrmw.global
; GFX12-NEXT: v_dual_mov_b32 v0, s0 :: v_dual_mov_b32 v1, s1
@@ -2473,9 +2499,11 @@ define amdgpu_kernel void @atomic_and_i64_ret_addr64(ptr %out, ptr %out2, i64 %i
; GFX12-NEXT: s_and_b32 s6, s6, exec_lo
; GFX12-NEXT: s_cselect_b32 s6, 1, 0
; GFX12-NEXT: s_cmp_lg_u32 s6, 1
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_scc1 .LBB15_5
; GFX12-NEXT: ; %bb.4: ; %atomicrmw.private
; GFX12-NEXT: s_cmp_lg_u64 s[0:1], 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s0, s0, -1
; GFX12-NEXT: scratch_load_b64 v[0:1], off, s0
; GFX12-NEXT: s_wait_loadcnt 0x0
@@ -2601,9 +2629,10 @@ define amdgpu_kernel void @atomic_sub_i64_offset(ptr %out, i64 %in) {
; GFX12-NEXT: s_mov_b64 s[6:7], src_private_base
; GFX12-NEXT: s_wait_kmcnt 0x0
; GFX12-NEXT: s_add_nc_u64 s[0:1], s[0:1], 32
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_eq_u32 s1, s7
; GFX12-NEXT: s_cselect_b32 s4, 1, 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_lg_u32 s4, 1
; GFX12-NEXT: s_mov_b32 s4, -1
; GFX12-NEXT: s_cbranch_scc0 .LBB16_2
@@ -2617,16 +2646,16 @@ define amdgpu_kernel void @atomic_sub_i64_offset(ptr %out, i64 %in) {
; GFX12-NEXT: .LBB16_2: ; %Flow
; GFX12-NEXT: s_and_b32 s4, s4, exec_lo
; GFX12-NEXT: s_cselect_b32 s4, 1, 0
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_lg_u32 s4, 1
; GFX12-NEXT: s_cbranch_scc1 .LBB16_4
; GFX12-NEXT: ; %bb.3: ; %atomicrmw.private
; GFX12-NEXT: s_cmp_lg_u64 s[0:1], 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s0, s0, -1
; GFX12-NEXT: scratch_load_b64 v[0:1], off, s0
; GFX12-NEXT: s_wait_loadcnt 0x0
; GFX12-NEXT: v_sub_co_u32 v0, vcc_lo, v0, s2
-; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-NEXT: v_subrev_co_ci_u32_e64 v1, null, s3, v1, vcc_lo
; GFX12-NEXT: scratch_store_b64 off, v[0:1], s0
; GFX12-NEXT: .LBB16_4: ; %atomicrmw.phi
@@ -2761,9 +2790,10 @@ define amdgpu_kernel void @atomic_sub_i64_ret_offset(ptr %out, ptr %out2, i64 %i
; GFX12-NEXT: s_mov_b64 s[6:7], src_private_base
; GFX12-NEXT: s_wait_kmcnt 0x0
; GFX12-NEXT: s_add_nc_u64 s[0:1], s[0:1], 32
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_eq_u32 s1, s7
; GFX12-NEXT: s_cselect_b32 s6, 1, 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_lg_u32 s6, 1
; GFX12-NEXT: s_cbranch_scc0 .LBB17_2
; GFX12-NEXT: ; %bb.1: ; %atomicrmw.global
@@ -2782,14 +2812,15 @@ define amdgpu_kernel void @atomic_sub_i64_ret_offset(ptr %out, ptr %out2, i64 %i
; GFX12-NEXT: s_and_b32 s6, s6, exec_lo
; GFX12-NEXT: s_cselect_b32 s6, 1, 0
; GFX12-NEXT: s_cmp_lg_u32 s6, 1
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_scc1 .LBB17_5
; GFX12-NEXT: ; %bb.4: ; %atomicrmw.private
; GFX12-NEXT: s_cmp_lg_u64 s[0:1], 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s0, s0, -1
; GFX12-NEXT: scratch_load_b64 v[0:1], off, s0
; GFX12-NEXT: s_wait_loadcnt 0x0
; GFX12-NEXT: v_sub_co_u32 v2, vcc_lo, v0, s4
-; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-NEXT: v_subrev_co_ci_u32_e64 v3, null, s5, v1, vcc_lo
; GFX12-NEXT: scratch_store_b64 off, v[2:3], s0
; GFX12-NEXT: .LBB17_5: ; %atomicrmw.end
@@ -2925,9 +2956,10 @@ define amdgpu_kernel void @atomic_sub_i64_addr64_offset(ptr %out, i64 %in, i64 %
; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_add_nc_u64 s[0:1], s[0:1], s[4:5]
; GFX12-NEXT: s_add_nc_u64 s[0:1], s[0:1], 32
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_eq_u32 s1, s7
; GFX12-NEXT: s_cselect_b32 s4, 1, 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_lg_u32 s4, 1
; GFX12-NEXT: s_mov_b32 s4, -1
; GFX12-NEXT: s_cbranch_scc0 .LBB18_2
@@ -2941,16 +2973,16 @@ define amdgpu_kernel void @atomic_sub_i64_addr64_offset(ptr %out, i64 %in, i64 %
; GFX12-NEXT: .LBB18_2: ; %Flow
; GFX12-NEXT: s_and_b32 s4, s4, exec_lo
; GFX12-NEXT: s_cselect_b32 s4, 1, 0
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_lg_u32 s4, 1
; GFX12-NEXT: s_cbranch_scc1 .LBB18_4
; GFX12-NEXT: ; %bb.3: ; %atomicrmw.private
; GFX12-NEXT: s_cmp_lg_u64 s[0:1], 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s0, s0, -1
; GFX12-NEXT: scratch_load_b64 v[0:1], off, s0
; GFX12-NEXT: s_wait_loadcnt 0x0
; GFX12-NEXT: v_sub_co_u32 v0, vcc_lo, v0, s2
-; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-NEXT: v_subrev_co_ci_u32_e64 v1, null, s3, v1, vcc_lo
; GFX12-NEXT: scratch_store_b64 off, v[0:1], s0
; GFX12-NEXT: .LBB18_4: ; %atomicrmw.phi
@@ -3091,9 +3123,10 @@ define amdgpu_kernel void @atomic_sub_i64_ret_addr64_offset(ptr %out, ptr %out2,
; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_add_nc_u64 s[0:1], s[0:1], s[6:7]
; GFX12-NEXT: s_add_nc_u64 s[0:1], s[0:1], 32
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_eq_u32 s1, s9
; GFX12-NEXT: s_cselect_b32 s6, 1, 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_lg_u32 s6, 1
; GFX12-NEXT: s_cbranch_scc0 .LBB19_2
; GFX12-NEXT: ; %bb.1: ; %atomicrmw.global
@@ -3112,14 +3145,15 @@ define amdgpu_kernel void @atomic_sub_i64_ret_addr64_offset(ptr %out, ptr %out2,
; GFX12-NEXT: s_and_b32 s6, s6, exec_lo
; GFX12-NEXT: s_cselect_b32 s6, 1, 0
; GFX12-NEXT: s_cmp_lg_u32 s6, 1
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_scc1 .LBB19_5
; GFX12-NEXT: ; %bb.4: ; %atomicrmw.private
; GFX12-NEXT: s_cmp_lg_u64 s[0:1], 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s0, s0, -1
; GFX12-NEXT: scratch_load_b64 v[0:1], off, s0
; GFX12-NEXT: s_wait_loadcnt 0x0
; GFX12-NEXT: v_sub_co_u32 v2, vcc_lo, v0, s4
-; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-NEXT: v_subrev_co_ci_u32_e64 v3, null, s5, v1, vcc_lo
; GFX12-NEXT: scratch_store_b64 off, v[2:3], s0
; GFX12-NEXT: .LBB19_5: ; %atomicrmw.end
@@ -3238,8 +3272,8 @@ define amdgpu_kernel void @atomic_sub_i64(ptr %out, i64 %in) {
; GFX12-NEXT: s_mov_b64 s[6:7], src_private_base
; GFX12-NEXT: s_wait_kmcnt 0x0
; GFX12-NEXT: s_cmp_eq_u32 s1, s7
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s4, 1, 0
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_lg_u32 s4, 1
; GFX12-NEXT: s_mov_b32 s4, -1
; GFX12-NEXT: s_cbranch_scc0 .LBB20_2
@@ -3253,16 +3287,16 @@ define amdgpu_kernel void @atomic_sub_i64(ptr %out, i64 %in) {
; GFX12-NEXT: .LBB20_2: ; %Flow
; GFX12-NEXT: s_and_b32 s4, s4, exec_lo
; GFX12-NEXT: s_cselect_b32 s4, 1, 0
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_lg_u32 s4, 1
; GFX12-NEXT: s_cbranch_scc1 .LBB20_4
; GFX12-NEXT: ; %bb.3: ; %atomicrmw.private
; GFX12-NEXT: s_cmp_lg_u64 s[0:1], 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s0, s0, -1
; GFX12-NEXT: scratch_load_b64 v[0:1], off, s0
; GFX12-NEXT: s_wait_loadcnt 0x0
; GFX12-NEXT: v_sub_co_u32 v0, vcc_lo, v0, s2
-; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-NEXT: v_subrev_co_ci_u32_e64 v1, null, s3, v1, vcc_lo
; GFX12-NEXT: scratch_store_b64 off, v[0:1], s0
; GFX12-NEXT: .LBB20_4: ; %atomicrmw.phi
@@ -3392,9 +3426,10 @@ define amdgpu_kernel void @atomic_sub_i64_ret(ptr %out, ptr %out2, i64 %in) {
; GFX12-NEXT: s_mov_b64 s[6:7], src_private_base
; GFX12-NEXT: s_wait_kmcnt 0x0
; GFX12-NEXT: s_cmp_eq_u32 s1, s7
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s6, 1, 0
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_lg_u32 s6, 1
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_scc0 .LBB21_2
; GFX12-NEXT: ; %bb.1: ; %atomicrmw.global
; GFX12-NEXT: v_dual_mov_b32 v0, s0 :: v_dual_mov_b32 v1, s1
@@ -3412,14 +3447,15 @@ define amdgpu_kernel void @atomic_sub_i64_ret(ptr %out, ptr %out2, i64 %in) {
; GFX12-NEXT: s_and_b32 s6, s6, exec_lo
; GFX12-NEXT: s_cselect_b32 s6, 1, 0
; GFX12-NEXT: s_cmp_lg_u32 s6, 1
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_scc1 .LBB21_5
; GFX12-NEXT: ; %bb.4: ; %atomicrmw.private
; GFX12-NEXT: s_cmp_lg_u64 s[0:1], 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s0, s0, -1
; GFX12-NEXT: scratch_load_b64 v[0:1], off, s0
; GFX12-NEXT: s_wait_loadcnt 0x0
; GFX12-NEXT: v_sub_co_u32 v2, vcc_lo, v0, s4
-; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-NEXT: v_subrev_co_ci_u32_e64 v3, null, s5, v1, vcc_lo
; GFX12-NEXT: scratch_store_b64 off, v[2:3], s0
; GFX12-NEXT: .LBB21_5: ; %atomicrmw.end
@@ -3549,8 +3585,8 @@ define amdgpu_kernel void @atomic_sub_i64_addr64(ptr %out, i64 %in, i64 %index)
; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_add_nc_u64 s[0:1], s[0:1], s[4:5]
; GFX12-NEXT: s_cmp_eq_u32 s1, s7
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s4, 1, 0
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_lg_u32 s4, 1
; GFX12-NEXT: s_mov_b32 s4, -1
; GFX12-NEXT: s_cbranch_scc0 .LBB22_2
@@ -3564,16 +3600,16 @@ define amdgpu_kernel void @atomic_sub_i64_addr64(ptr %out, i64 %in, i64 %index)
; GFX12-NEXT: .LBB22_2: ; %Flow
; GFX12-NEXT: s_and_b32 s4, s4, exec_lo
; GFX12-NEXT: s_cselect_b32 s4, 1, 0
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_lg_u32 s4, 1
; GFX12-NEXT: s_cbranch_scc1 .LBB22_4
; GFX12-NEXT: ; %bb.3: ; %atomicrmw.private
; GFX12-NEXT: s_cmp_lg_u64 s[0:1], 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s0, s0, -1
; GFX12-NEXT: scratch_load_b64 v[0:1], off, s0
; GFX12-NEXT: s_wait_loadcnt 0x0
; GFX12-NEXT: v_sub_co_u32 v0, vcc_lo, v0, s2
-; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-NEXT: v_subrev_co_ci_u32_e64 v1, null, s3, v1, vcc_lo
; GFX12-NEXT: scratch_store_b64 off, v[0:1], s0
; GFX12-NEXT: .LBB22_4: ; %atomicrmw.phi
@@ -3709,9 +3745,10 @@ define amdgpu_kernel void @atomic_sub_i64_ret_addr64(ptr %out, ptr %out2, i64 %i
; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_add_nc_u64 s[0:1], s[0:1], s[6:7]
; GFX12-NEXT: s_cmp_eq_u32 s1, s9
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s6, 1, 0
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_lg_u32 s6, 1
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_scc0 .LBB23_2
; GFX12-NEXT: ; %bb.1: ; %atomicrmw.global
; GFX12-NEXT: v_dual_mov_b32 v0, s0 :: v_dual_mov_b32 v1, s1
@@ -3729,14 +3766,15 @@ define amdgpu_kernel void @atomic_sub_i64_ret_addr64(ptr %out, ptr %out2, i64 %i
; GFX12-NEXT: s_and_b32 s6, s6, exec_lo
; GFX12-NEXT: s_cselect_b32 s6, 1, 0
; GFX12-NEXT: s_cmp_lg_u32 s6, 1
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_scc1 .LBB23_5
; GFX12-NEXT: ; %bb.4: ; %atomicrmw.private
; GFX12-NEXT: s_cmp_lg_u64 s[0:1], 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s0, s0, -1
; GFX12-NEXT: scratch_load_b64 v[0:1], off, s0
; GFX12-NEXT: s_wait_loadcnt 0x0
; GFX12-NEXT: v_sub_co_u32 v2, vcc_lo, v0, s4
-; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-NEXT: v_subrev_co_ci_u32_e64 v3, null, s5, v1, vcc_lo
; GFX12-NEXT: scratch_store_b64 off, v[2:3], s0
; GFX12-NEXT: .LBB23_5: ; %atomicrmw.end
@@ -3858,9 +3896,10 @@ define amdgpu_kernel void @atomic_max_i64_offset(ptr %out, i64 %in) {
; GFX12-NEXT: s_mov_b64 s[6:7], src_private_base
; GFX12-NEXT: s_wait_kmcnt 0x0
; GFX12-NEXT: s_add_nc_u64 s[0:1], s[0:1], 32
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_eq_u32 s1, s7
; GFX12-NEXT: s_cselect_b32 s4, 1, 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_lg_u32 s4, 1
; GFX12-NEXT: s_mov_b32 s4, -1
; GFX12-NEXT: s_cbranch_scc0 .LBB24_2
@@ -3874,11 +3913,12 @@ define amdgpu_kernel void @atomic_max_i64_offset(ptr %out, i64 %in) {
; GFX12-NEXT: .LBB24_2: ; %Flow
; GFX12-NEXT: s_and_b32 s4, s4, exec_lo
; GFX12-NEXT: s_cselect_b32 s4, 1, 0
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_lg_u32 s4, 1
; GFX12-NEXT: s_cbranch_scc1 .LBB24_4
; GFX12-NEXT: ; %bb.3: ; %atomicrmw.private
; GFX12-NEXT: s_cmp_lg_u64 s[0:1], 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s0, s0, -1
; GFX12-NEXT: scratch_load_b64 v[0:1], off, s0
; GFX12-NEXT: s_wait_loadcnt 0x0
@@ -4020,9 +4060,10 @@ define amdgpu_kernel void @atomic_max_i64_ret_offset(ptr %out, ptr %out2, i64 %i
; GFX12-NEXT: s_mov_b64 s[6:7], src_private_base
; GFX12-NEXT: s_wait_kmcnt 0x0
; GFX12-NEXT: s_add_nc_u64 s[0:1], s[0:1], 32
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_eq_u32 s1, s7
; GFX12-NEXT: s_cselect_b32 s6, 1, 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_lg_u32 s6, 1
; GFX12-NEXT: s_cbranch_scc0 .LBB25_2
; GFX12-NEXT: ; %bb.1: ; %atomicrmw.global
@@ -4041,9 +4082,11 @@ define amdgpu_kernel void @atomic_max_i64_ret_offset(ptr %out, ptr %out2, i64 %i
; GFX12-NEXT: s_and_b32 s6, s6, exec_lo
; GFX12-NEXT: s_cselect_b32 s6, 1, 0
; GFX12-NEXT: s_cmp_lg_u32 s6, 1
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_scc1 .LBB25_5
; GFX12-NEXT: ; %bb.4: ; %atomicrmw.private
; GFX12-NEXT: s_cmp_lg_u64 s[0:1], 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s0, s0, -1
; GFX12-NEXT: scratch_load_b64 v[0:1], off, s0
; GFX12-NEXT: s_wait_loadcnt 0x0
@@ -4184,9 +4227,10 @@ define amdgpu_kernel void @atomic_max_i64_addr64_offset(ptr %out, i64 %in, i64 %
; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_add_nc_u64 s[0:1], s[0:1], s[4:5]
; GFX12-NEXT: s_add_nc_u64 s[0:1], s[0:1], 32
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_eq_u32 s1, s7
; GFX12-NEXT: s_cselect_b32 s4, 1, 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_lg_u32 s4, 1
; GFX12-NEXT: s_mov_b32 s4, -1
; GFX12-NEXT: s_cbranch_scc0 .LBB26_2
@@ -4200,11 +4244,12 @@ define amdgpu_kernel void @atomic_max_i64_addr64_offset(ptr %out, i64 %in, i64 %
; GFX12-NEXT: .LBB26_2: ; %Flow
; GFX12-NEXT: s_and_b32 s4, s4, exec_lo
; GFX12-NEXT: s_cselect_b32 s4, 1, 0
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_lg_u32 s4, 1
; GFX12-NEXT: s_cbranch_scc1 .LBB26_4
; GFX12-NEXT: ; %bb.3: ; %atomicrmw.private
; GFX12-NEXT: s_cmp_lg_u64 s[0:1], 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s0, s0, -1
; GFX12-NEXT: scratch_load_b64 v[0:1], off, s0
; GFX12-NEXT: s_wait_loadcnt 0x0
@@ -4352,9 +4397,10 @@ define amdgpu_kernel void @atomic_max_i64_ret_addr64_offset(ptr %out, ptr %out2,
; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_add_nc_u64 s[0:1], s[0:1], s[6:7]
; GFX12-NEXT: s_add_nc_u64 s[0:1], s[0:1], 32
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_eq_u32 s1, s9
; GFX12-NEXT: s_cselect_b32 s6, 1, 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_lg_u32 s6, 1
; GFX12-NEXT: s_cbranch_scc0 .LBB27_2
; GFX12-NEXT: ; %bb.1: ; %atomicrmw.global
@@ -4373,9 +4419,11 @@ define amdgpu_kernel void @atomic_max_i64_ret_addr64_offset(ptr %out, ptr %out2,
; GFX12-NEXT: s_and_b32 s6, s6, exec_lo
; GFX12-NEXT: s_cselect_b32 s6, 1, 0
; GFX12-NEXT: s_cmp_lg_u32 s6, 1
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_scc1 .LBB27_5
; GFX12-NEXT: ; %bb.4: ; %atomicrmw.private
; GFX12-NEXT: s_cmp_lg_u64 s[0:1], 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s0, s0, -1
; GFX12-NEXT: scratch_load_b64 v[0:1], off, s0
; GFX12-NEXT: s_wait_loadcnt 0x0
@@ -4499,8 +4547,8 @@ define amdgpu_kernel void @atomic_max_i64(ptr %out, i64 %in) {
; GFX12-NEXT: s_mov_b64 s[6:7], src_private_base
; GFX12-NEXT: s_wait_kmcnt 0x0
; GFX12-NEXT: s_cmp_eq_u32 s1, s7
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s4, 1, 0
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_lg_u32 s4, 1
; GFX12-NEXT: s_mov_b32 s4, -1
; GFX12-NEXT: s_cbranch_scc0 .LBB28_2
@@ -4514,11 +4562,12 @@ define amdgpu_kernel void @atomic_max_i64(ptr %out, i64 %in) {
; GFX12-NEXT: .LBB28_2: ; %Flow
; GFX12-NEXT: s_and_b32 s4, s4, exec_lo
; GFX12-NEXT: s_cselect_b32 s4, 1, 0
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_lg_u32 s4, 1
; GFX12-NEXT: s_cbranch_scc1 .LBB28_4
; GFX12-NEXT: ; %bb.3: ; %atomicrmw.private
; GFX12-NEXT: s_cmp_lg_u64 s[0:1], 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s0, s0, -1
; GFX12-NEXT: scratch_load_b64 v[0:1], off, s0
; GFX12-NEXT: s_wait_loadcnt 0x0
@@ -4655,9 +4704,10 @@ define amdgpu_kernel void @atomic_max_i64_ret(ptr %out, ptr %out2, i64 %in) {
; GFX12-NEXT: s_mov_b64 s[6:7], src_private_base
; GFX12-NEXT: s_wait_kmcnt 0x0
; GFX12-NEXT: s_cmp_eq_u32 s1, s7
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s6, 1, 0
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_lg_u32 s6, 1
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_scc0 .LBB29_2
; GFX12-NEXT: ; %bb.1: ; %atomicrmw.global
; GFX12-NEXT: v_dual_mov_b32 v0, s0 :: v_dual_mov_b32 v1, s1
@@ -4675,9 +4725,11 @@ define amdgpu_kernel void @atomic_max_i64_ret(ptr %out, ptr %out2, i64 %in) {
; GFX12-NEXT: s_and_b32 s6, s6, exec_lo
; GFX12-NEXT: s_cselect_b32 s6, 1, 0
; GFX12-NEXT: s_cmp_lg_u32 s6, 1
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_scc1 .LBB29_5
; GFX12-NEXT: ; %bb.4: ; %atomicrmw.private
; GFX12-NEXT: s_cmp_lg_u64 s[0:1], 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s0, s0, -1
; GFX12-NEXT: scratch_load_b64 v[0:1], off, s0
; GFX12-NEXT: s_wait_loadcnt 0x0
@@ -4812,8 +4864,8 @@ define amdgpu_kernel void @atomic_max_i64_addr64(ptr %out, i64 %in, i64 %index)
; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_add_nc_u64 s[0:1], s[0:1], s[4:5]
; GFX12-NEXT: s_cmp_eq_u32 s1, s7
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s4, 1, 0
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_lg_u32 s4, 1
; GFX12-NEXT: s_mov_b32 s4, -1
; GFX12-NEXT: s_cbranch_scc0 .LBB30_2
@@ -4827,11 +4879,12 @@ define amdgpu_kernel void @atomic_max_i64_addr64(ptr %out, i64 %in, i64 %index)
; GFX12-NEXT: .LBB30_2: ; %Flow
; GFX12-NEXT: s_and_b32 s4, s4, exec_lo
; GFX12-NEXT: s_cselect_b32 s4, 1, 0
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_lg_u32 s4, 1
; GFX12-NEXT: s_cbranch_scc1 .LBB30_4
; GFX12-NEXT: ; %bb.3: ; %atomicrmw.private
; GFX12-NEXT: s_cmp_lg_u64 s[0:1], 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s0, s0, -1
; GFX12-NEXT: scratch_load_b64 v[0:1], off, s0
; GFX12-NEXT: s_wait_loadcnt 0x0
@@ -4974,9 +5027,10 @@ define amdgpu_kernel void @atomic_max_i64_ret_addr64(ptr %out, ptr %out2, i64 %i
; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_add_nc_u64 s[0:1], s[0:1], s[6:7]
; GFX12-NEXT: s_cmp_eq_u32 s1, s9
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s6, 1, 0
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_lg_u32 s6, 1
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_scc0 .LBB31_2
; GFX12-NEXT: ; %bb.1: ; %atomicrmw.global
; GFX12-NEXT: v_dual_mov_b32 v0, s0 :: v_dual_mov_b32 v1, s1
@@ -4994,9 +5048,11 @@ define amdgpu_kernel void @atomic_max_i64_ret_addr64(ptr %out, ptr %out2, i64 %i
; GFX12-NEXT: s_and_b32 s6, s6, exec_lo
; GFX12-NEXT: s_cselect_b32 s6, 1, 0
; GFX12-NEXT: s_cmp_lg_u32 s6, 1
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_scc1 .LBB31_5
; GFX12-NEXT: ; %bb.4: ; %atomicrmw.private
; GFX12-NEXT: s_cmp_lg_u64 s[0:1], 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s0, s0, -1
; GFX12-NEXT: scratch_load_b64 v[0:1], off, s0
; GFX12-NEXT: s_wait_loadcnt 0x0
@@ -5123,9 +5179,10 @@ define amdgpu_kernel void @atomic_umax_i64_offset(ptr %out, i64 %in) {
; GFX12-NEXT: s_mov_b64 s[6:7], src_private_base
; GFX12-NEXT: s_wait_kmcnt 0x0
; GFX12-NEXT: s_add_nc_u64 s[0:1], s[0:1], 32
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_eq_u32 s1, s7
; GFX12-NEXT: s_cselect_b32 s4, 1, 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_lg_u32 s4, 1
; GFX12-NEXT: s_mov_b32 s4, -1
; GFX12-NEXT: s_cbranch_scc0 .LBB32_2
@@ -5139,11 +5196,12 @@ define amdgpu_kernel void @atomic_umax_i64_offset(ptr %out, i64 %in) {
; GFX12-NEXT: .LBB32_2: ; %Flow
; GFX12-NEXT: s_and_b32 s4, s4, exec_lo
; GFX12-NEXT: s_cselect_b32 s4, 1, 0
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_lg_u32 s4, 1
; GFX12-NEXT: s_cbranch_scc1 .LBB32_4
; GFX12-NEXT: ; %bb.3: ; %atomicrmw.private
; GFX12-NEXT: s_cmp_lg_u64 s[0:1], 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s0, s0, -1
; GFX12-NEXT: scratch_load_b64 v[0:1], off, s0
; GFX12-NEXT: s_wait_loadcnt 0x0
@@ -5285,9 +5343,10 @@ define amdgpu_kernel void @atomic_umax_i64_ret_offset(ptr %out, ptr %out2, i64 %
; GFX12-NEXT: s_mov_b64 s[6:7], src_private_base
; GFX12-NEXT: s_wait_kmcnt 0x0
; GFX12-NEXT: s_add_nc_u64 s[0:1], s[0:1], 32
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_eq_u32 s1, s7
; GFX12-NEXT: s_cselect_b32 s6, 1, 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_lg_u32 s6, 1
; GFX12-NEXT: s_cbranch_scc0 .LBB33_2
; GFX12-NEXT: ; %bb.1: ; %atomicrmw.global
@@ -5306,9 +5365,11 @@ define amdgpu_kernel void @atomic_umax_i64_ret_offset(ptr %out, ptr %out2, i64 %
; GFX12-NEXT: s_and_b32 s6, s6, exec_lo
; GFX12-NEXT: s_cselect_b32 s6, 1, 0
; GFX12-NEXT: s_cmp_lg_u32 s6, 1
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_scc1 .LBB33_5
; GFX12-NEXT: ; %bb.4: ; %atomicrmw.private
; GFX12-NEXT: s_cmp_lg_u64 s[0:1], 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s0, s0, -1
; GFX12-NEXT: scratch_load_b64 v[0:1], off, s0
; GFX12-NEXT: s_wait_loadcnt 0x0
@@ -5449,9 +5510,10 @@ define amdgpu_kernel void @atomic_umax_i64_addr64_offset(ptr %out, i64 %in, i64
; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_add_nc_u64 s[0:1], s[0:1], s[4:5]
; GFX12-NEXT: s_add_nc_u64 s[0:1], s[0:1], 32
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_eq_u32 s1, s7
; GFX12-NEXT: s_cselect_b32 s4, 1, 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_lg_u32 s4, 1
; GFX12-NEXT: s_mov_b32 s4, -1
; GFX12-NEXT: s_cbranch_scc0 .LBB34_2
@@ -5465,11 +5527,12 @@ define amdgpu_kernel void @atomic_umax_i64_addr64_offset(ptr %out, i64 %in, i64
; GFX12-NEXT: .LBB34_2: ; %Flow
; GFX12-NEXT: s_and_b32 s4, s4, exec_lo
; GFX12-NEXT: s_cselect_b32 s4, 1, 0
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_lg_u32 s4, 1
; GFX12-NEXT: s_cbranch_scc1 .LBB34_4
; GFX12-NEXT: ; %bb.3: ; %atomicrmw.private
; GFX12-NEXT: s_cmp_lg_u64 s[0:1], 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s0, s0, -1
; GFX12-NEXT: scratch_load_b64 v[0:1], off, s0
; GFX12-NEXT: s_wait_loadcnt 0x0
@@ -5617,9 +5680,10 @@ define amdgpu_kernel void @atomic_umax_i64_ret_addr64_offset(ptr %out, ptr %out2
; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_add_nc_u64 s[0:1], s[0:1], s[6:7]
; GFX12-NEXT: s_add_nc_u64 s[0:1], s[0:1], 32
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_eq_u32 s1, s9
; GFX12-NEXT: s_cselect_b32 s6, 1, 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_lg_u32 s6, 1
; GFX12-NEXT: s_cbranch_scc0 .LBB35_2
; GFX12-NEXT: ; %bb.1: ; %atomicrmw.global
@@ -5638,9 +5702,11 @@ define amdgpu_kernel void @atomic_umax_i64_ret_addr64_offset(ptr %out, ptr %out2
; GFX12-NEXT: s_and_b32 s6, s6, exec_lo
; GFX12-NEXT: s_cselect_b32 s6, 1, 0
; GFX12-NEXT: s_cmp_lg_u32 s6, 1
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_scc1 .LBB35_5
; GFX12-NEXT: ; %bb.4: ; %atomicrmw.private
; GFX12-NEXT: s_cmp_lg_u64 s[0:1], 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s0, s0, -1
; GFX12-NEXT: scratch_load_b64 v[0:1], off, s0
; GFX12-NEXT: s_wait_loadcnt 0x0
@@ -5764,8 +5830,8 @@ define amdgpu_kernel void @atomic_umax_i64(ptr %out, i64 %in) {
; GFX12-NEXT: s_mov_b64 s[6:7], src_private_base
; GFX12-NEXT: s_wait_kmcnt 0x0
; GFX12-NEXT: s_cmp_eq_u32 s1, s7
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s4, 1, 0
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_lg_u32 s4, 1
; GFX12-NEXT: s_mov_b32 s4, -1
; GFX12-NEXT: s_cbranch_scc0 .LBB36_2
@@ -5779,11 +5845,12 @@ define amdgpu_kernel void @atomic_umax_i64(ptr %out, i64 %in) {
; GFX12-NEXT: .LBB36_2: ; %Flow
; GFX12-NEXT: s_and_b32 s4, s4, exec_lo
; GFX12-NEXT: s_cselect_b32 s4, 1, 0
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_lg_u32 s4, 1
; GFX12-NEXT: s_cbranch_scc1 .LBB36_4
; GFX12-NEXT: ; %bb.3: ; %atomicrmw.private
; GFX12-NEXT: s_cmp_lg_u64 s[0:1], 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s0, s0, -1
; GFX12-NEXT: scratch_load_b64 v[0:1], off, s0
; GFX12-NEXT: s_wait_loadcnt 0x0
@@ -5920,9 +5987,10 @@ define amdgpu_kernel void @atomic_umax_i64_ret(ptr %out, ptr %out2, i64 %in) {
; GFX12-NEXT: s_mov_b64 s[6:7], src_private_base
; GFX12-NEXT: s_wait_kmcnt 0x0
; GFX12-NEXT: s_cmp_eq_u32 s1, s7
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s6, 1, 0
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_lg_u32 s6, 1
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_scc0 .LBB37_2
; GFX12-NEXT: ; %bb.1: ; %atomicrmw.global
; GFX12-NEXT: v_dual_mov_b32 v0, s0 :: v_dual_mov_b32 v1, s1
@@ -5940,9 +6008,11 @@ define amdgpu_kernel void @atomic_umax_i64_ret(ptr %out, ptr %out2, i64 %in) {
; GFX12-NEXT: s_and_b32 s6, s6, exec_lo
; GFX12-NEXT: s_cselect_b32 s6, 1, 0
; GFX12-NEXT: s_cmp_lg_u32 s6, 1
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_scc1 .LBB37_5
; GFX12-NEXT: ; %bb.4: ; %atomicrmw.private
; GFX12-NEXT: s_cmp_lg_u64 s[0:1], 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s0, s0, -1
; GFX12-NEXT: scratch_load_b64 v[0:1], off, s0
; GFX12-NEXT: s_wait_loadcnt 0x0
@@ -6077,8 +6147,8 @@ define amdgpu_kernel void @atomic_umax_i64_addr64(ptr %out, i64 %in, i64 %index)
; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_add_nc_u64 s[0:1], s[0:1], s[4:5]
; GFX12-NEXT: s_cmp_eq_u32 s1, s7
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s4, 1, 0
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_lg_u32 s4, 1
; GFX12-NEXT: s_mov_b32 s4, -1
; GFX12-NEXT: s_cbranch_scc0 .LBB38_2
@@ -6092,11 +6162,12 @@ define amdgpu_kernel void @atomic_umax_i64_addr64(ptr %out, i64 %in, i64 %index)
; GFX12-NEXT: .LBB38_2: ; %Flow
; GFX12-NEXT: s_and_b32 s4, s4, exec_lo
; GFX12-NEXT: s_cselect_b32 s4, 1, 0
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_lg_u32 s4, 1
; GFX12-NEXT: s_cbranch_scc1 .LBB38_4
; GFX12-NEXT: ; %bb.3: ; %atomicrmw.private
; GFX12-NEXT: s_cmp_lg_u64 s[0:1], 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s0, s0, -1
; GFX12-NEXT: scratch_load_b64 v[0:1], off, s0
; GFX12-NEXT: s_wait_loadcnt 0x0
@@ -6239,9 +6310,10 @@ define amdgpu_kernel void @atomic_umax_i64_ret_addr64(ptr %out, ptr %out2, i64 %
; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_add_nc_u64 s[0:1], s[0:1], s[6:7]
; GFX12-NEXT: s_cmp_eq_u32 s1, s9
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s6, 1, 0
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_lg_u32 s6, 1
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_scc0 .LBB39_2
; GFX12-NEXT: ; %bb.1: ; %atomicrmw.global
; GFX12-NEXT: v_dual_mov_b32 v0, s0 :: v_dual_mov_b32 v1, s1
@@ -6259,9 +6331,11 @@ define amdgpu_kernel void @atomic_umax_i64_ret_addr64(ptr %out, ptr %out2, i64 %
; GFX12-NEXT: s_and_b32 s6, s6, exec_lo
; GFX12-NEXT: s_cselect_b32 s6, 1, 0
; GFX12-NEXT: s_cmp_lg_u32 s6, 1
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_scc1 .LBB39_5
; GFX12-NEXT: ; %bb.4: ; %atomicrmw.private
; GFX12-NEXT: s_cmp_lg_u64 s[0:1], 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s0, s0, -1
; GFX12-NEXT: scratch_load_b64 v[0:1], off, s0
; GFX12-NEXT: s_wait_loadcnt 0x0
@@ -6388,9 +6462,10 @@ define amdgpu_kernel void @atomic_min_i64_offset(ptr %out, i64 %in) {
; GFX12-NEXT: s_mov_b64 s[6:7], src_private_base
; GFX12-NEXT: s_wait_kmcnt 0x0
; GFX12-NEXT: s_add_nc_u64 s[0:1], s[0:1], 32
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_eq_u32 s1, s7
; GFX12-NEXT: s_cselect_b32 s4, 1, 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_lg_u32 s4, 1
; GFX12-NEXT: s_mov_b32 s4, -1
; GFX12-NEXT: s_cbranch_scc0 .LBB40_2
@@ -6404,11 +6479,12 @@ define amdgpu_kernel void @atomic_min_i64_offset(ptr %out, i64 %in) {
; GFX12-NEXT: .LBB40_2: ; %Flow
; GFX12-NEXT: s_and_b32 s4, s4, exec_lo
; GFX12-NEXT: s_cselect_b32 s4, 1, 0
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_lg_u32 s4, 1
; GFX12-NEXT: s_cbranch_scc1 .LBB40_4
; GFX12-NEXT: ; %bb.3: ; %atomicrmw.private
; GFX12-NEXT: s_cmp_lg_u64 s[0:1], 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s0, s0, -1
; GFX12-NEXT: scratch_load_b64 v[0:1], off, s0
; GFX12-NEXT: s_wait_loadcnt 0x0
@@ -6550,9 +6626,10 @@ define amdgpu_kernel void @atomic_min_i64_ret_offset(ptr %out, ptr %out2, i64 %i
; GFX12-NEXT: s_mov_b64 s[6:7], src_private_base
; GFX12-NEXT: s_wait_kmcnt 0x0
; GFX12-NEXT: s_add_nc_u64 s[0:1], s[0:1], 32
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_eq_u32 s1, s7
; GFX12-NEXT: s_cselect_b32 s6, 1, 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_lg_u32 s6, 1
; GFX12-NEXT: s_cbranch_scc0 .LBB41_2
; GFX12-NEXT: ; %bb.1: ; %atomicrmw.global
@@ -6571,9 +6648,11 @@ define amdgpu_kernel void @atomic_min_i64_ret_offset(ptr %out, ptr %out2, i64 %i
; GFX12-NEXT: s_and_b32 s6, s6, exec_lo
; GFX12-NEXT: s_cselect_b32 s6, 1, 0
; GFX12-NEXT: s_cmp_lg_u32 s6, 1
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_scc1 .LBB41_5
; GFX12-NEXT: ; %bb.4: ; %atomicrmw.private
; GFX12-NEXT: s_cmp_lg_u64 s[0:1], 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s0, s0, -1
; GFX12-NEXT: scratch_load_b64 v[0:1], off, s0
; GFX12-NEXT: s_wait_loadcnt 0x0
@@ -6714,9 +6793,10 @@ define amdgpu_kernel void @atomic_min_i64_addr64_offset(ptr %out, i64 %in, i64 %
; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_add_nc_u64 s[0:1], s[0:1], s[4:5]
; GFX12-NEXT: s_add_nc_u64 s[0:1], s[0:1], 32
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_eq_u32 s1, s7
; GFX12-NEXT: s_cselect_b32 s4, 1, 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_lg_u32 s4, 1
; GFX12-NEXT: s_mov_b32 s4, -1
; GFX12-NEXT: s_cbranch_scc0 .LBB42_2
@@ -6730,11 +6810,12 @@ define amdgpu_kernel void @atomic_min_i64_addr64_offset(ptr %out, i64 %in, i64 %
; GFX12-NEXT: .LBB42_2: ; %Flow
; GFX12-NEXT: s_and_b32 s4, s4, exec_lo
; GFX12-NEXT: s_cselect_b32 s4, 1, 0
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_lg_u32 s4, 1
; GFX12-NEXT: s_cbranch_scc1 .LBB42_4
; GFX12-NEXT: ; %bb.3: ; %atomicrmw.private
; GFX12-NEXT: s_cmp_lg_u64 s[0:1], 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s0, s0, -1
; GFX12-NEXT: scratch_load_b64 v[0:1], off, s0
; GFX12-NEXT: s_wait_loadcnt 0x0
@@ -6882,9 +6963,10 @@ define amdgpu_kernel void @atomic_min_i64_ret_addr64_offset(ptr %out, ptr %out2,
; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_add_nc_u64 s[0:1], s[0:1], s[6:7]
; GFX12-NEXT: s_add_nc_u64 s[0:1], s[0:1], 32
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_eq_u32 s1, s9
; GFX12-NEXT: s_cselect_b32 s6, 1, 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_lg_u32 s6, 1
; GFX12-NEXT: s_cbranch_scc0 .LBB43_2
; GFX12-NEXT: ; %bb.1: ; %atomicrmw.global
@@ -6903,9 +6985,11 @@ define amdgpu_kernel void @atomic_min_i64_ret_addr64_offset(ptr %out, ptr %out2,
; GFX12-NEXT: s_and_b32 s6, s6, exec_lo
; GFX12-NEXT: s_cselect_b32 s6, 1, 0
; GFX12-NEXT: s_cmp_lg_u32 s6, 1
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_scc1 .LBB43_5
; GFX12-NEXT: ; %bb.4: ; %atomicrmw.private
; GFX12-NEXT: s_cmp_lg_u64 s[0:1], 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s0, s0, -1
; GFX12-NEXT: scratch_load_b64 v[0:1], off, s0
; GFX12-NEXT: s_wait_loadcnt 0x0
@@ -7029,8 +7113,8 @@ define amdgpu_kernel void @atomic_min_i64(ptr %out, i64 %in) {
; GFX12-NEXT: s_mov_b64 s[6:7], src_private_base
; GFX12-NEXT: s_wait_kmcnt 0x0
; GFX12-NEXT: s_cmp_eq_u32 s1, s7
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s4, 1, 0
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_lg_u32 s4, 1
; GFX12-NEXT: s_mov_b32 s4, -1
; GFX12-NEXT: s_cbranch_scc0 .LBB44_2
@@ -7044,11 +7128,12 @@ define amdgpu_kernel void @atomic_min_i64(ptr %out, i64 %in) {
; GFX12-NEXT: .LBB44_2: ; %Flow
; GFX12-NEXT: s_and_b32 s4, s4, exec_lo
; GFX12-NEXT: s_cselect_b32 s4, 1, 0
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_lg_u32 s4, 1
; GFX12-NEXT: s_cbranch_scc1 .LBB44_4
; GFX12-NEXT: ; %bb.3: ; %atomicrmw.private
; GFX12-NEXT: s_cmp_lg_u64 s[0:1], 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s0, s0, -1
; GFX12-NEXT: scratch_load_b64 v[0:1], off, s0
; GFX12-NEXT: s_wait_loadcnt 0x0
@@ -7185,9 +7270,10 @@ define amdgpu_kernel void @atomic_min_i64_ret(ptr %out, ptr %out2, i64 %in) {
; GFX12-NEXT: s_mov_b64 s[6:7], src_private_base
; GFX12-NEXT: s_wait_kmcnt 0x0
; GFX12-NEXT: s_cmp_eq_u32 s1, s7
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s6, 1, 0
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_lg_u32 s6, 1
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_scc0 .LBB45_2
; GFX12-NEXT: ; %bb.1: ; %atomicrmw.global
; GFX12-NEXT: v_dual_mov_b32 v0, s0 :: v_dual_mov_b32 v1, s1
@@ -7205,9 +7291,11 @@ define amdgpu_kernel void @atomic_min_i64_ret(ptr %out, ptr %out2, i64 %in) {
; GFX12-NEXT: s_and_b32 s6, s6, exec_lo
; GFX12-NEXT: s_cselect_b32 s6, 1, 0
; GFX12-NEXT: s_cmp_lg_u32 s6, 1
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_scc1 .LBB45_5
; GFX12-NEXT: ; %bb.4: ; %atomicrmw.private
; GFX12-NEXT: s_cmp_lg_u64 s[0:1], 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s0, s0, -1
; GFX12-NEXT: scratch_load_b64 v[0:1], off, s0
; GFX12-NEXT: s_wait_loadcnt 0x0
@@ -7342,8 +7430,8 @@ define amdgpu_kernel void @atomic_min_i64_addr64(ptr %out, i64 %in, i64 %index)
; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_add_nc_u64 s[0:1], s[0:1], s[4:5]
; GFX12-NEXT: s_cmp_eq_u32 s1, s7
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s4, 1, 0
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_lg_u32 s4, 1
; GFX12-NEXT: s_mov_b32 s4, -1
; GFX12-NEXT: s_cbranch_scc0 .LBB46_2
@@ -7357,11 +7445,12 @@ define amdgpu_kernel void @atomic_min_i64_addr64(ptr %out, i64 %in, i64 %index)
; GFX12-NEXT: .LBB46_2: ; %Flow
; GFX12-NEXT: s_and_b32 s4, s4, exec_lo
; GFX12-NEXT: s_cselect_b32 s4, 1, 0
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_lg_u32 s4, 1
; GFX12-NEXT: s_cbranch_scc1 .LBB46_4
; GFX12-NEXT: ; %bb.3: ; %atomicrmw.private
; GFX12-NEXT: s_cmp_lg_u64 s[0:1], 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s0, s0, -1
; GFX12-NEXT: scratch_load_b64 v[0:1], off, s0
; GFX12-NEXT: s_wait_loadcnt 0x0
@@ -7504,9 +7593,10 @@ define amdgpu_kernel void @atomic_min_i64_ret_addr64(ptr %out, ptr %out2, i64 %i
; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_add_nc_u64 s[0:1], s[0:1], s[6:7]
; GFX12-NEXT: s_cmp_eq_u32 s1, s9
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s6, 1, 0
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_lg_u32 s6, 1
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_scc0 .LBB47_2
; GFX12-NEXT: ; %bb.1: ; %atomicrmw.global
; GFX12-NEXT: v_dual_mov_b32 v0, s0 :: v_dual_mov_b32 v1, s1
@@ -7524,9 +7614,11 @@ define amdgpu_kernel void @atomic_min_i64_ret_addr64(ptr %out, ptr %out2, i64 %i
; GFX12-NEXT: s_and_b32 s6, s6, exec_lo
; GFX12-NEXT: s_cselect_b32 s6, 1, 0
; GFX12-NEXT: s_cmp_lg_u32 s6, 1
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_scc1 .LBB47_5
; GFX12-NEXT: ; %bb.4: ; %atomicrmw.private
; GFX12-NEXT: s_cmp_lg_u64 s[0:1], 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s0, s0, -1
; GFX12-NEXT: scratch_load_b64 v[0:1], off, s0
; GFX12-NEXT: s_wait_loadcnt 0x0
@@ -7653,9 +7745,10 @@ define amdgpu_kernel void @atomic_umin_i64_offset(ptr %out, i64 %in) {
; GFX12-NEXT: s_mov_b64 s[6:7], src_private_base
; GFX12-NEXT: s_wait_kmcnt 0x0
; GFX12-NEXT: s_add_nc_u64 s[0:1], s[0:1], 32
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_eq_u32 s1, s7
; GFX12-NEXT: s_cselect_b32 s4, 1, 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_lg_u32 s4, 1
; GFX12-NEXT: s_mov_b32 s4, -1
; GFX12-NEXT: s_cbranch_scc0 .LBB48_2
@@ -7669,11 +7762,12 @@ define amdgpu_kernel void @atomic_umin_i64_offset(ptr %out, i64 %in) {
; GFX12-NEXT: .LBB48_2: ; %Flow
; GFX12-NEXT: s_and_b32 s4, s4, exec_lo
; GFX12-NEXT: s_cselect_b32 s4, 1, 0
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_lg_u32 s4, 1
; GFX12-NEXT: s_cbranch_scc1 .LBB48_4
; GFX12-NEXT: ; %bb.3: ; %atomicrmw.private
; GFX12-NEXT: s_cmp_lg_u64 s[0:1], 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s0, s0, -1
; GFX12-NEXT: scratch_load_b64 v[0:1], off, s0
; GFX12-NEXT: s_wait_loadcnt 0x0
@@ -7815,9 +7909,10 @@ define amdgpu_kernel void @atomic_umin_i64_ret_offset(ptr %out, ptr %out2, i64 %
; GFX12-NEXT: s_mov_b64 s[6:7], src_private_base
; GFX12-NEXT: s_wait_kmcnt 0x0
; GFX12-NEXT: s_add_nc_u64 s[0:1], s[0:1], 32
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_eq_u32 s1, s7
; GFX12-NEXT: s_cselect_b32 s6, 1, 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_lg_u32 s6, 1
; GFX12-NEXT: s_cbranch_scc0 .LBB49_2
; GFX12-NEXT: ; %bb.1: ; %atomicrmw.global
@@ -7836,9 +7931,11 @@ define amdgpu_kernel void @atomic_umin_i64_ret_offset(ptr %out, ptr %out2, i64 %
; GFX12-NEXT: s_and_b32 s6, s6, exec_lo
; GFX12-NEXT: s_cselect_b32 s6, 1, 0
; GFX12-NEXT: s_cmp_lg_u32 s6, 1
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_scc1 .LBB49_5
; GFX12-NEXT: ; %bb.4: ; %atomicrmw.private
; GFX12-NEXT: s_cmp_lg_u64 s[0:1], 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s0, s0, -1
; GFX12-NEXT: scratch_load_b64 v[0:1], off, s0
; GFX12-NEXT: s_wait_loadcnt 0x0
@@ -7979,9 +8076,10 @@ define amdgpu_kernel void @atomic_umin_i64_addr64_offset(ptr %out, i64 %in, i64
; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_add_nc_u64 s[0:1], s[0:1], s[4:5]
; GFX12-NEXT: s_add_nc_u64 s[0:1], s[0:1], 32
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_eq_u32 s1, s7
; GFX12-NEXT: s_cselect_b32 s4, 1, 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_lg_u32 s4, 1
; GFX12-NEXT: s_mov_b32 s4, -1
; GFX12-NEXT: s_cbranch_scc0 .LBB50_2
@@ -7995,11 +8093,12 @@ define amdgpu_kernel void @atomic_umin_i64_addr64_offset(ptr %out, i64 %in, i64
; GFX12-NEXT: .LBB50_2: ; %Flow
; GFX12-NEXT: s_and_b32 s4, s4, exec_lo
; GFX12-NEXT: s_cselect_b32 s4, 1, 0
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_lg_u32 s4, 1
; GFX12-NEXT: s_cbranch_scc1 .LBB50_4
; GFX12-NEXT: ; %bb.3: ; %atomicrmw.private
; GFX12-NEXT: s_cmp_lg_u64 s[0:1], 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s0, s0, -1
; GFX12-NEXT: scratch_load_b64 v[0:1], off, s0
; GFX12-NEXT: s_wait_loadcnt 0x0
@@ -8147,9 +8246,10 @@ define amdgpu_kernel void @atomic_umin_i64_ret_addr64_offset(ptr %out, ptr %out2
; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_add_nc_u64 s[0:1], s[0:1], s[6:7]
; GFX12-NEXT: s_add_nc_u64 s[0:1], s[0:1], 32
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_eq_u32 s1, s9
; GFX12-NEXT: s_cselect_b32 s6, 1, 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_lg_u32 s6, 1
; GFX12-NEXT: s_cbranch_scc0 .LBB51_2
; GFX12-NEXT: ; %bb.1: ; %atomicrmw.global
@@ -8168,9 +8268,11 @@ define amdgpu_kernel void @atomic_umin_i64_ret_addr64_offset(ptr %out, ptr %out2
; GFX12-NEXT: s_and_b32 s6, s6, exec_lo
; GFX12-NEXT: s_cselect_b32 s6, 1, 0
; GFX12-NEXT: s_cmp_lg_u32 s6, 1
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_scc1 .LBB51_5
; GFX12-NEXT: ; %bb.4: ; %atomicrmw.private
; GFX12-NEXT: s_cmp_lg_u64 s[0:1], 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s0, s0, -1
; GFX12-NEXT: scratch_load_b64 v[0:1], off, s0
; GFX12-NEXT: s_wait_loadcnt 0x0
@@ -8294,8 +8396,8 @@ define amdgpu_kernel void @atomic_umin_i64(ptr %out, i64 %in) {
; GFX12-NEXT: s_mov_b64 s[6:7], src_private_base
; GFX12-NEXT: s_wait_kmcnt 0x0
; GFX12-NEXT: s_cmp_eq_u32 s1, s7
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s4, 1, 0
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_lg_u32 s4, 1
; GFX12-NEXT: s_mov_b32 s4, -1
; GFX12-NEXT: s_cbranch_scc0 .LBB52_2
@@ -8309,11 +8411,12 @@ define amdgpu_kernel void @atomic_umin_i64(ptr %out, i64 %in) {
; GFX12-NEXT: .LBB52_2: ; %Flow
; GFX12-NEXT: s_and_b32 s4, s4, exec_lo
; GFX12-NEXT: s_cselect_b32 s4, 1, 0
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_lg_u32 s4, 1
; GFX12-NEXT: s_cbranch_scc1 .LBB52_4
; GFX12-NEXT: ; %bb.3: ; %atomicrmw.private
; GFX12-NEXT: s_cmp_lg_u64 s[0:1], 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s0, s0, -1
; GFX12-NEXT: scratch_load_b64 v[0:1], off, s0
; GFX12-NEXT: s_wait_loadcnt 0x0
@@ -8450,9 +8553,10 @@ define amdgpu_kernel void @atomic_umin_i64_ret(ptr %out, ptr %out2, i64 %in) {
; GFX12-NEXT: s_mov_b64 s[6:7], src_private_base
; GFX12-NEXT: s_wait_kmcnt 0x0
; GFX12-NEXT: s_cmp_eq_u32 s1, s7
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s6, 1, 0
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_lg_u32 s6, 1
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_scc0 .LBB53_2
; GFX12-NEXT: ; %bb.1: ; %atomicrmw.global
; GFX12-NEXT: v_dual_mov_b32 v0, s0 :: v_dual_mov_b32 v1, s1
@@ -8470,9 +8574,11 @@ define amdgpu_kernel void @atomic_umin_i64_ret(ptr %out, ptr %out2, i64 %in) {
; GFX12-NEXT: s_and_b32 s6, s6, exec_lo
; GFX12-NEXT: s_cselect_b32 s6, 1, 0
; GFX12-NEXT: s_cmp_lg_u32 s6, 1
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_scc1 .LBB53_5
; GFX12-NEXT: ; %bb.4: ; %atomicrmw.private
; GFX12-NEXT: s_cmp_lg_u64 s[0:1], 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s0, s0, -1
; GFX12-NEXT: scratch_load_b64 v[0:1], off, s0
; GFX12-NEXT: s_wait_loadcnt 0x0
@@ -8607,8 +8713,8 @@ define amdgpu_kernel void @atomic_umin_i64_addr64(ptr %out, i64 %in, i64 %index)
; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_add_nc_u64 s[0:1], s[0:1], s[4:5]
; GFX12-NEXT: s_cmp_eq_u32 s1, s7
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s4, 1, 0
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_lg_u32 s4, 1
; GFX12-NEXT: s_mov_b32 s4, -1
; GFX12-NEXT: s_cbranch_scc0 .LBB54_2
@@ -8622,11 +8728,12 @@ define amdgpu_kernel void @atomic_umin_i64_addr64(ptr %out, i64 %in, i64 %index)
; GFX12-NEXT: .LBB54_2: ; %Flow
; GFX12-NEXT: s_and_b32 s4, s4, exec_lo
; GFX12-NEXT: s_cselect_b32 s4, 1, 0
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_lg_u32 s4, 1
; GFX12-NEXT: s_cbranch_scc1 .LBB54_4
; GFX12-NEXT: ; %bb.3: ; %atomicrmw.private
; GFX12-NEXT: s_cmp_lg_u64 s[0:1], 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s0, s0, -1
; GFX12-NEXT: scratch_load_b64 v[0:1], off, s0
; GFX12-NEXT: s_wait_loadcnt 0x0
@@ -8769,9 +8876,10 @@ define amdgpu_kernel void @atomic_umin_i64_ret_addr64(ptr %out, ptr %out2, i64 %
; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_add_nc_u64 s[0:1], s[0:1], s[6:7]
; GFX12-NEXT: s_cmp_eq_u32 s1, s9
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s6, 1, 0
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_lg_u32 s6, 1
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_scc0 .LBB55_2
; GFX12-NEXT: ; %bb.1: ; %atomicrmw.global
; GFX12-NEXT: v_dual_mov_b32 v0, s0 :: v_dual_mov_b32 v1, s1
@@ -8789,9 +8897,11 @@ define amdgpu_kernel void @atomic_umin_i64_ret_addr64(ptr %out, ptr %out2, i64 %
; GFX12-NEXT: s_and_b32 s6, s6, exec_lo
; GFX12-NEXT: s_cselect_b32 s6, 1, 0
; GFX12-NEXT: s_cmp_lg_u32 s6, 1
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_scc1 .LBB55_5
; GFX12-NEXT: ; %bb.4: ; %atomicrmw.private
; GFX12-NEXT: s_cmp_lg_u64 s[0:1], 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s0, s0, -1
; GFX12-NEXT: scratch_load_b64 v[0:1], off, s0
; GFX12-NEXT: s_wait_loadcnt 0x0
@@ -8916,9 +9026,10 @@ define amdgpu_kernel void @atomic_or_i64_offset(ptr %out, i64 %in) {
; GFX12-NEXT: s_mov_b64 s[6:7], src_private_base
; GFX12-NEXT: s_wait_kmcnt 0x0
; GFX12-NEXT: s_add_nc_u64 s[0:1], s[0:1], 32
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_eq_u32 s1, s7
; GFX12-NEXT: s_cselect_b32 s4, 1, 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_lg_u32 s4, 1
; GFX12-NEXT: s_mov_b32 s4, -1
; GFX12-NEXT: s_cbranch_scc0 .LBB56_2
@@ -8932,11 +9043,12 @@ define amdgpu_kernel void @atomic_or_i64_offset(ptr %out, i64 %in) {
; GFX12-NEXT: .LBB56_2: ; %Flow
; GFX12-NEXT: s_and_b32 s4, s4, exec_lo
; GFX12-NEXT: s_cselect_b32 s4, 1, 0
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_lg_u32 s4, 1
; GFX12-NEXT: s_cbranch_scc1 .LBB56_4
; GFX12-NEXT: ; %bb.3: ; %atomicrmw.private
; GFX12-NEXT: s_cmp_lg_u64 s[0:1], 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s0, s0, -1
; GFX12-NEXT: scratch_load_b64 v[0:1], off, s0
; GFX12-NEXT: s_wait_loadcnt 0x0
@@ -9073,9 +9185,10 @@ define amdgpu_kernel void @atomic_or_i64_ret_offset(ptr %out, ptr %out2, i64 %in
; GFX12-NEXT: s_mov_b64 s[6:7], src_private_base
; GFX12-NEXT: s_wait_kmcnt 0x0
; GFX12-NEXT: s_add_nc_u64 s[0:1], s[0:1], 32
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_eq_u32 s1, s7
; GFX12-NEXT: s_cselect_b32 s6, 1, 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_lg_u32 s6, 1
; GFX12-NEXT: s_cbranch_scc0 .LBB57_2
; GFX12-NEXT: ; %bb.1: ; %atomicrmw.global
@@ -9094,9 +9207,11 @@ define amdgpu_kernel void @atomic_or_i64_ret_offset(ptr %out, ptr %out2, i64 %in
; GFX12-NEXT: s_and_b32 s6, s6, exec_lo
; GFX12-NEXT: s_cselect_b32 s6, 1, 0
; GFX12-NEXT: s_cmp_lg_u32 s6, 1
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_scc1 .LBB57_5
; GFX12-NEXT: ; %bb.4: ; %atomicrmw.private
; GFX12-NEXT: s_cmp_lg_u64 s[0:1], 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s0, s0, -1
; GFX12-NEXT: scratch_load_b64 v[0:1], off, s0
; GFX12-NEXT: s_wait_loadcnt 0x0
@@ -9234,9 +9349,10 @@ define amdgpu_kernel void @atomic_or_i64_addr64_offset(ptr %out, i64 %in, i64 %i
; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_add_nc_u64 s[0:1], s[0:1], s[4:5]
; GFX12-NEXT: s_add_nc_u64 s[0:1], s[0:1], 32
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_eq_u32 s1, s7
; GFX12-NEXT: s_cselect_b32 s4, 1, 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_lg_u32 s4, 1
; GFX12-NEXT: s_mov_b32 s4, -1
; GFX12-NEXT: s_cbranch_scc0 .LBB58_2
@@ -9250,11 +9366,12 @@ define amdgpu_kernel void @atomic_or_i64_addr64_offset(ptr %out, i64 %in, i64 %i
; GFX12-NEXT: .LBB58_2: ; %Flow
; GFX12-NEXT: s_and_b32 s4, s4, exec_lo
; GFX12-NEXT: s_cselect_b32 s4, 1, 0
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_lg_u32 s4, 1
; GFX12-NEXT: s_cbranch_scc1 .LBB58_4
; GFX12-NEXT: ; %bb.3: ; %atomicrmw.private
; GFX12-NEXT: s_cmp_lg_u64 s[0:1], 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s0, s0, -1
; GFX12-NEXT: scratch_load_b64 v[0:1], off, s0
; GFX12-NEXT: s_wait_loadcnt 0x0
@@ -9397,9 +9514,10 @@ define amdgpu_kernel void @atomic_or_i64_ret_addr64_offset(ptr %out, ptr %out2,
; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_add_nc_u64 s[0:1], s[0:1], s[6:7]
; GFX12-NEXT: s_add_nc_u64 s[0:1], s[0:1], 32
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_eq_u32 s1, s9
; GFX12-NEXT: s_cselect_b32 s6, 1, 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_lg_u32 s6, 1
; GFX12-NEXT: s_cbranch_scc0 .LBB59_2
; GFX12-NEXT: ; %bb.1: ; %atomicrmw.global
@@ -9418,9 +9536,11 @@ define amdgpu_kernel void @atomic_or_i64_ret_addr64_offset(ptr %out, ptr %out2,
; GFX12-NEXT: s_and_b32 s6, s6, exec_lo
; GFX12-NEXT: s_cselect_b32 s6, 1, 0
; GFX12-NEXT: s_cmp_lg_u32 s6, 1
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_scc1 .LBB59_5
; GFX12-NEXT: ; %bb.4: ; %atomicrmw.private
; GFX12-NEXT: s_cmp_lg_u64 s[0:1], 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s0, s0, -1
; GFX12-NEXT: scratch_load_b64 v[0:1], off, s0
; GFX12-NEXT: s_wait_loadcnt 0x0
@@ -9541,8 +9661,8 @@ define amdgpu_kernel void @atomic_or_i64(ptr %out, i64 %in) {
; GFX12-NEXT: s_mov_b64 s[6:7], src_private_base
; GFX12-NEXT: s_wait_kmcnt 0x0
; GFX12-NEXT: s_cmp_eq_u32 s1, s7
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s4, 1, 0
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_lg_u32 s4, 1
; GFX12-NEXT: s_mov_b32 s4, -1
; GFX12-NEXT: s_cbranch_scc0 .LBB60_2
@@ -9556,11 +9676,12 @@ define amdgpu_kernel void @atomic_or_i64(ptr %out, i64 %in) {
; GFX12-NEXT: .LBB60_2: ; %Flow
; GFX12-NEXT: s_and_b32 s4, s4, exec_lo
; GFX12-NEXT: s_cselect_b32 s4, 1, 0
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_lg_u32 s4, 1
; GFX12-NEXT: s_cbranch_scc1 .LBB60_4
; GFX12-NEXT: ; %bb.3: ; %atomicrmw.private
; GFX12-NEXT: s_cmp_lg_u64 s[0:1], 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s0, s0, -1
; GFX12-NEXT: scratch_load_b64 v[0:1], off, s0
; GFX12-NEXT: s_wait_loadcnt 0x0
@@ -9692,9 +9813,10 @@ define amdgpu_kernel void @atomic_or_i64_ret(ptr %out, ptr %out2, i64 %in) {
; GFX12-NEXT: s_mov_b64 s[6:7], src_private_base
; GFX12-NEXT: s_wait_kmcnt 0x0
; GFX12-NEXT: s_cmp_eq_u32 s1, s7
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s6, 1, 0
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_lg_u32 s6, 1
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_scc0 .LBB61_2
; GFX12-NEXT: ; %bb.1: ; %atomicrmw.global
; GFX12-NEXT: v_dual_mov_b32 v0, s0 :: v_dual_mov_b32 v1, s1
@@ -9712,9 +9834,11 @@ define amdgpu_kernel void @atomic_or_i64_ret(ptr %out, ptr %out2, i64 %in) {
; GFX12-NEXT: s_and_b32 s6, s6, exec_lo
; GFX12-NEXT: s_cselect_b32 s6, 1, 0
; GFX12-NEXT: s_cmp_lg_u32 s6, 1
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_scc1 .LBB61_5
; GFX12-NEXT: ; %bb.4: ; %atomicrmw.private
; GFX12-NEXT: s_cmp_lg_u64 s[0:1], 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s0, s0, -1
; GFX12-NEXT: scratch_load_b64 v[0:1], off, s0
; GFX12-NEXT: s_wait_loadcnt 0x0
@@ -9846,8 +9970,8 @@ define amdgpu_kernel void @atomic_or_i64_addr64(ptr %out, i64 %in, i64 %index) {
; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_add_nc_u64 s[0:1], s[0:1], s[4:5]
; GFX12-NEXT: s_cmp_eq_u32 s1, s7
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s4, 1, 0
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_lg_u32 s4, 1
; GFX12-NEXT: s_mov_b32 s4, -1
; GFX12-NEXT: s_cbranch_scc0 .LBB62_2
@@ -9861,11 +9985,12 @@ define amdgpu_kernel void @atomic_or_i64_addr64(ptr %out, i64 %in, i64 %index) {
; GFX12-NEXT: .LBB62_2: ; %Flow
; GFX12-NEXT: s_and_b32 s4, s4, exec_lo
; GFX12-NEXT: s_cselect_b32 s4, 1, 0
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_lg_u32 s4, 1
; GFX12-NEXT: s_cbranch_scc1 .LBB62_4
; GFX12-NEXT: ; %bb.3: ; %atomicrmw.private
; GFX12-NEXT: s_cmp_lg_u64 s[0:1], 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s0, s0, -1
; GFX12-NEXT: scratch_load_b64 v[0:1], off, s0
; GFX12-NEXT: s_wait_loadcnt 0x0
@@ -10003,9 +10128,10 @@ define amdgpu_kernel void @atomic_or_i64_ret_addr64(ptr %out, ptr %out2, i64 %in
; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_add_nc_u64 s[0:1], s[0:1], s[6:7]
; GFX12-NEXT: s_cmp_eq_u32 s1, s9
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s6, 1, 0
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_lg_u32 s6, 1
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_scc0 .LBB63_2
; GFX12-NEXT: ; %bb.1: ; %atomicrmw.global
; GFX12-NEXT: v_dual_mov_b32 v0, s0 :: v_dual_mov_b32 v1, s1
@@ -10023,9 +10149,11 @@ define amdgpu_kernel void @atomic_or_i64_ret_addr64(ptr %out, ptr %out2, i64 %in
; GFX12-NEXT: s_and_b32 s6, s6, exec_lo
; GFX12-NEXT: s_cselect_b32 s6, 1, 0
; GFX12-NEXT: s_cmp_lg_u32 s6, 1
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_scc1 .LBB63_5
; GFX12-NEXT: ; %bb.4: ; %atomicrmw.private
; GFX12-NEXT: s_cmp_lg_u64 s[0:1], 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s0, s0, -1
; GFX12-NEXT: scratch_load_b64 v[0:1], off, s0
; GFX12-NEXT: s_wait_loadcnt 0x0
@@ -10141,9 +10269,10 @@ define amdgpu_kernel void @atomic_xchg_i64_offset(ptr %out, i64 %in) {
; GFX12-NEXT: s_mov_b64 s[6:7], src_private_base
; GFX12-NEXT: s_wait_kmcnt 0x0
; GFX12-NEXT: s_add_nc_u64 s[0:1], s[0:1], 32
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_eq_u32 s1, s7
; GFX12-NEXT: s_cselect_b32 s4, 1, 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_lg_u32 s4, 1
; GFX12-NEXT: s_mov_b32 s4, -1
; GFX12-NEXT: s_cbranch_scc0 .LBB64_2
@@ -10157,12 +10286,13 @@ define amdgpu_kernel void @atomic_xchg_i64_offset(ptr %out, i64 %in) {
; GFX12-NEXT: .LBB64_2: ; %Flow
; GFX12-NEXT: s_and_b32 s4, s4, exec_lo
; GFX12-NEXT: s_cselect_b32 s4, 1, 0
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_lg_u32 s4, 1
; GFX12-NEXT: s_cbranch_scc1 .LBB64_4
; GFX12-NEXT: ; %bb.3: ; %atomicrmw.private
; GFX12-NEXT: v_dual_mov_b32 v0, s2 :: v_dual_mov_b32 v1, s3
; GFX12-NEXT: s_cmp_lg_u64 s[0:1], 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s0, s0, -1
; GFX12-NEXT: scratch_store_b64 off, v[0:1], s0
; GFX12-NEXT: .LBB64_4: ; %atomicrmw.phi
@@ -10271,9 +10401,10 @@ define amdgpu_kernel void @atomic_xchg_f64_offset(ptr %out, double %in) {
; GFX12-NEXT: s_mov_b64 s[6:7], src_private_base
; GFX12-NEXT: s_wait_kmcnt 0x0
; GFX12-NEXT: s_add_nc_u64 s[0:1], s[0:1], 32
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_eq_u32 s1, s7
; GFX12-NEXT: s_cselect_b32 s4, 1, 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_lg_u32 s4, 1
; GFX12-NEXT: s_mov_b32 s4, -1
; GFX12-NEXT: s_cbranch_scc0 .LBB65_2
@@ -10287,12 +10418,13 @@ define amdgpu_kernel void @atomic_xchg_f64_offset(ptr %out, double %in) {
; GFX12-NEXT: .LBB65_2: ; %Flow
; GFX12-NEXT: s_and_b32 s4, s4, exec_lo
; GFX12-NEXT: s_cselect_b32 s4, 1, 0
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_lg_u32 s4, 1
; GFX12-NEXT: s_cbranch_scc1 .LBB65_4
; GFX12-NEXT: ; %bb.3: ; %atomicrmw.private
; GFX12-NEXT: v_dual_mov_b32 v0, s2 :: v_dual_mov_b32 v1, s3
; GFX12-NEXT: s_cmp_lg_u64 s[0:1], 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s0, s0, -1
; GFX12-NEXT: scratch_store_b64 off, v[0:1], s0
; GFX12-NEXT: .LBB65_4: ; %atomicrmw.phi
@@ -10401,9 +10533,10 @@ define amdgpu_kernel void @atomic_xchg_pointer_offset(ptr %out, ptr %in) {
; GFX12-NEXT: s_mov_b64 s[6:7], src_private_base
; GFX12-NEXT: s_wait_kmcnt 0x0
; GFX12-NEXT: s_add_nc_u64 s[0:1], s[0:1], 32
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_eq_u32 s1, s7
; GFX12-NEXT: s_cselect_b32 s4, 1, 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_lg_u32 s4, 1
; GFX12-NEXT: s_mov_b32 s4, -1
; GFX12-NEXT: s_cbranch_scc0 .LBB66_2
@@ -10417,12 +10550,13 @@ define amdgpu_kernel void @atomic_xchg_pointer_offset(ptr %out, ptr %in) {
; GFX12-NEXT: .LBB66_2: ; %Flow
; GFX12-NEXT: s_and_b32 s4, s4, exec_lo
; GFX12-NEXT: s_cselect_b32 s4, 1, 0
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_lg_u32 s4, 1
; GFX12-NEXT: s_cbranch_scc1 .LBB66_4
; GFX12-NEXT: ; %bb.3: ; %atomicrmw.private
; GFX12-NEXT: v_dual_mov_b32 v0, s2 :: v_dual_mov_b32 v1, s3
; GFX12-NEXT: s_cmp_lg_u64 s[0:1], 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s0, s0, -1
; GFX12-NEXT: scratch_store_b64 off, v[0:1], s0
; GFX12-NEXT: .LBB66_4: ; %atomicrmw.phi
@@ -10553,9 +10687,10 @@ define amdgpu_kernel void @atomic_xchg_i64_ret_offset(ptr %out, ptr %out2, i64 %
; GFX12-NEXT: s_mov_b64 s[6:7], src_private_base
; GFX12-NEXT: s_wait_kmcnt 0x0
; GFX12-NEXT: s_add_nc_u64 s[0:1], s[0:1], 32
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_eq_u32 s1, s7
; GFX12-NEXT: s_cselect_b32 s6, 1, 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_lg_u32 s6, 1
; GFX12-NEXT: s_cbranch_scc0 .LBB67_2
; GFX12-NEXT: ; %bb.1: ; %atomicrmw.global
@@ -10574,6 +10709,7 @@ define amdgpu_kernel void @atomic_xchg_i64_ret_offset(ptr %out, ptr %out2, i64 %
; GFX12-NEXT: s_and_b32 s6, s6, exec_lo
; GFX12-NEXT: s_cselect_b32 s6, 1, 0
; GFX12-NEXT: s_cmp_lg_u32 s6, 1
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_scc1 .LBB67_5
; GFX12-NEXT: ; %bb.4: ; %atomicrmw.private
; GFX12-NEXT: s_cmp_lg_u64 s[0:1], 0
@@ -10705,9 +10841,10 @@ define amdgpu_kernel void @atomic_xchg_i64_addr64_offset(ptr %out, i64 %in, i64
; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_add_nc_u64 s[0:1], s[0:1], s[4:5]
; GFX12-NEXT: s_add_nc_u64 s[0:1], s[0:1], 32
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_eq_u32 s1, s7
; GFX12-NEXT: s_cselect_b32 s4, 1, 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_lg_u32 s4, 1
; GFX12-NEXT: s_mov_b32 s4, -1
; GFX12-NEXT: s_cbranch_scc0 .LBB68_2
@@ -10721,12 +10858,13 @@ define amdgpu_kernel void @atomic_xchg_i64_addr64_offset(ptr %out, i64 %in, i64
; GFX12-NEXT: .LBB68_2: ; %Flow
; GFX12-NEXT: s_and_b32 s4, s4, exec_lo
; GFX12-NEXT: s_cselect_b32 s4, 1, 0
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_lg_u32 s4, 1
; GFX12-NEXT: s_cbranch_scc1 .LBB68_4
; GFX12-NEXT: ; %bb.3: ; %atomicrmw.private
; GFX12-NEXT: v_dual_mov_b32 v0, s2 :: v_dual_mov_b32 v1, s3
; GFX12-NEXT: s_cmp_lg_u64 s[0:1], 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s0, s0, -1
; GFX12-NEXT: scratch_store_b64 off, v[0:1], s0
; GFX12-NEXT: .LBB68_4: ; %atomicrmw.phi
@@ -10863,9 +11001,10 @@ define amdgpu_kernel void @atomic_xchg_i64_ret_addr64_offset(ptr %out, ptr %out2
; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_add_nc_u64 s[0:1], s[0:1], s[6:7]
; GFX12-NEXT: s_add_nc_u64 s[0:1], s[0:1], 32
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_eq_u32 s1, s9
; GFX12-NEXT: s_cselect_b32 s6, 1, 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_lg_u32 s6, 1
; GFX12-NEXT: s_cbranch_scc0 .LBB69_2
; GFX12-NEXT: ; %bb.1: ; %atomicrmw.global
@@ -10884,6 +11023,7 @@ define amdgpu_kernel void @atomic_xchg_i64_ret_addr64_offset(ptr %out, ptr %out2
; GFX12-NEXT: s_and_b32 s6, s6, exec_lo
; GFX12-NEXT: s_cselect_b32 s6, 1, 0
; GFX12-NEXT: s_cmp_lg_u32 s6, 1
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_scc1 .LBB69_5
; GFX12-NEXT: ; %bb.4: ; %atomicrmw.private
; GFX12-NEXT: s_cmp_lg_u64 s[0:1], 0
@@ -10998,8 +11138,8 @@ define amdgpu_kernel void @atomic_xchg_i64(ptr %out, i64 %in) {
; GFX12-NEXT: s_mov_b64 s[6:7], src_private_base
; GFX12-NEXT: s_wait_kmcnt 0x0
; GFX12-NEXT: s_cmp_eq_u32 s1, s7
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s4, 1, 0
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_lg_u32 s4, 1
; GFX12-NEXT: s_mov_b32 s4, -1
; GFX12-NEXT: s_cbranch_scc0 .LBB70_2
@@ -11013,12 +11153,13 @@ define amdgpu_kernel void @atomic_xchg_i64(ptr %out, i64 %in) {
; GFX12-NEXT: .LBB70_2: ; %Flow
; GFX12-NEXT: s_and_b32 s4, s4, exec_lo
; GFX12-NEXT: s_cselect_b32 s4, 1, 0
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_lg_u32 s4, 1
; GFX12-NEXT: s_cbranch_scc1 .LBB70_4
; GFX12-NEXT: ; %bb.3: ; %atomicrmw.private
; GFX12-NEXT: v_dual_mov_b32 v0, s2 :: v_dual_mov_b32 v1, s3
; GFX12-NEXT: s_cmp_lg_u64 s[0:1], 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s0, s0, -1
; GFX12-NEXT: scratch_store_b64 off, v[0:1], s0
; GFX12-NEXT: .LBB70_4: ; %atomicrmw.phi
@@ -11144,9 +11285,10 @@ define amdgpu_kernel void @atomic_xchg_i64_ret(ptr %out, ptr %out2, i64 %in) {
; GFX12-NEXT: s_mov_b64 s[6:7], src_private_base
; GFX12-NEXT: s_wait_kmcnt 0x0
; GFX12-NEXT: s_cmp_eq_u32 s1, s7
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s6, 1, 0
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_lg_u32 s6, 1
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_scc0 .LBB71_2
; GFX12-NEXT: ; %bb.1: ; %atomicrmw.global
; GFX12-NEXT: v_dual_mov_b32 v0, s0 :: v_dual_mov_b32 v1, s1
@@ -11164,6 +11306,7 @@ define amdgpu_kernel void @atomic_xchg_i64_ret(ptr %out, ptr %out2, i64 %in) {
; GFX12-NEXT: s_and_b32 s6, s6, exec_lo
; GFX12-NEXT: s_cselect_b32 s6, 1, 0
; GFX12-NEXT: s_cmp_lg_u32 s6, 1
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_scc1 .LBB71_5
; GFX12-NEXT: ; %bb.4: ; %atomicrmw.private
; GFX12-NEXT: s_cmp_lg_u64 s[0:1], 0
@@ -11289,8 +11432,8 @@ define amdgpu_kernel void @atomic_xchg_i64_addr64(ptr %out, i64 %in, i64 %index)
; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_add_nc_u64 s[0:1], s[0:1], s[4:5]
; GFX12-NEXT: s_cmp_eq_u32 s1, s7
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s4, 1, 0
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_lg_u32 s4, 1
; GFX12-NEXT: s_mov_b32 s4, -1
; GFX12-NEXT: s_cbranch_scc0 .LBB72_2
@@ -11304,12 +11447,13 @@ define amdgpu_kernel void @atomic_xchg_i64_addr64(ptr %out, i64 %in, i64 %index)
; GFX12-NEXT: .LBB72_2: ; %Flow
; GFX12-NEXT: s_and_b32 s4, s4, exec_lo
; GFX12-NEXT: s_cselect_b32 s4, 1, 0
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_lg_u32 s4, 1
; GFX12-NEXT: s_cbranch_scc1 .LBB72_4
; GFX12-NEXT: ; %bb.3: ; %atomicrmw.private
; GFX12-NEXT: v_dual_mov_b32 v0, s2 :: v_dual_mov_b32 v1, s3
; GFX12-NEXT: s_cmp_lg_u64 s[0:1], 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s0, s0, -1
; GFX12-NEXT: scratch_store_b64 off, v[0:1], s0
; GFX12-NEXT: .LBB72_4: ; %atomicrmw.phi
@@ -11441,9 +11585,10 @@ define amdgpu_kernel void @atomic_xchg_i64_ret_addr64(ptr %out, ptr %out2, i64 %
; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_add_nc_u64 s[0:1], s[0:1], s[6:7]
; GFX12-NEXT: s_cmp_eq_u32 s1, s9
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s6, 1, 0
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_lg_u32 s6, 1
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_scc0 .LBB73_2
; GFX12-NEXT: ; %bb.1: ; %atomicrmw.global
; GFX12-NEXT: v_dual_mov_b32 v0, s0 :: v_dual_mov_b32 v1, s1
@@ -11461,6 +11606,7 @@ define amdgpu_kernel void @atomic_xchg_i64_ret_addr64(ptr %out, ptr %out2, i64 %
; GFX12-NEXT: s_and_b32 s6, s6, exec_lo
; GFX12-NEXT: s_cselect_b32 s6, 1, 0
; GFX12-NEXT: s_cmp_lg_u32 s6, 1
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_scc1 .LBB73_5
; GFX12-NEXT: ; %bb.4: ; %atomicrmw.private
; GFX12-NEXT: s_cmp_lg_u64 s[0:1], 0
@@ -11586,9 +11732,10 @@ define amdgpu_kernel void @atomic_xor_i64_offset(ptr %out, i64 %in) {
; GFX12-NEXT: s_mov_b64 s[6:7], src_private_base
; GFX12-NEXT: s_wait_kmcnt 0x0
; GFX12-NEXT: s_add_nc_u64 s[0:1], s[0:1], 32
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_eq_u32 s1, s7
; GFX12-NEXT: s_cselect_b32 s4, 1, 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_lg_u32 s4, 1
; GFX12-NEXT: s_mov_b32 s4, -1
; GFX12-NEXT: s_cbranch_scc0 .LBB74_2
@@ -11602,11 +11749,12 @@ define amdgpu_kernel void @atomic_xor_i64_offset(ptr %out, i64 %in) {
; GFX12-NEXT: .LBB74_2: ; %Flow
; GFX12-NEXT: s_and_b32 s4, s4, exec_lo
; GFX12-NEXT: s_cselect_b32 s4, 1, 0
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_lg_u32 s4, 1
; GFX12-NEXT: s_cbranch_scc1 .LBB74_4
; GFX12-NEXT: ; %bb.3: ; %atomicrmw.private
; GFX12-NEXT: s_cmp_lg_u64 s[0:1], 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s0, s0, -1
; GFX12-NEXT: scratch_load_b64 v[0:1], off, s0
; GFX12-NEXT: s_wait_loadcnt 0x0
@@ -11743,9 +11891,10 @@ define amdgpu_kernel void @atomic_xor_i64_ret_offset(ptr %out, ptr %out2, i64 %i
; GFX12-NEXT: s_mov_b64 s[6:7], src_private_base
; GFX12-NEXT: s_wait_kmcnt 0x0
; GFX12-NEXT: s_add_nc_u64 s[0:1], s[0:1], 32
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_eq_u32 s1, s7
; GFX12-NEXT: s_cselect_b32 s6, 1, 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_lg_u32 s6, 1
; GFX12-NEXT: s_cbranch_scc0 .LBB75_2
; GFX12-NEXT: ; %bb.1: ; %atomicrmw.global
@@ -11764,9 +11913,11 @@ define amdgpu_kernel void @atomic_xor_i64_ret_offset(ptr %out, ptr %out2, i64 %i
; GFX12-NEXT: s_and_b32 s6, s6, exec_lo
; GFX12-NEXT: s_cselect_b32 s6, 1, 0
; GFX12-NEXT: s_cmp_lg_u32 s6, 1
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_scc1 .LBB75_5
; GFX12-NEXT: ; %bb.4: ; %atomicrmw.private
; GFX12-NEXT: s_cmp_lg_u64 s[0:1], 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s0, s0, -1
; GFX12-NEXT: scratch_load_b64 v[0:1], off, s0
; GFX12-NEXT: s_wait_loadcnt 0x0
@@ -11904,9 +12055,10 @@ define amdgpu_kernel void @atomic_xor_i64_addr64_offset(ptr %out, i64 %in, i64 %
; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_add_nc_u64 s[0:1], s[0:1], s[4:5]
; GFX12-NEXT: s_add_nc_u64 s[0:1], s[0:1], 32
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_eq_u32 s1, s7
; GFX12-NEXT: s_cselect_b32 s4, 1, 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_lg_u32 s4, 1
; GFX12-NEXT: s_mov_b32 s4, -1
; GFX12-NEXT: s_cbranch_scc0 .LBB76_2
@@ -11920,11 +12072,12 @@ define amdgpu_kernel void @atomic_xor_i64_addr64_offset(ptr %out, i64 %in, i64 %
; GFX12-NEXT: .LBB76_2: ; %Flow
; GFX12-NEXT: s_and_b32 s4, s4, exec_lo
; GFX12-NEXT: s_cselect_b32 s4, 1, 0
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_lg_u32 s4, 1
; GFX12-NEXT: s_cbranch_scc1 .LBB76_4
; GFX12-NEXT: ; %bb.3: ; %atomicrmw.private
; GFX12-NEXT: s_cmp_lg_u64 s[0:1], 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s0, s0, -1
; GFX12-NEXT: scratch_load_b64 v[0:1], off, s0
; GFX12-NEXT: s_wait_loadcnt 0x0
@@ -12067,9 +12220,10 @@ define amdgpu_kernel void @atomic_xor_i64_ret_addr64_offset(ptr %out, ptr %out2,
; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_add_nc_u64 s[0:1], s[0:1], s[6:7]
; GFX12-NEXT: s_add_nc_u64 s[0:1], s[0:1], 32
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_eq_u32 s1, s9
; GFX12-NEXT: s_cselect_b32 s6, 1, 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_lg_u32 s6, 1
; GFX12-NEXT: s_cbranch_scc0 .LBB77_2
; GFX12-NEXT: ; %bb.1: ; %atomicrmw.global
@@ -12088,9 +12242,11 @@ define amdgpu_kernel void @atomic_xor_i64_ret_addr64_offset(ptr %out, ptr %out2,
; GFX12-NEXT: s_and_b32 s6, s6, exec_lo
; GFX12-NEXT: s_cselect_b32 s6, 1, 0
; GFX12-NEXT: s_cmp_lg_u32 s6, 1
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_scc1 .LBB77_5
; GFX12-NEXT: ; %bb.4: ; %atomicrmw.private
; GFX12-NEXT: s_cmp_lg_u64 s[0:1], 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s0, s0, -1
; GFX12-NEXT: scratch_load_b64 v[0:1], off, s0
; GFX12-NEXT: s_wait_loadcnt 0x0
@@ -12211,8 +12367,8 @@ define amdgpu_kernel void @atomic_xor_i64(ptr %out, i64 %in) {
; GFX12-NEXT: s_mov_b64 s[6:7], src_private_base
; GFX12-NEXT: s_wait_kmcnt 0x0
; GFX12-NEXT: s_cmp_eq_u32 s1, s7
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s4, 1, 0
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_lg_u32 s4, 1
; GFX12-NEXT: s_mov_b32 s4, -1
; GFX12-NEXT: s_cbranch_scc0 .LBB78_2
@@ -12226,11 +12382,12 @@ define amdgpu_kernel void @atomic_xor_i64(ptr %out, i64 %in) {
; GFX12-NEXT: .LBB78_2: ; %Flow
; GFX12-NEXT: s_and_b32 s4, s4, exec_lo
; GFX12-NEXT: s_cselect_b32 s4, 1, 0
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_lg_u32 s4, 1
; GFX12-NEXT: s_cbranch_scc1 .LBB78_4
; GFX12-NEXT: ; %bb.3: ; %atomicrmw.private
; GFX12-NEXT: s_cmp_lg_u64 s[0:1], 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s0, s0, -1
; GFX12-NEXT: scratch_load_b64 v[0:1], off, s0
; GFX12-NEXT: s_wait_loadcnt 0x0
@@ -12362,9 +12519,10 @@ define amdgpu_kernel void @atomic_xor_i64_ret(ptr %out, ptr %out2, i64 %in) {
; GFX12-NEXT: s_mov_b64 s[6:7], src_private_base
; GFX12-NEXT: s_wait_kmcnt 0x0
; GFX12-NEXT: s_cmp_eq_u32 s1, s7
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s6, 1, 0
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_lg_u32 s6, 1
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_scc0 .LBB79_2
; GFX12-NEXT: ; %bb.1: ; %atomicrmw.global
; GFX12-NEXT: v_dual_mov_b32 v0, s0 :: v_dual_mov_b32 v1, s1
@@ -12382,9 +12540,11 @@ define amdgpu_kernel void @atomic_xor_i64_ret(ptr %out, ptr %out2, i64 %in) {
; GFX12-NEXT: s_and_b32 s6, s6, exec_lo
; GFX12-NEXT: s_cselect_b32 s6, 1, 0
; GFX12-NEXT: s_cmp_lg_u32 s6, 1
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_scc1 .LBB79_5
; GFX12-NEXT: ; %bb.4: ; %atomicrmw.private
; GFX12-NEXT: s_cmp_lg_u64 s[0:1], 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s0, s0, -1
; GFX12-NEXT: scratch_load_b64 v[0:1], off, s0
; GFX12-NEXT: s_wait_loadcnt 0x0
@@ -12516,8 +12676,8 @@ define amdgpu_kernel void @atomic_xor_i64_addr64(ptr %out, i64 %in, i64 %index)
; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_add_nc_u64 s[0:1], s[0:1], s[4:5]
; GFX12-NEXT: s_cmp_eq_u32 s1, s7
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s4, 1, 0
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_lg_u32 s4, 1
; GFX12-NEXT: s_mov_b32 s4, -1
; GFX12-NEXT: s_cbranch_scc0 .LBB80_2
@@ -12531,11 +12691,12 @@ define amdgpu_kernel void @atomic_xor_i64_addr64(ptr %out, i64 %in, i64 %index)
; GFX12-NEXT: .LBB80_2: ; %Flow
; GFX12-NEXT: s_and_b32 s4, s4, exec_lo
; GFX12-NEXT: s_cselect_b32 s4, 1, 0
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_lg_u32 s4, 1
; GFX12-NEXT: s_cbranch_scc1 .LBB80_4
; GFX12-NEXT: ; %bb.3: ; %atomicrmw.private
; GFX12-NEXT: s_cmp_lg_u64 s[0:1], 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s0, s0, -1
; GFX12-NEXT: scratch_load_b64 v[0:1], off, s0
; GFX12-NEXT: s_wait_loadcnt 0x0
@@ -12673,9 +12834,10 @@ define amdgpu_kernel void @atomic_xor_i64_ret_addr64(ptr %out, ptr %out2, i64 %i
; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_add_nc_u64 s[0:1], s[0:1], s[6:7]
; GFX12-NEXT: s_cmp_eq_u32 s1, s9
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s6, 1, 0
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_lg_u32 s6, 1
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_scc0 .LBB81_2
; GFX12-NEXT: ; %bb.1: ; %atomicrmw.global
; GFX12-NEXT: v_dual_mov_b32 v0, s0 :: v_dual_mov_b32 v1, s1
@@ -12693,9 +12855,11 @@ define amdgpu_kernel void @atomic_xor_i64_ret_addr64(ptr %out, ptr %out2, i64 %i
; GFX12-NEXT: s_and_b32 s6, s6, exec_lo
; GFX12-NEXT: s_cselect_b32 s6, 1, 0
; GFX12-NEXT: s_cmp_lg_u32 s6, 1
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_scc1 .LBB81_5
; GFX12-NEXT: ; %bb.4: ; %atomicrmw.private
; GFX12-NEXT: s_cmp_lg_u64 s[0:1], 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s0, s0, -1
; GFX12-NEXT: scratch_load_b64 v[0:1], off, s0
; GFX12-NEXT: s_wait_loadcnt 0x0
@@ -13243,9 +13407,10 @@ define amdgpu_kernel void @atomic_cmpxchg_i64_offset(ptr %out, i64 %in, i64 %old
; GFX12-NEXT: s_mov_b64 s[6:7], src_private_base
; GFX12-NEXT: s_wait_kmcnt 0x0
; GFX12-NEXT: s_add_nc_u64 s[0:1], s[0:1], 32
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_eq_u32 s1, s7
; GFX12-NEXT: s_cselect_b32 s6, 1, 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_lg_u32 s6, 1
; GFX12-NEXT: s_mov_b32 s6, -1
; GFX12-NEXT: s_cbranch_scc0 .LBB90_2
@@ -13260,11 +13425,12 @@ define amdgpu_kernel void @atomic_cmpxchg_i64_offset(ptr %out, i64 %in, i64 %old
; GFX12-NEXT: .LBB90_2: ; %Flow
; GFX12-NEXT: s_and_b32 s6, s6, exec_lo
; GFX12-NEXT: s_cselect_b32 s6, 1, 0
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_lg_u32 s6, 1
; GFX12-NEXT: s_cbranch_scc1 .LBB90_4
; GFX12-NEXT: ; %bb.3: ; %atomicrmw.private
; GFX12-NEXT: s_cmp_lg_u64 s[0:1], 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s0, s0, -1
; GFX12-NEXT: scratch_load_b64 v[0:1], off, s0
; GFX12-NEXT: s_wait_loadcnt 0x0
@@ -13398,9 +13564,10 @@ define amdgpu_kernel void @atomic_cmpxchg_i64_soffset(ptr %out, i64 %in, i64 %ol
; GFX12-NEXT: s_mov_b64 s[6:7], src_private_base
; GFX12-NEXT: s_wait_kmcnt 0x0
; GFX12-NEXT: s_add_nc_u64 s[0:1], s[0:1], 0x11940
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_eq_u32 s1, s7
; GFX12-NEXT: s_cselect_b32 s6, 1, 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_lg_u32 s6, 1
; GFX12-NEXT: s_mov_b32 s6, -1
; GFX12-NEXT: s_cbranch_scc0 .LBB91_2
@@ -13415,11 +13582,12 @@ define amdgpu_kernel void @atomic_cmpxchg_i64_soffset(ptr %out, i64 %in, i64 %ol
; GFX12-NEXT: .LBB91_2: ; %Flow
; GFX12-NEXT: s_and_b32 s6, s6, exec_lo
; GFX12-NEXT: s_cselect_b32 s6, 1, 0
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_lg_u32 s6, 1
; GFX12-NEXT: s_cbranch_scc1 .LBB91_4
; GFX12-NEXT: ; %bb.3: ; %atomicrmw.private
; GFX12-NEXT: s_cmp_lg_u64 s[0:1], 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s0, s0, -1
; GFX12-NEXT: scratch_load_b64 v[0:1], off, s0
; GFX12-NEXT: s_wait_loadcnt 0x0
@@ -13561,9 +13729,10 @@ define amdgpu_kernel void @atomic_cmpxchg_i64_ret_offset(ptr %out, ptr %out2, i6
; GFX12-NEXT: s_mov_b64 s[8:9], src_private_base
; GFX12-NEXT: s_wait_kmcnt 0x0
; GFX12-NEXT: s_add_nc_u64 s[0:1], s[0:1], 32
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_eq_u32 s1, s9
; GFX12-NEXT: s_cselect_b32 s8, 1, 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_lg_u32 s8, 1
; GFX12-NEXT: s_cbranch_scc0 .LBB92_2
; GFX12-NEXT: ; %bb.1: ; %atomicrmw.global
@@ -13583,9 +13752,11 @@ define amdgpu_kernel void @atomic_cmpxchg_i64_ret_offset(ptr %out, ptr %out2, i6
; GFX12-NEXT: s_and_b32 s8, s8, exec_lo
; GFX12-NEXT: s_cselect_b32 s8, 1, 0
; GFX12-NEXT: s_cmp_lg_u32 s8, 1
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_scc1 .LBB92_5
; GFX12-NEXT: ; %bb.4: ; %atomicrmw.private
; GFX12-NEXT: s_cmp_lg_u64 s[0:1], 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s0, s0, -1
; GFX12-NEXT: scratch_load_b64 v[0:1], off, s0
; GFX12-NEXT: s_wait_loadcnt 0x0
@@ -13728,9 +13899,10 @@ define amdgpu_kernel void @atomic_cmpxchg_i64_addr64_offset(ptr %out, i64 %in, i
; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_add_nc_u64 s[0:1], s[0:1], s[4:5]
; GFX12-NEXT: s_add_nc_u64 s[0:1], s[0:1], 32
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_eq_u32 s1, s9
; GFX12-NEXT: s_cselect_b32 s4, 1, 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_lg_u32 s4, 1
; GFX12-NEXT: s_mov_b32 s4, -1
; GFX12-NEXT: s_cbranch_scc0 .LBB93_2
@@ -13745,11 +13917,12 @@ define amdgpu_kernel void @atomic_cmpxchg_i64_addr64_offset(ptr %out, i64 %in, i
; GFX12-NEXT: .LBB93_2: ; %Flow
; GFX12-NEXT: s_and_b32 s4, s4, exec_lo
; GFX12-NEXT: s_cselect_b32 s4, 1, 0
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_lg_u32 s4, 1
; GFX12-NEXT: s_cbranch_scc1 .LBB93_4
; GFX12-NEXT: ; %bb.3: ; %atomicrmw.private
; GFX12-NEXT: s_cmp_lg_u64 s[0:1], 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s0, s0, -1
; GFX12-NEXT: scratch_load_b64 v[0:1], off, s0
; GFX12-NEXT: s_wait_loadcnt 0x0
@@ -13905,9 +14078,10 @@ define amdgpu_kernel void @atomic_cmpxchg_i64_ret_addr64_offset(ptr %out, ptr %o
; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_add_nc_u64 s[2:3], s[8:9], s[2:3]
; GFX12-NEXT: s_add_nc_u64 s[2:3], s[2:3], 32
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_eq_u32 s3, s7
; GFX12-NEXT: s_cselect_b32 s4, 1, 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_lg_u32 s4, 1
; GFX12-NEXT: s_cbranch_scc0 .LBB94_2
; GFX12-NEXT: ; %bb.1: ; %atomicrmw.global
@@ -13927,9 +14101,11 @@ define amdgpu_kernel void @atomic_cmpxchg_i64_ret_addr64_offset(ptr %out, ptr %o
; GFX12-NEXT: s_and_b32 s4, s4, exec_lo
; GFX12-NEXT: s_cselect_b32 s4, 1, 0
; GFX12-NEXT: s_cmp_lg_u32 s4, 1
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_scc1 .LBB94_5
; GFX12-NEXT: ; %bb.4: ; %atomicrmw.private
; GFX12-NEXT: s_cmp_lg_u64 s[2:3], 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s2, s2, -1
; GFX12-NEXT: scratch_load_b64 v[0:1], off, s2
; GFX12-NEXT: s_wait_loadcnt 0x0
@@ -14064,8 +14240,8 @@ define amdgpu_kernel void @atomic_cmpxchg_i64(ptr %out, i64 %in, i64 %old) {
; GFX12-NEXT: s_mov_b64 s[6:7], src_private_base
; GFX12-NEXT: s_wait_kmcnt 0x0
; GFX12-NEXT: s_cmp_eq_u32 s1, s7
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s6, 1, 0
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_lg_u32 s6, 1
; GFX12-NEXT: s_mov_b32 s6, -1
; GFX12-NEXT: s_cbranch_scc0 .LBB95_2
@@ -14080,11 +14256,12 @@ define amdgpu_kernel void @atomic_cmpxchg_i64(ptr %out, i64 %in, i64 %old) {
; GFX12-NEXT: .LBB95_2: ; %Flow
; GFX12-NEXT: s_and_b32 s6, s6, exec_lo
; GFX12-NEXT: s_cselect_b32 s6, 1, 0
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_lg_u32 s6, 1
; GFX12-NEXT: s_cbranch_scc1 .LBB95_4
; GFX12-NEXT: ; %bb.3: ; %atomicrmw.private
; GFX12-NEXT: s_cmp_lg_u64 s[0:1], 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s0, s0, -1
; GFX12-NEXT: scratch_load_b64 v[0:1], off, s0
; GFX12-NEXT: s_wait_loadcnt 0x0
@@ -14221,9 +14398,10 @@ define amdgpu_kernel void @atomic_cmpxchg_i64_ret(ptr %out, ptr %out2, i64 %in,
; GFX12-NEXT: s_mov_b64 s[8:9], src_private_base
; GFX12-NEXT: s_wait_kmcnt 0x0
; GFX12-NEXT: s_cmp_eq_u32 s1, s9
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s8, 1, 0
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_lg_u32 s8, 1
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_scc0 .LBB96_2
; GFX12-NEXT: ; %bb.1: ; %atomicrmw.global
; GFX12-NEXT: v_dual_mov_b32 v0, s4 :: v_dual_mov_b32 v1, s5
@@ -14242,9 +14420,11 @@ define amdgpu_kernel void @atomic_cmpxchg_i64_ret(ptr %out, ptr %out2, i64 %in,
; GFX12-NEXT: s_and_b32 s8, s8, exec_lo
; GFX12-NEXT: s_cselect_b32 s8, 1, 0
; GFX12-NEXT: s_cmp_lg_u32 s8, 1
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_scc1 .LBB96_5
; GFX12-NEXT: ; %bb.4: ; %atomicrmw.private
; GFX12-NEXT: s_cmp_lg_u64 s[0:1], 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s0, s0, -1
; GFX12-NEXT: scratch_load_b64 v[0:1], off, s0
; GFX12-NEXT: s_wait_loadcnt 0x0
@@ -14382,8 +14562,8 @@ define amdgpu_kernel void @atomic_cmpxchg_i64_addr64(ptr %out, i64 %in, i64 %ind
; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_add_nc_u64 s[0:1], s[0:1], s[4:5]
; GFX12-NEXT: s_cmp_eq_u32 s1, s9
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s4, 1, 0
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_lg_u32 s4, 1
; GFX12-NEXT: s_mov_b32 s4, -1
; GFX12-NEXT: s_cbranch_scc0 .LBB97_2
@@ -14398,11 +14578,12 @@ define amdgpu_kernel void @atomic_cmpxchg_i64_addr64(ptr %out, i64 %in, i64 %ind
; GFX12-NEXT: .LBB97_2: ; %Flow
; GFX12-NEXT: s_and_b32 s4, s4, exec_lo
; GFX12-NEXT: s_cselect_b32 s4, 1, 0
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_lg_u32 s4, 1
; GFX12-NEXT: s_cbranch_scc1 .LBB97_4
; GFX12-NEXT: ; %bb.3: ; %atomicrmw.private
; GFX12-NEXT: s_cmp_lg_u64 s[0:1], 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s0, s0, -1
; GFX12-NEXT: scratch_load_b64 v[0:1], off, s0
; GFX12-NEXT: s_wait_loadcnt 0x0
@@ -14553,9 +14734,10 @@ define amdgpu_kernel void @atomic_cmpxchg_i64_ret_addr64(ptr %out, ptr %out2, i6
; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_add_nc_u64 s[2:3], s[8:9], s[2:3]
; GFX12-NEXT: s_cmp_eq_u32 s3, s7
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s4, 1, 0
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_lg_u32 s4, 1
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_scc0 .LBB98_2
; GFX12-NEXT: ; %bb.1: ; %atomicrmw.global
; GFX12-NEXT: v_dual_mov_b32 v0, s12 :: v_dual_mov_b32 v1, s13
@@ -14574,9 +14756,11 @@ define amdgpu_kernel void @atomic_cmpxchg_i64_ret_addr64(ptr %out, ptr %out2, i6
; GFX12-NEXT: s_and_b32 s4, s4, exec_lo
; GFX12-NEXT: s_cselect_b32 s4, 1, 0
; GFX12-NEXT: s_cmp_lg_u32 s4, 1
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_scc1 .LBB98_5
; GFX12-NEXT: ; %bb.4: ; %atomicrmw.private
; GFX12-NEXT: s_cmp_lg_u64 s[2:3], 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s2, s2, -1
; GFX12-NEXT: scratch_load_b64 v[0:1], off, s2
; GFX12-NEXT: s_wait_loadcnt 0x0
@@ -15120,9 +15304,10 @@ define amdgpu_kernel void @atomic_inc_i64_offset(ptr %out, i64 %in) {
; GFX12-NEXT: s_mov_b64 s[6:7], src_private_base
; GFX12-NEXT: s_wait_kmcnt 0x0
; GFX12-NEXT: s_add_nc_u64 s[0:1], s[0:1], 32
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_eq_u32 s1, s7
; GFX12-NEXT: s_cselect_b32 s4, 1, 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_lg_u32 s4, 1
; GFX12-NEXT: s_mov_b32 s4, -1
; GFX12-NEXT: s_cbranch_scc0 .LBB107_2
@@ -15136,19 +15321,20 @@ define amdgpu_kernel void @atomic_inc_i64_offset(ptr %out, i64 %in) {
; GFX12-NEXT: .LBB107_2: ; %Flow
; GFX12-NEXT: s_and_b32 s4, s4, exec_lo
; GFX12-NEXT: s_cselect_b32 s4, 1, 0
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_lg_u32 s4, 1
; GFX12-NEXT: s_cbranch_scc1 .LBB107_4
; GFX12-NEXT: ; %bb.3: ; %atomicrmw.private
; GFX12-NEXT: s_cmp_lg_u64 s[0:1], 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s0, s0, -1
; GFX12-NEXT: scratch_load_b64 v[0:1], off, s0
; GFX12-NEXT: s_wait_loadcnt 0x0
; GFX12-NEXT: v_add_co_u32 v2, vcc_lo, v0, 1
-; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_2) | instid1(VALU_DEP_2)
; GFX12-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v1, vcc_lo
; GFX12-NEXT: v_cmp_gt_u64_e32 vcc_lo, s[2:3], v[0:1]
; GFX12-NEXT: s_wait_alu depctr_va_vcc(0)
+; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX12-NEXT: v_dual_cndmask_b32 v1, 0, v3 :: v_dual_cndmask_b32 v0, 0, v2
; GFX12-NEXT: scratch_store_b64 off, v[0:1], s0
; GFX12-NEXT: .LBB107_4: ; %atomicrmw.phi
@@ -15287,9 +15473,10 @@ define amdgpu_kernel void @atomic_inc_i64_ret_offset(ptr %out, ptr %out2, i64 %i
; GFX12-NEXT: s_mov_b64 s[6:7], src_private_base
; GFX12-NEXT: s_wait_kmcnt 0x0
; GFX12-NEXT: s_add_nc_u64 s[0:1], s[0:1], 32
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_eq_u32 s1, s7
; GFX12-NEXT: s_cselect_b32 s6, 1, 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_lg_u32 s6, 1
; GFX12-NEXT: s_cbranch_scc0 .LBB108_2
; GFX12-NEXT: ; %bb.1: ; %atomicrmw.global
@@ -15308,17 +15495,19 @@ define amdgpu_kernel void @atomic_inc_i64_ret_offset(ptr %out, ptr %out2, i64 %i
; GFX12-NEXT: s_and_b32 s6, s6, exec_lo
; GFX12-NEXT: s_cselect_b32 s6, 1, 0
; GFX12-NEXT: s_cmp_lg_u32 s6, 1
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_scc1 .LBB108_5
; GFX12-NEXT: ; %bb.4: ; %atomicrmw.private
; GFX12-NEXT: s_cmp_lg_u64 s[0:1], 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s0, s0, -1
; GFX12-NEXT: scratch_load_b64 v[0:1], off, s0
; GFX12-NEXT: s_wait_loadcnt 0x0
; GFX12-NEXT: v_add_co_u32 v2, vcc_lo, v0, 1
-; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_2) | instid1(VALU_DEP_2)
; GFX12-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v1, vcc_lo
; GFX12-NEXT: v_cmp_gt_u64_e32 vcc_lo, s[4:5], v[0:1]
; GFX12-NEXT: s_wait_alu depctr_va_vcc(0)
+; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX12-NEXT: v_dual_cndmask_b32 v3, 0, v3 :: v_dual_cndmask_b32 v2, 0, v2
; GFX12-NEXT: scratch_store_b64 off, v[2:3], s0
; GFX12-NEXT: .LBB108_5: ; %atomicrmw.end
@@ -15458,9 +15647,10 @@ define amdgpu_kernel void @atomic_inc_i64_incr64_offset(ptr %out, i64 %in, i64 %
; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_add_nc_u64 s[0:1], s[0:1], s[4:5]
; GFX12-NEXT: s_add_nc_u64 s[0:1], s[0:1], 32
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_eq_u32 s1, s7
; GFX12-NEXT: s_cselect_b32 s4, 1, 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_lg_u32 s4, 1
; GFX12-NEXT: s_mov_b32 s4, -1
; GFX12-NEXT: s_cbranch_scc0 .LBB109_2
@@ -15474,19 +15664,20 @@ define amdgpu_kernel void @atomic_inc_i64_incr64_offset(ptr %out, i64 %in, i64 %
; GFX12-NEXT: .LBB109_2: ; %Flow
; GFX12-NEXT: s_and_b32 s4, s4, exec_lo
; GFX12-NEXT: s_cselect_b32 s4, 1, 0
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_lg_u32 s4, 1
; GFX12-NEXT: s_cbranch_scc1 .LBB109_4
; GFX12-NEXT: ; %bb.3: ; %atomicrmw.private
; GFX12-NEXT: s_cmp_lg_u64 s[0:1], 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s0, s0, -1
; GFX12-NEXT: scratch_load_b64 v[0:1], off, s0
; GFX12-NEXT: s_wait_loadcnt 0x0
; GFX12-NEXT: v_add_co_u32 v2, vcc_lo, v0, 1
-; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_2) | instid1(VALU_DEP_2)
; GFX12-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v1, vcc_lo
; GFX12-NEXT: v_cmp_gt_u64_e32 vcc_lo, s[2:3], v[0:1]
; GFX12-NEXT: s_wait_alu depctr_va_vcc(0)
+; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX12-NEXT: v_dual_cndmask_b32 v1, 0, v3 :: v_dual_cndmask_b32 v0, 0, v2
; GFX12-NEXT: scratch_store_b64 off, v[0:1], s0
; GFX12-NEXT: .LBB109_4: ; %atomicrmw.phi
@@ -15631,9 +15822,10 @@ define amdgpu_kernel void @atomic_inc_i64_ret_incr64_offset(ptr %out, ptr %out2,
; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_add_nc_u64 s[0:1], s[0:1], s[6:7]
; GFX12-NEXT: s_add_nc_u64 s[0:1], s[0:1], 32
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_eq_u32 s1, s9
; GFX12-NEXT: s_cselect_b32 s6, 1, 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_lg_u32 s6, 1
; GFX12-NEXT: s_cbranch_scc0 .LBB110_2
; GFX12-NEXT: ; %bb.1: ; %atomicrmw.global
@@ -15652,17 +15844,19 @@ define amdgpu_kernel void @atomic_inc_i64_ret_incr64_offset(ptr %out, ptr %out2,
; GFX12-NEXT: s_and_b32 s6, s6, exec_lo
; GFX12-NEXT: s_cselect_b32 s6, 1, 0
; GFX12-NEXT: s_cmp_lg_u32 s6, 1
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_scc1 .LBB110_5
; GFX12-NEXT: ; %bb.4: ; %atomicrmw.private
; GFX12-NEXT: s_cmp_lg_u64 s[0:1], 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s0, s0, -1
; GFX12-NEXT: scratch_load_b64 v[0:1], off, s0
; GFX12-NEXT: s_wait_loadcnt 0x0
; GFX12-NEXT: v_add_co_u32 v2, vcc_lo, v0, 1
-; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_2) | instid1(VALU_DEP_2)
; GFX12-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v1, vcc_lo
; GFX12-NEXT: v_cmp_gt_u64_e32 vcc_lo, s[4:5], v[0:1]
; GFX12-NEXT: s_wait_alu depctr_va_vcc(0)
+; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX12-NEXT: v_dual_cndmask_b32 v3, 0, v3 :: v_dual_cndmask_b32 v2, 0, v2
; GFX12-NEXT: scratch_store_b64 off, v[2:3], s0
; GFX12-NEXT: .LBB110_5: ; %atomicrmw.end
@@ -15785,8 +15979,8 @@ define amdgpu_kernel void @atomic_inc_i64(ptr %out, i64 %in) {
; GFX12-NEXT: s_mov_b64 s[6:7], src_private_base
; GFX12-NEXT: s_wait_kmcnt 0x0
; GFX12-NEXT: s_cmp_eq_u32 s1, s7
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s4, 1, 0
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_lg_u32 s4, 1
; GFX12-NEXT: s_mov_b32 s4, -1
; GFX12-NEXT: s_cbranch_scc0 .LBB111_2
@@ -15800,19 +15994,20 @@ define amdgpu_kernel void @atomic_inc_i64(ptr %out, i64 %in) {
; GFX12-NEXT: .LBB111_2: ; %Flow
; GFX12-NEXT: s_and_b32 s4, s4, exec_lo
; GFX12-NEXT: s_cselect_b32 s4, 1, 0
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_lg_u32 s4, 1
; GFX12-NEXT: s_cbranch_scc1 .LBB111_4
; GFX12-NEXT: ; %bb.3: ; %atomicrmw.private
; GFX12-NEXT: s_cmp_lg_u64 s[0:1], 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s0, s0, -1
; GFX12-NEXT: scratch_load_b64 v[0:1], off, s0
; GFX12-NEXT: s_wait_loadcnt 0x0
; GFX12-NEXT: v_add_co_u32 v2, vcc_lo, v0, 1
-; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_2) | instid1(VALU_DEP_2)
; GFX12-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v1, vcc_lo
; GFX12-NEXT: v_cmp_gt_u64_e32 vcc_lo, s[2:3], v[0:1]
; GFX12-NEXT: s_wait_alu depctr_va_vcc(0)
+; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX12-NEXT: v_dual_cndmask_b32 v1, 0, v3 :: v_dual_cndmask_b32 v0, 0, v2
; GFX12-NEXT: scratch_store_b64 off, v[0:1], s0
; GFX12-NEXT: .LBB111_4: ; %atomicrmw.phi
@@ -15946,9 +16141,10 @@ define amdgpu_kernel void @atomic_inc_i64_ret(ptr %out, ptr %out2, i64 %in) {
; GFX12-NEXT: s_mov_b64 s[6:7], src_private_base
; GFX12-NEXT: s_wait_kmcnt 0x0
; GFX12-NEXT: s_cmp_eq_u32 s1, s7
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s6, 1, 0
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_lg_u32 s6, 1
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_scc0 .LBB112_2
; GFX12-NEXT: ; %bb.1: ; %atomicrmw.global
; GFX12-NEXT: v_dual_mov_b32 v0, s0 :: v_dual_mov_b32 v1, s1
@@ -15966,17 +16162,19 @@ define amdgpu_kernel void @atomic_inc_i64_ret(ptr %out, ptr %out2, i64 %in) {
; GFX12-NEXT: s_and_b32 s6, s6, exec_lo
; GFX12-NEXT: s_cselect_b32 s6, 1, 0
; GFX12-NEXT: s_cmp_lg_u32 s6, 1
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_scc1 .LBB112_5
; GFX12-NEXT: ; %bb.4: ; %atomicrmw.private
; GFX12-NEXT: s_cmp_lg_u64 s[0:1], 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s0, s0, -1
; GFX12-NEXT: scratch_load_b64 v[0:1], off, s0
; GFX12-NEXT: s_wait_loadcnt 0x0
; GFX12-NEXT: v_add_co_u32 v2, vcc_lo, v0, 1
-; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_2) | instid1(VALU_DEP_2)
; GFX12-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v1, vcc_lo
; GFX12-NEXT: v_cmp_gt_u64_e32 vcc_lo, s[4:5], v[0:1]
; GFX12-NEXT: s_wait_alu depctr_va_vcc(0)
+; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX12-NEXT: v_dual_cndmask_b32 v3, 0, v3 :: v_dual_cndmask_b32 v2, 0, v2
; GFX12-NEXT: scratch_store_b64 off, v[2:3], s0
; GFX12-NEXT: .LBB112_5: ; %atomicrmw.end
@@ -16110,8 +16308,8 @@ define amdgpu_kernel void @atomic_inc_i64_incr64(ptr %out, i64 %in, i64 %index)
; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_add_nc_u64 s[0:1], s[0:1], s[4:5]
; GFX12-NEXT: s_cmp_eq_u32 s1, s7
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s4, 1, 0
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_lg_u32 s4, 1
; GFX12-NEXT: s_mov_b32 s4, -1
; GFX12-NEXT: s_cbranch_scc0 .LBB113_2
@@ -16125,19 +16323,20 @@ define amdgpu_kernel void @atomic_inc_i64_incr64(ptr %out, i64 %in, i64 %index)
; GFX12-NEXT: .LBB113_2: ; %Flow
; GFX12-NEXT: s_and_b32 s4, s4, exec_lo
; GFX12-NEXT: s_cselect_b32 s4, 1, 0
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_lg_u32 s4, 1
; GFX12-NEXT: s_cbranch_scc1 .LBB113_4
; GFX12-NEXT: ; %bb.3: ; %atomicrmw.private
; GFX12-NEXT: s_cmp_lg_u64 s[0:1], 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s0, s0, -1
; GFX12-NEXT: scratch_load_b64 v[0:1], off, s0
; GFX12-NEXT: s_wait_loadcnt 0x0
; GFX12-NEXT: v_add_co_u32 v2, vcc_lo, v0, 1
-; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_2) | instid1(VALU_DEP_2)
; GFX12-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v1, vcc_lo
; GFX12-NEXT: v_cmp_gt_u64_e32 vcc_lo, s[2:3], v[0:1]
; GFX12-NEXT: s_wait_alu depctr_va_vcc(0)
+; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX12-NEXT: v_dual_cndmask_b32 v1, 0, v3 :: v_dual_cndmask_b32 v0, 0, v2
; GFX12-NEXT: scratch_store_b64 off, v[0:1], s0
; GFX12-NEXT: .LBB113_4: ; %atomicrmw.phi
@@ -16277,9 +16476,10 @@ define amdgpu_kernel void @atomic_inc_i64_ret_incr64(ptr %out, ptr %out2, i64 %i
; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_add_nc_u64 s[0:1], s[0:1], s[6:7]
; GFX12-NEXT: s_cmp_eq_u32 s1, s9
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s6, 1, 0
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_lg_u32 s6, 1
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_scc0 .LBB114_2
; GFX12-NEXT: ; %bb.1: ; %atomicrmw.global
; GFX12-NEXT: v_dual_mov_b32 v0, s0 :: v_dual_mov_b32 v1, s1
@@ -16297,17 +16497,19 @@ define amdgpu_kernel void @atomic_inc_i64_ret_incr64(ptr %out, ptr %out2, i64 %i
; GFX12-NEXT: s_and_b32 s6, s6, exec_lo
; GFX12-NEXT: s_cselect_b32 s6, 1, 0
; GFX12-NEXT: s_cmp_lg_u32 s6, 1
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_scc1 .LBB114_5
; GFX12-NEXT: ; %bb.4: ; %atomicrmw.private
; GFX12-NEXT: s_cmp_lg_u64 s[0:1], 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s0, s0, -1
; GFX12-NEXT: scratch_load_b64 v[0:1], off, s0
; GFX12-NEXT: s_wait_loadcnt 0x0
; GFX12-NEXT: v_add_co_u32 v2, vcc_lo, v0, 1
-; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_2) | instid1(VALU_DEP_2)
; GFX12-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v1, vcc_lo
; GFX12-NEXT: v_cmp_gt_u64_e32 vcc_lo, s[4:5], v[0:1]
; GFX12-NEXT: s_wait_alu depctr_va_vcc(0)
+; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX12-NEXT: v_dual_cndmask_b32 v3, 0, v3 :: v_dual_cndmask_b32 v2, 0, v2
; GFX12-NEXT: scratch_store_b64 off, v[2:3], s0
; GFX12-NEXT: .LBB114_5: ; %atomicrmw.end
@@ -16437,9 +16639,10 @@ define amdgpu_kernel void @atomic_dec_i64_offset(ptr %out, i64 %in) {
; GFX12-NEXT: s_mov_b64 s[6:7], src_private_base
; GFX12-NEXT: s_wait_kmcnt 0x0
; GFX12-NEXT: s_add_nc_u64 s[0:1], s[0:1], 32
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_eq_u32 s1, s7
; GFX12-NEXT: s_cselect_b32 s4, 1, 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_lg_u32 s4, 1
; GFX12-NEXT: s_mov_b32 s4, -1
; GFX12-NEXT: s_cbranch_scc0 .LBB115_2
@@ -16453,11 +16656,12 @@ define amdgpu_kernel void @atomic_dec_i64_offset(ptr %out, i64 %in) {
; GFX12-NEXT: .LBB115_2: ; %Flow
; GFX12-NEXT: s_and_b32 s4, s4, exec_lo
; GFX12-NEXT: s_cselect_b32 s4, 1, 0
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_lg_u32 s4, 1
; GFX12-NEXT: s_cbranch_scc1 .LBB115_4
; GFX12-NEXT: ; %bb.3: ; %atomicrmw.private
; GFX12-NEXT: s_cmp_lg_u64 s[0:1], 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s1, s0, -1
; GFX12-NEXT: scratch_load_b64 v[0:1], off, s1
; GFX12-NEXT: s_wait_loadcnt 0x0
@@ -16612,9 +16816,10 @@ define amdgpu_kernel void @atomic_dec_i64_ret_offset(ptr %out, ptr %out2, i64 %i
; GFX12-NEXT: s_mov_b64 s[6:7], src_private_base
; GFX12-NEXT: s_wait_kmcnt 0x0
; GFX12-NEXT: s_add_nc_u64 s[0:1], s[0:1], 32
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_eq_u32 s1, s7
; GFX12-NEXT: s_cselect_b32 s6, 1, 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_lg_u32 s6, 1
; GFX12-NEXT: s_cbranch_scc0 .LBB116_2
; GFX12-NEXT: ; %bb.1: ; %atomicrmw.global
@@ -16633,9 +16838,11 @@ define amdgpu_kernel void @atomic_dec_i64_ret_offset(ptr %out, ptr %out2, i64 %i
; GFX12-NEXT: s_and_b32 s6, s6, exec_lo
; GFX12-NEXT: s_cselect_b32 s6, 1, 0
; GFX12-NEXT: s_cmp_lg_u32 s6, 1
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_scc1 .LBB116_5
; GFX12-NEXT: ; %bb.4: ; %atomicrmw.private
; GFX12-NEXT: s_cmp_lg_u64 s[0:1], 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s1, s0, -1
; GFX12-NEXT: scratch_load_b64 v[0:1], off, s1
; GFX12-NEXT: s_wait_loadcnt 0x0
@@ -16789,9 +16996,10 @@ define amdgpu_kernel void @atomic_dec_i64_decr64_offset(ptr %out, i64 %in, i64 %
; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_add_nc_u64 s[0:1], s[0:1], s[4:5]
; GFX12-NEXT: s_add_nc_u64 s[0:1], s[0:1], 32
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_eq_u32 s1, s7
; GFX12-NEXT: s_cselect_b32 s4, 1, 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_lg_u32 s4, 1
; GFX12-NEXT: s_mov_b32 s4, -1
; GFX12-NEXT: s_cbranch_scc0 .LBB117_2
@@ -16805,11 +17013,12 @@ define amdgpu_kernel void @atomic_dec_i64_decr64_offset(ptr %out, i64 %in, i64 %
; GFX12-NEXT: .LBB117_2: ; %Flow
; GFX12-NEXT: s_and_b32 s4, s4, exec_lo
; GFX12-NEXT: s_cselect_b32 s4, 1, 0
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_lg_u32 s4, 1
; GFX12-NEXT: s_cbranch_scc1 .LBB117_4
; GFX12-NEXT: ; %bb.3: ; %atomicrmw.private
; GFX12-NEXT: s_cmp_lg_u64 s[0:1], 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s1, s0, -1
; GFX12-NEXT: scratch_load_b64 v[0:1], off, s1
; GFX12-NEXT: s_wait_loadcnt 0x0
@@ -16970,9 +17179,10 @@ define amdgpu_kernel void @atomic_dec_i64_ret_decr64_offset(ptr %out, ptr %out2,
; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_add_nc_u64 s[0:1], s[0:1], s[6:7]
; GFX12-NEXT: s_add_nc_u64 s[0:1], s[0:1], 32
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_eq_u32 s1, s9
; GFX12-NEXT: s_cselect_b32 s6, 1, 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_lg_u32 s6, 1
; GFX12-NEXT: s_cbranch_scc0 .LBB118_2
; GFX12-NEXT: ; %bb.1: ; %atomicrmw.global
@@ -16991,9 +17201,11 @@ define amdgpu_kernel void @atomic_dec_i64_ret_decr64_offset(ptr %out, ptr %out2,
; GFX12-NEXT: s_and_b32 s6, s6, exec_lo
; GFX12-NEXT: s_cselect_b32 s6, 1, 0
; GFX12-NEXT: s_cmp_lg_u32 s6, 1
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_scc1 .LBB118_5
; GFX12-NEXT: ; %bb.4: ; %atomicrmw.private
; GFX12-NEXT: s_cmp_lg_u64 s[0:1], 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s1, s0, -1
; GFX12-NEXT: scratch_load_b64 v[0:1], off, s1
; GFX12-NEXT: s_wait_loadcnt 0x0
@@ -17130,8 +17342,8 @@ define amdgpu_kernel void @atomic_dec_i64(ptr %out, i64 %in) {
; GFX12-NEXT: s_mov_b64 s[6:7], src_private_base
; GFX12-NEXT: s_wait_kmcnt 0x0
; GFX12-NEXT: s_cmp_eq_u32 s1, s7
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s4, 1, 0
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_lg_u32 s4, 1
; GFX12-NEXT: s_mov_b32 s4, -1
; GFX12-NEXT: s_cbranch_scc0 .LBB119_2
@@ -17145,11 +17357,12 @@ define amdgpu_kernel void @atomic_dec_i64(ptr %out, i64 %in) {
; GFX12-NEXT: .LBB119_2: ; %Flow
; GFX12-NEXT: s_and_b32 s4, s4, exec_lo
; GFX12-NEXT: s_cselect_b32 s4, 1, 0
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_lg_u32 s4, 1
; GFX12-NEXT: s_cbranch_scc1 .LBB119_4
; GFX12-NEXT: ; %bb.3: ; %atomicrmw.private
; GFX12-NEXT: s_cmp_lg_u64 s[0:1], 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s1, s0, -1
; GFX12-NEXT: scratch_load_b64 v[0:1], off, s1
; GFX12-NEXT: s_wait_loadcnt 0x0
@@ -17299,9 +17512,10 @@ define amdgpu_kernel void @atomic_dec_i64_ret(ptr %out, ptr %out2, i64 %in) {
; GFX12-NEXT: s_mov_b64 s[6:7], src_private_base
; GFX12-NEXT: s_wait_kmcnt 0x0
; GFX12-NEXT: s_cmp_eq_u32 s1, s7
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s6, 1, 0
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_lg_u32 s6, 1
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_scc0 .LBB120_2
; GFX12-NEXT: ; %bb.1: ; %atomicrmw.global
; GFX12-NEXT: v_dual_mov_b32 v0, s0 :: v_dual_mov_b32 v1, s1
@@ -17319,9 +17533,11 @@ define amdgpu_kernel void @atomic_dec_i64_ret(ptr %out, ptr %out2, i64 %in) {
; GFX12-NEXT: s_and_b32 s6, s6, exec_lo
; GFX12-NEXT: s_cselect_b32 s6, 1, 0
; GFX12-NEXT: s_cmp_lg_u32 s6, 1
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_scc1 .LBB120_5
; GFX12-NEXT: ; %bb.4: ; %atomicrmw.private
; GFX12-NEXT: s_cmp_lg_u64 s[0:1], 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s1, s0, -1
; GFX12-NEXT: scratch_load_b64 v[0:1], off, s1
; GFX12-NEXT: s_wait_loadcnt 0x0
@@ -17469,8 +17685,8 @@ define amdgpu_kernel void @atomic_dec_i64_decr64(ptr %out, i64 %in, i64 %index)
; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_add_nc_u64 s[0:1], s[0:1], s[4:5]
; GFX12-NEXT: s_cmp_eq_u32 s1, s7
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s4, 1, 0
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_lg_u32 s4, 1
; GFX12-NEXT: s_mov_b32 s4, -1
; GFX12-NEXT: s_cbranch_scc0 .LBB121_2
@@ -17484,11 +17700,12 @@ define amdgpu_kernel void @atomic_dec_i64_decr64(ptr %out, i64 %in, i64 %index)
; GFX12-NEXT: .LBB121_2: ; %Flow
; GFX12-NEXT: s_and_b32 s4, s4, exec_lo
; GFX12-NEXT: s_cselect_b32 s4, 1, 0
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_lg_u32 s4, 1
; GFX12-NEXT: s_cbranch_scc1 .LBB121_4
; GFX12-NEXT: ; %bb.3: ; %atomicrmw.private
; GFX12-NEXT: s_cmp_lg_u64 s[0:1], 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s1, s0, -1
; GFX12-NEXT: scratch_load_b64 v[0:1], off, s1
; GFX12-NEXT: s_wait_loadcnt 0x0
@@ -17644,9 +17861,10 @@ define amdgpu_kernel void @atomic_dec_i64_ret_decr64(ptr %out, ptr %out2, i64 %i
; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_add_nc_u64 s[0:1], s[0:1], s[6:7]
; GFX12-NEXT: s_cmp_eq_u32 s1, s9
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s6, 1, 0
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_lg_u32 s6, 1
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_scc0 .LBB122_2
; GFX12-NEXT: ; %bb.1: ; %atomicrmw.global
; GFX12-NEXT: v_dual_mov_b32 v0, s0 :: v_dual_mov_b32 v1, s1
@@ -17664,9 +17882,11 @@ define amdgpu_kernel void @atomic_dec_i64_ret_decr64(ptr %out, ptr %out2, i64 %i
; GFX12-NEXT: s_and_b32 s6, s6, exec_lo
; GFX12-NEXT: s_cselect_b32 s6, 1, 0
; GFX12-NEXT: s_cmp_lg_u32 s6, 1
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_scc1 .LBB122_5
; GFX12-NEXT: ; %bb.4: ; %atomicrmw.private
; GFX12-NEXT: s_cmp_lg_u64 s[0:1], 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s1, s0, -1
; GFX12-NEXT: scratch_load_b64 v[0:1], off, s1
; GFX12-NEXT: s_wait_loadcnt 0x0
diff --git a/llvm/test/CodeGen/AMDGPU/float-sopc-vopc.ll b/llvm/test/CodeGen/AMDGPU/float-sopc-vopc.ll
index 40c256b106a2a2..6045402b099032 100644
--- a/llvm/test/CodeGen/AMDGPU/float-sopc-vopc.ll
+++ b/llvm/test/CodeGen/AMDGPU/float-sopc-vopc.ll
@@ -8,8 +8,8 @@ define amdgpu_vs void @f32_olt(ptr addrspace(1) inreg %out, float inreg %a, floa
; SDAG-GFX1150-LABEL: f32_olt:
; SDAG-GFX1150: ; %bb.0: ; %entry
; SDAG-GFX1150-NEXT: s_cmp_lt_f32 s2, s3
+; SDAG-GFX1150-NEXT: s_delay_alu instid0(SALU_CYCLE_3) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; SDAG-GFX1150-NEXT: s_cselect_b32 s2, -1, 0
-; SDAG-GFX1150-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; SDAG-GFX1150-NEXT: v_dual_mov_b32 v0, 0 :: v_dual_mov_b32 v1, s2
; SDAG-GFX1150-NEXT: global_store_b32 v0, v1, s[0:1]
; SDAG-GFX1150-NEXT: s_endpgm
@@ -18,8 +18,8 @@ define amdgpu_vs void @f32_olt(ptr addrspace(1) inreg %out, float inreg %a, floa
; GISEL-GFX1150: ; %bb.0: ; %entry
; GISEL-GFX1150-NEXT: s_cmp_lt_f32 s2, s3
; GISEL-GFX1150-NEXT: v_mov_b32_e32 v1, 0
+; GISEL-GFX1150-NEXT: s_delay_alu instid0(SALU_CYCLE_2) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GISEL-GFX1150-NEXT: s_cselect_b32 s2, -1, 0
-; GISEL-GFX1150-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GISEL-GFX1150-NEXT: v_mov_b32_e32 v0, s2
; GISEL-GFX1150-NEXT: global_store_b32 v1, v0, s[0:1]
; GISEL-GFX1150-NEXT: s_endpgm
@@ -27,8 +27,8 @@ define amdgpu_vs void @f32_olt(ptr addrspace(1) inreg %out, float inreg %a, floa
; SDAG-GFX1200-LABEL: f32_olt:
; SDAG-GFX1200: ; %bb.0: ; %entry
; SDAG-GFX1200-NEXT: s_cmp_lt_f32 s2, s3
+; SDAG-GFX1200-NEXT: s_delay_alu instid0(SALU_CYCLE_3) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; SDAG-GFX1200-NEXT: s_cselect_b32 s2, -1, 0
-; SDAG-GFX1200-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; SDAG-GFX1200-NEXT: v_dual_mov_b32 v0, 0 :: v_dual_mov_b32 v1, s2
; SDAG-GFX1200-NEXT: global_store_b32 v0, v1, s[0:1]
; SDAG-GFX1200-NEXT: s_endpgm
@@ -37,8 +37,8 @@ define amdgpu_vs void @f32_olt(ptr addrspace(1) inreg %out, float inreg %a, floa
; GISEL-GFX1200: ; %bb.0: ; %entry
; GISEL-GFX1200-NEXT: s_cmp_lt_f32 s2, s3
; GISEL-GFX1200-NEXT: v_mov_b32_e32 v1, 0
+; GISEL-GFX1200-NEXT: s_delay_alu instid0(SALU_CYCLE_2) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GISEL-GFX1200-NEXT: s_cselect_b32 s2, -1, 0
-; GISEL-GFX1200-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GISEL-GFX1200-NEXT: v_mov_b32_e32 v0, s2
; GISEL-GFX1200-NEXT: global_store_b32 v1, v0, s[0:1]
; GISEL-GFX1200-NEXT: s_endpgm
@@ -53,8 +53,8 @@ define amdgpu_vs void @f32_oeq(ptr addrspace(1) inreg %out, float inreg %a, floa
; SDAG-GFX1150-LABEL: f32_oeq:
; SDAG-GFX1150: ; %bb.0: ; %entry
; SDAG-GFX1150-NEXT: s_cmp_eq_f32 s2, s3
+; SDAG-GFX1150-NEXT: s_delay_alu instid0(SALU_CYCLE_3) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; SDAG-GFX1150-NEXT: s_cselect_b32 s2, -1, 0
-; SDAG-GFX1150-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; SDAG-GFX1150-NEXT: v_dual_mov_b32 v0, 0 :: v_dual_mov_b32 v1, s2
; SDAG-GFX1150-NEXT: global_store_b32 v0, v1, s[0:1]
; SDAG-GFX1150-NEXT: s_endpgm
@@ -63,8 +63,8 @@ define amdgpu_vs void @f32_oeq(ptr addrspace(1) inreg %out, float inreg %a, floa
; GISEL-GFX1150: ; %bb.0: ; %entry
; GISEL-GFX1150-NEXT: s_cmp_eq_f32 s2, s3
; GISEL-GFX1150-NEXT: v_mov_b32_e32 v1, 0
+; GISEL-GFX1150-NEXT: s_delay_alu instid0(SALU_CYCLE_2) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GISEL-GFX1150-NEXT: s_cselect_b32 s2, -1, 0
-; GISEL-GFX1150-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GISEL-GFX1150-NEXT: v_mov_b32_e32 v0, s2
; GISEL-GFX1150-NEXT: global_store_b32 v1, v0, s[0:1]
; GISEL-GFX1150-NEXT: s_endpgm
@@ -72,8 +72,8 @@ define amdgpu_vs void @f32_oeq(ptr addrspace(1) inreg %out, float inreg %a, floa
; SDAG-GFX1200-LABEL: f32_oeq:
; SDAG-GFX1200: ; %bb.0: ; %entry
; SDAG-GFX1200-NEXT: s_cmp_eq_f32 s2, s3
+; SDAG-GFX1200-NEXT: s_delay_alu instid0(SALU_CYCLE_3) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; SDAG-GFX1200-NEXT: s_cselect_b32 s2, -1, 0
-; SDAG-GFX1200-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; SDAG-GFX1200-NEXT: v_dual_mov_b32 v0, 0 :: v_dual_mov_b32 v1, s2
; SDAG-GFX1200-NEXT: global_store_b32 v0, v1, s[0:1]
; SDAG-GFX1200-NEXT: s_endpgm
@@ -82,8 +82,8 @@ define amdgpu_vs void @f32_oeq(ptr addrspace(1) inreg %out, float inreg %a, floa
; GISEL-GFX1200: ; %bb.0: ; %entry
; GISEL-GFX1200-NEXT: s_cmp_eq_f32 s2, s3
; GISEL-GFX1200-NEXT: v_mov_b32_e32 v1, 0
+; GISEL-GFX1200-NEXT: s_delay_alu instid0(SALU_CYCLE_2) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GISEL-GFX1200-NEXT: s_cselect_b32 s2, -1, 0
-; GISEL-GFX1200-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GISEL-GFX1200-NEXT: v_mov_b32_e32 v0, s2
; GISEL-GFX1200-NEXT: global_store_b32 v1, v0, s[0:1]
; GISEL-GFX1200-NEXT: s_endpgm
@@ -98,8 +98,8 @@ define amdgpu_vs void @f32_ole(ptr addrspace(1) inreg %out, float inreg %a, floa
; SDAG-GFX1150-LABEL: f32_ole:
; SDAG-GFX1150: ; %bb.0: ; %entry
; SDAG-GFX1150-NEXT: s_cmp_le_f32 s2, s3
+; SDAG-GFX1150-NEXT: s_delay_alu instid0(SALU_CYCLE_3) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; SDAG-GFX1150-NEXT: s_cselect_b32 s2, -1, 0
-; SDAG-GFX1150-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; SDAG-GFX1150-NEXT: v_dual_mov_b32 v0, 0 :: v_dual_mov_b32 v1, s2
; SDAG-GFX1150-NEXT: global_store_b32 v0, v1, s[0:1]
; SDAG-GFX1150-NEXT: s_endpgm
@@ -108,8 +108,8 @@ define amdgpu_vs void @f32_ole(ptr addrspace(1) inreg %out, float inreg %a, floa
; GISEL-GFX1150: ; %bb.0: ; %entry
; GISEL-GFX1150-NEXT: s_cmp_le_f32 s2, s3
; GISEL-GFX1150-NEXT: v_mov_b32_e32 v1, 0
+; GISEL-GFX1150-NEXT: s_delay_alu instid0(SALU_CYCLE_2) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GISEL-GFX1150-NEXT: s_cselect_b32 s2, -1, 0
-; GISEL-GFX1150-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GISEL-GFX1150-NEXT: v_mov_b32_e32 v0, s2
; GISEL-GFX1150-NEXT: global_store_b32 v1, v0, s[0:1]
; GISEL-GFX1150-NEXT: s_endpgm
@@ -117,8 +117,8 @@ define amdgpu_vs void @f32_ole(ptr addrspace(1) inreg %out, float inreg %a, floa
; SDAG-GFX1200-LABEL: f32_ole:
; SDAG-GFX1200: ; %bb.0: ; %entry
; SDAG-GFX1200-NEXT: s_cmp_le_f32 s2, s3
+; SDAG-GFX1200-NEXT: s_delay_alu instid0(SALU_CYCLE_3) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; SDAG-GFX1200-NEXT: s_cselect_b32 s2, -1, 0
-; SDAG-GFX1200-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; SDAG-GFX1200-NEXT: v_dual_mov_b32 v0, 0 :: v_dual_mov_b32 v1, s2
; SDAG-GFX1200-NEXT: global_store_b32 v0, v1, s[0:1]
; SDAG-GFX1200-NEXT: s_endpgm
@@ -127,8 +127,8 @@ define amdgpu_vs void @f32_ole(ptr addrspace(1) inreg %out, float inreg %a, floa
; GISEL-GFX1200: ; %bb.0: ; %entry
; GISEL-GFX1200-NEXT: s_cmp_le_f32 s2, s3
; GISEL-GFX1200-NEXT: v_mov_b32_e32 v1, 0
+; GISEL-GFX1200-NEXT: s_delay_alu instid0(SALU_CYCLE_2) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GISEL-GFX1200-NEXT: s_cselect_b32 s2, -1, 0
-; GISEL-GFX1200-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GISEL-GFX1200-NEXT: v_mov_b32_e32 v0, s2
; GISEL-GFX1200-NEXT: global_store_b32 v1, v0, s[0:1]
; GISEL-GFX1200-NEXT: s_endpgm
@@ -143,8 +143,8 @@ define amdgpu_vs void @f32_ogt(ptr addrspace(1) inreg %out, float inreg %a, floa
; SDAG-GFX1150-LABEL: f32_ogt:
; SDAG-GFX1150: ; %bb.0: ; %entry
; SDAG-GFX1150-NEXT: s_cmp_gt_f32 s2, s3
+; SDAG-GFX1150-NEXT: s_delay_alu instid0(SALU_CYCLE_3) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; SDAG-GFX1150-NEXT: s_cselect_b32 s2, -1, 0
-; SDAG-GFX1150-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; SDAG-GFX1150-NEXT: v_dual_mov_b32 v0, 0 :: v_dual_mov_b32 v1, s2
; SDAG-GFX1150-NEXT: global_store_b32 v0, v1, s[0:1]
; SDAG-GFX1150-NEXT: s_endpgm
@@ -153,8 +153,8 @@ define amdgpu_vs void @f32_ogt(ptr addrspace(1) inreg %out, float inreg %a, floa
; GISEL-GFX1150: ; %bb.0: ; %entry
; GISEL-GFX1150-NEXT: s_cmp_gt_f32 s2, s3
; GISEL-GFX1150-NEXT: v_mov_b32_e32 v1, 0
+; GISEL-GFX1150-NEXT: s_delay_alu instid0(SALU_CYCLE_2) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GISEL-GFX1150-NEXT: s_cselect_b32 s2, -1, 0
-; GISEL-GFX1150-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GISEL-GFX1150-NEXT: v_mov_b32_e32 v0, s2
; GISEL-GFX1150-NEXT: global_store_b32 v1, v0, s[0:1]
; GISEL-GFX1150-NEXT: s_endpgm
@@ -162,8 +162,8 @@ define amdgpu_vs void @f32_ogt(ptr addrspace(1) inreg %out, float inreg %a, floa
; SDAG-GFX1200-LABEL: f32_ogt:
; SDAG-GFX1200: ; %bb.0: ; %entry
; SDAG-GFX1200-NEXT: s_cmp_gt_f32 s2, s3
+; SDAG-GFX1200-NEXT: s_delay_alu instid0(SALU_CYCLE_3) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; SDAG-GFX1200-NEXT: s_cselect_b32 s2, -1, 0
-; SDAG-GFX1200-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; SDAG-GFX1200-NEXT: v_dual_mov_b32 v0, 0 :: v_dual_mov_b32 v1, s2
; SDAG-GFX1200-NEXT: global_store_b32 v0, v1, s[0:1]
; SDAG-GFX1200-NEXT: s_endpgm
@@ -172,8 +172,8 @@ define amdgpu_vs void @f32_ogt(ptr addrspace(1) inreg %out, float inreg %a, floa
; GISEL-GFX1200: ; %bb.0: ; %entry
; GISEL-GFX1200-NEXT: s_cmp_gt_f32 s2, s3
; GISEL-GFX1200-NEXT: v_mov_b32_e32 v1, 0
+; GISEL-GFX1200-NEXT: s_delay_alu instid0(SALU_CYCLE_2) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GISEL-GFX1200-NEXT: s_cselect_b32 s2, -1, 0
-; GISEL-GFX1200-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GISEL-GFX1200-NEXT: v_mov_b32_e32 v0, s2
; GISEL-GFX1200-NEXT: global_store_b32 v1, v0, s[0:1]
; GISEL-GFX1200-NEXT: s_endpgm
@@ -188,8 +188,8 @@ define amdgpu_vs void @f32_one(ptr addrspace(1) inreg %out, float inreg %a, floa
; SDAG-GFX1150-LABEL: f32_one:
; SDAG-GFX1150: ; %bb.0: ; %entry
; SDAG-GFX1150-NEXT: s_cmp_lg_f32 s2, s3
+; SDAG-GFX1150-NEXT: s_delay_alu instid0(SALU_CYCLE_3) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; SDAG-GFX1150-NEXT: s_cselect_b32 s2, -1, 0
-; SDAG-GFX1150-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; SDAG-GFX1150-NEXT: v_dual_mov_b32 v0, 0 :: v_dual_mov_b32 v1, s2
; SDAG-GFX1150-NEXT: global_store_b32 v0, v1, s[0:1]
; SDAG-GFX1150-NEXT: s_endpgm
@@ -198,8 +198,8 @@ define amdgpu_vs void @f32_one(ptr addrspace(1) inreg %out, float inreg %a, floa
; GISEL-GFX1150: ; %bb.0: ; %entry
; GISEL-GFX1150-NEXT: s_cmp_lg_f32 s2, s3
; GISEL-GFX1150-NEXT: v_mov_b32_e32 v1, 0
+; GISEL-GFX1150-NEXT: s_delay_alu instid0(SALU_CYCLE_2) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GISEL-GFX1150-NEXT: s_cselect_b32 s2, -1, 0
-; GISEL-GFX1150-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GISEL-GFX1150-NEXT: v_mov_b32_e32 v0, s2
; GISEL-GFX1150-NEXT: global_store_b32 v1, v0, s[0:1]
; GISEL-GFX1150-NEXT: s_endpgm
@@ -207,8 +207,8 @@ define amdgpu_vs void @f32_one(ptr addrspace(1) inreg %out, float inreg %a, floa
; SDAG-GFX1200-LABEL: f32_one:
; SDAG-GFX1200: ; %bb.0: ; %entry
; SDAG-GFX1200-NEXT: s_cmp_lg_f32 s2, s3
+; SDAG-GFX1200-NEXT: s_delay_alu instid0(SALU_CYCLE_3) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; SDAG-GFX1200-NEXT: s_cselect_b32 s2, -1, 0
-; SDAG-GFX1200-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; SDAG-GFX1200-NEXT: v_dual_mov_b32 v0, 0 :: v_dual_mov_b32 v1, s2
; SDAG-GFX1200-NEXT: global_store_b32 v0, v1, s[0:1]
; SDAG-GFX1200-NEXT: s_endpgm
@@ -217,8 +217,8 @@ define amdgpu_vs void @f32_one(ptr addrspace(1) inreg %out, float inreg %a, floa
; GISEL-GFX1200: ; %bb.0: ; %entry
; GISEL-GFX1200-NEXT: s_cmp_lg_f32 s2, s3
; GISEL-GFX1200-NEXT: v_mov_b32_e32 v1, 0
+; GISEL-GFX1200-NEXT: s_delay_alu instid0(SALU_CYCLE_2) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GISEL-GFX1200-NEXT: s_cselect_b32 s2, -1, 0
-; GISEL-GFX1200-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GISEL-GFX1200-NEXT: v_mov_b32_e32 v0, s2
; GISEL-GFX1200-NEXT: global_store_b32 v1, v0, s[0:1]
; GISEL-GFX1200-NEXT: s_endpgm
@@ -233,8 +233,8 @@ define amdgpu_vs void @f32_oge(ptr addrspace(1) inreg %out, float inreg %a, floa
; SDAG-GFX1150-LABEL: f32_oge:
; SDAG-GFX1150: ; %bb.0: ; %entry
; SDAG-GFX1150-NEXT: s_cmp_ge_f32 s2, s3
+; SDAG-GFX1150-NEXT: s_delay_alu instid0(SALU_CYCLE_3) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; SDAG-GFX1150-NEXT: s_cselect_b32 s2, -1, 0
-; SDAG-GFX1150-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; SDAG-GFX1150-NEXT: v_dual_mov_b32 v0, 0 :: v_dual_mov_b32 v1, s2
; SDAG-GFX1150-NEXT: global_store_b32 v0, v1, s[0:1]
; SDAG-GFX1150-NEXT: s_endpgm
@@ -243,8 +243,8 @@ define amdgpu_vs void @f32_oge(ptr addrspace(1) inreg %out, float inreg %a, floa
; GISEL-GFX1150: ; %bb.0: ; %entry
; GISEL-GFX1150-NEXT: s_cmp_ge_f32 s2, s3
; GISEL-GFX1150-NEXT: v_mov_b32_e32 v1, 0
+; GISEL-GFX1150-NEXT: s_delay_alu instid0(SALU_CYCLE_2) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GISEL-GFX1150-NEXT: s_cselect_b32 s2, -1, 0
-; GISEL-GFX1150-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GISEL-GFX1150-NEXT: v_mov_b32_e32 v0, s2
; GISEL-GFX1150-NEXT: global_store_b32 v1, v0, s[0:1]
; GISEL-GFX1150-NEXT: s_endpgm
@@ -252,8 +252,8 @@ define amdgpu_vs void @f32_oge(ptr addrspace(1) inreg %out, float inreg %a, floa
; SDAG-GFX1200-LABEL: f32_oge:
; SDAG-GFX1200: ; %bb.0: ; %entry
; SDAG-GFX1200-NEXT: s_cmp_ge_f32 s2, s3
+; SDAG-GFX1200-NEXT: s_delay_alu instid0(SALU_CYCLE_3) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; SDAG-GFX1200-NEXT: s_cselect_b32 s2, -1, 0
-; SDAG-GFX1200-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; SDAG-GFX1200-NEXT: v_dual_mov_b32 v0, 0 :: v_dual_mov_b32 v1, s2
; SDAG-GFX1200-NEXT: global_store_b32 v0, v1, s[0:1]
; SDAG-GFX1200-NEXT: s_endpgm
@@ -262,8 +262,8 @@ define amdgpu_vs void @f32_oge(ptr addrspace(1) inreg %out, float inreg %a, floa
; GISEL-GFX1200: ; %bb.0: ; %entry
; GISEL-GFX1200-NEXT: s_cmp_ge_f32 s2, s3
; GISEL-GFX1200-NEXT: v_mov_b32_e32 v1, 0
+; GISEL-GFX1200-NEXT: s_delay_alu instid0(SALU_CYCLE_2) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GISEL-GFX1200-NEXT: s_cselect_b32 s2, -1, 0
-; GISEL-GFX1200-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GISEL-GFX1200-NEXT: v_mov_b32_e32 v0, s2
; GISEL-GFX1200-NEXT: global_store_b32 v1, v0, s[0:1]
; GISEL-GFX1200-NEXT: s_endpgm
@@ -278,8 +278,8 @@ define amdgpu_vs void @f32_ord(ptr addrspace(1) inreg %out, float inreg %a, floa
; SDAG-GFX1150-LABEL: f32_ord:
; SDAG-GFX1150: ; %bb.0: ; %entry
; SDAG-GFX1150-NEXT: s_cmp_o_f32 s2, s3
+; SDAG-GFX1150-NEXT: s_delay_alu instid0(SALU_CYCLE_3) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; SDAG-GFX1150-NEXT: s_cselect_b32 s2, -1, 0
-; SDAG-GFX1150-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; SDAG-GFX1150-NEXT: v_dual_mov_b32 v0, 0 :: v_dual_mov_b32 v1, s2
; SDAG-GFX1150-NEXT: global_store_b32 v0, v1, s[0:1]
; SDAG-GFX1150-NEXT: s_endpgm
@@ -288,8 +288,8 @@ define amdgpu_vs void @f32_ord(ptr addrspace(1) inreg %out, float inreg %a, floa
; GISEL-GFX1150: ; %bb.0: ; %entry
; GISEL-GFX1150-NEXT: s_cmp_o_f32 s2, s3
; GISEL-GFX1150-NEXT: v_mov_b32_e32 v1, 0
+; GISEL-GFX1150-NEXT: s_delay_alu instid0(SALU_CYCLE_2) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GISEL-GFX1150-NEXT: s_cselect_b32 s2, -1, 0
-; GISEL-GFX1150-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GISEL-GFX1150-NEXT: v_mov_b32_e32 v0, s2
; GISEL-GFX1150-NEXT: global_store_b32 v1, v0, s[0:1]
; GISEL-GFX1150-NEXT: s_endpgm
@@ -297,8 +297,8 @@ define amdgpu_vs void @f32_ord(ptr addrspace(1) inreg %out, float inreg %a, floa
; SDAG-GFX1200-LABEL: f32_ord:
; SDAG-GFX1200: ; %bb.0: ; %entry
; SDAG-GFX1200-NEXT: s_cmp_o_f32 s2, s3
+; SDAG-GFX1200-NEXT: s_delay_alu instid0(SALU_CYCLE_3) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; SDAG-GFX1200-NEXT: s_cselect_b32 s2, -1, 0
-; SDAG-GFX1200-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; SDAG-GFX1200-NEXT: v_dual_mov_b32 v0, 0 :: v_dual_mov_b32 v1, s2
; SDAG-GFX1200-NEXT: global_store_b32 v0, v1, s[0:1]
; SDAG-GFX1200-NEXT: s_endpgm
@@ -307,8 +307,8 @@ define amdgpu_vs void @f32_ord(ptr addrspace(1) inreg %out, float inreg %a, floa
; GISEL-GFX1200: ; %bb.0: ; %entry
; GISEL-GFX1200-NEXT: s_cmp_o_f32 s2, s3
; GISEL-GFX1200-NEXT: v_mov_b32_e32 v1, 0
+; GISEL-GFX1200-NEXT: s_delay_alu instid0(SALU_CYCLE_2) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GISEL-GFX1200-NEXT: s_cselect_b32 s2, -1, 0
-; GISEL-GFX1200-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GISEL-GFX1200-NEXT: v_mov_b32_e32 v0, s2
; GISEL-GFX1200-NEXT: global_store_b32 v1, v0, s[0:1]
; GISEL-GFX1200-NEXT: s_endpgm
@@ -323,8 +323,8 @@ define amdgpu_vs void @f32_uno(ptr addrspace(1) inreg %out, float inreg %a, floa
; SDAG-GFX1150-LABEL: f32_uno:
; SDAG-GFX1150: ; %bb.0: ; %entry
; SDAG-GFX1150-NEXT: s_cmp_u_f32 s2, s3
+; SDAG-GFX1150-NEXT: s_delay_alu instid0(SALU_CYCLE_3) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; SDAG-GFX1150-NEXT: s_cselect_b32 s2, -1, 0
-; SDAG-GFX1150-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; SDAG-GFX1150-NEXT: v_dual_mov_b32 v0, 0 :: v_dual_mov_b32 v1, s2
; SDAG-GFX1150-NEXT: global_store_b32 v0, v1, s[0:1]
; SDAG-GFX1150-NEXT: s_endpgm
@@ -333,8 +333,8 @@ define amdgpu_vs void @f32_uno(ptr addrspace(1) inreg %out, float inreg %a, floa
; GISEL-GFX1150: ; %bb.0: ; %entry
; GISEL-GFX1150-NEXT: s_cmp_u_f32 s2, s3
; GISEL-GFX1150-NEXT: v_mov_b32_e32 v1, 0
+; GISEL-GFX1150-NEXT: s_delay_alu instid0(SALU_CYCLE_2) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GISEL-GFX1150-NEXT: s_cselect_b32 s2, -1, 0
-; GISEL-GFX1150-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GISEL-GFX1150-NEXT: v_mov_b32_e32 v0, s2
; GISEL-GFX1150-NEXT: global_store_b32 v1, v0, s[0:1]
; GISEL-GFX1150-NEXT: s_endpgm
@@ -342,8 +342,8 @@ define amdgpu_vs void @f32_uno(ptr addrspace(1) inreg %out, float inreg %a, floa
; SDAG-GFX1200-LABEL: f32_uno:
; SDAG-GFX1200: ; %bb.0: ; %entry
; SDAG-GFX1200-NEXT: s_cmp_u_f32 s2, s3
+; SDAG-GFX1200-NEXT: s_delay_alu instid0(SALU_CYCLE_3) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; SDAG-GFX1200-NEXT: s_cselect_b32 s2, -1, 0
-; SDAG-GFX1200-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; SDAG-GFX1200-NEXT: v_dual_mov_b32 v0, 0 :: v_dual_mov_b32 v1, s2
; SDAG-GFX1200-NEXT: global_store_b32 v0, v1, s[0:1]
; SDAG-GFX1200-NEXT: s_endpgm
@@ -352,8 +352,8 @@ define amdgpu_vs void @f32_uno(ptr addrspace(1) inreg %out, float inreg %a, floa
; GISEL-GFX1200: ; %bb.0: ; %entry
; GISEL-GFX1200-NEXT: s_cmp_u_f32 s2, s3
; GISEL-GFX1200-NEXT: v_mov_b32_e32 v1, 0
+; GISEL-GFX1200-NEXT: s_delay_alu instid0(SALU_CYCLE_2) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GISEL-GFX1200-NEXT: s_cselect_b32 s2, -1, 0
-; GISEL-GFX1200-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GISEL-GFX1200-NEXT: v_mov_b32_e32 v0, s2
; GISEL-GFX1200-NEXT: global_store_b32 v1, v0, s[0:1]
; GISEL-GFX1200-NEXT: s_endpgm
@@ -368,8 +368,8 @@ define amdgpu_vs void @f32_ult(ptr addrspace(1) inreg %out, float inreg %a, floa
; SDAG-GFX1150-LABEL: f32_ult:
; SDAG-GFX1150: ; %bb.0: ; %entry
; SDAG-GFX1150-NEXT: s_cmp_nge_f32 s2, s3
+; SDAG-GFX1150-NEXT: s_delay_alu instid0(SALU_CYCLE_3) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; SDAG-GFX1150-NEXT: s_cselect_b32 s2, -1, 0
-; SDAG-GFX1150-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; SDAG-GFX1150-NEXT: v_dual_mov_b32 v0, 0 :: v_dual_mov_b32 v1, s2
; SDAG-GFX1150-NEXT: global_store_b32 v0, v1, s[0:1]
; SDAG-GFX1150-NEXT: s_endpgm
@@ -378,8 +378,8 @@ define amdgpu_vs void @f32_ult(ptr addrspace(1) inreg %out, float inreg %a, floa
; GISEL-GFX1150: ; %bb.0: ; %entry
; GISEL-GFX1150-NEXT: s_cmp_nge_f32 s2, s3
; GISEL-GFX1150-NEXT: v_mov_b32_e32 v1, 0
+; GISEL-GFX1150-NEXT: s_delay_alu instid0(SALU_CYCLE_2) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GISEL-GFX1150-NEXT: s_cselect_b32 s2, -1, 0
-; GISEL-GFX1150-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GISEL-GFX1150-NEXT: v_mov_b32_e32 v0, s2
; GISEL-GFX1150-NEXT: global_store_b32 v1, v0, s[0:1]
; GISEL-GFX1150-NEXT: s_endpgm
@@ -387,8 +387,8 @@ define amdgpu_vs void @f32_ult(ptr addrspace(1) inreg %out, float inreg %a, floa
; SDAG-GFX1200-LABEL: f32_ult:
; SDAG-GFX1200: ; %bb.0: ; %entry
; SDAG-GFX1200-NEXT: s_cmp_nge_f32 s2, s3
+; SDAG-GFX1200-NEXT: s_delay_alu instid0(SALU_CYCLE_3) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; SDAG-GFX1200-NEXT: s_cselect_b32 s2, -1, 0
-; SDAG-GFX1200-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; SDAG-GFX1200-NEXT: v_dual_mov_b32 v0, 0 :: v_dual_mov_b32 v1, s2
; SDAG-GFX1200-NEXT: global_store_b32 v0, v1, s[0:1]
; SDAG-GFX1200-NEXT: s_endpgm
@@ -397,8 +397,8 @@ define amdgpu_vs void @f32_ult(ptr addrspace(1) inreg %out, float inreg %a, floa
; GISEL-GFX1200: ; %bb.0: ; %entry
; GISEL-GFX1200-NEXT: s_cmp_nge_f32 s2, s3
; GISEL-GFX1200-NEXT: v_mov_b32_e32 v1, 0
+; GISEL-GFX1200-NEXT: s_delay_alu instid0(SALU_CYCLE_2) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GISEL-GFX1200-NEXT: s_cselect_b32 s2, -1, 0
-; GISEL-GFX1200-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GISEL-GFX1200-NEXT: v_mov_b32_e32 v0, s2
; GISEL-GFX1200-NEXT: global_store_b32 v1, v0, s[0:1]
; GISEL-GFX1200-NEXT: s_endpgm
@@ -413,8 +413,8 @@ define amdgpu_vs void @f32_ueq(ptr addrspace(1) inreg %out, float inreg %a, floa
; SDAG-GFX1150-LABEL: f32_ueq:
; SDAG-GFX1150: ; %bb.0: ; %entry
; SDAG-GFX1150-NEXT: s_cmp_nlg_f32 s2, s3
+; SDAG-GFX1150-NEXT: s_delay_alu instid0(SALU_CYCLE_3) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; SDAG-GFX1150-NEXT: s_cselect_b32 s2, -1, 0
-; SDAG-GFX1150-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; SDAG-GFX1150-NEXT: v_dual_mov_b32 v0, 0 :: v_dual_mov_b32 v1, s2
; SDAG-GFX1150-NEXT: global_store_b32 v0, v1, s[0:1]
; SDAG-GFX1150-NEXT: s_endpgm
@@ -423,8 +423,8 @@ define amdgpu_vs void @f32_ueq(ptr addrspace(1) inreg %out, float inreg %a, floa
; GISEL-GFX1150: ; %bb.0: ; %entry
; GISEL-GFX1150-NEXT: s_cmp_nlg_f32 s2, s3
; GISEL-GFX1150-NEXT: v_mov_b32_e32 v1, 0
+; GISEL-GFX1150-NEXT: s_delay_alu instid0(SALU_CYCLE_2) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GISEL-GFX1150-NEXT: s_cselect_b32 s2, -1, 0
-; GISEL-GFX1150-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GISEL-GFX1150-NEXT: v_mov_b32_e32 v0, s2
; GISEL-GFX1150-NEXT: global_store_b32 v1, v0, s[0:1]
; GISEL-GFX1150-NEXT: s_endpgm
@@ -432,8 +432,8 @@ define amdgpu_vs void @f32_ueq(ptr addrspace(1) inreg %out, float inreg %a, floa
; SDAG-GFX1200-LABEL: f32_ueq:
; SDAG-GFX1200: ; %bb.0: ; %entry
; SDAG-GFX1200-NEXT: s_cmp_nlg_f32 s2, s3
+; SDAG-GFX1200-NEXT: s_delay_alu instid0(SALU_CYCLE_3) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; SDAG-GFX1200-NEXT: s_cselect_b32 s2, -1, 0
-; SDAG-GFX1200-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; SDAG-GFX1200-NEXT: v_dual_mov_b32 v0, 0 :: v_dual_mov_b32 v1, s2
; SDAG-GFX1200-NEXT: global_store_b32 v0, v1, s[0:1]
; SDAG-GFX1200-NEXT: s_endpgm
@@ -442,8 +442,8 @@ define amdgpu_vs void @f32_ueq(ptr addrspace(1) inreg %out, float inreg %a, floa
; GISEL-GFX1200: ; %bb.0: ; %entry
; GISEL-GFX1200-NEXT: s_cmp_nlg_f32 s2, s3
; GISEL-GFX1200-NEXT: v_mov_b32_e32 v1, 0
+; GISEL-GFX1200-NEXT: s_delay_alu instid0(SALU_CYCLE_2) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GISEL-GFX1200-NEXT: s_cselect_b32 s2, -1, 0
-; GISEL-GFX1200-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GISEL-GFX1200-NEXT: v_mov_b32_e32 v0, s2
; GISEL-GFX1200-NEXT: global_store_b32 v1, v0, s[0:1]
; GISEL-GFX1200-NEXT: s_endpgm
@@ -458,8 +458,8 @@ define amdgpu_vs void @f32_ule(ptr addrspace(1) inreg %out, float inreg %a, floa
; SDAG-GFX1150-LABEL: f32_ule:
; SDAG-GFX1150: ; %bb.0: ; %entry
; SDAG-GFX1150-NEXT: s_cmp_ngt_f32 s2, s3
+; SDAG-GFX1150-NEXT: s_delay_alu instid0(SALU_CYCLE_3) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; SDAG-GFX1150-NEXT: s_cselect_b32 s2, -1, 0
-; SDAG-GFX1150-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; SDAG-GFX1150-NEXT: v_dual_mov_b32 v0, 0 :: v_dual_mov_b32 v1, s2
; SDAG-GFX1150-NEXT: global_store_b32 v0, v1, s[0:1]
; SDAG-GFX1150-NEXT: s_endpgm
@@ -468,8 +468,8 @@ define amdgpu_vs void @f32_ule(ptr addrspace(1) inreg %out, float inreg %a, floa
; GISEL-GFX1150: ; %bb.0: ; %entry
; GISEL-GFX1150-NEXT: s_cmp_ngt_f32 s2, s3
; GISEL-GFX1150-NEXT: v_mov_b32_e32 v1, 0
+; GISEL-GFX1150-NEXT: s_delay_alu instid0(SALU_CYCLE_2) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GISEL-GFX1150-NEXT: s_cselect_b32 s2, -1, 0
-; GISEL-GFX1150-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GISEL-GFX1150-NEXT: v_mov_b32_e32 v0, s2
; GISEL-GFX1150-NEXT: global_store_b32 v1, v0, s[0:1]
; GISEL-GFX1150-NEXT: s_endpgm
@@ -477,8 +477,8 @@ define amdgpu_vs void @f32_ule(ptr addrspace(1) inreg %out, float inreg %a, floa
; SDAG-GFX1200-LABEL: f32_ule:
; SDAG-GFX1200: ; %bb.0: ; %entry
; SDAG-GFX1200-NEXT: s_cmp_ngt_f32 s2, s3
+; SDAG-GFX1200-NEXT: s_delay_alu instid0(SALU_CYCLE_3) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; SDAG-GFX1200-NEXT: s_cselect_b32 s2, -1, 0
-; SDAG-GFX1200-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; SDAG-GFX1200-NEXT: v_dual_mov_b32 v0, 0 :: v_dual_mov_b32 v1, s2
; SDAG-GFX1200-NEXT: global_store_b32 v0, v1, s[0:1]
; SDAG-GFX1200-NEXT: s_endpgm
@@ -487,8 +487,8 @@ define amdgpu_vs void @f32_ule(ptr addrspace(1) inreg %out, float inreg %a, floa
; GISEL-GFX1200: ; %bb.0: ; %entry
; GISEL-GFX1200-NEXT: s_cmp_ngt_f32 s2, s3
; GISEL-GFX1200-NEXT: v_mov_b32_e32 v1, 0
+; GISEL-GFX1200-NEXT: s_delay_alu instid0(SALU_CYCLE_2) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GISEL-GFX1200-NEXT: s_cselect_b32 s2, -1, 0
-; GISEL-GFX1200-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GISEL-GFX1200-NEXT: v_mov_b32_e32 v0, s2
; GISEL-GFX1200-NEXT: global_store_b32 v1, v0, s[0:1]
; GISEL-GFX1200-NEXT: s_endpgm
@@ -503,8 +503,8 @@ define amdgpu_vs void @f32_ugt(ptr addrspace(1) inreg %out, float inreg %a, floa
; SDAG-GFX1150-LABEL: f32_ugt:
; SDAG-GFX1150: ; %bb.0: ; %entry
; SDAG-GFX1150-NEXT: s_cmp_nle_f32 s2, s3
+; SDAG-GFX1150-NEXT: s_delay_alu instid0(SALU_CYCLE_3) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; SDAG-GFX1150-NEXT: s_cselect_b32 s2, -1, 0
-; SDAG-GFX1150-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; SDAG-GFX1150-NEXT: v_dual_mov_b32 v0, 0 :: v_dual_mov_b32 v1, s2
; SDAG-GFX1150-NEXT: global_store_b32 v0, v1, s[0:1]
; SDAG-GFX1150-NEXT: s_endpgm
@@ -513,8 +513,8 @@ define amdgpu_vs void @f32_ugt(ptr addrspace(1) inreg %out, float inreg %a, floa
; GISEL-GFX1150: ; %bb.0: ; %entry
; GISEL-GFX1150-NEXT: s_cmp_nle_f32 s2, s3
; GISEL-GFX1150-NEXT: v_mov_b32_e32 v1, 0
+; GISEL-GFX1150-NEXT: s_delay_alu instid0(SALU_CYCLE_2) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GISEL-GFX1150-NEXT: s_cselect_b32 s2, -1, 0
-; GISEL-GFX1150-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GISEL-GFX1150-NEXT: v_mov_b32_e32 v0, s2
; GISEL-GFX1150-NEXT: global_store_b32 v1, v0, s[0:1]
; GISEL-GFX1150-NEXT: s_endpgm
@@ -522,8 +522,8 @@ define amdgpu_vs void @f32_ugt(ptr addrspace(1) inreg %out, float inreg %a, floa
; SDAG-GFX1200-LABEL: f32_ugt:
; SDAG-GFX1200: ; %bb.0: ; %entry
; SDAG-GFX1200-NEXT: s_cmp_nle_f32 s2, s3
+; SDAG-GFX1200-NEXT: s_delay_alu instid0(SALU_CYCLE_3) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; SDAG-GFX1200-NEXT: s_cselect_b32 s2, -1, 0
-; SDAG-GFX1200-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; SDAG-GFX1200-NEXT: v_dual_mov_b32 v0, 0 :: v_dual_mov_b32 v1, s2
; SDAG-GFX1200-NEXT: global_store_b32 v0, v1, s[0:1]
; SDAG-GFX1200-NEXT: s_endpgm
@@ -532,8 +532,8 @@ define amdgpu_vs void @f32_ugt(ptr addrspace(1) inreg %out, float inreg %a, floa
; GISEL-GFX1200: ; %bb.0: ; %entry
; GISEL-GFX1200-NEXT: s_cmp_nle_f32 s2, s3
; GISEL-GFX1200-NEXT: v_mov_b32_e32 v1, 0
+; GISEL-GFX1200-NEXT: s_delay_alu instid0(SALU_CYCLE_2) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GISEL-GFX1200-NEXT: s_cselect_b32 s2, -1, 0
-; GISEL-GFX1200-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GISEL-GFX1200-NEXT: v_mov_b32_e32 v0, s2
; GISEL-GFX1200-NEXT: global_store_b32 v1, v0, s[0:1]
; GISEL-GFX1200-NEXT: s_endpgm
@@ -548,8 +548,8 @@ define amdgpu_vs void @f32_une(ptr addrspace(1) inreg %out, float inreg %a, floa
; SDAG-GFX1150-LABEL: f32_une:
; SDAG-GFX1150: ; %bb.0: ; %entry
; SDAG-GFX1150-NEXT: s_cmp_neq_f32 s2, s3
+; SDAG-GFX1150-NEXT: s_delay_alu instid0(SALU_CYCLE_3) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; SDAG-GFX1150-NEXT: s_cselect_b32 s2, -1, 0
-; SDAG-GFX1150-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; SDAG-GFX1150-NEXT: v_dual_mov_b32 v0, 0 :: v_dual_mov_b32 v1, s2
; SDAG-GFX1150-NEXT: global_store_b32 v0, v1, s[0:1]
; SDAG-GFX1150-NEXT: s_endpgm
@@ -558,8 +558,8 @@ define amdgpu_vs void @f32_une(ptr addrspace(1) inreg %out, float inreg %a, floa
; GISEL-GFX1150: ; %bb.0: ; %entry
; GISEL-GFX1150-NEXT: s_cmp_neq_f32 s2, s3
; GISEL-GFX1150-NEXT: v_mov_b32_e32 v1, 0
+; GISEL-GFX1150-NEXT: s_delay_alu instid0(SALU_CYCLE_2) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GISEL-GFX1150-NEXT: s_cselect_b32 s2, -1, 0
-; GISEL-GFX1150-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GISEL-GFX1150-NEXT: v_mov_b32_e32 v0, s2
; GISEL-GFX1150-NEXT: global_store_b32 v1, v0, s[0:1]
; GISEL-GFX1150-NEXT: s_endpgm
@@ -567,8 +567,8 @@ define amdgpu_vs void @f32_une(ptr addrspace(1) inreg %out, float inreg %a, floa
; SDAG-GFX1200-LABEL: f32_une:
; SDAG-GFX1200: ; %bb.0: ; %entry
; SDAG-GFX1200-NEXT: s_cmp_neq_f32 s2, s3
+; SDAG-GFX1200-NEXT: s_delay_alu instid0(SALU_CYCLE_3) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; SDAG-GFX1200-NEXT: s_cselect_b32 s2, -1, 0
-; SDAG-GFX1200-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; SDAG-GFX1200-NEXT: v_dual_mov_b32 v0, 0 :: v_dual_mov_b32 v1, s2
; SDAG-GFX1200-NEXT: global_store_b32 v0, v1, s[0:1]
; SDAG-GFX1200-NEXT: s_endpgm
@@ -577,8 +577,8 @@ define amdgpu_vs void @f32_une(ptr addrspace(1) inreg %out, float inreg %a, floa
; GISEL-GFX1200: ; %bb.0: ; %entry
; GISEL-GFX1200-NEXT: s_cmp_neq_f32 s2, s3
; GISEL-GFX1200-NEXT: v_mov_b32_e32 v1, 0
+; GISEL-GFX1200-NEXT: s_delay_alu instid0(SALU_CYCLE_2) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GISEL-GFX1200-NEXT: s_cselect_b32 s2, -1, 0
-; GISEL-GFX1200-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GISEL-GFX1200-NEXT: v_mov_b32_e32 v0, s2
; GISEL-GFX1200-NEXT: global_store_b32 v1, v0, s[0:1]
; GISEL-GFX1200-NEXT: s_endpgm
@@ -593,8 +593,8 @@ define amdgpu_vs void @f32_uge(ptr addrspace(1) inreg %out, float inreg %a, floa
; SDAG-GFX1150-LABEL: f32_uge:
; SDAG-GFX1150: ; %bb.0: ; %entry
; SDAG-GFX1150-NEXT: s_cmp_nlt_f32 s2, s3
+; SDAG-GFX1150-NEXT: s_delay_alu instid0(SALU_CYCLE_3) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; SDAG-GFX1150-NEXT: s_cselect_b32 s2, -1, 0
-; SDAG-GFX1150-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; SDAG-GFX1150-NEXT: v_dual_mov_b32 v0, 0 :: v_dual_mov_b32 v1, s2
; SDAG-GFX1150-NEXT: global_store_b32 v0, v1, s[0:1]
; SDAG-GFX1150-NEXT: s_endpgm
@@ -603,8 +603,8 @@ define amdgpu_vs void @f32_uge(ptr addrspace(1) inreg %out, float inreg %a, floa
; GISEL-GFX1150: ; %bb.0: ; %entry
; GISEL-GFX1150-NEXT: s_cmp_nlt_f32 s2, s3
; GISEL-GFX1150-NEXT: v_mov_b32_e32 v1, 0
+; GISEL-GFX1150-NEXT: s_delay_alu instid0(SALU_CYCLE_2) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GISEL-GFX1150-NEXT: s_cselect_b32 s2, -1, 0
-; GISEL-GFX1150-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GISEL-GFX1150-NEXT: v_mov_b32_e32 v0, s2
; GISEL-GFX1150-NEXT: global_store_b32 v1, v0, s[0:1]
; GISEL-GFX1150-NEXT: s_endpgm
@@ -612,8 +612,8 @@ define amdgpu_vs void @f32_uge(ptr addrspace(1) inreg %out, float inreg %a, floa
; SDAG-GFX1200-LABEL: f32_uge:
; SDAG-GFX1200: ; %bb.0: ; %entry
; SDAG-GFX1200-NEXT: s_cmp_nlt_f32 s2, s3
+; SDAG-GFX1200-NEXT: s_delay_alu instid0(SALU_CYCLE_3) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; SDAG-GFX1200-NEXT: s_cselect_b32 s2, -1, 0
-; SDAG-GFX1200-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; SDAG-GFX1200-NEXT: v_dual_mov_b32 v0, 0 :: v_dual_mov_b32 v1, s2
; SDAG-GFX1200-NEXT: global_store_b32 v0, v1, s[0:1]
; SDAG-GFX1200-NEXT: s_endpgm
@@ -622,8 +622,8 @@ define amdgpu_vs void @f32_uge(ptr addrspace(1) inreg %out, float inreg %a, floa
; GISEL-GFX1200: ; %bb.0: ; %entry
; GISEL-GFX1200-NEXT: s_cmp_nlt_f32 s2, s3
; GISEL-GFX1200-NEXT: v_mov_b32_e32 v1, 0
+; GISEL-GFX1200-NEXT: s_delay_alu instid0(SALU_CYCLE_2) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GISEL-GFX1200-NEXT: s_cselect_b32 s2, -1, 0
-; GISEL-GFX1200-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GISEL-GFX1200-NEXT: v_mov_b32_e32 v0, s2
; GISEL-GFX1200-NEXT: global_store_b32 v1, v0, s[0:1]
; GISEL-GFX1200-NEXT: s_endpgm
@@ -638,8 +638,8 @@ define amdgpu_vs void @f16_olt(ptr addrspace(1) inreg %out, half inreg %a, half
; SDAG-GFX1150-LABEL: f16_olt:
; SDAG-GFX1150: ; %bb.0: ; %entry
; SDAG-GFX1150-NEXT: s_cmp_lt_f16 s2, s3
+; SDAG-GFX1150-NEXT: s_delay_alu instid0(SALU_CYCLE_3) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; SDAG-GFX1150-NEXT: s_cselect_b32 s2, -1, 0
-; SDAG-GFX1150-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; SDAG-GFX1150-NEXT: v_dual_mov_b32 v0, 0 :: v_dual_mov_b32 v1, s2
; SDAG-GFX1150-NEXT: global_store_b32 v0, v1, s[0:1]
; SDAG-GFX1150-NEXT: s_endpgm
@@ -648,8 +648,8 @@ define amdgpu_vs void @f16_olt(ptr addrspace(1) inreg %out, half inreg %a, half
; GISEL-GFX1150: ; %bb.0: ; %entry
; GISEL-GFX1150-NEXT: s_cmp_lt_f16 s2, s3
; GISEL-GFX1150-NEXT: v_mov_b32_e32 v1, 0
+; GISEL-GFX1150-NEXT: s_delay_alu instid0(SALU_CYCLE_2) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GISEL-GFX1150-NEXT: s_cselect_b32 s2, -1, 0
-; GISEL-GFX1150-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GISEL-GFX1150-NEXT: v_mov_b32_e32 v0, s2
; GISEL-GFX1150-NEXT: global_store_b32 v1, v0, s[0:1]
; GISEL-GFX1150-NEXT: s_endpgm
@@ -657,8 +657,8 @@ define amdgpu_vs void @f16_olt(ptr addrspace(1) inreg %out, half inreg %a, half
; SDAG-GFX1200-LABEL: f16_olt:
; SDAG-GFX1200: ; %bb.0: ; %entry
; SDAG-GFX1200-NEXT: s_cmp_lt_f16 s2, s3
+; SDAG-GFX1200-NEXT: s_delay_alu instid0(SALU_CYCLE_3) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; SDAG-GFX1200-NEXT: s_cselect_b32 s2, -1, 0
-; SDAG-GFX1200-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; SDAG-GFX1200-NEXT: v_dual_mov_b32 v0, 0 :: v_dual_mov_b32 v1, s2
; SDAG-GFX1200-NEXT: global_store_b32 v0, v1, s[0:1]
; SDAG-GFX1200-NEXT: s_endpgm
@@ -667,8 +667,8 @@ define amdgpu_vs void @f16_olt(ptr addrspace(1) inreg %out, half inreg %a, half
; GISEL-GFX1200: ; %bb.0: ; %entry
; GISEL-GFX1200-NEXT: s_cmp_lt_f16 s2, s3
; GISEL-GFX1200-NEXT: v_mov_b32_e32 v1, 0
+; GISEL-GFX1200-NEXT: s_delay_alu instid0(SALU_CYCLE_2) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GISEL-GFX1200-NEXT: s_cselect_b32 s2, -1, 0
-; GISEL-GFX1200-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GISEL-GFX1200-NEXT: v_mov_b32_e32 v0, s2
; GISEL-GFX1200-NEXT: global_store_b32 v1, v0, s[0:1]
; GISEL-GFX1200-NEXT: s_endpgm
@@ -683,8 +683,8 @@ define amdgpu_vs void @f16_oeq(ptr addrspace(1) inreg %out, half inreg %a, half
; SDAG-GFX1150-LABEL: f16_oeq:
; SDAG-GFX1150: ; %bb.0: ; %entry
; SDAG-GFX1150-NEXT: s_cmp_eq_f16 s2, s3
+; SDAG-GFX1150-NEXT: s_delay_alu instid0(SALU_CYCLE_3) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; SDAG-GFX1150-NEXT: s_cselect_b32 s2, -1, 0
-; SDAG-GFX1150-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; SDAG-GFX1150-NEXT: v_dual_mov_b32 v0, 0 :: v_dual_mov_b32 v1, s2
; SDAG-GFX1150-NEXT: global_store_b32 v0, v1, s[0:1]
; SDAG-GFX1150-NEXT: s_endpgm
@@ -693,8 +693,8 @@ define amdgpu_vs void @f16_oeq(ptr addrspace(1) inreg %out, half inreg %a, half
; GISEL-GFX1150: ; %bb.0: ; %entry
; GISEL-GFX1150-NEXT: s_cmp_eq_f16 s2, s3
; GISEL-GFX1150-NEXT: v_mov_b32_e32 v1, 0
+; GISEL-GFX1150-NEXT: s_delay_alu instid0(SALU_CYCLE_2) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GISEL-GFX1150-NEXT: s_cselect_b32 s2, -1, 0
-; GISEL-GFX1150-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GISEL-GFX1150-NEXT: v_mov_b32_e32 v0, s2
; GISEL-GFX1150-NEXT: global_store_b32 v1, v0, s[0:1]
; GISEL-GFX1150-NEXT: s_endpgm
@@ -702,8 +702,8 @@ define amdgpu_vs void @f16_oeq(ptr addrspace(1) inreg %out, half inreg %a, half
; SDAG-GFX1200-LABEL: f16_oeq:
; SDAG-GFX1200: ; %bb.0: ; %entry
; SDAG-GFX1200-NEXT: s_cmp_eq_f16 s2, s3
+; SDAG-GFX1200-NEXT: s_delay_alu instid0(SALU_CYCLE_3) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; SDAG-GFX1200-NEXT: s_cselect_b32 s2, -1, 0
-; SDAG-GFX1200-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; SDAG-GFX1200-NEXT: v_dual_mov_b32 v0, 0 :: v_dual_mov_b32 v1, s2
; SDAG-GFX1200-NEXT: global_store_b32 v0, v1, s[0:1]
; SDAG-GFX1200-NEXT: s_endpgm
@@ -712,8 +712,8 @@ define amdgpu_vs void @f16_oeq(ptr addrspace(1) inreg %out, half inreg %a, half
; GISEL-GFX1200: ; %bb.0: ; %entry
; GISEL-GFX1200-NEXT: s_cmp_eq_f16 s2, s3
; GISEL-GFX1200-NEXT: v_mov_b32_e32 v1, 0
+; GISEL-GFX1200-NEXT: s_delay_alu instid0(SALU_CYCLE_2) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GISEL-GFX1200-NEXT: s_cselect_b32 s2, -1, 0
-; GISEL-GFX1200-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GISEL-GFX1200-NEXT: v_mov_b32_e32 v0, s2
; GISEL-GFX1200-NEXT: global_store_b32 v1, v0, s[0:1]
; GISEL-GFX1200-NEXT: s_endpgm
@@ -728,8 +728,8 @@ define amdgpu_vs void @f16_ole(ptr addrspace(1) inreg %out, half inreg %a, half
; SDAG-GFX1150-LABEL: f16_ole:
; SDAG-GFX1150: ; %bb.0: ; %entry
; SDAG-GFX1150-NEXT: s_cmp_le_f16 s2, s3
+; SDAG-GFX1150-NEXT: s_delay_alu instid0(SALU_CYCLE_3) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; SDAG-GFX1150-NEXT: s_cselect_b32 s2, -1, 0
-; SDAG-GFX1150-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; SDAG-GFX1150-NEXT: v_dual_mov_b32 v0, 0 :: v_dual_mov_b32 v1, s2
; SDAG-GFX1150-NEXT: global_store_b32 v0, v1, s[0:1]
; SDAG-GFX1150-NEXT: s_endpgm
@@ -738,8 +738,8 @@ define amdgpu_vs void @f16_ole(ptr addrspace(1) inreg %out, half inreg %a, half
; GISEL-GFX1150: ; %bb.0: ; %entry
; GISEL-GFX1150-NEXT: s_cmp_le_f16 s2, s3
; GISEL-GFX1150-NEXT: v_mov_b32_e32 v1, 0
+; GISEL-GFX1150-NEXT: s_delay_alu instid0(SALU_CYCLE_2) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GISEL-GFX1150-NEXT: s_cselect_b32 s2, -1, 0
-; GISEL-GFX1150-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GISEL-GFX1150-NEXT: v_mov_b32_e32 v0, s2
; GISEL-GFX1150-NEXT: global_store_b32 v1, v0, s[0:1]
; GISEL-GFX1150-NEXT: s_endpgm
@@ -747,8 +747,8 @@ define amdgpu_vs void @f16_ole(ptr addrspace(1) inreg %out, half inreg %a, half
; SDAG-GFX1200-LABEL: f16_ole:
; SDAG-GFX1200: ; %bb.0: ; %entry
; SDAG-GFX1200-NEXT: s_cmp_le_f16 s2, s3
+; SDAG-GFX1200-NEXT: s_delay_alu instid0(SALU_CYCLE_3) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; SDAG-GFX1200-NEXT: s_cselect_b32 s2, -1, 0
-; SDAG-GFX1200-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; SDAG-GFX1200-NEXT: v_dual_mov_b32 v0, 0 :: v_dual_mov_b32 v1, s2
; SDAG-GFX1200-NEXT: global_store_b32 v0, v1, s[0:1]
; SDAG-GFX1200-NEXT: s_endpgm
@@ -757,8 +757,8 @@ define amdgpu_vs void @f16_ole(ptr addrspace(1) inreg %out, half inreg %a, half
; GISEL-GFX1200: ; %bb.0: ; %entry
; GISEL-GFX1200-NEXT: s_cmp_le_f16 s2, s3
; GISEL-GFX1200-NEXT: v_mov_b32_e32 v1, 0
+; GISEL-GFX1200-NEXT: s_delay_alu instid0(SALU_CYCLE_2) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GISEL-GFX1200-NEXT: s_cselect_b32 s2, -1, 0
-; GISEL-GFX1200-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GISEL-GFX1200-NEXT: v_mov_b32_e32 v0, s2
; GISEL-GFX1200-NEXT: global_store_b32 v1, v0, s[0:1]
; GISEL-GFX1200-NEXT: s_endpgm
@@ -773,8 +773,8 @@ define amdgpu_vs void @f16_ogt(ptr addrspace(1) inreg %out, half inreg %a, half
; SDAG-GFX1150-LABEL: f16_ogt:
; SDAG-GFX1150: ; %bb.0: ; %entry
; SDAG-GFX1150-NEXT: s_cmp_gt_f16 s2, s3
+; SDAG-GFX1150-NEXT: s_delay_alu instid0(SALU_CYCLE_3) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; SDAG-GFX1150-NEXT: s_cselect_b32 s2, -1, 0
-; SDAG-GFX1150-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; SDAG-GFX1150-NEXT: v_dual_mov_b32 v0, 0 :: v_dual_mov_b32 v1, s2
; SDAG-GFX1150-NEXT: global_store_b32 v0, v1, s[0:1]
; SDAG-GFX1150-NEXT: s_endpgm
@@ -783,8 +783,8 @@ define amdgpu_vs void @f16_ogt(ptr addrspace(1) inreg %out, half inreg %a, half
; GISEL-GFX1150: ; %bb.0: ; %entry
; GISEL-GFX1150-NEXT: s_cmp_gt_f16 s2, s3
; GISEL-GFX1150-NEXT: v_mov_b32_e32 v1, 0
+; GISEL-GFX1150-NEXT: s_delay_alu instid0(SALU_CYCLE_2) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GISEL-GFX1150-NEXT: s_cselect_b32 s2, -1, 0
-; GISEL-GFX1150-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GISEL-GFX1150-NEXT: v_mov_b32_e32 v0, s2
; GISEL-GFX1150-NEXT: global_store_b32 v1, v0, s[0:1]
; GISEL-GFX1150-NEXT: s_endpgm
@@ -792,8 +792,8 @@ define amdgpu_vs void @f16_ogt(ptr addrspace(1) inreg %out, half inreg %a, half
; SDAG-GFX1200-LABEL: f16_ogt:
; SDAG-GFX1200: ; %bb.0: ; %entry
; SDAG-GFX1200-NEXT: s_cmp_gt_f16 s2, s3
+; SDAG-GFX1200-NEXT: s_delay_alu instid0(SALU_CYCLE_3) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; SDAG-GFX1200-NEXT: s_cselect_b32 s2, -1, 0
-; SDAG-GFX1200-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; SDAG-GFX1200-NEXT: v_dual_mov_b32 v0, 0 :: v_dual_mov_b32 v1, s2
; SDAG-GFX1200-NEXT: global_store_b32 v0, v1, s[0:1]
; SDAG-GFX1200-NEXT: s_endpgm
@@ -802,8 +802,8 @@ define amdgpu_vs void @f16_ogt(ptr addrspace(1) inreg %out, half inreg %a, half
; GISEL-GFX1200: ; %bb.0: ; %entry
; GISEL-GFX1200-NEXT: s_cmp_gt_f16 s2, s3
; GISEL-GFX1200-NEXT: v_mov_b32_e32 v1, 0
+; GISEL-GFX1200-NEXT: s_delay_alu instid0(SALU_CYCLE_2) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GISEL-GFX1200-NEXT: s_cselect_b32 s2, -1, 0
-; GISEL-GFX1200-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GISEL-GFX1200-NEXT: v_mov_b32_e32 v0, s2
; GISEL-GFX1200-NEXT: global_store_b32 v1, v0, s[0:1]
; GISEL-GFX1200-NEXT: s_endpgm
@@ -818,8 +818,8 @@ define amdgpu_vs void @f16_one(ptr addrspace(1) inreg %out, half inreg %a, half
; SDAG-GFX1150-LABEL: f16_one:
; SDAG-GFX1150: ; %bb.0: ; %entry
; SDAG-GFX1150-NEXT: s_cmp_lg_f16 s2, s3
+; SDAG-GFX1150-NEXT: s_delay_alu instid0(SALU_CYCLE_3) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; SDAG-GFX1150-NEXT: s_cselect_b32 s2, -1, 0
-; SDAG-GFX1150-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; SDAG-GFX1150-NEXT: v_dual_mov_b32 v0, 0 :: v_dual_mov_b32 v1, s2
; SDAG-GFX1150-NEXT: global_store_b32 v0, v1, s[0:1]
; SDAG-GFX1150-NEXT: s_endpgm
@@ -828,8 +828,8 @@ define amdgpu_vs void @f16_one(ptr addrspace(1) inreg %out, half inreg %a, half
; GISEL-GFX1150: ; %bb.0: ; %entry
; GISEL-GFX1150-NEXT: s_cmp_lg_f16 s2, s3
; GISEL-GFX1150-NEXT: v_mov_b32_e32 v1, 0
+; GISEL-GFX1150-NEXT: s_delay_alu instid0(SALU_CYCLE_2) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GISEL-GFX1150-NEXT: s_cselect_b32 s2, -1, 0
-; GISEL-GFX1150-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GISEL-GFX1150-NEXT: v_mov_b32_e32 v0, s2
; GISEL-GFX1150-NEXT: global_store_b32 v1, v0, s[0:1]
; GISEL-GFX1150-NEXT: s_endpgm
@@ -837,8 +837,8 @@ define amdgpu_vs void @f16_one(ptr addrspace(1) inreg %out, half inreg %a, half
; SDAG-GFX1200-LABEL: f16_one:
; SDAG-GFX1200: ; %bb.0: ; %entry
; SDAG-GFX1200-NEXT: s_cmp_lg_f16 s2, s3
+; SDAG-GFX1200-NEXT: s_delay_alu instid0(SALU_CYCLE_3) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; SDAG-GFX1200-NEXT: s_cselect_b32 s2, -1, 0
-; SDAG-GFX1200-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; SDAG-GFX1200-NEXT: v_dual_mov_b32 v0, 0 :: v_dual_mov_b32 v1, s2
; SDAG-GFX1200-NEXT: global_store_b32 v0, v1, s[0:1]
; SDAG-GFX1200-NEXT: s_endpgm
@@ -847,8 +847,8 @@ define amdgpu_vs void @f16_one(ptr addrspace(1) inreg %out, half inreg %a, half
; GISEL-GFX1200: ; %bb.0: ; %entry
; GISEL-GFX1200-NEXT: s_cmp_lg_f16 s2, s3
; GISEL-GFX1200-NEXT: v_mov_b32_e32 v1, 0
+; GISEL-GFX1200-NEXT: s_delay_alu instid0(SALU_CYCLE_2) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GISEL-GFX1200-NEXT: s_cselect_b32 s2, -1, 0
-; GISEL-GFX1200-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GISEL-GFX1200-NEXT: v_mov_b32_e32 v0, s2
; GISEL-GFX1200-NEXT: global_store_b32 v1, v0, s[0:1]
; GISEL-GFX1200-NEXT: s_endpgm
@@ -863,8 +863,8 @@ define amdgpu_vs void @f16_oge(ptr addrspace(1) inreg %out, half inreg %a, half
; SDAG-GFX1150-LABEL: f16_oge:
; SDAG-GFX1150: ; %bb.0: ; %entry
; SDAG-GFX1150-NEXT: s_cmp_ge_f16 s2, s3
+; SDAG-GFX1150-NEXT: s_delay_alu instid0(SALU_CYCLE_3) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; SDAG-GFX1150-NEXT: s_cselect_b32 s2, -1, 0
-; SDAG-GFX1150-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; SDAG-GFX1150-NEXT: v_dual_mov_b32 v0, 0 :: v_dual_mov_b32 v1, s2
; SDAG-GFX1150-NEXT: global_store_b32 v0, v1, s[0:1]
; SDAG-GFX1150-NEXT: s_endpgm
@@ -873,8 +873,8 @@ define amdgpu_vs void @f16_oge(ptr addrspace(1) inreg %out, half inreg %a, half
; GISEL-GFX1150: ; %bb.0: ; %entry
; GISEL-GFX1150-NEXT: s_cmp_ge_f16 s2, s3
; GISEL-GFX1150-NEXT: v_mov_b32_e32 v1, 0
+; GISEL-GFX1150-NEXT: s_delay_alu instid0(SALU_CYCLE_2) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GISEL-GFX1150-NEXT: s_cselect_b32 s2, -1, 0
-; GISEL-GFX1150-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GISEL-GFX1150-NEXT: v_mov_b32_e32 v0, s2
; GISEL-GFX1150-NEXT: global_store_b32 v1, v0, s[0:1]
; GISEL-GFX1150-NEXT: s_endpgm
@@ -882,8 +882,8 @@ define amdgpu_vs void @f16_oge(ptr addrspace(1) inreg %out, half inreg %a, half
; SDAG-GFX1200-LABEL: f16_oge:
; SDAG-GFX1200: ; %bb.0: ; %entry
; SDAG-GFX1200-NEXT: s_cmp_ge_f16 s2, s3
+; SDAG-GFX1200-NEXT: s_delay_alu instid0(SALU_CYCLE_3) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; SDAG-GFX1200-NEXT: s_cselect_b32 s2, -1, 0
-; SDAG-GFX1200-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; SDAG-GFX1200-NEXT: v_dual_mov_b32 v0, 0 :: v_dual_mov_b32 v1, s2
; SDAG-GFX1200-NEXT: global_store_b32 v0, v1, s[0:1]
; SDAG-GFX1200-NEXT: s_endpgm
@@ -892,8 +892,8 @@ define amdgpu_vs void @f16_oge(ptr addrspace(1) inreg %out, half inreg %a, half
; GISEL-GFX1200: ; %bb.0: ; %entry
; GISEL-GFX1200-NEXT: s_cmp_ge_f16 s2, s3
; GISEL-GFX1200-NEXT: v_mov_b32_e32 v1, 0
+; GISEL-GFX1200-NEXT: s_delay_alu instid0(SALU_CYCLE_2) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GISEL-GFX1200-NEXT: s_cselect_b32 s2, -1, 0
-; GISEL-GFX1200-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GISEL-GFX1200-NEXT: v_mov_b32_e32 v0, s2
; GISEL-GFX1200-NEXT: global_store_b32 v1, v0, s[0:1]
; GISEL-GFX1200-NEXT: s_endpgm
@@ -908,8 +908,8 @@ define amdgpu_vs void @f16_ord(ptr addrspace(1) inreg %out, half inreg %a, half
; SDAG-GFX1150-LABEL: f16_ord:
; SDAG-GFX1150: ; %bb.0: ; %entry
; SDAG-GFX1150-NEXT: s_cmp_o_f16 s2, s3
+; SDAG-GFX1150-NEXT: s_delay_alu instid0(SALU_CYCLE_3) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; SDAG-GFX1150-NEXT: s_cselect_b32 s2, -1, 0
-; SDAG-GFX1150-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; SDAG-GFX1150-NEXT: v_dual_mov_b32 v0, 0 :: v_dual_mov_b32 v1, s2
; SDAG-GFX1150-NEXT: global_store_b32 v0, v1, s[0:1]
; SDAG-GFX1150-NEXT: s_endpgm
@@ -918,8 +918,8 @@ define amdgpu_vs void @f16_ord(ptr addrspace(1) inreg %out, half inreg %a, half
; GISEL-GFX1150: ; %bb.0: ; %entry
; GISEL-GFX1150-NEXT: s_cmp_o_f16 s2, s3
; GISEL-GFX1150-NEXT: v_mov_b32_e32 v1, 0
+; GISEL-GFX1150-NEXT: s_delay_alu instid0(SALU_CYCLE_2) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GISEL-GFX1150-NEXT: s_cselect_b32 s2, -1, 0
-; GISEL-GFX1150-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GISEL-GFX1150-NEXT: v_mov_b32_e32 v0, s2
; GISEL-GFX1150-NEXT: global_store_b32 v1, v0, s[0:1]
; GISEL-GFX1150-NEXT: s_endpgm
@@ -927,8 +927,8 @@ define amdgpu_vs void @f16_ord(ptr addrspace(1) inreg %out, half inreg %a, half
; SDAG-GFX1200-LABEL: f16_ord:
; SDAG-GFX1200: ; %bb.0: ; %entry
; SDAG-GFX1200-NEXT: s_cmp_o_f16 s2, s3
+; SDAG-GFX1200-NEXT: s_delay_alu instid0(SALU_CYCLE_3) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; SDAG-GFX1200-NEXT: s_cselect_b32 s2, -1, 0
-; SDAG-GFX1200-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; SDAG-GFX1200-NEXT: v_dual_mov_b32 v0, 0 :: v_dual_mov_b32 v1, s2
; SDAG-GFX1200-NEXT: global_store_b32 v0, v1, s[0:1]
; SDAG-GFX1200-NEXT: s_endpgm
@@ -937,8 +937,8 @@ define amdgpu_vs void @f16_ord(ptr addrspace(1) inreg %out, half inreg %a, half
; GISEL-GFX1200: ; %bb.0: ; %entry
; GISEL-GFX1200-NEXT: s_cmp_o_f16 s2, s3
; GISEL-GFX1200-NEXT: v_mov_b32_e32 v1, 0
+; GISEL-GFX1200-NEXT: s_delay_alu instid0(SALU_CYCLE_2) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GISEL-GFX1200-NEXT: s_cselect_b32 s2, -1, 0
-; GISEL-GFX1200-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GISEL-GFX1200-NEXT: v_mov_b32_e32 v0, s2
; GISEL-GFX1200-NEXT: global_store_b32 v1, v0, s[0:1]
; GISEL-GFX1200-NEXT: s_endpgm
@@ -953,8 +953,8 @@ define amdgpu_vs void @f16_uno(ptr addrspace(1) inreg %out, half inreg %a, half
; SDAG-GFX1150-LABEL: f16_uno:
; SDAG-GFX1150: ; %bb.0: ; %entry
; SDAG-GFX1150-NEXT: s_cmp_u_f16 s2, s3
+; SDAG-GFX1150-NEXT: s_delay_alu instid0(SALU_CYCLE_3) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; SDAG-GFX1150-NEXT: s_cselect_b32 s2, -1, 0
-; SDAG-GFX1150-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; SDAG-GFX1150-NEXT: v_dual_mov_b32 v0, 0 :: v_dual_mov_b32 v1, s2
; SDAG-GFX1150-NEXT: global_store_b32 v0, v1, s[0:1]
; SDAG-GFX1150-NEXT: s_endpgm
@@ -963,8 +963,8 @@ define amdgpu_vs void @f16_uno(ptr addrspace(1) inreg %out, half inreg %a, half
; GISEL-GFX1150: ; %bb.0: ; %entry
; GISEL-GFX1150-NEXT: s_cmp_u_f16 s2, s3
; GISEL-GFX1150-NEXT: v_mov_b32_e32 v1, 0
+; GISEL-GFX1150-NEXT: s_delay_alu instid0(SALU_CYCLE_2) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GISEL-GFX1150-NEXT: s_cselect_b32 s2, -1, 0
-; GISEL-GFX1150-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GISEL-GFX1150-NEXT: v_mov_b32_e32 v0, s2
; GISEL-GFX1150-NEXT: global_store_b32 v1, v0, s[0:1]
; GISEL-GFX1150-NEXT: s_endpgm
@@ -972,8 +972,8 @@ define amdgpu_vs void @f16_uno(ptr addrspace(1) inreg %out, half inreg %a, half
; SDAG-GFX1200-LABEL: f16_uno:
; SDAG-GFX1200: ; %bb.0: ; %entry
; SDAG-GFX1200-NEXT: s_cmp_u_f16 s2, s3
+; SDAG-GFX1200-NEXT: s_delay_alu instid0(SALU_CYCLE_3) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; SDAG-GFX1200-NEXT: s_cselect_b32 s2, -1, 0
-; SDAG-GFX1200-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; SDAG-GFX1200-NEXT: v_dual_mov_b32 v0, 0 :: v_dual_mov_b32 v1, s2
; SDAG-GFX1200-NEXT: global_store_b32 v0, v1, s[0:1]
; SDAG-GFX1200-NEXT: s_endpgm
@@ -982,8 +982,8 @@ define amdgpu_vs void @f16_uno(ptr addrspace(1) inreg %out, half inreg %a, half
; GISEL-GFX1200: ; %bb.0: ; %entry
; GISEL-GFX1200-NEXT: s_cmp_u_f16 s2, s3
; GISEL-GFX1200-NEXT: v_mov_b32_e32 v1, 0
+; GISEL-GFX1200-NEXT: s_delay_alu instid0(SALU_CYCLE_2) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GISEL-GFX1200-NEXT: s_cselect_b32 s2, -1, 0
-; GISEL-GFX1200-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GISEL-GFX1200-NEXT: v_mov_b32_e32 v0, s2
; GISEL-GFX1200-NEXT: global_store_b32 v1, v0, s[0:1]
; GISEL-GFX1200-NEXT: s_endpgm
@@ -998,8 +998,8 @@ define amdgpu_vs void @f16_ult(ptr addrspace(1) inreg %out, half inreg %a, half
; SDAG-GFX1150-LABEL: f16_ult:
; SDAG-GFX1150: ; %bb.0: ; %entry
; SDAG-GFX1150-NEXT: s_cmp_nge_f16 s2, s3
+; SDAG-GFX1150-NEXT: s_delay_alu instid0(SALU_CYCLE_3) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; SDAG-GFX1150-NEXT: s_cselect_b32 s2, -1, 0
-; SDAG-GFX1150-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; SDAG-GFX1150-NEXT: v_dual_mov_b32 v0, 0 :: v_dual_mov_b32 v1, s2
; SDAG-GFX1150-NEXT: global_store_b32 v0, v1, s[0:1]
; SDAG-GFX1150-NEXT: s_endpgm
@@ -1008,8 +1008,8 @@ define amdgpu_vs void @f16_ult(ptr addrspace(1) inreg %out, half inreg %a, half
; GISEL-GFX1150: ; %bb.0: ; %entry
; GISEL-GFX1150-NEXT: s_cmp_nge_f16 s2, s3
; GISEL-GFX1150-NEXT: v_mov_b32_e32 v1, 0
+; GISEL-GFX1150-NEXT: s_delay_alu instid0(SALU_CYCLE_2) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GISEL-GFX1150-NEXT: s_cselect_b32 s2, -1, 0
-; GISEL-GFX1150-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GISEL-GFX1150-NEXT: v_mov_b32_e32 v0, s2
; GISEL-GFX1150-NEXT: global_store_b32 v1, v0, s[0:1]
; GISEL-GFX1150-NEXT: s_endpgm
@@ -1017,8 +1017,8 @@ define amdgpu_vs void @f16_ult(ptr addrspace(1) inreg %out, half inreg %a, half
; SDAG-GFX1200-LABEL: f16_ult:
; SDAG-GFX1200: ; %bb.0: ; %entry
; SDAG-GFX1200-NEXT: s_cmp_nge_f16 s2, s3
+; SDAG-GFX1200-NEXT: s_delay_alu instid0(SALU_CYCLE_3) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; SDAG-GFX1200-NEXT: s_cselect_b32 s2, -1, 0
-; SDAG-GFX1200-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; SDAG-GFX1200-NEXT: v_dual_mov_b32 v0, 0 :: v_dual_mov_b32 v1, s2
; SDAG-GFX1200-NEXT: global_store_b32 v0, v1, s[0:1]
; SDAG-GFX1200-NEXT: s_endpgm
@@ -1027,8 +1027,8 @@ define amdgpu_vs void @f16_ult(ptr addrspace(1) inreg %out, half inreg %a, half
; GISEL-GFX1200: ; %bb.0: ; %entry
; GISEL-GFX1200-NEXT: s_cmp_nge_f16 s2, s3
; GISEL-GFX1200-NEXT: v_mov_b32_e32 v1, 0
+; GISEL-GFX1200-NEXT: s_delay_alu instid0(SALU_CYCLE_2) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GISEL-GFX1200-NEXT: s_cselect_b32 s2, -1, 0
-; GISEL-GFX1200-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GISEL-GFX1200-NEXT: v_mov_b32_e32 v0, s2
; GISEL-GFX1200-NEXT: global_store_b32 v1, v0, s[0:1]
; GISEL-GFX1200-NEXT: s_endpgm
@@ -1043,8 +1043,8 @@ define amdgpu_vs void @f16_ueq(ptr addrspace(1) inreg %out, half inreg %a, half
; SDAG-GFX1150-LABEL: f16_ueq:
; SDAG-GFX1150: ; %bb.0: ; %entry
; SDAG-GFX1150-NEXT: s_cmp_nlg_f16 s2, s3
+; SDAG-GFX1150-NEXT: s_delay_alu instid0(SALU_CYCLE_3) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; SDAG-GFX1150-NEXT: s_cselect_b32 s2, -1, 0
-; SDAG-GFX1150-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; SDAG-GFX1150-NEXT: v_dual_mov_b32 v0, 0 :: v_dual_mov_b32 v1, s2
; SDAG-GFX1150-NEXT: global_store_b32 v0, v1, s[0:1]
; SDAG-GFX1150-NEXT: s_endpgm
@@ -1053,8 +1053,8 @@ define amdgpu_vs void @f16_ueq(ptr addrspace(1) inreg %out, half inreg %a, half
; GISEL-GFX1150: ; %bb.0: ; %entry
; GISEL-GFX1150-NEXT: s_cmp_nlg_f16 s2, s3
; GISEL-GFX1150-NEXT: v_mov_b32_e32 v1, 0
+; GISEL-GFX1150-NEXT: s_delay_alu instid0(SALU_CYCLE_2) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GISEL-GFX1150-NEXT: s_cselect_b32 s2, -1, 0
-; GISEL-GFX1150-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GISEL-GFX1150-NEXT: v_mov_b32_e32 v0, s2
; GISEL-GFX1150-NEXT: global_store_b32 v1, v0, s[0:1]
; GISEL-GFX1150-NEXT: s_endpgm
@@ -1062,8 +1062,8 @@ define amdgpu_vs void @f16_ueq(ptr addrspace(1) inreg %out, half inreg %a, half
; SDAG-GFX1200-LABEL: f16_ueq:
; SDAG-GFX1200: ; %bb.0: ; %entry
; SDAG-GFX1200-NEXT: s_cmp_nlg_f16 s2, s3
+; SDAG-GFX1200-NEXT: s_delay_alu instid0(SALU_CYCLE_3) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; SDAG-GFX1200-NEXT: s_cselect_b32 s2, -1, 0
-; SDAG-GFX1200-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; SDAG-GFX1200-NEXT: v_dual_mov_b32 v0, 0 :: v_dual_mov_b32 v1, s2
; SDAG-GFX1200-NEXT: global_store_b32 v0, v1, s[0:1]
; SDAG-GFX1200-NEXT: s_endpgm
@@ -1072,8 +1072,8 @@ define amdgpu_vs void @f16_ueq(ptr addrspace(1) inreg %out, half inreg %a, half
; GISEL-GFX1200: ; %bb.0: ; %entry
; GISEL-GFX1200-NEXT: s_cmp_nlg_f16 s2, s3
; GISEL-GFX1200-NEXT: v_mov_b32_e32 v1, 0
+; GISEL-GFX1200-NEXT: s_delay_alu instid0(SALU_CYCLE_2) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GISEL-GFX1200-NEXT: s_cselect_b32 s2, -1, 0
-; GISEL-GFX1200-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GISEL-GFX1200-NEXT: v_mov_b32_e32 v0, s2
; GISEL-GFX1200-NEXT: global_store_b32 v1, v0, s[0:1]
; GISEL-GFX1200-NEXT: s_endpgm
@@ -1088,8 +1088,8 @@ define amdgpu_vs void @f16_ule(ptr addrspace(1) inreg %out, half inreg %a, half
; SDAG-GFX1150-LABEL: f16_ule:
; SDAG-GFX1150: ; %bb.0: ; %entry
; SDAG-GFX1150-NEXT: s_cmp_ngt_f16 s2, s3
+; SDAG-GFX1150-NEXT: s_delay_alu instid0(SALU_CYCLE_3) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; SDAG-GFX1150-NEXT: s_cselect_b32 s2, -1, 0
-; SDAG-GFX1150-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; SDAG-GFX1150-NEXT: v_dual_mov_b32 v0, 0 :: v_dual_mov_b32 v1, s2
; SDAG-GFX1150-NEXT: global_store_b32 v0, v1, s[0:1]
; SDAG-GFX1150-NEXT: s_endpgm
@@ -1098,8 +1098,8 @@ define amdgpu_vs void @f16_ule(ptr addrspace(1) inreg %out, half inreg %a, half
; GISEL-GFX1150: ; %bb.0: ; %entry
; GISEL-GFX1150-NEXT: s_cmp_ngt_f16 s2, s3
; GISEL-GFX1150-NEXT: v_mov_b32_e32 v1, 0
+; GISEL-GFX1150-NEXT: s_delay_alu instid0(SALU_CYCLE_2) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GISEL-GFX1150-NEXT: s_cselect_b32 s2, -1, 0
-; GISEL-GFX1150-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GISEL-GFX1150-NEXT: v_mov_b32_e32 v0, s2
; GISEL-GFX1150-NEXT: global_store_b32 v1, v0, s[0:1]
; GISEL-GFX1150-NEXT: s_endpgm
@@ -1107,8 +1107,8 @@ define amdgpu_vs void @f16_ule(ptr addrspace(1) inreg %out, half inreg %a, half
; SDAG-GFX1200-LABEL: f16_ule:
; SDAG-GFX1200: ; %bb.0: ; %entry
; SDAG-GFX1200-NEXT: s_cmp_ngt_f16 s2, s3
+; SDAG-GFX1200-NEXT: s_delay_alu instid0(SALU_CYCLE_3) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; SDAG-GFX1200-NEXT: s_cselect_b32 s2, -1, 0
-; SDAG-GFX1200-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; SDAG-GFX1200-NEXT: v_dual_mov_b32 v0, 0 :: v_dual_mov_b32 v1, s2
; SDAG-GFX1200-NEXT: global_store_b32 v0, v1, s[0:1]
; SDAG-GFX1200-NEXT: s_endpgm
@@ -1117,8 +1117,8 @@ define amdgpu_vs void @f16_ule(ptr addrspace(1) inreg %out, half inreg %a, half
; GISEL-GFX1200: ; %bb.0: ; %entry
; GISEL-GFX1200-NEXT: s_cmp_ngt_f16 s2, s3
; GISEL-GFX1200-NEXT: v_mov_b32_e32 v1, 0
+; GISEL-GFX1200-NEXT: s_delay_alu instid0(SALU_CYCLE_2) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GISEL-GFX1200-NEXT: s_cselect_b32 s2, -1, 0
-; GISEL-GFX1200-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GISEL-GFX1200-NEXT: v_mov_b32_e32 v0, s2
; GISEL-GFX1200-NEXT: global_store_b32 v1, v0, s[0:1]
; GISEL-GFX1200-NEXT: s_endpgm
@@ -1133,8 +1133,8 @@ define amdgpu_vs void @f16_ugt(ptr addrspace(1) inreg %out, half inreg %a, half
; SDAG-GFX1150-LABEL: f16_ugt:
; SDAG-GFX1150: ; %bb.0: ; %entry
; SDAG-GFX1150-NEXT: s_cmp_nle_f16 s2, s3
+; SDAG-GFX1150-NEXT: s_delay_alu instid0(SALU_CYCLE_3) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; SDAG-GFX1150-NEXT: s_cselect_b32 s2, -1, 0
-; SDAG-GFX1150-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; SDAG-GFX1150-NEXT: v_dual_mov_b32 v0, 0 :: v_dual_mov_b32 v1, s2
; SDAG-GFX1150-NEXT: global_store_b32 v0, v1, s[0:1]
; SDAG-GFX1150-NEXT: s_endpgm
@@ -1143,8 +1143,8 @@ define amdgpu_vs void @f16_ugt(ptr addrspace(1) inreg %out, half inreg %a, half
; GISEL-GFX1150: ; %bb.0: ; %entry
; GISEL-GFX1150-NEXT: s_cmp_nle_f16 s2, s3
; GISEL-GFX1150-NEXT: v_mov_b32_e32 v1, 0
+; GISEL-GFX1150-NEXT: s_delay_alu instid0(SALU_CYCLE_2) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GISEL-GFX1150-NEXT: s_cselect_b32 s2, -1, 0
-; GISEL-GFX1150-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GISEL-GFX1150-NEXT: v_mov_b32_e32 v0, s2
; GISEL-GFX1150-NEXT: global_store_b32 v1, v0, s[0:1]
; GISEL-GFX1150-NEXT: s_endpgm
@@ -1152,8 +1152,8 @@ define amdgpu_vs void @f16_ugt(ptr addrspace(1) inreg %out, half inreg %a, half
; SDAG-GFX1200-LABEL: f16_ugt:
; SDAG-GFX1200: ; %bb.0: ; %entry
; SDAG-GFX1200-NEXT: s_cmp_nle_f16 s2, s3
+; SDAG-GFX1200-NEXT: s_delay_alu instid0(SALU_CYCLE_3) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; SDAG-GFX1200-NEXT: s_cselect_b32 s2, -1, 0
-; SDAG-GFX1200-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; SDAG-GFX1200-NEXT: v_dual_mov_b32 v0, 0 :: v_dual_mov_b32 v1, s2
; SDAG-GFX1200-NEXT: global_store_b32 v0, v1, s[0:1]
; SDAG-GFX1200-NEXT: s_endpgm
@@ -1162,8 +1162,8 @@ define amdgpu_vs void @f16_ugt(ptr addrspace(1) inreg %out, half inreg %a, half
; GISEL-GFX1200: ; %bb.0: ; %entry
; GISEL-GFX1200-NEXT: s_cmp_nle_f16 s2, s3
; GISEL-GFX1200-NEXT: v_mov_b32_e32 v1, 0
+; GISEL-GFX1200-NEXT: s_delay_alu instid0(SALU_CYCLE_2) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GISEL-GFX1200-NEXT: s_cselect_b32 s2, -1, 0
-; GISEL-GFX1200-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GISEL-GFX1200-NEXT: v_mov_b32_e32 v0, s2
; GISEL-GFX1200-NEXT: global_store_b32 v1, v0, s[0:1]
; GISEL-GFX1200-NEXT: s_endpgm
@@ -1178,8 +1178,8 @@ define amdgpu_vs void @f16_une(ptr addrspace(1) inreg %out, half inreg %a, half
; SDAG-GFX1150-LABEL: f16_une:
; SDAG-GFX1150: ; %bb.0: ; %entry
; SDAG-GFX1150-NEXT: s_cmp_neq_f16 s2, s3
+; SDAG-GFX1150-NEXT: s_delay_alu instid0(SALU_CYCLE_3) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; SDAG-GFX1150-NEXT: s_cselect_b32 s2, -1, 0
-; SDAG-GFX1150-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; SDAG-GFX1150-NEXT: v_dual_mov_b32 v0, 0 :: v_dual_mov_b32 v1, s2
; SDAG-GFX1150-NEXT: global_store_b32 v0, v1, s[0:1]
; SDAG-GFX1150-NEXT: s_endpgm
@@ -1188,8 +1188,8 @@ define amdgpu_vs void @f16_une(ptr addrspace(1) inreg %out, half inreg %a, half
; GISEL-GFX1150: ; %bb.0: ; %entry
; GISEL-GFX1150-NEXT: s_cmp_neq_f16 s2, s3
; GISEL-GFX1150-NEXT: v_mov_b32_e32 v1, 0
+; GISEL-GFX1150-NEXT: s_delay_alu instid0(SALU_CYCLE_2) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GISEL-GFX1150-NEXT: s_cselect_b32 s2, -1, 0
-; GISEL-GFX1150-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GISEL-GFX1150-NEXT: v_mov_b32_e32 v0, s2
; GISEL-GFX1150-NEXT: global_store_b32 v1, v0, s[0:1]
; GISEL-GFX1150-NEXT: s_endpgm
@@ -1197,8 +1197,8 @@ define amdgpu_vs void @f16_une(ptr addrspace(1) inreg %out, half inreg %a, half
; SDAG-GFX1200-LABEL: f16_une:
; SDAG-GFX1200: ; %bb.0: ; %entry
; SDAG-GFX1200-NEXT: s_cmp_neq_f16 s2, s3
+; SDAG-GFX1200-NEXT: s_delay_alu instid0(SALU_CYCLE_3) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; SDAG-GFX1200-NEXT: s_cselect_b32 s2, -1, 0
-; SDAG-GFX1200-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; SDAG-GFX1200-NEXT: v_dual_mov_b32 v0, 0 :: v_dual_mov_b32 v1, s2
; SDAG-GFX1200-NEXT: global_store_b32 v0, v1, s[0:1]
; SDAG-GFX1200-NEXT: s_endpgm
@@ -1207,8 +1207,8 @@ define amdgpu_vs void @f16_une(ptr addrspace(1) inreg %out, half inreg %a, half
; GISEL-GFX1200: ; %bb.0: ; %entry
; GISEL-GFX1200-NEXT: s_cmp_neq_f16 s2, s3
; GISEL-GFX1200-NEXT: v_mov_b32_e32 v1, 0
+; GISEL-GFX1200-NEXT: s_delay_alu instid0(SALU_CYCLE_2) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GISEL-GFX1200-NEXT: s_cselect_b32 s2, -1, 0
-; GISEL-GFX1200-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GISEL-GFX1200-NEXT: v_mov_b32_e32 v0, s2
; GISEL-GFX1200-NEXT: global_store_b32 v1, v0, s[0:1]
; GISEL-GFX1200-NEXT: s_endpgm
@@ -1223,8 +1223,8 @@ define amdgpu_vs void @f16_uge(ptr addrspace(1) inreg %out, half inreg %a, half
; SDAG-GFX1150-LABEL: f16_uge:
; SDAG-GFX1150: ; %bb.0: ; %entry
; SDAG-GFX1150-NEXT: s_cmp_nlt_f16 s2, s3
+; SDAG-GFX1150-NEXT: s_delay_alu instid0(SALU_CYCLE_3) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; SDAG-GFX1150-NEXT: s_cselect_b32 s2, -1, 0
-; SDAG-GFX1150-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; SDAG-GFX1150-NEXT: v_dual_mov_b32 v0, 0 :: v_dual_mov_b32 v1, s2
; SDAG-GFX1150-NEXT: global_store_b32 v0, v1, s[0:1]
; SDAG-GFX1150-NEXT: s_endpgm
@@ -1233,8 +1233,8 @@ define amdgpu_vs void @f16_uge(ptr addrspace(1) inreg %out, half inreg %a, half
; GISEL-GFX1150: ; %bb.0: ; %entry
; GISEL-GFX1150-NEXT: s_cmp_nlt_f16 s2, s3
; GISEL-GFX1150-NEXT: v_mov_b32_e32 v1, 0
+; GISEL-GFX1150-NEXT: s_delay_alu instid0(SALU_CYCLE_2) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GISEL-GFX1150-NEXT: s_cselect_b32 s2, -1, 0
-; GISEL-GFX1150-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GISEL-GFX1150-NEXT: v_mov_b32_e32 v0, s2
; GISEL-GFX1150-NEXT: global_store_b32 v1, v0, s[0:1]
; GISEL-GFX1150-NEXT: s_endpgm
@@ -1242,8 +1242,8 @@ define amdgpu_vs void @f16_uge(ptr addrspace(1) inreg %out, half inreg %a, half
; SDAG-GFX1200-LABEL: f16_uge:
; SDAG-GFX1200: ; %bb.0: ; %entry
; SDAG-GFX1200-NEXT: s_cmp_nlt_f16 s2, s3
+; SDAG-GFX1200-NEXT: s_delay_alu instid0(SALU_CYCLE_3) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; SDAG-GFX1200-NEXT: s_cselect_b32 s2, -1, 0
-; SDAG-GFX1200-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; SDAG-GFX1200-NEXT: v_dual_mov_b32 v0, 0 :: v_dual_mov_b32 v1, s2
; SDAG-GFX1200-NEXT: global_store_b32 v0, v1, s[0:1]
; SDAG-GFX1200-NEXT: s_endpgm
@@ -1252,8 +1252,8 @@ define amdgpu_vs void @f16_uge(ptr addrspace(1) inreg %out, half inreg %a, half
; GISEL-GFX1200: ; %bb.0: ; %entry
; GISEL-GFX1200-NEXT: s_cmp_nlt_f16 s2, s3
; GISEL-GFX1200-NEXT: v_mov_b32_e32 v1, 0
+; GISEL-GFX1200-NEXT: s_delay_alu instid0(SALU_CYCLE_2) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GISEL-GFX1200-NEXT: s_cselect_b32 s2, -1, 0
-; GISEL-GFX1200-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GISEL-GFX1200-NEXT: v_mov_b32_e32 v0, s2
; GISEL-GFX1200-NEXT: global_store_b32 v1, v0, s[0:1]
; GISEL-GFX1200-NEXT: s_endpgm
@@ -1314,23 +1314,26 @@ define <8 x i1> @vector_f32_ole() {
; GISEL-GFX1150-NEXT: v_readfirstlane_b32 s7, v7
; GISEL-GFX1150-NEXT: s_cselect_b32 s0, 1, 0
; GISEL-GFX1150-NEXT: s_cmp_le_f32 s1, 0
+; GISEL-GFX1150-NEXT: s_delay_alu instid0(SALU_CYCLE_3) | instskip(SKIP_2) | instid1(SALU_CYCLE_2)
; GISEL-GFX1150-NEXT: s_cselect_b32 s1, 1, 0
; GISEL-GFX1150-NEXT: s_cmp_le_f32 s2, 0
; GISEL-GFX1150-NEXT: v_dual_mov_b32 v0, s0 :: v_dual_mov_b32 v1, s1
; GISEL-GFX1150-NEXT: s_cselect_b32 s2, 1, 0
; GISEL-GFX1150-NEXT: s_cmp_le_f32 s3, 0
+; GISEL-GFX1150-NEXT: s_delay_alu instid0(SALU_CYCLE_3) | instskip(SKIP_2) | instid1(SALU_CYCLE_2)
; GISEL-GFX1150-NEXT: s_cselect_b32 s3, 1, 0
; GISEL-GFX1150-NEXT: s_cmp_le_f32 s4, 0
; GISEL-GFX1150-NEXT: v_dual_mov_b32 v2, s2 :: v_dual_mov_b32 v3, s3
; GISEL-GFX1150-NEXT: s_cselect_b32 s4, 1, 0
; GISEL-GFX1150-NEXT: s_cmp_le_f32 s5, 0
+; GISEL-GFX1150-NEXT: s_delay_alu instid0(SALU_CYCLE_3) | instskip(SKIP_2) | instid1(SALU_CYCLE_2)
; GISEL-GFX1150-NEXT: s_cselect_b32 s5, 1, 0
; GISEL-GFX1150-NEXT: s_cmp_le_f32 s6, 0
; GISEL-GFX1150-NEXT: v_dual_mov_b32 v4, s4 :: v_dual_mov_b32 v5, s5
; GISEL-GFX1150-NEXT: s_cselect_b32 s6, 1, 0
; GISEL-GFX1150-NEXT: s_cmp_le_f32 s7, 0
+; GISEL-GFX1150-NEXT: s_delay_alu instid0(SALU_CYCLE_3) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GISEL-GFX1150-NEXT: s_cselect_b32 s7, 1, 0
-; GISEL-GFX1150-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GISEL-GFX1150-NEXT: v_dual_mov_b32 v6, s6 :: v_dual_mov_b32 v7, s7
; GISEL-GFX1150-NEXT: s_setpc_b64 s[30:31]
;
@@ -1399,24 +1402,28 @@ define <8 x i1> @vector_f32_ole() {
; GISEL-GFX1200-NEXT: v_readfirstlane_b32 s7, v7
; GISEL-GFX1200-NEXT: s_cselect_b32 s0, 1, 0
; GISEL-GFX1200-NEXT: s_cmp_le_f32 s1, 0
+; GISEL-GFX1200-NEXT: s_delay_alu instid0(SALU_CYCLE_3) | instskip(SKIP_3) | instid1(SALU_CYCLE_1)
; GISEL-GFX1200-NEXT: s_cselect_b32 s1, 1, 0
; GISEL-GFX1200-NEXT: s_cmp_le_f32 s2, 0
; GISEL-GFX1200-NEXT: s_wait_alu depctr_sa_sdst(0)
; GISEL-GFX1200-NEXT: v_dual_mov_b32 v0, s0 :: v_dual_mov_b32 v1, s1
; GISEL-GFX1200-NEXT: s_cselect_b32 s2, 1, 0
; GISEL-GFX1200-NEXT: s_cmp_le_f32 s3, 0
+; GISEL-GFX1200-NEXT: s_delay_alu instid0(SALU_CYCLE_3) | instskip(SKIP_3) | instid1(SALU_CYCLE_1)
; GISEL-GFX1200-NEXT: s_cselect_b32 s3, 1, 0
; GISEL-GFX1200-NEXT: s_cmp_le_f32 s4, 0
; GISEL-GFX1200-NEXT: s_wait_alu depctr_sa_sdst(0)
; GISEL-GFX1200-NEXT: v_dual_mov_b32 v2, s2 :: v_dual_mov_b32 v3, s3
; GISEL-GFX1200-NEXT: s_cselect_b32 s4, 1, 0
; GISEL-GFX1200-NEXT: s_cmp_le_f32 s5, 0
+; GISEL-GFX1200-NEXT: s_delay_alu instid0(SALU_CYCLE_3) | instskip(SKIP_3) | instid1(SALU_CYCLE_1)
; GISEL-GFX1200-NEXT: s_cselect_b32 s5, 1, 0
; GISEL-GFX1200-NEXT: s_cmp_le_f32 s6, 0
; GISEL-GFX1200-NEXT: s_wait_alu depctr_sa_sdst(0)
; GISEL-GFX1200-NEXT: v_dual_mov_b32 v4, s4 :: v_dual_mov_b32 v5, s5
; GISEL-GFX1200-NEXT: s_cselect_b32 s6, 1, 0
; GISEL-GFX1200-NEXT: s_cmp_le_f32 s7, 0
+; GISEL-GFX1200-NEXT: s_delay_alu instid0(SALU_CYCLE_3)
; GISEL-GFX1200-NEXT: s_cselect_b32 s7, 1, 0
; GISEL-GFX1200-NEXT: s_wait_alu depctr_sa_sdst(0)
; GISEL-GFX1200-NEXT: v_dual_mov_b32 v6, s6 :: v_dual_mov_b32 v7, s7
@@ -1455,11 +1462,13 @@ define <4 x i1> @vector_f32_ogt() {
; GISEL-GFX1150-NEXT: v_readfirstlane_b32 s2, v2
; GISEL-GFX1150-NEXT: v_readfirstlane_b32 s3, v3
; GISEL-GFX1150-NEXT: s_cmp_gt_f32 s0, 0
+; GISEL-GFX1150-NEXT: s_delay_alu instid0(SALU_CYCLE_3) | instskip(SKIP_1) | instid1(SALU_CYCLE_3)
; GISEL-GFX1150-NEXT: s_cselect_b32 s0, 1, 0
; GISEL-GFX1150-NEXT: s_cmp_gt_f32 s1, 0
; GISEL-GFX1150-NEXT: s_cselect_b32 s1, 1, 0
; GISEL-GFX1150-NEXT: s_cmp_gt_f32 s2, 0
; GISEL-GFX1150-NEXT: v_dual_mov_b32 v0, s0 :: v_dual_mov_b32 v1, s1
+; GISEL-GFX1150-NEXT: s_delay_alu instid0(SALU_CYCLE_2) | instskip(SKIP_1) | instid1(SALU_CYCLE_3)
; GISEL-GFX1150-NEXT: s_cselect_b32 s2, 1, 0
; GISEL-GFX1150-NEXT: s_cmp_gt_f32 s3, 0
; GISEL-GFX1150-NEXT: s_cselect_b32 s3, 1, 0
@@ -1506,12 +1515,14 @@ define <4 x i1> @vector_f32_ogt() {
; GISEL-GFX1200-NEXT: v_readfirstlane_b32 s2, v2
; GISEL-GFX1200-NEXT: v_readfirstlane_b32 s3, v3
; GISEL-GFX1200-NEXT: s_cmp_gt_f32 s0, 0
+; GISEL-GFX1200-NEXT: s_delay_alu instid0(SALU_CYCLE_3) | instskip(SKIP_1) | instid1(SALU_CYCLE_3)
; GISEL-GFX1200-NEXT: s_cselect_b32 s0, 1, 0
; GISEL-GFX1200-NEXT: s_cmp_gt_f32 s1, 0
; GISEL-GFX1200-NEXT: s_cselect_b32 s1, 1, 0
; GISEL-GFX1200-NEXT: s_cmp_gt_f32 s2, 0
; GISEL-GFX1200-NEXT: s_wait_alu depctr_sa_sdst(0)
; GISEL-GFX1200-NEXT: v_dual_mov_b32 v0, s0 :: v_dual_mov_b32 v1, s1
+; GISEL-GFX1200-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_3)
; GISEL-GFX1200-NEXT: s_cselect_b32 s2, 1, 0
; GISEL-GFX1200-NEXT: s_cmp_gt_f32 s3, 0
; GISEL-GFX1200-NEXT: s_cselect_b32 s3, 1, 0
@@ -1704,6 +1715,7 @@ define <32 x i1> @vector_f32_ueq() {
; GISEL-GFX1150-NEXT: s_cmp_nlg_f32 s9, 0
; GISEL-GFX1150-NEXT: v_dual_mov_b32 v0, s0 :: v_dual_mov_b32 v1, s1
; GISEL-GFX1150-NEXT: v_dual_mov_b32 v2, s2 :: v_dual_mov_b32 v3, s3
+; GISEL-GFX1150-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_3) | instid1(SALU_CYCLE_1)
; GISEL-GFX1150-NEXT: s_cselect_b32 s9, 1, 0
; GISEL-GFX1150-NEXT: s_cmp_nlg_f32 s10, 0
; GISEL-GFX1150-NEXT: v_dual_mov_b32 v4, s4 :: v_dual_mov_b32 v5, s5
@@ -1711,58 +1723,68 @@ define <32 x i1> @vector_f32_ueq() {
; GISEL-GFX1150-NEXT: s_cselect_b32 s10, 1, 0
; GISEL-GFX1150-NEXT: s_cmp_nlg_f32 s11, 0
; GISEL-GFX1150-NEXT: v_dual_mov_b32 v8, s8 :: v_dual_mov_b32 v9, s9
+; GISEL-GFX1150-NEXT: s_delay_alu instid0(SALU_CYCLE_2) | instskip(SKIP_2) | instid1(SALU_CYCLE_2)
; GISEL-GFX1150-NEXT: s_cselect_b32 s11, 1, 0
; GISEL-GFX1150-NEXT: s_cmp_nlg_f32 s12, 0
; GISEL-GFX1150-NEXT: v_dual_mov_b32 v10, s10 :: v_dual_mov_b32 v11, s11
; GISEL-GFX1150-NEXT: s_cselect_b32 s12, 1, 0
; GISEL-GFX1150-NEXT: s_cmp_nlg_f32 s13, 0
+; GISEL-GFX1150-NEXT: s_delay_alu instid0(SALU_CYCLE_3) | instskip(SKIP_2) | instid1(SALU_CYCLE_2)
; GISEL-GFX1150-NEXT: s_cselect_b32 s13, 1, 0
; GISEL-GFX1150-NEXT: s_cmp_nlg_f32 s14, 0
; GISEL-GFX1150-NEXT: v_dual_mov_b32 v12, s12 :: v_dual_mov_b32 v13, s13
; GISEL-GFX1150-NEXT: s_cselect_b32 s14, 1, 0
; GISEL-GFX1150-NEXT: s_cmp_nlg_f32 s15, 0
+; GISEL-GFX1150-NEXT: s_delay_alu instid0(SALU_CYCLE_3) | instskip(SKIP_2) | instid1(SALU_CYCLE_2)
; GISEL-GFX1150-NEXT: s_cselect_b32 s15, 1, 0
; GISEL-GFX1150-NEXT: s_cmp_nlg_f32 s16, 0
; GISEL-GFX1150-NEXT: v_dual_mov_b32 v14, s14 :: v_dual_mov_b32 v15, s15
; GISEL-GFX1150-NEXT: s_cselect_b32 s16, 1, 0
; GISEL-GFX1150-NEXT: s_cmp_nlg_f32 s17, 0
+; GISEL-GFX1150-NEXT: s_delay_alu instid0(SALU_CYCLE_3) | instskip(SKIP_2) | instid1(SALU_CYCLE_2)
; GISEL-GFX1150-NEXT: s_cselect_b32 s17, 1, 0
; GISEL-GFX1150-NEXT: s_cmp_nlg_f32 s18, 0
; GISEL-GFX1150-NEXT: v_dual_mov_b32 v16, s16 :: v_dual_mov_b32 v17, s17
; GISEL-GFX1150-NEXT: s_cselect_b32 s18, 1, 0
; GISEL-GFX1150-NEXT: s_cmp_nlg_f32 s19, 0
+; GISEL-GFX1150-NEXT: s_delay_alu instid0(SALU_CYCLE_3) | instskip(SKIP_2) | instid1(SALU_CYCLE_2)
; GISEL-GFX1150-NEXT: s_cselect_b32 s19, 1, 0
; GISEL-GFX1150-NEXT: s_cmp_nlg_f32 s20, 0
; GISEL-GFX1150-NEXT: v_dual_mov_b32 v18, s18 :: v_dual_mov_b32 v19, s19
; GISEL-GFX1150-NEXT: s_cselect_b32 s20, 1, 0
; GISEL-GFX1150-NEXT: s_cmp_nlg_f32 s21, 0
+; GISEL-GFX1150-NEXT: s_delay_alu instid0(SALU_CYCLE_3) | instskip(SKIP_2) | instid1(SALU_CYCLE_2)
; GISEL-GFX1150-NEXT: s_cselect_b32 s21, 1, 0
; GISEL-GFX1150-NEXT: s_cmp_nlg_f32 s22, 0
; GISEL-GFX1150-NEXT: v_dual_mov_b32 v20, s20 :: v_dual_mov_b32 v21, s21
; GISEL-GFX1150-NEXT: s_cselect_b32 s22, 1, 0
; GISEL-GFX1150-NEXT: s_cmp_nlg_f32 s23, 0
+; GISEL-GFX1150-NEXT: s_delay_alu instid0(SALU_CYCLE_3) | instskip(SKIP_2) | instid1(SALU_CYCLE_2)
; GISEL-GFX1150-NEXT: s_cselect_b32 s23, 1, 0
; GISEL-GFX1150-NEXT: s_cmp_nlg_f32 s24, 0
; GISEL-GFX1150-NEXT: v_dual_mov_b32 v22, s22 :: v_dual_mov_b32 v23, s23
; GISEL-GFX1150-NEXT: s_cselect_b32 s24, 1, 0
; GISEL-GFX1150-NEXT: s_cmp_nlg_f32 s25, 0
+; GISEL-GFX1150-NEXT: s_delay_alu instid0(SALU_CYCLE_3) | instskip(SKIP_2) | instid1(SALU_CYCLE_2)
; GISEL-GFX1150-NEXT: s_cselect_b32 s25, 1, 0
; GISEL-GFX1150-NEXT: s_cmp_nlg_f32 s26, 0
; GISEL-GFX1150-NEXT: v_dual_mov_b32 v24, s24 :: v_dual_mov_b32 v25, s25
; GISEL-GFX1150-NEXT: s_cselect_b32 s26, 1, 0
; GISEL-GFX1150-NEXT: s_cmp_nlg_f32 s27, 0
+; GISEL-GFX1150-NEXT: s_delay_alu instid0(SALU_CYCLE_3) | instskip(SKIP_2) | instid1(SALU_CYCLE_2)
; GISEL-GFX1150-NEXT: s_cselect_b32 s27, 1, 0
; GISEL-GFX1150-NEXT: s_cmp_nlg_f32 s28, 0
; GISEL-GFX1150-NEXT: v_dual_mov_b32 v26, s26 :: v_dual_mov_b32 v27, s27
; GISEL-GFX1150-NEXT: s_cselect_b32 s28, 1, 0
; GISEL-GFX1150-NEXT: s_cmp_nlg_f32 s29, 0
+; GISEL-GFX1150-NEXT: s_delay_alu instid0(SALU_CYCLE_3) | instskip(SKIP_2) | instid1(SALU_CYCLE_2)
; GISEL-GFX1150-NEXT: s_cselect_b32 s29, 1, 0
; GISEL-GFX1150-NEXT: s_cmp_nlg_f32 s40, 0
; GISEL-GFX1150-NEXT: v_dual_mov_b32 v28, s28 :: v_dual_mov_b32 v29, s29
; GISEL-GFX1150-NEXT: s_cselect_b32 s40, 1, 0
; GISEL-GFX1150-NEXT: s_cmp_nlg_f32 s41, 0
+; GISEL-GFX1150-NEXT: s_delay_alu instid0(SALU_CYCLE_3) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GISEL-GFX1150-NEXT: s_cselect_b32 s41, 1, 0
-; GISEL-GFX1150-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GISEL-GFX1150-NEXT: v_dual_mov_b32 v30, s40 :: v_dual_mov_b32 v31, s41
; GISEL-GFX1150-NEXT: s_setpc_b64 s[30:31]
;
@@ -1990,6 +2012,7 @@ define <32 x i1> @vector_f32_ueq() {
; GISEL-GFX1200-NEXT: s_cmp_nlg_f32 s10, 0
; GISEL-GFX1200-NEXT: v_dual_mov_b32 v4, s4 :: v_dual_mov_b32 v5, s5
; GISEL-GFX1200-NEXT: v_dual_mov_b32 v6, s6 :: v_dual_mov_b32 v7, s7
+; GISEL-GFX1200-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_3) | instid1(SALU_CYCLE_1)
; GISEL-GFX1200-NEXT: s_cselect_b32 s10, 1, 0
; GISEL-GFX1200-NEXT: s_cmp_nlg_f32 s11, 0
; GISEL-GFX1200-NEXT: s_wait_alu depctr_sa_sdst(0)
@@ -1998,60 +2021,70 @@ define <32 x i1> @vector_f32_ueq() {
; GISEL-GFX1200-NEXT: s_cmp_nlg_f32 s12, 0
; GISEL-GFX1200-NEXT: s_wait_alu depctr_sa_sdst(0)
; GISEL-GFX1200-NEXT: v_dual_mov_b32 v10, s10 :: v_dual_mov_b32 v11, s11
+; GISEL-GFX1200-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_3)
; GISEL-GFX1200-NEXT: s_cselect_b32 s12, 1, 0
; GISEL-GFX1200-NEXT: s_cmp_nlg_f32 s13, 0
; GISEL-GFX1200-NEXT: s_cselect_b32 s13, 1, 0
; GISEL-GFX1200-NEXT: s_cmp_nlg_f32 s14, 0
; GISEL-GFX1200-NEXT: s_wait_alu depctr_sa_sdst(0)
; GISEL-GFX1200-NEXT: v_dual_mov_b32 v12, s12 :: v_dual_mov_b32 v13, s13
+; GISEL-GFX1200-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_3)
; GISEL-GFX1200-NEXT: s_cselect_b32 s14, 1, 0
; GISEL-GFX1200-NEXT: s_cmp_nlg_f32 s15, 0
; GISEL-GFX1200-NEXT: s_cselect_b32 s15, 1, 0
; GISEL-GFX1200-NEXT: s_cmp_nlg_f32 s16, 0
; GISEL-GFX1200-NEXT: s_wait_alu depctr_sa_sdst(0)
; GISEL-GFX1200-NEXT: v_dual_mov_b32 v14, s14 :: v_dual_mov_b32 v15, s15
+; GISEL-GFX1200-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_3)
; GISEL-GFX1200-NEXT: s_cselect_b32 s16, 1, 0
; GISEL-GFX1200-NEXT: s_cmp_nlg_f32 s17, 0
; GISEL-GFX1200-NEXT: s_cselect_b32 s17, 1, 0
; GISEL-GFX1200-NEXT: s_cmp_nlg_f32 s18, 0
; GISEL-GFX1200-NEXT: s_wait_alu depctr_sa_sdst(0)
; GISEL-GFX1200-NEXT: v_dual_mov_b32 v16, s16 :: v_dual_mov_b32 v17, s17
+; GISEL-GFX1200-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_3)
; GISEL-GFX1200-NEXT: s_cselect_b32 s18, 1, 0
; GISEL-GFX1200-NEXT: s_cmp_nlg_f32 s19, 0
; GISEL-GFX1200-NEXT: s_cselect_b32 s19, 1, 0
; GISEL-GFX1200-NEXT: s_cmp_nlg_f32 s20, 0
; GISEL-GFX1200-NEXT: s_wait_alu depctr_sa_sdst(0)
; GISEL-GFX1200-NEXT: v_dual_mov_b32 v18, s18 :: v_dual_mov_b32 v19, s19
+; GISEL-GFX1200-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_3)
; GISEL-GFX1200-NEXT: s_cselect_b32 s20, 1, 0
; GISEL-GFX1200-NEXT: s_cmp_nlg_f32 s21, 0
; GISEL-GFX1200-NEXT: s_cselect_b32 s21, 1, 0
; GISEL-GFX1200-NEXT: s_cmp_nlg_f32 s22, 0
; GISEL-GFX1200-NEXT: s_wait_alu depctr_sa_sdst(0)
; GISEL-GFX1200-NEXT: v_dual_mov_b32 v20, s20 :: v_dual_mov_b32 v21, s21
+; GISEL-GFX1200-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_3)
; GISEL-GFX1200-NEXT: s_cselect_b32 s22, 1, 0
; GISEL-GFX1200-NEXT: s_cmp_nlg_f32 s23, 0
; GISEL-GFX1200-NEXT: s_cselect_b32 s23, 1, 0
; GISEL-GFX1200-NEXT: s_cmp_nlg_f32 s24, 0
; GISEL-GFX1200-NEXT: s_wait_alu depctr_sa_sdst(0)
; GISEL-GFX1200-NEXT: v_dual_mov_b32 v22, s22 :: v_dual_mov_b32 v23, s23
+; GISEL-GFX1200-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_3)
; GISEL-GFX1200-NEXT: s_cselect_b32 s24, 1, 0
; GISEL-GFX1200-NEXT: s_cmp_nlg_f32 s25, 0
; GISEL-GFX1200-NEXT: s_cselect_b32 s25, 1, 0
; GISEL-GFX1200-NEXT: s_cmp_nlg_f32 s26, 0
; GISEL-GFX1200-NEXT: s_wait_alu depctr_sa_sdst(0)
; GISEL-GFX1200-NEXT: v_dual_mov_b32 v24, s24 :: v_dual_mov_b32 v25, s25
+; GISEL-GFX1200-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_3)
; GISEL-GFX1200-NEXT: s_cselect_b32 s26, 1, 0
; GISEL-GFX1200-NEXT: s_cmp_nlg_f32 s27, 0
; GISEL-GFX1200-NEXT: s_cselect_b32 s27, 1, 0
; GISEL-GFX1200-NEXT: s_cmp_nlg_f32 s28, 0
; GISEL-GFX1200-NEXT: s_wait_alu depctr_sa_sdst(0)
; GISEL-GFX1200-NEXT: v_dual_mov_b32 v26, s26 :: v_dual_mov_b32 v27, s27
+; GISEL-GFX1200-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_3)
; GISEL-GFX1200-NEXT: s_cselect_b32 s28, 1, 0
; GISEL-GFX1200-NEXT: s_cmp_nlg_f32 s29, 0
; GISEL-GFX1200-NEXT: s_cselect_b32 s29, 1, 0
; GISEL-GFX1200-NEXT: s_cmp_nlg_f32 s40, 0
; GISEL-GFX1200-NEXT: s_wait_alu depctr_sa_sdst(0)
; GISEL-GFX1200-NEXT: v_dual_mov_b32 v28, s28 :: v_dual_mov_b32 v29, s29
+; GISEL-GFX1200-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_3)
; GISEL-GFX1200-NEXT: s_cselect_b32 s40, 1, 0
; GISEL-GFX1200-NEXT: s_cmp_nlg_f32 s41, 0
; GISEL-GFX1200-NEXT: s_cselect_b32 s41, 1, 0
@@ -2092,11 +2125,13 @@ define <4 x i1> @vector_f32_ugt() {
; GISEL-GFX1150-NEXT: v_readfirstlane_b32 s2, v2
; GISEL-GFX1150-NEXT: v_readfirstlane_b32 s3, v3
; GISEL-GFX1150-NEXT: s_cmp_nle_f32 s0, 0
+; GISEL-GFX1150-NEXT: s_delay_alu instid0(SALU_CYCLE_3) | instskip(SKIP_1) | instid1(SALU_CYCLE_3)
; GISEL-GFX1150-NEXT: s_cselect_b32 s0, 1, 0
; GISEL-GFX1150-NEXT: s_cmp_nle_f32 s1, 0
; GISEL-GFX1150-NEXT: s_cselect_b32 s1, 1, 0
; GISEL-GFX1150-NEXT: s_cmp_nle_f32 s2, 0
; GISEL-GFX1150-NEXT: v_dual_mov_b32 v0, s0 :: v_dual_mov_b32 v1, s1
+; GISEL-GFX1150-NEXT: s_delay_alu instid0(SALU_CYCLE_2) | instskip(SKIP_1) | instid1(SALU_CYCLE_3)
; GISEL-GFX1150-NEXT: s_cselect_b32 s2, 1, 0
; GISEL-GFX1150-NEXT: s_cmp_nle_f32 s3, 0
; GISEL-GFX1150-NEXT: s_cselect_b32 s3, 1, 0
@@ -2143,12 +2178,14 @@ define <4 x i1> @vector_f32_ugt() {
; GISEL-GFX1200-NEXT: v_readfirstlane_b32 s2, v2
; GISEL-GFX1200-NEXT: v_readfirstlane_b32 s3, v3
; GISEL-GFX1200-NEXT: s_cmp_nle_f32 s0, 0
+; GISEL-GFX1200-NEXT: s_delay_alu instid0(SALU_CYCLE_3) | instskip(SKIP_1) | instid1(SALU_CYCLE_3)
; GISEL-GFX1200-NEXT: s_cselect_b32 s0, 1, 0
; GISEL-GFX1200-NEXT: s_cmp_nle_f32 s1, 0
; GISEL-GFX1200-NEXT: s_cselect_b32 s1, 1, 0
; GISEL-GFX1200-NEXT: s_cmp_nle_f32 s2, 0
; GISEL-GFX1200-NEXT: s_wait_alu depctr_sa_sdst(0)
; GISEL-GFX1200-NEXT: v_dual_mov_b32 v0, s0 :: v_dual_mov_b32 v1, s1
+; GISEL-GFX1200-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_3)
; GISEL-GFX1200-NEXT: s_cselect_b32 s2, 1, 0
; GISEL-GFX1200-NEXT: s_cmp_nle_f32 s3, 0
; GISEL-GFX1200-NEXT: s_cselect_b32 s3, 1, 0
diff --git a/llvm/test/CodeGen/AMDGPU/float-to-arbitrary-fp-fp8-f16-hw.ll b/llvm/test/CodeGen/AMDGPU/float-to-arbitrary-fp-fp8-f16-hw.ll
index d4288ae2362e70..e3d6d9a1170f68 100644
--- a/llvm/test/CodeGen/AMDGPU/float-to-arbitrary-fp-fp8-f16-hw.ll
+++ b/llvm/test/CodeGen/AMDGPU/float-to-arbitrary-fp-fp8-f16-hw.ll
@@ -1680,21 +1680,21 @@ define i8 @to_fp8_f16_towardzero(half %x) {
; GFX1250-TRUE16-NEXT: v_and_b16 v2.l, 0x80, v2.l
; GFX1250-TRUE16-NEXT: v_cndmask_b16 v1.l, v1.l, 0, vcc_lo
; GFX1250-TRUE16-NEXT: v_add_nc_u16 v1.h, v1.h, v3.l
-; GFX1250-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(SKIP_2) | instid1(VALU_DEP_4)
; GFX1250-TRUE16-NEXT: v_cndmask_b16 v0.h, v0.h, 0, s0
; GFX1250-TRUE16-NEXT: v_cndmask_b16 v2.h, 0, 8, s0
; GFX1250-TRUE16-NEXT: v_cmp_eq_f16_e64 s0, 0, v0.l
+; GFX1250-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_3)
; GFX1250-TRUE16-NEXT: v_add_nc_u16 v1.h, v1.h, 6
-; GFX1250-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-TRUE16-NEXT: v_bitop3_b16 v0.h, v2.l, v0.h, v2.h bitop3:0xfe
+; GFX1250-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX1250-TRUE16-NEXT: v_lshlrev_b16 v3.l, 3, v1.h
; GFX1250-TRUE16-NEXT: v_cmp_gt_i16_e32 vcc_lo, 1, v1.h
-; GFX1250-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1250-TRUE16-NEXT: v_bitop3_b16 v1.l, v2.l, v1.l, v3.l bitop3:0xfe
+; GFX1250-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX1250-TRUE16-NEXT: v_cndmask_b16 v0.h, v1.l, v0.h, vcc_lo
; GFX1250-TRUE16-NEXT: v_cmp_o_f16_e32 vcc_lo, v0.l, v0.l
-; GFX1250-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1250-TRUE16-NEXT: v_cndmask_b16 v0.h, v0.h, v2.l, s0
+; GFX1250-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1250-TRUE16-NEXT: v_cndmask_b16 v0.l, 0x7f, v0.h, vcc_lo
; GFX1250-TRUE16-NEXT: s_set_pc_i64 s[30:31]
;
@@ -1900,10 +1900,9 @@ define <2 x i8> @to_fp8_v2f16_saturate(<2 x half> %x) {
; GFX1250-TRUE16-NEXT: v_cmp_lt_i16_e64 s1, 7, v3.h
; GFX1250-TRUE16-NEXT: v_mov_b16_e32 v8.l, v14.l
; GFX1250-TRUE16-NEXT: v_cmp_lt_i16_e64 s2, 7, v3.l
-; GFX1250-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
+; GFX1250-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(SKIP_1) | instid1(VALU_DEP_4)
; GFX1250-TRUE16-NEXT: v_lshlrev_b16 v5.l, 3, v1.h
; GFX1250-TRUE16-NEXT: v_cndmask_b16 v2.l, v3.h, 0, s1
-; GFX1250-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
; GFX1250-TRUE16-NEXT: v_bitop3_b16 v1.l, v1.l, v8.l, v5.h bitop3:0xe0
; GFX1250-TRUE16-NEXT: v_cndmask_b32_e64 v8, 0, 1, s2
; GFX1250-TRUE16-NEXT: v_cndmask_b16 v3.h, 0, 8, s1
@@ -2049,12 +2048,11 @@ define <2 x i8> @to_fp8_v2f16_saturate(<2 x half> %x) {
; GFX1250-FAKE16-NEXT: v_cmp_lt_i16_e32 vcc_lo, 15, v2
; GFX1250-FAKE16-NEXT: v_add_nc_u16 v9, v20, v9
; GFX1250-FAKE16-NEXT: s_or_b32 vcc_lo, vcc_lo, s3
-; GFX1250-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_3) | instid1(VALU_DEP_3)
; GFX1250-FAKE16-NEXT: v_cmp_lt_i16_e64 s0, 7, v9
; GFX1250-FAKE16-NEXT: v_cndmask_b32_e64 v11, 0, 1, s0
; GFX1250-FAKE16-NEXT: v_cndmask_b32_e64 v9, v9, 0, s0
; GFX1250-FAKE16-NEXT: v_cmp_lt_i16_e64 s0, 7, v6
-; GFX1250-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_2) | instid1(VALU_DEP_4)
; GFX1250-FAKE16-NEXT: v_add_nc_u16 v8, v10, v11
; GFX1250-FAKE16-NEXT: v_bitop3_b16 v10, v7, v1, v12 bitop3:0xfe
; GFX1250-FAKE16-NEXT: v_lshrrev_b32_e32 v1, 31, v0
@@ -2063,10 +2061,9 @@ define <2 x i8> @to_fp8_v2f16_saturate(<2 x half> %x) {
; GFX1250-FAKE16-NEXT: v_add_nc_u16 v8, v8, 6
; GFX1250-FAKE16-NEXT: v_cmp_gt_i16_e64 s0, 1, v2
; GFX1250-FAKE16-NEXT: v_lshlrev_b16 v1, 7, v1
-; GFX1250-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX1250-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_1) | instid1(VALU_DEP_3)
; GFX1250-FAKE16-NEXT: v_lshlrev_b16 v12, 3, v8
; GFX1250-FAKE16-NEXT: v_cndmask_b32_e64 v5, v10, v5, s0
-; GFX1250-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX1250-FAKE16-NEXT: v_bitop3_b16 v6, v1, v6, v11 bitop3:0xfe
; GFX1250-FAKE16-NEXT: v_cmp_gt_i16_e64 s2, 1, v8
; GFX1250-FAKE16-NEXT: v_cmp_lt_i16_e64 s0, 6, v9
@@ -2082,14 +2079,14 @@ define <2 x i8> @to_fp8_v2f16_saturate(<2 x half> %x) {
; GFX1250-FAKE16-NEXT: s_or_b32 s0, s2, s0
; GFX1250-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX1250-FAKE16-NEXT: v_cndmask_b32_e64 v2, v2, v6, s0
-; GFX1250-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_3) | instid1(VALU_DEP_4)
+; GFX1250-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_3) | instid1(VALU_DEP_1)
; GFX1250-FAKE16-NEXT: v_cndmask_b32_e32 v1, v2, v1, vcc_lo
; GFX1250-FAKE16-NEXT: v_cmp_eq_f16_e32 vcc_lo, 0, v0
; GFX1250-FAKE16-NEXT: v_cndmask_b32_e32 v2, v5, v7, vcc_lo
; GFX1250-FAKE16-NEXT: v_cmp_class_f16_e64 vcc_lo, v4, 0x204
; GFX1250-FAKE16-NEXT: v_cndmask_b32_e32 v1, v1, v6, vcc_lo
; GFX1250-FAKE16-NEXT: v_cmp_class_f16_e64 vcc_lo, v0, 0x204
-; GFX1250-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(SKIP_1) | instid1(VALU_DEP_4)
+; GFX1250-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_4)
; GFX1250-FAKE16-NEXT: v_cndmask_b32_e32 v2, v2, v3, vcc_lo
; GFX1250-FAKE16-NEXT: v_cmp_o_f16_e32 vcc_lo, v4, v4
; GFX1250-FAKE16-NEXT: v_cndmask_b32_e32 v1, 0x7f, v1, vcc_lo
@@ -2326,12 +2323,11 @@ define <2 x i8> @to_fp8_v2f16_saturate(<2 x half> %x) {
; GFX1310-NEXT: v_cmp_lt_i16_e32 vcc_lo, 15, v2
; GFX1310-NEXT: v_add_nc_u16 v9, v20, v9
; GFX1310-NEXT: s_or_b32 vcc_lo, vcc_lo, s3
-; GFX1310-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1310-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_3) | instid1(VALU_DEP_3)
; GFX1310-NEXT: v_cmp_lt_i16_e64 s0, 7, v9
; GFX1310-NEXT: v_cndmask_b32_e64 v11, 0, 1, s0
; GFX1310-NEXT: v_cndmask_b32_e64 v9, v9, 0, s0
; GFX1310-NEXT: v_cmp_lt_i16_e64 s0, 7, v6
-; GFX1310-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_2) | instid1(VALU_DEP_4)
; GFX1310-NEXT: v_add_nc_u16 v8, v10, v11
; GFX1310-NEXT: v_bitop3_b16 v10, v7, v1, v12 bitop3:0xfe
; GFX1310-NEXT: v_lshrrev_b32_e32 v1, 31, v0
@@ -2340,10 +2336,9 @@ define <2 x i8> @to_fp8_v2f16_saturate(<2 x half> %x) {
; GFX1310-NEXT: v_add_nc_u16 v8, v8, 6
; GFX1310-NEXT: v_cmp_gt_i16_e64 s0, 1, v2
; GFX1310-NEXT: v_lshlrev_b16 v1, 7, v1
-; GFX1310-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX1310-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_1) | instid1(VALU_DEP_3)
; GFX1310-NEXT: v_lshlrev_b16 v12, 3, v8
; GFX1310-NEXT: v_cndmask_b32_e64 v5, v10, v5, s0
-; GFX1310-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX1310-NEXT: v_bitop3_b16 v6, v1, v6, v11 bitop3:0xfe
; GFX1310-NEXT: v_cmp_gt_i16_e64 s2, 1, v8
; GFX1310-NEXT: v_cmp_lt_i16_e64 s0, 6, v9
@@ -2359,14 +2354,14 @@ define <2 x i8> @to_fp8_v2f16_saturate(<2 x half> %x) {
; GFX1310-NEXT: s_or_b32 s0, s2, s0
; GFX1310-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX1310-NEXT: v_cndmask_b32_e64 v2, v2, v6, s0
-; GFX1310-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_3) | instid1(VALU_DEP_4)
+; GFX1310-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_3) | instid1(VALU_DEP_1)
; GFX1310-NEXT: v_cndmask_b32_e32 v1, v2, v1, vcc_lo
; GFX1310-NEXT: v_cmp_eq_f16_e32 vcc_lo, 0, v0
; GFX1310-NEXT: v_cndmask_b32_e32 v2, v5, v7, vcc_lo
; GFX1310-NEXT: v_cmp_class_f16_e64 vcc_lo, v4, 0x204
; GFX1310-NEXT: v_cndmask_b32_e32 v1, v1, v6, vcc_lo
; GFX1310-NEXT: v_cmp_class_f16_e64 vcc_lo, v0, 0x204
-; GFX1310-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(SKIP_1) | instid1(VALU_DEP_4)
+; GFX1310-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_4)
; GFX1310-NEXT: v_cndmask_b32_e32 v2, v2, v3, vcc_lo
; GFX1310-NEXT: v_cmp_o_f16_e32 vcc_lo, v4, v4
; GFX1310-NEXT: v_cndmask_b32_e32 v1, 0x7f, v1, vcc_lo
diff --git a/llvm/test/CodeGen/AMDGPU/float-to-arbitrary-fp-fp8-hw.ll b/llvm/test/CodeGen/AMDGPU/float-to-arbitrary-fp-fp8-hw.ll
index 547724a407ca8a..80c8b91fe23272 100644
--- a/llvm/test/CodeGen/AMDGPU/float-to-arbitrary-fp-fp8-hw.ll
+++ b/llvm/test/CodeGen/AMDGPU/float-to-arbitrary-fp-fp8-hw.ll
@@ -603,7 +603,7 @@ define i8 @to_fp8_f32_saturate(float %x) {
; GFX1200-NEXT: v_cndmask_b32_e32 v1, v1, v2, vcc_lo
; GFX1200-NEXT: v_cmp_eq_f32_e32 vcc_lo, 0, v0
; GFX1200-NEXT: s_wait_alu depctr_va_vcc(0)
-; GFX1200-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_2) | instid1(VALU_DEP_2)
+; GFX1200-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_2) | instid1(VALU_DEP_1)
; GFX1200-NEXT: v_cndmask_b32_e32 v1, v1, v4, vcc_lo
; GFX1200-NEXT: v_cmp_class_f32_e64 vcc_lo, v0, 0x204
; GFX1200-NEXT: s_wait_alu depctr_va_vcc(0)
@@ -673,7 +673,7 @@ define i8 @to_fp8_f32_saturate(float %x) {
; GFX1250-NEXT: v_cmp_eq_f32_e32 vcc_lo, 0, v0
; GFX1250-NEXT: v_cndmask_b32_e32 v1, v1, v4, vcc_lo
; GFX1250-NEXT: v_cmp_class_f32_e64 vcc_lo, v0, 0x204
-; GFX1250-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_2)
+; GFX1250-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX1250-NEXT: v_cndmask_b32_e32 v1, v1, v2, vcc_lo
; GFX1250-NEXT: v_cmp_o_f32_e32 vcc_lo, v0, v0
; GFX1250-NEXT: v_cndmask_b32_e32 v0, 0x7f, v1, vcc_lo
@@ -741,7 +741,7 @@ define i8 @to_fp8_f32_saturate(float %x) {
; GFX1310-NEXT: v_cmp_eq_f32_e32 vcc_lo, 0, v0
; GFX1310-NEXT: v_cndmask_b32_e32 v1, v1, v4, vcc_lo
; GFX1310-NEXT: v_cmp_class_f32_e64 vcc_lo, v0, 0x204
-; GFX1310-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_2)
+; GFX1310-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX1310-NEXT: v_cndmask_b32_e32 v1, v1, v2, vcc_lo
; GFX1310-NEXT: v_cmp_o_f32_e32 vcc_lo, v0, v0
; GFX1310-NEXT: v_cndmask_b32_e32 v0, 0x7f, v1, vcc_lo
@@ -1006,26 +1006,26 @@ define i4 @to_fp4_f32(float %x) {
; GFX1200-NEXT: v_and_b32_e32 v4, 8, v4
; GFX1200-NEXT: v_and_b32_e32 v3, v3, v6
; GFX1200-NEXT: v_cmp_lt_i32_e64 s0, 1, v2
-; GFX1200-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_2)
+; GFX1200-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_3) | instid1(VALU_DEP_3)
; GFX1200-NEXT: v_add_nc_u32_e32 v3, v8, v3
; GFX1200-NEXT: s_wait_alu depctr_va_sdst(0)
; GFX1200-NEXT: v_cndmask_b32_e64 v2, v2, 0, s0
; GFX1200-NEXT: v_cndmask_b32_e64 v5, 0, 2, s0
-; GFX1200-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1200-NEXT: v_cmp_lt_i32_e32 vcc_lo, 1, v3
+; GFX1200-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_3) | instid1(VALU_DEP_2)
; GFX1200-NEXT: v_or3_b32 v2, v4, v5, v2
; GFX1200-NEXT: s_wait_alu depctr_va_vcc(0)
; GFX1200-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1200-NEXT: v_cndmask_b32_e64 v3, v3, 0, vcc_lo
-; GFX1200-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX1200-NEXT: v_lshlrev_b32_e32 v6, 1, v1
; GFX1200-NEXT: v_cmp_gt_i32_e32 vcc_lo, 1, v1
+; GFX1200-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX1200-NEXT: v_or3_b32 v3, v4, v6, v3
; GFX1200-NEXT: s_wait_alu depctr_va_vcc(0)
-; GFX1200-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_2) | instid1(VALU_DEP_2)
; GFX1200-NEXT: v_cndmask_b32_e32 v1, v3, v2, vcc_lo
; GFX1200-NEXT: v_cmp_eq_f32_e32 vcc_lo, 0, v0
; GFX1200-NEXT: s_wait_alu depctr_va_vcc(0)
+; GFX1200-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX1200-NEXT: v_cndmask_b32_e32 v0, v1, v4, vcc_lo
; GFX1200-NEXT: s_setpc_b64 s[30:31]
;
@@ -1060,7 +1060,7 @@ define i4 @to_fp4_f32(float %x) {
; GFX1250-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-NEXT: v_dual_add_nc_u32 v2, v4, v2 :: v_dual_lshrrev_b32 v4, 28, v0
; GFX1250-NEXT: v_cmp_lt_i32_e32 vcc_lo, 1, v3
-; GFX1250-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_3) | instid1(VALU_DEP_4)
+; GFX1250-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX1250-NEXT: v_cmp_lt_i32_e64 s0, 1, v2
; GFX1250-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1250-NEXT: v_cndmask_b32_e64 v3, v3, 0, vcc_lo
@@ -1112,7 +1112,7 @@ define i4 @to_fp4_f32(float %x) {
; GFX1310-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1310-NEXT: v_dual_add_nc_u32 v2, v4, v2 :: v_dual_lshrrev_b32 v4, 28, v0
; GFX1310-NEXT: v_cmp_lt_i32_e32 vcc_lo, 1, v3
-; GFX1310-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_3) | instid1(VALU_DEP_4)
+; GFX1310-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX1310-NEXT: v_cmp_lt_i32_e64 s0, 1, v2
; GFX1310-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1310-NEXT: v_cndmask_b32_e64 v3, v3, 0, vcc_lo
@@ -1285,7 +1285,7 @@ define i8 @to_fp8_f64(double %x) {
; GFX1200-NEXT: v_cndmask_b32_e64 v3, v11, 0, vcc_lo
; GFX1200-NEXT: v_lshrrev_b32_e32 v11, 24, v1
; GFX1200-NEXT: v_and_b32_e32 v2, v2, v8
-; GFX1200-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(SKIP_4) | instid1(VALU_DEP_1)
+; GFX1200-NEXT: s_delay_alu instid0(VALU_DEP_4)
; GFX1200-NEXT: v_add_co_u32 v8, s0, v12, v9
; GFX1200-NEXT: s_wait_alu depctr_va_sdst(0)
; GFX1200-NEXT: v_add_co_ci_u32_e64 v9, null, 0, v13, s0
@@ -1328,7 +1328,7 @@ define i8 @to_fp8_f64(double %x) {
; GFX1200-NEXT: v_cndmask_b32_e32 v2, v2, v9, vcc_lo
; GFX1200-NEXT: v_cmp_class_f64_e64 vcc_lo, v[0:1], 0x204
; GFX1200-NEXT: s_wait_alu depctr_va_vcc(0)
-; GFX1200-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_2) | instid1(VALU_DEP_2)
+; GFX1200-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_2) | instid1(VALU_DEP_2)
; GFX1200-NEXT: v_cndmask_b32_e32 v2, v2, v3, vcc_lo
; GFX1200-NEXT: v_cmp_o_f64_e32 vcc_lo, v[0:1], v[0:1]
; GFX1200-NEXT: s_wait_alu depctr_va_vcc(0)
@@ -1388,16 +1388,15 @@ define i8 @to_fp8_f64(double %x) {
; GFX1250-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
; GFX1250-NEXT: v_and_b32_e32 v7, 0x80, v10
; GFX1250-NEXT: v_cmp_lt_i64_e32 vcc_lo, 3, v[4:5]
-; GFX1250-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_3) | instid1(VALU_DEP_3)
; GFX1250-NEXT: v_cndmask_b32_e64 v6, v6, 0, s0
; GFX1250-NEXT: v_cndmask_b32_e64 v8, 0, 1, vcc_lo
; GFX1250-NEXT: v_cndmask_b32_e64 v5, v5, 0, vcc_lo
; GFX1250-NEXT: v_cndmask_b32_e64 v4, v4, 0, vcc_lo
+; GFX1250-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-NEXT: v_add_nc_u64_e32 v[2:3], v[2:3], v[8:9]
-; GFX1250-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-NEXT: v_cmp_lt_i64_e32 vcc_lo, 3, v[4:5]
+; GFX1250-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1250-NEXT: v_add_nc_u64_e32 v[2:3], 14, v[2:3]
-; GFX1250-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1250-NEXT: v_lshlrev_b64_e32 v[8:9], 2, v[2:3]
; GFX1250-NEXT: v_cndmask_b32_e64 v9, 0, 4, s0
; GFX1250-NEXT: v_cmp_gt_i64_e64 s2, 1, v[2:3]
@@ -1416,7 +1415,7 @@ define i8 @to_fp8_f64(double %x) {
; GFX1250-NEXT: v_cmp_eq_f64_e32 vcc_lo, 0, v[0:1]
; GFX1250-NEXT: v_cndmask_b32_e32 v2, v2, v7, vcc_lo
; GFX1250-NEXT: v_cmp_class_f64_e64 vcc_lo, v[0:1], 0x204
-; GFX1250-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_2)
+; GFX1250-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX1250-NEXT: v_cndmask_b32_e32 v2, v2, v3, vcc_lo
; GFX1250-NEXT: v_cmp_o_f64_e32 vcc_lo, v[0:1], v[0:1]
; GFX1250-NEXT: v_cndmask_b32_e32 v0, 0x7e, v2, vcc_lo
@@ -1456,46 +1455,44 @@ define i8 @to_fp8_f64(double %x) {
; GFX1310-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1310-NEXT: v_or_b32_e32 v10, v10, v11
; GFX1310-NEXT: v_add_co_u32 v11, vcc_lo, v6, -1
-; GFX1310-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX1310-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_3)
; GFX1310-NEXT: v_add_co_ci_u32_e64 v16, null, -1, v7, vcc_lo
; GFX1310-NEXT: v_and_b32_e32 v10, v15, v10
; GFX1310-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_1) | instid1(VALU_DEP_4)
; GFX1310-NEXT: v_and_b32_e32 v2, v2, v11
; GFX1310-NEXT: v_lshrrev_b64 v[6:7], v4, v[8:9]
; GFX1310-NEXT: v_and_b32_e32 v3, v9, v16
-; GFX1310-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1310-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(SKIP_1) | instid1(VALU_DEP_3)
; GFX1310-NEXT: v_add_co_u32 v10, s0, v17, v10
; GFX1310-NEXT: v_add_co_ci_u32_e64 v11, null, 0, 0, s0
-; GFX1310-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_4) | instid1(VALU_DEP_2)
; GFX1310-NEXT: v_cmp_ne_u64_e32 vcc_lo, 0, v[2:3]
; GFX1310-NEXT: v_and_b32_e32 v15, 1, v6
; GFX1310-NEXT: v_lshrrev_b64 v[2:3], v14, v[8:9]
; GFX1310-NEXT: v_cndmask_b32_e64 v16, 0, 1, vcc_lo
; GFX1310-NEXT: v_cmp_lt_i64_e32 vcc_lo, 3, v[10:11]
+; GFX1310-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_2) | instid1(VALU_DEP_3)
; GFX1310-NEXT: v_or_b32_e32 v8, v16, v15
; GFX1310-NEXT: v_cndmask_b32_e64 v9, 0, 1, vcc_lo
; GFX1310-NEXT: v_cndmask_b32_e64 v3, v11, 0, vcc_lo
-; GFX1310-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_3)
; GFX1310-NEXT: v_dual_lshrrev_b32 v11, 24, v1 :: v_dual_bitop2_b32 v2, v2, v8 bitop3:0x40
+; GFX1310-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_2) | instid1(VALU_DEP_4)
; GFX1310-NEXT: v_add_co_u32 v8, s0, v12, v9
-; GFX1310-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX1310-NEXT: v_add_co_ci_u32_e64 v9, null, 0, v13, s0
; GFX1310-NEXT: v_cmp_ne_u64_e64 s0, 0, v[4:5]
; GFX1310-NEXT: v_cndmask_b32_e64 v2, 0, v2, s0
-; GFX1310-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1310-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
; GFX1310-NEXT: v_add_co_u32 v4, s0, v8, 14
; GFX1310-NEXT: v_add_co_ci_u32_e64 v5, null, 0, v9, s0
-; GFX1310-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1310-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_1) | instid1(VALU_DEP_3)
; GFX1310-NEXT: v_add_co_u32 v6, s0, v6, v2
; GFX1310-NEXT: v_add_co_ci_u32_e64 v7, null, 0, v7, s0
-; GFX1310-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_2) | instid1(VALU_DEP_4)
; GFX1310-NEXT: v_lshlrev_b64_e32 v[8:9], 2, v[4:5]
; GFX1310-NEXT: v_and_b32_e32 v9, 0x80, v11
; GFX1310-NEXT: v_cndmask_b32_e64 v2, v10, 0, vcc_lo
+; GFX1310-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(SKIP_2) | instid1(VALU_DEP_4)
; GFX1310-NEXT: v_cmp_lt_i64_e64 s0, 3, v[6:7]
; GFX1310-NEXT: v_cmp_gt_i64_e64 s2, 1, v[4:5]
; GFX1310-NEXT: v_cmp_lt_i64_e64 s1, 30, v[4:5]
-; GFX1310-NEXT: s_delay_alu instid0(VALU_DEP_4)
; GFX1310-NEXT: v_cmp_lt_i64_e32 vcc_lo, 3, v[2:3]
; GFX1310-NEXT: v_or_b32_e32 v3, 0x7c, v9
; GFX1310-NEXT: v_or_b32_e32 v7, v8, v9
@@ -1514,7 +1511,7 @@ define i8 @to_fp8_f64(double %x) {
; GFX1310-NEXT: v_cmp_eq_f64_e32 vcc_lo, 0, v[0:1]
; GFX1310-NEXT: v_cndmask_b32_e32 v2, v2, v9, vcc_lo
; GFX1310-NEXT: v_cmp_class_f64_e64 vcc_lo, v[0:1], 0x204
-; GFX1310-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_2)
+; GFX1310-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX1310-NEXT: v_cndmask_b32_e32 v2, v2, v3, vcc_lo
; GFX1310-NEXT: v_cmp_o_f64_e32 vcc_lo, v[0:1], v[0:1]
; GFX1310-NEXT: v_cndmask_b32_e32 v0, 0x7e, v2, vcc_lo
@@ -1643,7 +1640,7 @@ define i8 @to_fp8_bf16(bfloat %x) {
; GFX1200-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1200-NEXT: v_or_b16 v3.l, v4.l, v3.l
; GFX1200-NEXT: v_cmp_lt_i16_e64 s0, 7, v2.l
-; GFX1200-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_2)
+; GFX1200-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX1200-NEXT: v_and_b16 v1.l, v1.h, v3.l
; GFX1200-NEXT: s_wait_alu depctr_va_sdst(0)
; GFX1200-NEXT: v_cndmask_b32_e64 v3, 0, 1, s0
@@ -1722,34 +1719,33 @@ define i8 @to_fp8_bf16(bfloat %x) {
; GFX1250-TRUE16-NEXT: v_cndmask_b32_e64 v5, 0, 1, vcc_lo
; GFX1250-TRUE16-NEXT: v_cmp_ne_u16_e32 vcc_lo, 0, v1.l
; GFX1250-TRUE16-NEXT: v_cmp_lt_i16_e64 s0, 7, v2.l
-; GFX1250-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX1250-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX1250-TRUE16-NEXT: v_mov_b16_e32 v4.l, v5.l
; GFX1250-TRUE16-NEXT: v_cndmask_b32_e64 v5, 0, 1, s0
-; GFX1250-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_2) | instid1(VALU_DEP_4)
; GFX1250-TRUE16-NEXT: v_bitop3_b16 v1.l, v1.h, v4.l, v2.h bitop3:0xe0
; GFX1250-TRUE16-NEXT: v_cndmask_b16 v1.h, v2.l, 0, s0
; GFX1250-TRUE16-NEXT: v_cmp_eq_f32_e64 s0, 0, v6
+; GFX1250-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
; GFX1250-TRUE16-NEXT: v_mov_b16_e32 v4.l, v5.l
-; GFX1250-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-TRUE16-NEXT: v_cndmask_b16 v1.l, 0, v1.l, vcc_lo
-; GFX1250-TRUE16-NEXT: v_add_nc_u16 v0.h, v7.l, v4.l
; GFX1250-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX1250-TRUE16-NEXT: v_add_nc_u16 v0.h, v7.l, v4.l
; GFX1250-TRUE16-NEXT: v_add_nc_u16 v1.l, v3.l, v1.l
-; GFX1250-TRUE16-NEXT: v_add_nc_u16 v0.h, v0.h, 6
; GFX1250-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX1250-TRUE16-NEXT: v_add_nc_u16 v0.h, v0.h, 6
; GFX1250-TRUE16-NEXT: v_cmp_lt_i16_e32 vcc_lo, 7, v1.l
+; GFX1250-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_3) | instid1(VALU_DEP_4)
; GFX1250-TRUE16-NEXT: v_lshlrev_b16 v2.l, 3, v0.h
; GFX1250-TRUE16-NEXT: v_cndmask_b16 v1.l, v1.l, 0, vcc_lo
; GFX1250-TRUE16-NEXT: v_cndmask_b16 v2.h, 0, 8, vcc_lo
; GFX1250-TRUE16-NEXT: v_cmp_gt_i16_e32 vcc_lo, 1, v0.h
-; GFX1250-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_3)
; GFX1250-TRUE16-NEXT: v_bitop3_b16 v0.h, v0.l, v1.h, v2.l bitop3:0xfe
+; GFX1250-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1250-TRUE16-NEXT: v_bitop3_b16 v1.l, v0.l, v1.l, v2.h bitop3:0xfe
-; GFX1250-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX1250-TRUE16-NEXT: v_cndmask_b16 v0.h, v0.h, v1.l, vcc_lo
; GFX1250-TRUE16-NEXT: v_cmp_o_f32_e32 vcc_lo, v6, v6
+; GFX1250-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1250-TRUE16-NEXT: v_cndmask_b16 v0.l, v0.h, v0.l, s0
-; GFX1250-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1250-TRUE16-NEXT: v_cndmask_b16 v0.l, 0x7f, v0.l, vcc_lo
; GFX1250-TRUE16-NEXT: s_set_pc_i64 s[30:31]
;
@@ -1991,7 +1987,7 @@ define i8 @to_e5m3fnu_f32(float %x) {
; GFX1200-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1200-NEXT: v_add_nc_u32_e32 v3, v9, v3
; GFX1200-NEXT: v_cmp_lt_i32_e32 vcc_lo, 7, v2
-; GFX1200-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_3) | instid1(VALU_DEP_2)
+; GFX1200-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX1200-NEXT: v_cmp_lt_i32_e64 s0, 7, v3
; GFX1200-NEXT: s_wait_alu depctr_va_vcc(0)
; GFX1200-NEXT: v_cndmask_b32_e64 v2, v2, 0, vcc_lo
@@ -2065,22 +2061,21 @@ define i8 @to_e5m3fnu_f32(float %x) {
; GFX1310-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1310-NEXT: v_add_nc_u32_e32 v3, v5, v3
; GFX1310-NEXT: v_cmp_lt_i32_e64 s0, 7, v2
-; GFX1310-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX1310-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_4) | instid1(VALU_DEP_2)
; GFX1310-NEXT: v_cmp_lt_i32_e32 vcc_lo, 7, v3
; GFX1310-NEXT: v_cndmask_b32_e64 v2, v2, 0, s0
; GFX1310-NEXT: v_cndmask_b32_e64 v3, v3, 0, vcc_lo
; GFX1310-NEXT: v_cndmask_b32_e64 v4, 0, 8, vcc_lo
; GFX1310-NEXT: v_add_co_ci_u32_e64 v1, null, 14, v1, s0
-; GFX1310-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1310-NEXT: v_or_b32_e32 v3, v4, v3
+; GFX1310-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX1310-NEXT: v_lshl_or_b32 v2, v1, 3, v2
; GFX1310-NEXT: v_cmp_gt_i32_e32 vcc_lo, 1, v1
-; GFX1310-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX1310-NEXT: v_cndmask_b32_e32 v1, v2, v3, vcc_lo
; GFX1310-NEXT: v_cmp_neq_f32_e32 vcc_lo, 0, v0
+; GFX1310-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX1310-NEXT: v_cndmask_b32_e32 v1, 0, v1, vcc_lo
; GFX1310-NEXT: v_cmp_o_f32_e32 vcc_lo, v0, v0
-; GFX1310-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX1310-NEXT: v_cndmask_b32_e32 v0, 0xff, v1, vcc_lo
; GFX1310-NEXT: s_set_pc_i64 s[30:31]
%r = call i8 @llvm.convert.to.arbitrary.fp.i8.f32(float %x, metadata !"Float8E5M3FNU", metadata !"round.tonearest", i1 false)
@@ -2376,43 +2371,42 @@ define <2 x i8> @to_e5m3fnu_v2f32(<2 x float> %x) {
; GFX1310-NEXT: v_bitop3_b32 v9, v12, v13, v11 bitop3:0xe0
; GFX1310-NEXT: v_cndmask_b32_e32 v8, 0, v17, vcc_lo
; GFX1310-NEXT: v_cmp_lt_i32_e32 vcc_lo, 7, v6
-; GFX1310-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(SKIP_1) | instid1(VALU_DEP_4)
; GFX1310-NEXT: v_cndmask_b32_e64 v3, v3, 0, s0
; GFX1310-NEXT: v_add_co_ci_u32_e64 v2, null, 14, v2, s0
+; GFX1310-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(SKIP_2) | instid1(VALU_DEP_4)
; GFX1310-NEXT: v_dual_add_nc_u32 v5, v5, v9 :: v_dual_add_nc_u32 v7, v10, v8
; GFX1310-NEXT: v_cndmask_b32_e64 v6, v6, 0, vcc_lo
; GFX1310-NEXT: v_cndmask_b32_e64 v8, 0, 8, vcc_lo
-; GFX1310-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
; GFX1310-NEXT: v_lshl_or_b32 v3, v2, 3, v3
+; GFX1310-NEXT: s_delay_alu instid0(VALU_DEP_4)
; GFX1310-NEXT: v_cmp_lt_i32_e64 s0, 7, v5
; GFX1310-NEXT: v_cmp_lt_i32_e32 vcc_lo, 7, v7
-; GFX1310-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_4) | instid1(VALU_DEP_3)
; GFX1310-NEXT: v_cndmask_b32_e64 v5, v5, 0, s0
; GFX1310-NEXT: v_cndmask_b32_e64 v7, v7, 0, vcc_lo
; GFX1310-NEXT: v_cndmask_b32_e64 v9, 0, 8, vcc_lo
; GFX1310-NEXT: v_add_co_ci_u32_e64 v4, null, 14, v4, s0
; GFX1310-NEXT: v_cmp_gt_i32_e32 vcc_lo, 1, v2
+; GFX1310-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_3)
; GFX1310-NEXT: v_or_b32_e32 v7, v9, v7
-; GFX1310-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX1310-NEXT: v_lshl_or_b32 v5, v4, 3, v5
; GFX1310-NEXT: v_or_b32_e32 v6, v8, v6
+; GFX1310-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_4)
; GFX1310-NEXT: v_cndmask_b32_e32 v2, v3, v6, vcc_lo
; GFX1310-NEXT: v_cmp_gt_i32_e32 vcc_lo, 1, v4
-; GFX1310-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(SKIP_1) | instid1(VALU_DEP_4)
; GFX1310-NEXT: v_cndmask_b32_e32 v3, v5, v7, vcc_lo
; GFX1310-NEXT: v_cmp_neq_f32_e32 vcc_lo, 0, v1
+; GFX1310-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(SKIP_1) | instid1(VALU_DEP_4)
; GFX1310-NEXT: v_cndmask_b32_e32 v2, 0, v2, vcc_lo
; GFX1310-NEXT: v_cmp_neq_f32_e32 vcc_lo, 0, v0
-; GFX1310-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(SKIP_1) | instid1(VALU_DEP_4)
; GFX1310-NEXT: v_cndmask_b32_e32 v3, 0, v3, vcc_lo
; GFX1310-NEXT: v_cmp_o_f32_e32 vcc_lo, v1, v1
+; GFX1310-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(SKIP_1) | instid1(VALU_DEP_4)
; GFX1310-NEXT: v_cndmask_b32_e32 v1, 0xff, v2, vcc_lo
; GFX1310-NEXT: v_cmp_o_f32_e32 vcc_lo, v0, v0
-; GFX1310-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_3)
; GFX1310-NEXT: v_cndmask_b32_e32 v0, 0xff, v3, vcc_lo
+; GFX1310-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX1310-NEXT: v_lshlrev_b16 v2, 8, v1
; GFX1310-NEXT: v_and_b32_e32 v1, 0xff, v1
-; GFX1310-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX1310-NEXT: v_bitop3_b16 v0, v0, v2, 0xff bitop3:0xec
; GFX1310-NEXT: s_set_pc_i64 s[30:31]
%r = call <2 x i8> @llvm.convert.to.arbitrary.fp.v2i8.v2f32(<2 x float> %x, metadata !"Float8E5M3FNU", metadata !"round.tonearest", i1 false)
@@ -2912,8 +2906,8 @@ define <4 x i8> @to_e5m3fnu_v4f32(<4 x float> %x) {
; GFX1310-NEXT: v_cmp_lt_i32_e64 s0, 7, v7
; GFX1310-NEXT: v_cndmask_b32_e64 v12, 0, 1, vcc_lo
; GFX1310-NEXT: v_cmp_lt_i32_e32 vcc_lo, 7, v5
-; GFX1310-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_3)
; GFX1310-NEXT: v_cndmask_b32_e64 v7, v7, 0, s0
+; GFX1310-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX1310-NEXT: v_bitop3_b32 v11, v11, v12, v13 bitop3:0xe0
; GFX1310-NEXT: v_cndmask_b32_e64 v5, v5, 0, vcc_lo
; GFX1310-NEXT: v_add_co_ci_u32_e64 v4, null, 14, v4, vcc_lo
@@ -2984,48 +2978,47 @@ define <4 x i8> @to_e5m3fnu_v4f32(<4 x float> %x) {
; GFX1310-NEXT: v_cmp_ne_u32_e64 s0, 0, v11
; GFX1310-NEXT: v_add_nc_u32_e32 v7, v8, v7
; GFX1310-NEXT: v_cmp_ne_u32_e32 vcc_lo, 0, v21
-; GFX1310-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
+; GFX1310-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX1310-NEXT: v_or_b32_e32 v5, v19, v5
; GFX1310-NEXT: v_cndmask_b32_e64 v11, 0, 1, s0
-; GFX1310-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_3) | instid1(VALU_DEP_3)
; GFX1310-NEXT: v_bitop3_b32 v11, v18, v11, v17 bitop3:0xe0
; GFX1310-NEXT: v_cndmask_b32_e64 v17, 0, 1, vcc_lo
; GFX1310-NEXT: v_cmp_ne_u32_e32 vcc_lo, 0, v16
; GFX1310-NEXT: v_lshrrev_b32_e32 v16, v16, v20
+; GFX1310-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_3) | instid1(VALU_DEP_2)
; GFX1310-NEXT: v_bitop3_b32 v12, v12, v17, v13 bitop3:0xe0
; GFX1310-NEXT: v_cndmask_b32_e32 v11, 0, v11, vcc_lo
; GFX1310-NEXT: v_bfe_u32 v13, v15, 20, 3
; GFX1310-NEXT: v_cmp_lt_i32_e32 vcc_lo, 7, v10
-; GFX1310-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_2) | instid1(VALU_DEP_3)
; GFX1310-NEXT: v_dual_add_nc_u32 v11, v16, v11 :: v_dual_add_nc_u32 v12, v13, v12
; GFX1310-NEXT: v_cndmask_b32_e64 v10, v10, 0, vcc_lo
; GFX1310-NEXT: v_add_co_ci_u32_e64 v9, null, 14, v9, vcc_lo
+; GFX1310-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_4)
; GFX1310-NEXT: v_cmp_lt_i32_e32 vcc_lo, 7, v11
-; GFX1310-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_3)
; GFX1310-NEXT: v_cmp_lt_i32_e64 s0, 7, v12
+; GFX1310-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX1310-NEXT: v_lshl_or_b32 v8, v9, 3, v10
; GFX1310-NEXT: v_cndmask_b32_e64 v10, v11, 0, vcc_lo
; GFX1310-NEXT: v_cndmask_b32_e64 v11, 0, 8, vcc_lo
-; GFX1310-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(SKIP_3) | instid1(VALU_DEP_3)
; GFX1310-NEXT: v_cndmask_b32_e64 v12, v12, 0, s0
; GFX1310-NEXT: v_add_co_ci_u32_e64 v13, null, 14, v14, s0
; GFX1310-NEXT: v_cmp_lt_i32_e32 vcc_lo, 7, v7
; GFX1310-NEXT: v_cmp_gt_i32_e64 s0, 1, v9
+; GFX1310-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_1) | instid1(VALU_DEP_3)
; GFX1310-NEXT: v_lshl_or_b32 v9, v13, 3, v12
; GFX1310-NEXT: v_cndmask_b32_e64 v7, v7, 0, vcc_lo
-; GFX1310-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_2) | instid1(VALU_DEP_2)
; GFX1310-NEXT: v_dual_cndmask_b32 v5, v8, v5, s0 :: v_dual_bitop2_b32 v8, v11, v10 bitop3:0x54
; GFX1310-NEXT: v_add_co_ci_u32_e64 v6, null, 14, v6, vcc_lo
; GFX1310-NEXT: v_cmp_neq_f32_e32 vcc_lo, 0, v3
+; GFX1310-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_4)
; GFX1310-NEXT: v_lshl_or_b32 v7, v6, 3, v7
-; GFX1310-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(SKIP_3) | instid1(VALU_DEP_4)
; GFX1310-NEXT: v_cndmask_b32_e32 v5, 0, v5, vcc_lo
; GFX1310-NEXT: v_cmp_gt_i32_e32 vcc_lo, 1, v13
; GFX1310-NEXT: v_cndmask_b32_e32 v8, v9, v8, vcc_lo
; GFX1310-NEXT: v_cmp_o_f32_e32 vcc_lo, v3, v3
+; GFX1310-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(SKIP_1) | instid1(VALU_DEP_4)
; GFX1310-NEXT: v_cndmask_b32_e32 v3, 0xff, v5, vcc_lo
; GFX1310-NEXT: v_cmp_neq_f32_e32 vcc_lo, 0, v2
-; GFX1310-NEXT: s_delay_alu instid0(VALU_DEP_4)
; GFX1310-NEXT: v_cndmask_b32_e32 v5, 0, v8, vcc_lo
; GFX1310-NEXT: v_cmp_gt_i32_e32 vcc_lo, 1, v6
; GFX1310-NEXT: v_cndmask_b32_e32 v4, v7, v4, vcc_lo
@@ -3158,7 +3151,7 @@ define i8 @to_e5m3fnu_f32_sat(float %x) {
; GFX1200-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1200-NEXT: v_add_nc_u32_e32 v3, v9, v3
; GFX1200-NEXT: v_cmp_lt_i32_e32 vcc_lo, 7, v2
-; GFX1200-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_3) | instid1(VALU_DEP_2)
+; GFX1200-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX1200-NEXT: v_cmp_lt_i32_e64 s0, 7, v3
; GFX1200-NEXT: s_wait_alu depctr_va_vcc(0)
; GFX1200-NEXT: v_cndmask_b32_e64 v2, v2, 0, vcc_lo
@@ -3169,7 +3162,7 @@ define i8 @to_e5m3fnu_f32_sat(float %x) {
; GFX1200-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_3)
; GFX1200-NEXT: v_cmp_lt_i32_e32 vcc_lo, 6, v3
; GFX1200-NEXT: v_or_b32_e32 v2, v4, v2
-; GFX1200-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_4) | instid1(VALU_DEP_3)
+; GFX1200-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX1200-NEXT: v_lshl_or_b32 v4, v1, 3, v3
; GFX1200-NEXT: v_cmp_gt_i32_e64 s2, 1, v1
; GFX1200-NEXT: v_cmp_eq_u32_e64 s0, 31, v1
@@ -3247,20 +3240,20 @@ define i8 @to_e5m3fnu_f32_sat(float %x) {
; GFX1310-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1310-NEXT: v_add_nc_u32_e32 v3, v5, v3
; GFX1310-NEXT: v_cmp_lt_i32_e64 s0, 7, v2
-; GFX1310-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX1310-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_4) | instid1(VALU_DEP_4)
; GFX1310-NEXT: v_cmp_lt_i32_e32 vcc_lo, 7, v3
; GFX1310-NEXT: v_cndmask_b32_e64 v2, v2, 0, s0
; GFX1310-NEXT: v_cndmask_b32_e64 v3, v3, 0, vcc_lo
; GFX1310-NEXT: v_cndmask_b32_e64 v4, 0, 8, vcc_lo
; GFX1310-NEXT: v_add_co_ci_u32_e64 v1, null, 14, v1, s0
-; GFX1310-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_3)
; GFX1310-NEXT: v_cmp_lt_i32_e32 vcc_lo, 6, v2
+; GFX1310-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_3)
; GFX1310-NEXT: v_or_b32_e32 v3, v4, v3
-; GFX1310-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_3) | instid1(VALU_DEP_3)
; GFX1310-NEXT: v_lshl_or_b32 v4, v1, 3, v2
; GFX1310-NEXT: v_cmp_gt_i32_e64 s2, 1, v1
; GFX1310-NEXT: v_cmp_eq_u32_e64 s0, 31, v1
; GFX1310-NEXT: v_cmp_lt_i32_e64 s1, 31, v1
+; GFX1310-NEXT: s_delay_alu instid0(VALU_DEP_4)
; GFX1310-NEXT: v_cndmask_b32_e64 v1, v4, v3, s2
; GFX1310-NEXT: s_and_b32 s0, s0, vcc_lo
; GFX1310-NEXT: v_cmp_neq_f32_e32 vcc_lo, 0, v0
@@ -3333,28 +3326,28 @@ define i8 @to_e5m3fnu_f32_rtz(float %x) {
; GFX1200-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1200-NEXT: v_min_u32_e32 v3, 31, v3
; GFX1200-NEXT: v_cmp_lt_i32_e64 s0, 7, v2
-; GFX1200-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_2)
+; GFX1200-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_3) | instid1(VALU_DEP_3)
; GFX1200-NEXT: v_lshrrev_b32_e32 v3, v3, v4
; GFX1200-NEXT: s_wait_alu depctr_va_sdst(0)
; GFX1200-NEXT: v_cndmask_b32_e64 v2, v2, 0, s0
; GFX1200-NEXT: v_add_co_ci_u32_e64 v1, null, 14, v1, s0
-; GFX1200-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1200-NEXT: v_cmp_lt_i32_e32 vcc_lo, 7, v3
+; GFX1200-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_4) | instid1(VALU_DEP_2)
; GFX1200-NEXT: v_lshl_or_b32 v2, v1, 3, v2
; GFX1200-NEXT: s_wait_alu depctr_va_vcc(0)
; GFX1200-NEXT: v_cndmask_b32_e64 v3, v3, 0, vcc_lo
; GFX1200-NEXT: v_cndmask_b32_e64 v4, 0, 8, vcc_lo
; GFX1200-NEXT: v_cmp_gt_i32_e32 vcc_lo, 1, v1
-; GFX1200-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX1200-NEXT: v_or_b32_e32 v3, v4, v3
; GFX1200-NEXT: s_wait_alu depctr_va_vcc(0)
+; GFX1200-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_2) | instid1(VALU_DEP_2)
; GFX1200-NEXT: v_cndmask_b32_e32 v1, v2, v3, vcc_lo
; GFX1200-NEXT: v_cmp_neq_f32_e32 vcc_lo, 0, v0
; GFX1200-NEXT: s_wait_alu depctr_va_vcc(0)
-; GFX1200-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_2) | instid1(VALU_DEP_2)
; GFX1200-NEXT: v_cndmask_b32_e32 v1, 0, v1, vcc_lo
; GFX1200-NEXT: v_cmp_o_f32_e32 vcc_lo, v0, v0
; GFX1200-NEXT: s_wait_alu depctr_va_vcc(0)
+; GFX1200-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX1200-NEXT: v_cndmask_b32_e32 v0, 0xff, v1, vcc_lo
; GFX1200-NEXT: s_setpc_b64 s[30:31]
;
@@ -3372,23 +3365,23 @@ define i8 @to_e5m3fnu_f32_rtz(float %x) {
; GFX1250-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-NEXT: v_min_u32_e32 v3, 31, v3
; GFX1250-NEXT: v_cmp_lt_i32_e64 s0, 7, v2
-; GFX1250-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX1250-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_2) | instid1(VALU_DEP_3)
; GFX1250-NEXT: v_lshrrev_b32_e32 v3, v3, v4
; GFX1250-NEXT: v_cndmask_b32_e64 v2, v2, 0, s0
; GFX1250-NEXT: v_add_co_ci_u32_e64 v1, null, 14, v1, s0
-; GFX1250-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-NEXT: v_cmp_lt_i32_e32 vcc_lo, 7, v3
+; GFX1250-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_3) | instid1(VALU_DEP_2)
; GFX1250-NEXT: v_lshl_or_b32 v2, v1, 3, v2
; GFX1250-NEXT: v_cndmask_b32_e64 v3, v3, 0, vcc_lo
; GFX1250-NEXT: v_cndmask_b32_e64 v4, 0, 8, vcc_lo
; GFX1250-NEXT: v_cmp_gt_i32_e32 vcc_lo, 1, v1
-; GFX1250-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1250-NEXT: v_or_b32_e32 v3, v4, v3
+; GFX1250-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX1250-NEXT: v_cndmask_b32_e32 v1, v2, v3, vcc_lo
; GFX1250-NEXT: v_cmp_neq_f32_e32 vcc_lo, 0, v0
-; GFX1250-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX1250-NEXT: v_cndmask_b32_e32 v1, 0, v1, vcc_lo
; GFX1250-NEXT: v_cmp_o_f32_e32 vcc_lo, v0, v0
+; GFX1250-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX1250-NEXT: v_cndmask_b32_e32 v0, 0xff, v1, vcc_lo
; GFX1250-NEXT: s_set_pc_i64 s[30:31]
;
@@ -3409,23 +3402,23 @@ define i8 @to_e5m3fnu_f32_rtz(float %x) {
; GFX1310-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1310-NEXT: v_min_u32_e32 v3, 31, v3
; GFX1310-NEXT: v_cmp_lt_i32_e64 s0, 7, v2
-; GFX1310-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX1310-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_2) | instid1(VALU_DEP_3)
; GFX1310-NEXT: v_lshrrev_b32_e32 v3, v3, v4
; GFX1310-NEXT: v_cndmask_b32_e64 v2, v2, 0, s0
; GFX1310-NEXT: v_add_co_ci_u32_e64 v1, null, 14, v1, s0
-; GFX1310-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1310-NEXT: v_cmp_lt_i32_e32 vcc_lo, 7, v3
+; GFX1310-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_3) | instid1(VALU_DEP_2)
; GFX1310-NEXT: v_lshl_or_b32 v2, v1, 3, v2
; GFX1310-NEXT: v_cndmask_b32_e64 v3, v3, 0, vcc_lo
; GFX1310-NEXT: v_cndmask_b32_e64 v4, 0, 8, vcc_lo
; GFX1310-NEXT: v_cmp_gt_i32_e32 vcc_lo, 1, v1
-; GFX1310-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1310-NEXT: v_or_b32_e32 v3, v4, v3
+; GFX1310-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX1310-NEXT: v_cndmask_b32_e32 v1, v2, v3, vcc_lo
; GFX1310-NEXT: v_cmp_neq_f32_e32 vcc_lo, 0, v0
-; GFX1310-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX1310-NEXT: v_cndmask_b32_e32 v1, 0, v1, vcc_lo
; GFX1310-NEXT: v_cmp_o_f32_e32 vcc_lo, v0, v0
+; GFX1310-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX1310-NEXT: v_cndmask_b32_e32 v0, 0xff, v1, vcc_lo
; GFX1310-NEXT: s_set_pc_i64 s[30:31]
%r = call i8 @llvm.convert.to.arbitrary.fp.i8.f32(float %x, metadata !"Float8E5M3FNU", metadata !"round.towardzero", i1 false)
diff --git a/llvm/test/CodeGen/AMDGPU/float-to-arbitrary-fp-widen.ll b/llvm/test/CodeGen/AMDGPU/float-to-arbitrary-fp-widen.ll
index 3aec6c15ee0c8c..4a393fdd9b4f45 100644
--- a/llvm/test/CodeGen/AMDGPU/float-to-arbitrary-fp-widen.ll
+++ b/llvm/test/CodeGen/AMDGPU/float-to-arbitrary-fp-widen.ll
@@ -2350,12 +2350,11 @@ define <7 x i8> @to_fp8_v7f16(<7 x half> %x) {
; GFX1200-NEXT: v_add_nc_u16 v4.l, v4.l, 6
; GFX1200-NEXT: v_cndmask_b16 v4.h, v4.h, 0, s2
; GFX1200-NEXT: v_cmp_lt_i16_e64 s1, 7, v6.l
-; GFX1200-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(SKIP_2) | instid1(VALU_DEP_3)
+; GFX1200-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(SKIP_3) | instid1(VALU_DEP_3)
; GFX1200-NEXT: v_mov_b16_e32 v10.l, v11.l
; GFX1200-NEXT: v_mov_b16_e32 v11.l, v12.l
; GFX1200-NEXT: s_wait_alu depctr_va_sdst(0)
; GFX1200-NEXT: v_cndmask_b16 v9.h, 0, 8, s1
-; GFX1200-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX1200-NEXT: v_or_b16 v8.h, v10.l, v9.l
; GFX1200-NEXT: v_lshlrev_b16 v9.l, 3, v4.l
; GFX1200-NEXT: v_or_b16 v6.h, v11.l, v6.h
@@ -2585,7 +2584,6 @@ define <7 x i8> @to_fp8_v7f16(<7 x half> %x) {
; GFX1200-NEXT: v_mov_b16_e32 v9.l, v11.l
; GFX1200-NEXT: v_or_b16 v8.l, v9.h, v8.l
; GFX1200-NEXT: s_wait_alu depctr_va_sdst(0)
-; GFX1200-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX1200-NEXT: v_cndmask_b32_e64 v12, 0, 1, s0
; GFX1200-NEXT: v_cmp_ne_u16_e64 s0, 0, v10.l
; GFX1200-NEXT: v_and_b16 v10.l, v10.h, 1
@@ -3538,7 +3536,6 @@ define <5 x i8> @to_bf8_v5f16(<5 x half> %x) {
; GFX1200-NEXT: v_or_b16 v4.l, 0x7c, v6.h
; GFX1200-NEXT: v_and_b16 v4.h, v2.h, 1
; GFX1200-NEXT: s_wait_alu depctr_va_sdst(0)
-; GFX1200-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX1200-NEXT: v_cndmask_b32_e64 v9, 0, 1, s0
; GFX1200-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX1200-NEXT: s_or_b32 s0, vcc_lo, s1
diff --git a/llvm/test/CodeGen/AMDGPU/fmax3-maximumnum.ll b/llvm/test/CodeGen/AMDGPU/fmax3-maximumnum.ll
index cb64b3de060ff9..228b9ddac7453a 100644
--- a/llvm/test/CodeGen/AMDGPU/fmax3-maximumnum.ll
+++ b/llvm/test/CodeGen/AMDGPU/fmax3-maximumnum.ll
@@ -1980,28 +1980,29 @@ define bfloat @v_max3_bf16_maximumnum_maximumnum__v_v_v_0(bfloat %a, bfloat %b,
; GFX11-SDAG-FAKE16-NEXT: v_cndmask_b32_e32 v1, v1, v0, vcc_lo
; GFX11-SDAG-FAKE16-NEXT: v_cmp_eq_u16_e32 vcc_lo, 0, v0
; GFX11-SDAG-FAKE16-NEXT: v_lshlrev_b32_e32 v3, 16, v1
-; GFX11-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_2) | instid1(VALU_DEP_1)
+; GFX11-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX11-SDAG-FAKE16-NEXT: v_cmp_eq_f32_e64 s0, 0, v3
; GFX11-SDAG-FAKE16-NEXT: s_and_b32 vcc_lo, s0, vcc_lo
; GFX11-SDAG-FAKE16-NEXT: v_cndmask_b32_e32 v0, v1, v0, vcc_lo
+; GFX11-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-SDAG-FAKE16-NEXT: v_lshlrev_b32_e32 v1, 16, v0
-; GFX11-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_2) | instid1(VALU_DEP_1)
; GFX11-SDAG-FAKE16-NEXT: v_cmp_u_f32_e32 vcc_lo, v1, v1
; GFX11-SDAG-FAKE16-NEXT: v_cndmask_b32_e32 v0, v0, v2, vcc_lo
; GFX11-SDAG-FAKE16-NEXT: v_lshlrev_b32_e32 v3, 16, v2
+; GFX11-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_3)
; GFX11-SDAG-FAKE16-NEXT: v_cmp_u_f32_e32 vcc_lo, v3, v3
-; GFX11-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX11-SDAG-FAKE16-NEXT: v_cndmask_b32_e32 v1, v2, v0, vcc_lo
; GFX11-SDAG-FAKE16-NEXT: v_lshlrev_b32_e32 v2, 16, v0
+; GFX11-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-SDAG-FAKE16-NEXT: v_lshlrev_b32_e32 v3, 16, v1
-; GFX11-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_2) | instid1(VALU_DEP_2)
; GFX11-SDAG-FAKE16-NEXT: v_cmp_gt_f32_e32 vcc_lo, v2, v3
; GFX11-SDAG-FAKE16-NEXT: v_cndmask_b32_e32 v1, v1, v0, vcc_lo
; GFX11-SDAG-FAKE16-NEXT: v_cmp_eq_u16_e32 vcc_lo, 0, v0
+; GFX11-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-SDAG-FAKE16-NEXT: v_lshlrev_b32_e32 v2, 16, v1
-; GFX11-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-SDAG-FAKE16-NEXT: v_cmp_eq_f32_e64 s0, 0, v2
; GFX11-SDAG-FAKE16-NEXT: s_and_b32 vcc_lo, s0, vcc_lo
+; GFX11-SDAG-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-SDAG-FAKE16-NEXT: v_cndmask_b32_e32 v0, v1, v0, vcc_lo
; GFX11-SDAG-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -2478,24 +2479,23 @@ define <2 x bfloat> @v_max3_v2bf16_maximumnum_maximumnum__v_v_v_0(<2 x bfloat> %
; GFX11-SDAG-TRUE16-NEXT: v_cmp_u_f32_e64 s1, v7, v7
; GFX11-SDAG-TRUE16-NEXT: v_cmp_u_f32_e64 s2, v8, v8
; GFX11-SDAG-TRUE16-NEXT: v_cndmask_b16 v3.l, v6.l, v4.l, vcc_lo
-; GFX11-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-SDAG-TRUE16-NEXT: v_cndmask_b16 v0.l, v0.l, v1.l, s0
+; GFX11-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-SDAG-TRUE16-NEXT: v_cndmask_b16 v4.l, v4.l, v3.l, s1
-; GFX11-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_2) | instid1(VALU_DEP_4)
; GFX11-SDAG-TRUE16-NEXT: v_cndmask_b16 v1.l, v1.l, v0.l, s2
; GFX11-SDAG-TRUE16-NEXT: v_mov_b16_e32 v5.l, v3.l
; GFX11-SDAG-TRUE16-NEXT: v_mov_b16_e32 v6.l, v0.l
-; GFX11-SDAG-TRUE16-NEXT: v_mov_b16_e32 v7.l, v4.l
; GFX11-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
+; GFX11-SDAG-TRUE16-NEXT: v_mov_b16_e32 v7.l, v4.l
; GFX11-SDAG-TRUE16-NEXT: v_mov_b16_e32 v8.l, v1.l
-; GFX11-SDAG-TRUE16-NEXT: v_lshlrev_b32_e32 v5, 16, v5
; GFX11-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
+; GFX11-SDAG-TRUE16-NEXT: v_lshlrev_b32_e32 v5, 16, v5
; GFX11-SDAG-TRUE16-NEXT: v_lshlrev_b32_e32 v6, 16, v6
+; GFX11-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
; GFX11-SDAG-TRUE16-NEXT: v_lshlrev_b32_e32 v7, 16, v7
-; GFX11-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-SDAG-TRUE16-NEXT: v_lshlrev_b32_e32 v8, 16, v8
+; GFX11-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-SDAG-TRUE16-NEXT: v_cmp_gt_f32_e32 vcc_lo, v5, v7
-; GFX11-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_2) | instid1(VALU_DEP_3)
; GFX11-SDAG-TRUE16-NEXT: v_cmp_gt_f32_e64 s0, v6, v8
; GFX11-SDAG-TRUE16-NEXT: v_lshlrev_b32_e32 v7, 16, v2
; GFX11-SDAG-TRUE16-NEXT: v_cndmask_b16 v4.l, v4.l, v3.l, vcc_lo
@@ -2529,24 +2529,23 @@ define <2 x bfloat> @v_max3_v2bf16_maximumnum_maximumnum__v_v_v_0(<2 x bfloat> %
; GFX11-SDAG-TRUE16-NEXT: v_cmp_u_f32_e32 vcc_lo, v1, v1
; GFX11-SDAG-TRUE16-NEXT: v_cmp_u_f32_e64 s0, v4, v4
; GFX11-SDAG-TRUE16-NEXT: v_cndmask_b16 v1.l, v3.l, v5.l, vcc_lo
-; GFX11-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-SDAG-TRUE16-NEXT: v_cndmask_b16 v0.l, v0.l, v2.l, s0
+; GFX11-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-SDAG-TRUE16-NEXT: v_cndmask_b16 v3.l, v5.l, v1.l, s1
-; GFX11-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_2) | instid1(VALU_DEP_4)
; GFX11-SDAG-TRUE16-NEXT: v_cndmask_b16 v2.l, v2.l, v0.l, s2
; GFX11-SDAG-TRUE16-NEXT: v_mov_b16_e32 v4.l, v1.l
; GFX11-SDAG-TRUE16-NEXT: v_mov_b16_e32 v5.l, v0.l
-; GFX11-SDAG-TRUE16-NEXT: v_mov_b16_e32 v6.l, v3.l
; GFX11-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
+; GFX11-SDAG-TRUE16-NEXT: v_mov_b16_e32 v6.l, v3.l
; GFX11-SDAG-TRUE16-NEXT: v_mov_b16_e32 v7.l, v2.l
-; GFX11-SDAG-TRUE16-NEXT: v_lshlrev_b32_e32 v4, 16, v4
; GFX11-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
+; GFX11-SDAG-TRUE16-NEXT: v_lshlrev_b32_e32 v4, 16, v4
; GFX11-SDAG-TRUE16-NEXT: v_lshlrev_b32_e32 v5, 16, v5
+; GFX11-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
; GFX11-SDAG-TRUE16-NEXT: v_lshlrev_b32_e32 v6, 16, v6
-; GFX11-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-SDAG-TRUE16-NEXT: v_lshlrev_b32_e32 v7, 16, v7
+; GFX11-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-SDAG-TRUE16-NEXT: v_cmp_gt_f32_e32 vcc_lo, v4, v6
-; GFX11-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX11-SDAG-TRUE16-NEXT: v_cmp_gt_f32_e64 s0, v5, v7
; GFX11-SDAG-TRUE16-NEXT: v_cndmask_b16 v3.l, v3.l, v1.l, vcc_lo
; GFX11-SDAG-TRUE16-NEXT: v_cndmask_b16 v2.l, v2.l, v0.l, s0
@@ -2596,7 +2595,7 @@ define <2 x bfloat> @v_max3_v2bf16_maximumnum_maximumnum__v_v_v_0(<2 x bfloat> %
; GFX11-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-SDAG-FAKE16-NEXT: v_lshlrev_b32_e32 v5, 16, v4
; GFX11-SDAG-FAKE16-NEXT: v_cmp_gt_f32_e32 vcc_lo, v7, v8
-; GFX11-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2)
+; GFX11-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_3) | instid1(SALU_CYCLE_1)
; GFX11-SDAG-FAKE16-NEXT: v_cmp_eq_f32_e64 s0, 0, v5
; GFX11-SDAG-FAKE16-NEXT: v_cndmask_b32_e32 v1, v1, v0, vcc_lo
; GFX11-SDAG-FAKE16-NEXT: v_cmp_eq_u16_e32 vcc_lo, 0, v3
@@ -2604,43 +2603,45 @@ define <2 x bfloat> @v_max3_v2bf16_maximumnum_maximumnum__v_v_v_0(<2 x bfloat> %
; GFX11-SDAG-FAKE16-NEXT: v_cndmask_b32_e32 v3, v4, v3, vcc_lo
; GFX11-SDAG-FAKE16-NEXT: v_lshrrev_b32_e32 v4, 16, v2
; GFX11-SDAG-FAKE16-NEXT: v_lshlrev_b32_e32 v6, 16, v1
-; GFX11-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_2) | instid1(VALU_DEP_1)
+; GFX11-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX11-SDAG-FAKE16-NEXT: v_cmp_eq_f32_e64 s2, 0, v6
; GFX11-SDAG-FAKE16-NEXT: s_and_b32 vcc_lo, s2, s1
; GFX11-SDAG-FAKE16-NEXT: v_dual_cndmask_b32 v0, v1, v0 :: v_dual_lshlrev_b32 v1, 16, v3
+; GFX11-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-SDAG-FAKE16-NEXT: v_lshlrev_b32_e32 v5, 16, v0
-; GFX11-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_3)
; GFX11-SDAG-FAKE16-NEXT: v_cmp_u_f32_e32 vcc_lo, v1, v1
; GFX11-SDAG-FAKE16-NEXT: v_dual_cndmask_b32 v1, v3, v4 :: v_dual_and_b32 v6, 0xffff0000, v2
+; GFX11-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_1) | instid1(VALU_DEP_3)
; GFX11-SDAG-FAKE16-NEXT: v_cmp_u_f32_e32 vcc_lo, v5, v5
; GFX11-SDAG-FAKE16-NEXT: v_cndmask_b32_e32 v0, v0, v2, vcc_lo
-; GFX11-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_1) | instid1(VALU_DEP_3)
; GFX11-SDAG-FAKE16-NEXT: v_cmp_u_f32_e32 vcc_lo, v6, v6
; GFX11-SDAG-FAKE16-NEXT: v_lshlrev_b32_e32 v7, 16, v2
+; GFX11-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_1) | instid1(VALU_DEP_3)
; GFX11-SDAG-FAKE16-NEXT: v_cmp_eq_u16_e64 s1, 0, v0
; GFX11-SDAG-FAKE16-NEXT: v_cndmask_b32_e32 v3, v4, v1, vcc_lo
-; GFX11-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_1) | instid1(VALU_DEP_3)
; GFX11-SDAG-FAKE16-NEXT: v_cmp_u_f32_e32 vcc_lo, v7, v7
; GFX11-SDAG-FAKE16-NEXT: v_lshlrev_b32_e32 v4, 16, v1
+; GFX11-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-SDAG-FAKE16-NEXT: v_dual_cndmask_b32 v2, v2, v0 :: v_dual_lshlrev_b32 v5, 16, v3
-; GFX11-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-SDAG-FAKE16-NEXT: v_cmp_gt_f32_e32 vcc_lo, v4, v5
+; GFX11-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-SDAG-FAKE16-NEXT: v_lshlrev_b32_e32 v7, 16, v2
; GFX11-SDAG-FAKE16-NEXT: v_dual_cndmask_b32 v3, v3, v1 :: v_dual_lshlrev_b32 v6, 16, v0
-; GFX11-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-SDAG-FAKE16-NEXT: v_lshlrev_b32_e32 v4, 16, v3
+; GFX11-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-SDAG-FAKE16-NEXT: v_cmp_gt_f32_e32 vcc_lo, v6, v7
-; GFX11-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_2) | instid1(VALU_DEP_2)
; GFX11-SDAG-FAKE16-NEXT: v_cmp_eq_f32_e64 s0, 0, v4
; GFX11-SDAG-FAKE16-NEXT: v_cndmask_b32_e32 v2, v2, v0, vcc_lo
; GFX11-SDAG-FAKE16-NEXT: v_cmp_eq_u16_e32 vcc_lo, 0, v1
+; GFX11-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX11-SDAG-FAKE16-NEXT: v_lshlrev_b32_e32 v5, 16, v2
; GFX11-SDAG-FAKE16-NEXT: s_and_b32 vcc_lo, s0, vcc_lo
; GFX11-SDAG-FAKE16-NEXT: v_cndmask_b32_e32 v1, v3, v1, vcc_lo
-; GFX11-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_2) | instid1(VALU_DEP_1)
+; GFX11-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX11-SDAG-FAKE16-NEXT: v_cmp_eq_f32_e64 s2, 0, v5
; GFX11-SDAG-FAKE16-NEXT: s_and_b32 vcc_lo, s2, s1
; GFX11-SDAG-FAKE16-NEXT: v_cndmask_b32_e32 v0, v2, v0, vcc_lo
+; GFX11-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-SDAG-FAKE16-NEXT: v_perm_b32 v0, v1, v0, 0x5040100
; GFX11-SDAG-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -2720,24 +2721,23 @@ define <2 x bfloat> @v_max3_v2bf16_maximumnum_maximumnum__v_v_v_0(<2 x bfloat> %
; GFX12-SDAG-TRUE16-NEXT: s_wait_alu depctr_va_vcc(0)
; GFX12-SDAG-TRUE16-NEXT: v_cndmask_b16 v1.l, v3.l, v5.l, vcc_lo
; GFX12-SDAG-TRUE16-NEXT: s_wait_alu depctr_va_sdst(0)
-; GFX12-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX12-SDAG-TRUE16-NEXT: v_cndmask_b16 v0.l, v0.l, v2.l, s0
+; GFX12-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX12-SDAG-TRUE16-NEXT: v_cndmask_b16 v3.l, v5.l, v1.l, s1
-; GFX12-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_2) | instid1(VALU_DEP_4)
; GFX12-SDAG-TRUE16-NEXT: v_cndmask_b16 v2.l, v2.l, v0.l, s2
; GFX12-SDAG-TRUE16-NEXT: v_mov_b16_e32 v4.l, v1.l
; GFX12-SDAG-TRUE16-NEXT: v_mov_b16_e32 v5.l, v0.l
-; GFX12-SDAG-TRUE16-NEXT: v_mov_b16_e32 v6.l, v3.l
; GFX12-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
+; GFX12-SDAG-TRUE16-NEXT: v_mov_b16_e32 v6.l, v3.l
; GFX12-SDAG-TRUE16-NEXT: v_mov_b16_e32 v7.l, v2.l
-; GFX12-SDAG-TRUE16-NEXT: v_lshlrev_b32_e32 v4, 16, v4
; GFX12-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
+; GFX12-SDAG-TRUE16-NEXT: v_lshlrev_b32_e32 v4, 16, v4
; GFX12-SDAG-TRUE16-NEXT: v_lshlrev_b32_e32 v5, 16, v5
+; GFX12-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
; GFX12-SDAG-TRUE16-NEXT: v_lshlrev_b32_e32 v6, 16, v6
-; GFX12-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX12-SDAG-TRUE16-NEXT: v_lshlrev_b32_e32 v7, 16, v7
+; GFX12-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX12-SDAG-TRUE16-NEXT: v_cmp_gt_f32_e32 vcc_lo, v4, v6
-; GFX12-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_3) | instid1(VALU_DEP_2)
; GFX12-SDAG-TRUE16-NEXT: v_cmp_gt_f32_e64 s0, v5, v7
; GFX12-SDAG-TRUE16-NEXT: s_wait_alu depctr_va_vcc(0)
; GFX12-SDAG-TRUE16-NEXT: v_cndmask_b16 v3.l, v3.l, v1.l, vcc_lo
diff --git a/llvm/test/CodeGen/AMDGPU/fmax_legacy.f16.ll b/llvm/test/CodeGen/AMDGPU/fmax_legacy.f16.ll
index 0785510b255099..781018d8018931 100644
--- a/llvm/test/CodeGen/AMDGPU/fmax_legacy.f16.ll
+++ b/llvm/test/CodeGen/AMDGPU/fmax_legacy.f16.ll
@@ -138,7 +138,7 @@ define <2 x half> @test_fmax_legacy_ugt_v2f16(<2 x half> %a, <2 x half> %b) #0 {
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v2, 16, v1
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v3, 16, v0
; GFX11-TRUE16-NEXT: v_cmp_nle_f16_e64 s0, v0.l, v1.l
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX11-TRUE16-NEXT: v_cmp_nle_f16_e32 vcc_lo, v3.l, v2.l
; GFX11-TRUE16-NEXT: v_cndmask_b16 v0.l, v1.l, v0.l, s0
; GFX11-TRUE16-NEXT: v_cndmask_b16 v0.h, v2.l, v3.l, vcc_lo
@@ -263,11 +263,10 @@ define <3 x half> @test_fmax_legacy_ugt_v3f16(<3 x half> %a, <3 x half> %b) #0 {
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v5, 16, v0
; GFX11-TRUE16-NEXT: v_cmp_nle_f16_e32 vcc_lo, v0.l, v2.l
; GFX11-TRUE16-NEXT: v_cmp_nle_f16_e64 s1, v1.l, v3.l
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_1) | instid1(VALU_DEP_3)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX11-TRUE16-NEXT: v_cmp_nle_f16_e64 s0, v5.l, v4.l
; GFX11-TRUE16-NEXT: v_cndmask_b16 v0.l, v2.l, v0.l, vcc_lo
; GFX11-TRUE16-NEXT: v_cndmask_b16 v1.l, v3.l, v1.l, s1
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX11-TRUE16-NEXT: v_cndmask_b16 v0.h, v4.l, v5.l, s0
; GFX11-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -422,11 +421,10 @@ define <4 x half> @test_fmax_legacy_ugt_v4f16(<4 x half> %a, <4 x half> %b) #0 {
; GFX11-TRUE16-NEXT: v_cmp_nle_f16_e32 vcc_lo, v1.l, v3.l
; GFX11-TRUE16-NEXT: v_cmp_nle_f16_e64 s0, v0.l, v2.l
; GFX11-TRUE16-NEXT: v_cmp_nle_f16_e64 s1, v5.l, v4.l
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(SKIP_1) | instid1(VALU_DEP_4)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_4)
; GFX11-TRUE16-NEXT: v_cmp_nle_f16_e64 s2, v7.l, v6.l
; GFX11-TRUE16-NEXT: v_cndmask_b16 v1.l, v3.l, v1.l, vcc_lo
; GFX11-TRUE16-NEXT: v_cndmask_b16 v0.l, v2.l, v0.l, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
; GFX11-TRUE16-NEXT: v_cndmask_b16 v0.h, v4.l, v5.l, s1
; GFX11-TRUE16-NEXT: v_cndmask_b16 v1.h, v6.l, v7.l, s2
; GFX11-TRUE16-NEXT: s_setpc_b64 s[30:31]
diff --git a/llvm/test/CodeGen/AMDGPU/fmed3.bf16.ll b/llvm/test/CodeGen/AMDGPU/fmed3.bf16.ll
index 7638f32cc61ac3..8a63240eeb75bc 100644
--- a/llvm/test/CodeGen/AMDGPU/fmed3.bf16.ll
+++ b/llvm/test/CodeGen/AMDGPU/fmed3.bf16.ll
@@ -201,20 +201,19 @@ define <2 x bfloat> @v_test_fmed3_r_i_i_v2bf16_minimumnum_maximumnum(<2 x bfloat
; GFX11-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_3)
; GFX11-SDAG-TRUE16-NEXT: v_cmp_o_f32_e32 vcc_lo, v1, v1
; GFX11-SDAG-TRUE16-NEXT: v_cmp_o_f32_e64 s0, v2, v2
-; GFX11-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX11-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX11-SDAG-TRUE16-NEXT: v_cndmask_b16 v1.l, 0x4000, v3.l, vcc_lo
; GFX11-SDAG-TRUE16-NEXT: v_cndmask_b16 v0.l, 0x4000, v0.l, s0
-; GFX11-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-SDAG-TRUE16-NEXT: v_mov_b16_e32 v2.l, v1.l
-; GFX11-SDAG-TRUE16-NEXT: v_mov_b16_e32 v3.l, v0.l
; GFX11-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX11-SDAG-TRUE16-NEXT: v_mov_b16_e32 v3.l, v0.l
; GFX11-SDAG-TRUE16-NEXT: v_lshlrev_b32_e32 v2, 16, v2
-; GFX11-SDAG-TRUE16-NEXT: v_lshlrev_b32_e32 v3, 16, v3
; GFX11-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX11-SDAG-TRUE16-NEXT: v_lshlrev_b32_e32 v3, 16, v3
; GFX11-SDAG-TRUE16-NEXT: v_cmp_lt_f32_e32 vcc_lo, 2.0, v2
+; GFX11-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_2) | instid1(VALU_DEP_2)
; GFX11-SDAG-TRUE16-NEXT: v_cmp_lt_f32_e64 s0, 2.0, v3
; GFX11-SDAG-TRUE16-NEXT: v_cndmask_b16 v1.l, 0x4000, v1.l, vcc_lo
-; GFX11-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-SDAG-TRUE16-NEXT: v_cndmask_b16 v0.l, 0x4000, v0.l, s0
; GFX11-SDAG-TRUE16-NEXT: v_mov_b16_e32 v2.l, v1.l
; GFX11-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
@@ -223,7 +222,7 @@ define <2 x bfloat> @v_test_fmed3_r_i_i_v2bf16_minimumnum_maximumnum(<2 x bfloat
; GFX11-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-SDAG-TRUE16-NEXT: v_lshlrev_b32_e32 v3, 16, v3
; GFX11-SDAG-TRUE16-NEXT: v_cmp_gt_f32_e32 vcc_lo, 4.0, v2
-; GFX11-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_2)
+; GFX11-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX11-SDAG-TRUE16-NEXT: v_cmp_gt_f32_e64 s0, 4.0, v3
; GFX11-SDAG-TRUE16-NEXT: v_cndmask_b16 v0.h, 0x4080, v1.l, vcc_lo
; GFX11-SDAG-TRUE16-NEXT: v_cndmask_b16 v0.l, 0x4080, v0.l, s0
diff --git a/llvm/test/CodeGen/AMDGPU/fmin3-minimumnum.ll b/llvm/test/CodeGen/AMDGPU/fmin3-minimumnum.ll
index da7f46a2a06137..1410cc5339b2a7 100644
--- a/llvm/test/CodeGen/AMDGPU/fmin3-minimumnum.ll
+++ b/llvm/test/CodeGen/AMDGPU/fmin3-minimumnum.ll
@@ -1982,28 +1982,29 @@ define bfloat @v_min3_bf16_minimumnum_minimumnum__v_v_v_0(bfloat %a, bfloat %b,
; GFX11-SDAG-FAKE16-NEXT: v_cndmask_b32_e32 v1, v1, v0, vcc_lo
; GFX11-SDAG-FAKE16-NEXT: v_cmp_eq_u16_e32 vcc_lo, 0x8000, v0
; GFX11-SDAG-FAKE16-NEXT: v_lshlrev_b32_e32 v3, 16, v1
-; GFX11-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_2) | instid1(VALU_DEP_1)
+; GFX11-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX11-SDAG-FAKE16-NEXT: v_cmp_eq_f32_e64 s0, 0, v3
; GFX11-SDAG-FAKE16-NEXT: s_and_b32 vcc_lo, s0, vcc_lo
; GFX11-SDAG-FAKE16-NEXT: v_cndmask_b32_e32 v0, v1, v0, vcc_lo
+; GFX11-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-SDAG-FAKE16-NEXT: v_lshlrev_b32_e32 v1, 16, v0
-; GFX11-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_2) | instid1(VALU_DEP_1)
; GFX11-SDAG-FAKE16-NEXT: v_cmp_u_f32_e32 vcc_lo, v1, v1
; GFX11-SDAG-FAKE16-NEXT: v_cndmask_b32_e32 v0, v0, v2, vcc_lo
; GFX11-SDAG-FAKE16-NEXT: v_lshlrev_b32_e32 v3, 16, v2
+; GFX11-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_3)
; GFX11-SDAG-FAKE16-NEXT: v_cmp_u_f32_e32 vcc_lo, v3, v3
-; GFX11-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX11-SDAG-FAKE16-NEXT: v_cndmask_b32_e32 v1, v2, v0, vcc_lo
; GFX11-SDAG-FAKE16-NEXT: v_lshlrev_b32_e32 v2, 16, v0
+; GFX11-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-SDAG-FAKE16-NEXT: v_lshlrev_b32_e32 v3, 16, v1
-; GFX11-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_2) | instid1(VALU_DEP_2)
; GFX11-SDAG-FAKE16-NEXT: v_cmp_lt_f32_e32 vcc_lo, v2, v3
; GFX11-SDAG-FAKE16-NEXT: v_cndmask_b32_e32 v1, v1, v0, vcc_lo
; GFX11-SDAG-FAKE16-NEXT: v_cmp_eq_u16_e32 vcc_lo, 0x8000, v0
+; GFX11-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-SDAG-FAKE16-NEXT: v_lshlrev_b32_e32 v2, 16, v1
-; GFX11-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-SDAG-FAKE16-NEXT: v_cmp_eq_f32_e64 s0, 0, v2
; GFX11-SDAG-FAKE16-NEXT: s_and_b32 vcc_lo, s0, vcc_lo
+; GFX11-SDAG-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-SDAG-FAKE16-NEXT: v_cndmask_b32_e32 v0, v1, v0, vcc_lo
; GFX11-SDAG-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -2483,24 +2484,23 @@ define <2 x bfloat> @v_min3_v2bf16_minimumnum_minimumnum__v_v_v_0(<2 x bfloat> %
; GFX11-SDAG-TRUE16-NEXT: v_cmp_u_f32_e64 s1, v7, v7
; GFX11-SDAG-TRUE16-NEXT: v_cmp_u_f32_e64 s2, v8, v8
; GFX11-SDAG-TRUE16-NEXT: v_cndmask_b16 v3.l, v6.l, v4.l, vcc_lo
-; GFX11-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-SDAG-TRUE16-NEXT: v_cndmask_b16 v0.l, v0.l, v1.l, s0
+; GFX11-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-SDAG-TRUE16-NEXT: v_cndmask_b16 v4.l, v4.l, v3.l, s1
-; GFX11-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_2) | instid1(VALU_DEP_4)
; GFX11-SDAG-TRUE16-NEXT: v_cndmask_b16 v1.l, v1.l, v0.l, s2
; GFX11-SDAG-TRUE16-NEXT: v_mov_b16_e32 v5.l, v3.l
; GFX11-SDAG-TRUE16-NEXT: v_mov_b16_e32 v6.l, v0.l
-; GFX11-SDAG-TRUE16-NEXT: v_mov_b16_e32 v7.l, v4.l
; GFX11-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
+; GFX11-SDAG-TRUE16-NEXT: v_mov_b16_e32 v7.l, v4.l
; GFX11-SDAG-TRUE16-NEXT: v_mov_b16_e32 v8.l, v1.l
-; GFX11-SDAG-TRUE16-NEXT: v_lshlrev_b32_e32 v5, 16, v5
; GFX11-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
+; GFX11-SDAG-TRUE16-NEXT: v_lshlrev_b32_e32 v5, 16, v5
; GFX11-SDAG-TRUE16-NEXT: v_lshlrev_b32_e32 v6, 16, v6
+; GFX11-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
; GFX11-SDAG-TRUE16-NEXT: v_lshlrev_b32_e32 v7, 16, v7
-; GFX11-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-SDAG-TRUE16-NEXT: v_lshlrev_b32_e32 v8, 16, v8
+; GFX11-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-SDAG-TRUE16-NEXT: v_cmp_lt_f32_e32 vcc_lo, v5, v7
-; GFX11-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_2) | instid1(VALU_DEP_3)
; GFX11-SDAG-TRUE16-NEXT: v_cmp_lt_f32_e64 s0, v6, v8
; GFX11-SDAG-TRUE16-NEXT: v_lshlrev_b32_e32 v7, 16, v2
; GFX11-SDAG-TRUE16-NEXT: v_cndmask_b16 v4.l, v4.l, v3.l, vcc_lo
@@ -2534,24 +2534,23 @@ define <2 x bfloat> @v_min3_v2bf16_minimumnum_minimumnum__v_v_v_0(<2 x bfloat> %
; GFX11-SDAG-TRUE16-NEXT: v_cmp_u_f32_e32 vcc_lo, v1, v1
; GFX11-SDAG-TRUE16-NEXT: v_cmp_u_f32_e64 s0, v4, v4
; GFX11-SDAG-TRUE16-NEXT: v_cndmask_b16 v1.l, v3.l, v5.l, vcc_lo
-; GFX11-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-SDAG-TRUE16-NEXT: v_cndmask_b16 v0.l, v0.l, v2.l, s0
+; GFX11-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-SDAG-TRUE16-NEXT: v_cndmask_b16 v3.l, v5.l, v1.l, s1
-; GFX11-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_2) | instid1(VALU_DEP_4)
; GFX11-SDAG-TRUE16-NEXT: v_cndmask_b16 v2.l, v2.l, v0.l, s2
; GFX11-SDAG-TRUE16-NEXT: v_mov_b16_e32 v4.l, v1.l
; GFX11-SDAG-TRUE16-NEXT: v_mov_b16_e32 v5.l, v0.l
-; GFX11-SDAG-TRUE16-NEXT: v_mov_b16_e32 v6.l, v3.l
; GFX11-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
+; GFX11-SDAG-TRUE16-NEXT: v_mov_b16_e32 v6.l, v3.l
; GFX11-SDAG-TRUE16-NEXT: v_mov_b16_e32 v7.l, v2.l
-; GFX11-SDAG-TRUE16-NEXT: v_lshlrev_b32_e32 v4, 16, v4
; GFX11-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
+; GFX11-SDAG-TRUE16-NEXT: v_lshlrev_b32_e32 v4, 16, v4
; GFX11-SDAG-TRUE16-NEXT: v_lshlrev_b32_e32 v5, 16, v5
+; GFX11-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
; GFX11-SDAG-TRUE16-NEXT: v_lshlrev_b32_e32 v6, 16, v6
-; GFX11-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-SDAG-TRUE16-NEXT: v_lshlrev_b32_e32 v7, 16, v7
+; GFX11-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-SDAG-TRUE16-NEXT: v_cmp_lt_f32_e32 vcc_lo, v4, v6
-; GFX11-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX11-SDAG-TRUE16-NEXT: v_cmp_lt_f32_e64 s0, v5, v7
; GFX11-SDAG-TRUE16-NEXT: v_cndmask_b16 v3.l, v3.l, v1.l, vcc_lo
; GFX11-SDAG-TRUE16-NEXT: v_cndmask_b16 v2.l, v2.l, v0.l, s0
@@ -2601,7 +2600,7 @@ define <2 x bfloat> @v_min3_v2bf16_minimumnum_minimumnum__v_v_v_0(<2 x bfloat> %
; GFX11-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-SDAG-FAKE16-NEXT: v_lshlrev_b32_e32 v5, 16, v4
; GFX11-SDAG-FAKE16-NEXT: v_cmp_lt_f32_e32 vcc_lo, v7, v8
-; GFX11-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2)
+; GFX11-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_3) | instid1(SALU_CYCLE_1)
; GFX11-SDAG-FAKE16-NEXT: v_cmp_eq_f32_e64 s0, 0, v5
; GFX11-SDAG-FAKE16-NEXT: v_cndmask_b32_e32 v1, v1, v0, vcc_lo
; GFX11-SDAG-FAKE16-NEXT: v_cmp_eq_u16_e32 vcc_lo, 0x8000, v3
@@ -2609,43 +2608,45 @@ define <2 x bfloat> @v_min3_v2bf16_minimumnum_minimumnum__v_v_v_0(<2 x bfloat> %
; GFX11-SDAG-FAKE16-NEXT: v_cndmask_b32_e32 v3, v4, v3, vcc_lo
; GFX11-SDAG-FAKE16-NEXT: v_lshrrev_b32_e32 v4, 16, v2
; GFX11-SDAG-FAKE16-NEXT: v_lshlrev_b32_e32 v6, 16, v1
-; GFX11-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_2) | instid1(VALU_DEP_1)
+; GFX11-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX11-SDAG-FAKE16-NEXT: v_cmp_eq_f32_e64 s2, 0, v6
; GFX11-SDAG-FAKE16-NEXT: s_and_b32 vcc_lo, s2, s1
; GFX11-SDAG-FAKE16-NEXT: v_dual_cndmask_b32 v0, v1, v0 :: v_dual_lshlrev_b32 v1, 16, v3
+; GFX11-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-SDAG-FAKE16-NEXT: v_lshlrev_b32_e32 v5, 16, v0
-; GFX11-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_3)
; GFX11-SDAG-FAKE16-NEXT: v_cmp_u_f32_e32 vcc_lo, v1, v1
; GFX11-SDAG-FAKE16-NEXT: v_dual_cndmask_b32 v1, v3, v4 :: v_dual_and_b32 v6, 0xffff0000, v2
+; GFX11-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_1) | instid1(VALU_DEP_3)
; GFX11-SDAG-FAKE16-NEXT: v_cmp_u_f32_e32 vcc_lo, v5, v5
; GFX11-SDAG-FAKE16-NEXT: v_cndmask_b32_e32 v0, v0, v2, vcc_lo
-; GFX11-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_1) | instid1(VALU_DEP_3)
; GFX11-SDAG-FAKE16-NEXT: v_cmp_u_f32_e32 vcc_lo, v6, v6
; GFX11-SDAG-FAKE16-NEXT: v_lshlrev_b32_e32 v7, 16, v2
+; GFX11-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_1) | instid1(VALU_DEP_3)
; GFX11-SDAG-FAKE16-NEXT: v_cmp_eq_u16_e64 s1, 0x8000, v0
; GFX11-SDAG-FAKE16-NEXT: v_cndmask_b32_e32 v3, v4, v1, vcc_lo
-; GFX11-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_1) | instid1(VALU_DEP_3)
; GFX11-SDAG-FAKE16-NEXT: v_cmp_u_f32_e32 vcc_lo, v7, v7
; GFX11-SDAG-FAKE16-NEXT: v_lshlrev_b32_e32 v4, 16, v1
+; GFX11-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-SDAG-FAKE16-NEXT: v_dual_cndmask_b32 v2, v2, v0 :: v_dual_lshlrev_b32 v5, 16, v3
-; GFX11-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-SDAG-FAKE16-NEXT: v_cmp_lt_f32_e32 vcc_lo, v4, v5
+; GFX11-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-SDAG-FAKE16-NEXT: v_lshlrev_b32_e32 v7, 16, v2
; GFX11-SDAG-FAKE16-NEXT: v_dual_cndmask_b32 v3, v3, v1 :: v_dual_lshlrev_b32 v6, 16, v0
-; GFX11-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-SDAG-FAKE16-NEXT: v_lshlrev_b32_e32 v4, 16, v3
+; GFX11-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-SDAG-FAKE16-NEXT: v_cmp_lt_f32_e32 vcc_lo, v6, v7
-; GFX11-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_2) | instid1(VALU_DEP_2)
; GFX11-SDAG-FAKE16-NEXT: v_cmp_eq_f32_e64 s0, 0, v4
; GFX11-SDAG-FAKE16-NEXT: v_cndmask_b32_e32 v2, v2, v0, vcc_lo
; GFX11-SDAG-FAKE16-NEXT: v_cmp_eq_u16_e32 vcc_lo, 0x8000, v1
+; GFX11-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX11-SDAG-FAKE16-NEXT: v_lshlrev_b32_e32 v5, 16, v2
; GFX11-SDAG-FAKE16-NEXT: s_and_b32 vcc_lo, s0, vcc_lo
; GFX11-SDAG-FAKE16-NEXT: v_cndmask_b32_e32 v1, v3, v1, vcc_lo
-; GFX11-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_2) | instid1(VALU_DEP_1)
+; GFX11-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX11-SDAG-FAKE16-NEXT: v_cmp_eq_f32_e64 s2, 0, v5
; GFX11-SDAG-FAKE16-NEXT: s_and_b32 vcc_lo, s2, s1
; GFX11-SDAG-FAKE16-NEXT: v_cndmask_b32_e32 v0, v2, v0, vcc_lo
+; GFX11-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-SDAG-FAKE16-NEXT: v_perm_b32 v0, v1, v0, 0x5040100
; GFX11-SDAG-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -2725,24 +2726,23 @@ define <2 x bfloat> @v_min3_v2bf16_minimumnum_minimumnum__v_v_v_0(<2 x bfloat> %
; GFX12-SDAG-TRUE16-NEXT: s_wait_alu depctr_va_vcc(0)
; GFX12-SDAG-TRUE16-NEXT: v_cndmask_b16 v1.l, v3.l, v5.l, vcc_lo
; GFX12-SDAG-TRUE16-NEXT: s_wait_alu depctr_va_sdst(0)
-; GFX12-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX12-SDAG-TRUE16-NEXT: v_cndmask_b16 v0.l, v0.l, v2.l, s0
+; GFX12-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX12-SDAG-TRUE16-NEXT: v_cndmask_b16 v3.l, v5.l, v1.l, s1
-; GFX12-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_2) | instid1(VALU_DEP_4)
; GFX12-SDAG-TRUE16-NEXT: v_cndmask_b16 v2.l, v2.l, v0.l, s2
; GFX12-SDAG-TRUE16-NEXT: v_mov_b16_e32 v4.l, v1.l
; GFX12-SDAG-TRUE16-NEXT: v_mov_b16_e32 v5.l, v0.l
-; GFX12-SDAG-TRUE16-NEXT: v_mov_b16_e32 v6.l, v3.l
; GFX12-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
+; GFX12-SDAG-TRUE16-NEXT: v_mov_b16_e32 v6.l, v3.l
; GFX12-SDAG-TRUE16-NEXT: v_mov_b16_e32 v7.l, v2.l
-; GFX12-SDAG-TRUE16-NEXT: v_lshlrev_b32_e32 v4, 16, v4
; GFX12-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
+; GFX12-SDAG-TRUE16-NEXT: v_lshlrev_b32_e32 v4, 16, v4
; GFX12-SDAG-TRUE16-NEXT: v_lshlrev_b32_e32 v5, 16, v5
+; GFX12-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
; GFX12-SDAG-TRUE16-NEXT: v_lshlrev_b32_e32 v6, 16, v6
-; GFX12-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX12-SDAG-TRUE16-NEXT: v_lshlrev_b32_e32 v7, 16, v7
+; GFX12-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX12-SDAG-TRUE16-NEXT: v_cmp_lt_f32_e32 vcc_lo, v4, v6
-; GFX12-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_3) | instid1(VALU_DEP_2)
; GFX12-SDAG-TRUE16-NEXT: v_cmp_lt_f32_e64 s0, v5, v7
; GFX12-SDAG-TRUE16-NEXT: s_wait_alu depctr_va_vcc(0)
; GFX12-SDAG-TRUE16-NEXT: v_cndmask_b16 v3.l, v3.l, v1.l, vcc_lo
diff --git a/llvm/test/CodeGen/AMDGPU/fmin_legacy.f16.ll b/llvm/test/CodeGen/AMDGPU/fmin_legacy.f16.ll
index c8f4d72abc2949..d31a00190dee4e 100644
--- a/llvm/test/CodeGen/AMDGPU/fmin_legacy.f16.ll
+++ b/llvm/test/CodeGen/AMDGPU/fmin_legacy.f16.ll
@@ -139,7 +139,7 @@ define <2 x half> @test_fmin_legacy_ule_v2f16(<2 x half> %a, <2 x half> %b) #0 {
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v2, 16, v1
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v3, 16, v0
; GFX11-TRUE16-NEXT: v_cmp_ngt_f16_e64 s0, v0.l, v1.l
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX11-TRUE16-NEXT: v_cmp_ngt_f16_e32 vcc_lo, v3.l, v2.l
; GFX11-TRUE16-NEXT: v_cndmask_b16 v0.l, v1.l, v0.l, s0
; GFX11-TRUE16-NEXT: v_cndmask_b16 v0.h, v2.l, v3.l, vcc_lo
@@ -264,11 +264,10 @@ define <3 x half> @test_fmin_legacy_ule_v3f16(<3 x half> %a, <3 x half> %b) #0 {
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v5, 16, v0
; GFX11-TRUE16-NEXT: v_cmp_ngt_f16_e32 vcc_lo, v0.l, v2.l
; GFX11-TRUE16-NEXT: v_cmp_ngt_f16_e64 s1, v1.l, v3.l
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_1) | instid1(VALU_DEP_3)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX11-TRUE16-NEXT: v_cmp_ngt_f16_e64 s0, v5.l, v4.l
; GFX11-TRUE16-NEXT: v_cndmask_b16 v0.l, v2.l, v0.l, vcc_lo
; GFX11-TRUE16-NEXT: v_cndmask_b16 v1.l, v3.l, v1.l, s1
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX11-TRUE16-NEXT: v_cndmask_b16 v0.h, v4.l, v5.l, s0
; GFX11-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -423,11 +422,10 @@ define <4 x half> @test_fmin_legacy_ule_v4f16(<4 x half> %a, <4 x half> %b) #0 {
; GFX11-TRUE16-NEXT: v_cmp_ngt_f16_e32 vcc_lo, v1.l, v3.l
; GFX11-TRUE16-NEXT: v_cmp_ngt_f16_e64 s0, v0.l, v2.l
; GFX11-TRUE16-NEXT: v_cmp_ngt_f16_e64 s1, v5.l, v4.l
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(SKIP_1) | instid1(VALU_DEP_4)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_4)
; GFX11-TRUE16-NEXT: v_cmp_ngt_f16_e64 s2, v7.l, v6.l
; GFX11-TRUE16-NEXT: v_cndmask_b16 v1.l, v3.l, v1.l, vcc_lo
; GFX11-TRUE16-NEXT: v_cndmask_b16 v0.l, v2.l, v0.l, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
; GFX11-TRUE16-NEXT: v_cndmask_b16 v0.h, v4.l, v5.l, s1
; GFX11-TRUE16-NEXT: v_cndmask_b16 v1.h, v6.l, v7.l, s2
; GFX11-TRUE16-NEXT: s_setpc_b64 s[30:31]
diff --git a/llvm/test/CodeGen/AMDGPU/fmul-2-combine-multi-use.ll b/llvm/test/CodeGen/AMDGPU/fmul-2-combine-multi-use.ll
index 3d7a5833123d92..aae7581bfbea4d 100644
--- a/llvm/test/CodeGen/AMDGPU/fmul-2-combine-multi-use.ll
+++ b/llvm/test/CodeGen/AMDGPU/fmul-2-combine-multi-use.ll
@@ -454,13 +454,12 @@ define amdgpu_kernel void @multiple_fadd_use_test_f16(ptr addrspace(1) %out, i16
; GFX11-DENORM-TRUE16-NEXT: v_add_f16_e64 v0.h, s0, -1.0
; GFX11-DENORM-TRUE16-NEXT: v_add_f16_e64 v0.l, s1, -1.0
; GFX11-DENORM-TRUE16-NEXT: s_load_b64 s[0:1], s[4:5], 0x0
-; GFX11-DENORM-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX11-DENORM-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-DENORM-TRUE16-NEXT: v_cmp_gt_f16_e64 s2, |v0.l|, |v0.h|
; GFX11-DENORM-TRUE16-NEXT: v_cndmask_b16 v0.l, v0.h, v0.l, s2
-; GFX11-DENORM-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-DENORM-TRUE16-NEXT: v_add_f16_e64 v0.l, |v0.l|, |v0.l|
+; GFX11-DENORM-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-DENORM-TRUE16-NEXT: v_mul_f16_e32 v0.h, v0.l, v0.l
-; GFX11-DENORM-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-DENORM-TRUE16-NEXT: v_fma_f16 v0.l, -v0.h, v0.l, 1.0
; GFX11-DENORM-TRUE16-NEXT: s_waitcnt lgkmcnt(0)
; GFX11-DENORM-TRUE16-NEXT: global_store_b16 v1, v0, s[0:1]
@@ -494,15 +493,15 @@ define amdgpu_kernel void @multiple_fadd_use_test_f16(ptr addrspace(1) %out, i16
; GFX11-FLUSH-TRUE16-NEXT: s_lshr_b32 s1, s0, 16
; GFX11-FLUSH-TRUE16-NEXT: v_add_f16_e64 v0.h, s0, -1.0
; GFX11-FLUSH-TRUE16-NEXT: v_add_f16_e64 v0.l, s1, -1.0
-; GFX11-FLUSH-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX11-FLUSH-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_2) | instid1(VALU_DEP_1)
; GFX11-FLUSH-TRUE16-NEXT: v_cmp_gt_f16_e64 s0, |v0.l|, |v0.h|
; GFX11-FLUSH-TRUE16-NEXT: v_cndmask_b16 v0.l, v0.h, v0.l, s0
; GFX11-FLUSH-TRUE16-NEXT: s_load_b64 s[0:1], s[4:5], 0x0
-; GFX11-FLUSH-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-FLUSH-TRUE16-NEXT: v_add_f16_e64 v0.l, |v0.l|, |v0.l|
-; GFX11-FLUSH-TRUE16-NEXT: v_mul_f16_e32 v0.h, v0.l, v0.l
; GFX11-FLUSH-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX11-FLUSH-TRUE16-NEXT: v_mul_f16_e32 v0.h, v0.l, v0.l
; GFX11-FLUSH-TRUE16-NEXT: v_mul_f16_e32 v0.l, v0.h, v0.l
+; GFX11-FLUSH-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-FLUSH-TRUE16-NEXT: v_sub_f16_e32 v0.l, 1.0, v0.l
; GFX11-FLUSH-TRUE16-NEXT: s_waitcnt lgkmcnt(0)
; GFX11-FLUSH-TRUE16-NEXT: global_store_b16 v1, v0, s[0:1]
diff --git a/llvm/test/CodeGen/AMDGPU/fneg-combines.f16.ll b/llvm/test/CodeGen/AMDGPU/fneg-combines.f16.ll
index 791b9440c06830..d5cdadbe0d244a 100644
--- a/llvm/test/CodeGen/AMDGPU/fneg-combines.f16.ll
+++ b/llvm/test/CodeGen/AMDGPU/fneg-combines.f16.ll
@@ -670,10 +670,9 @@ define amdgpu_ps half @fneg_fadd_0_nsz_f16(half inreg %tmp2, half inreg %tmp6, <
; GFX11-NEXT: v_rcp_f16_e32 v0, s1
; GFX11-NEXT: s_waitcnt_depctr depctr_va_vdst(0)
; GFX11-NEXT: v_mul_f16_e32 v0, 0x8000, v0
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_cmp_nlt_f16_e64 s1, -v0, s0
; GFX11-NEXT: v_cndmask_b32_e64 v0, v0, s0, s1
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_cmp_nlt_f16_e32 vcc_lo, 0, v0
; GFX11-NEXT: v_cndmask_b32_e64 v0, 0x7e00, 0, vcc_lo
; GFX11-NEXT: ; return to shader part epilog
@@ -4762,11 +4761,11 @@ define half @v_fneg_round_f16(half %a) #0 {
; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_sub_f16_e32 v2, v0, v1
; GFX11-NEXT: v_cmp_ge_f16_e64 s0, |v2|, 0.5
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_cndmask_b32_e64 v2, 0, 0x3c00, s0
-; GFX11-NEXT: v_bfi_b32 v0, 0x7fff, v2, v0
; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX11-NEXT: v_bfi_b32 v0, 0x7fff, v2, v0
; GFX11-NEXT: v_add_f16_e32 v0, v1, v0
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_xor_b32_e32 v0, 0x8000, v0
; GFX11-NEXT: s_setpc_b64 s[30:31]
%round = call half @llvm.round.f16(half %a)
@@ -4810,10 +4809,9 @@ define half @v_fneg_round_f16_nsz(half %a) #0 {
; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_sub_f16_e32 v2, v0, v1
; GFX11-NEXT: v_cmp_ge_f16_e64 s0, |v2|, 0.5
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_cndmask_b32_e64 v2, 0, 0x3c00, s0
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_bfi_b32 v0, 0x7fff, v2, v0
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_sub_f16_e64 v0, -v1, v0
; GFX11-NEXT: s_setpc_b64 s[30:31]
%round = call nsz half @llvm.round.f16(half %a)
@@ -5004,9 +5002,9 @@ define void @v_fneg_copytoreg_f16(ptr addrspace(1) %out, half %a, half %b, half
; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_lshlrev_b32_e32 v6, 1, v6
; GFX11-NEXT: v_add_co_u32 v0, vcc_lo, v0, v6
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-NEXT: v_cmpx_eq_u32_e32 0, v5
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_cbranch_execz .LBB101_2
; GFX11-NEXT: ; %bb.1: ; %if
; GFX11-NEXT: v_mul_f16_e64 v3, -v2, v4
diff --git a/llvm/test/CodeGen/AMDGPU/fneg-modifier-casting.ll b/llvm/test/CodeGen/AMDGPU/fneg-modifier-casting.ll
index 1a3bd8b68f9644..70a1522eea7179 100644
--- a/llvm/test/CodeGen/AMDGPU/fneg-modifier-casting.ll
+++ b/llvm/test/CodeGen/AMDGPU/fneg-modifier-casting.ll
@@ -152,7 +152,6 @@ define <2 x i64> @fneg_xor_select_v2i64(<2 x i1> %cond, <2 x i64> %arg0, <2 x i6
; GFX11-TRUE16-NEXT: v_cmp_eq_u16_e32 vcc_lo, 1, v0.l
; GFX11-TRUE16-NEXT: v_cmp_eq_u16_e64 s0, 1, v0.h
; GFX11-TRUE16-NEXT: v_cndmask_b32_e32 v0, v6, v2, vcc_lo
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX11-TRUE16-NEXT: v_cndmask_b32_e64 v2, v8, v4, s0
; GFX11-TRUE16-NEXT: v_cndmask_b32_e64 v1, -v7, -v3, vcc_lo
; GFX11-TRUE16-NEXT: v_cndmask_b32_e64 v3, -v9, -v5, s0
@@ -167,7 +166,6 @@ define <2 x i64> @fneg_xor_select_v2i64(<2 x i1> %cond, <2 x i64> %arg0, <2 x i6
; GFX11-FAKE16-NEXT: v_dual_cndmask_b32 v0, v6, v2 :: v_dual_and_b32 v1, 1, v1
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e64 s0, 1, v1
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v1, -v7, -v3, vcc_lo
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v2, v8, v4, s0
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v3, -v9, -v5, s0
; GFX11-FAKE16-NEXT: s_setpc_b64 s[30:31]
@@ -250,10 +248,9 @@ define <2 x i16> @fneg_xor_select_v2i16(<2 x i1> %cond, <2 x i16> %arg0, <2 x i1
; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
; GFX11-TRUE16-NEXT: v_cmp_eq_u16_e32 vcc_lo, 1, v0.h
; GFX11-TRUE16-NEXT: v_cmp_eq_u16_e64 s0, 1, v0.l
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_cndmask_b16 v0.h, v4.l, v1.l, vcc_lo
; GFX11-TRUE16-NEXT: v_cndmask_b16 v0.l, v3.l, v2.l, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_xor_b32_e32 v0, 0x80008000, v0
; GFX11-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -773,19 +770,18 @@ define <2 x half> @select_fneg_select_v2f16(<2 x i1> %cond0, <2 x i1> %cond1, <2
; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v6, 16, v4
; GFX11-TRUE16-NEXT: v_cmp_eq_u16_e32 vcc_lo, 1, v0.h
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(SKIP_3) | instid1(VALU_DEP_4)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(SKIP_4) | instid1(VALU_DEP_4)
; GFX11-TRUE16-NEXT: v_cmp_eq_u16_e64 s0, 1, v0.l
; GFX11-TRUE16-NEXT: v_and_b16 v0.l, 1, v3.l
; GFX11-TRUE16-NEXT: v_and_b16 v0.h, 1, v2.l
; GFX11-TRUE16-NEXT: v_cndmask_b16 v1.h, v6.l, v1.l, vcc_lo
; GFX11-TRUE16-NEXT: v_cndmask_b16 v1.l, v4.l, v5.l, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
; GFX11-TRUE16-NEXT: v_cmp_eq_u16_e32 vcc_lo, 1, v0.l
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_3)
; GFX11-TRUE16-NEXT: v_cmp_eq_u16_e64 s0, 1, v0.h
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_xor_b32_e32 v3, 0x80008000, v1
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v2, 16, v3
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-TRUE16-NEXT: v_cndmask_b16 v0.l, v1.l, v3.l, s0
; GFX11-TRUE16-NEXT: v_cndmask_b16 v0.h, v1.h, v2.l, vcc_lo
; GFX11-TRUE16-NEXT: s_setpc_b64 s[30:31]
@@ -884,19 +880,18 @@ define <2 x i16> @select_fneg_xor_select_v2i16(<2 x i1> %cond0, <2 x i1> %cond1,
; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v6, 16, v4
; GFX11-TRUE16-NEXT: v_cmp_eq_u16_e32 vcc_lo, 1, v0.h
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(SKIP_3) | instid1(VALU_DEP_4)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(SKIP_4) | instid1(VALU_DEP_4)
; GFX11-TRUE16-NEXT: v_cmp_eq_u16_e64 s0, 1, v0.l
; GFX11-TRUE16-NEXT: v_and_b16 v0.l, 1, v3.l
; GFX11-TRUE16-NEXT: v_and_b16 v0.h, 1, v2.l
; GFX11-TRUE16-NEXT: v_cndmask_b16 v1.h, v6.l, v1.l, vcc_lo
; GFX11-TRUE16-NEXT: v_cndmask_b16 v1.l, v4.l, v5.l, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
; GFX11-TRUE16-NEXT: v_cmp_eq_u16_e32 vcc_lo, 1, v0.l
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_3)
; GFX11-TRUE16-NEXT: v_cmp_eq_u16_e64 s0, 1, v0.h
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_xor_b32_e32 v3, 0x80008000, v1
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v2, 16, v3
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-TRUE16-NEXT: v_cndmask_b16 v0.l, v1.l, v3.l, s0
; GFX11-TRUE16-NEXT: v_cndmask_b16 v0.h, v1.h, v2.l, vcc_lo
; GFX11-TRUE16-NEXT: s_setpc_b64 s[30:31]
@@ -1053,12 +1048,12 @@ define float @cospiD_pattern0_half(i16 %arg, float %arg1, float %arg2) {
; GFX11-TRUE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-TRUE16-NEXT: v_and_b16 v0.h, v0.l, 1
; GFX11-TRUE16-NEXT: v_cmp_lt_i16_e32 vcc_lo, 1, v0.l
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_2) | instid1(VALU_DEP_2)
; GFX11-TRUE16-NEXT: v_cmp_eq_u16_e64 s0, 0, v0.h
; GFX11-TRUE16-NEXT: v_cndmask_b32_e64 v0, v1, v2, s0
; GFX11-TRUE16-NEXT: v_cndmask_b16 v1.l, 0, 0x8000, vcc_lo
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v2, 16, v0
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_xor_b16 v0.h, v2.l, v1.l
; GFX11-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -1667,12 +1662,14 @@ define amdgpu_kernel void @multiple_uses_fneg_select_f64(double %x, double %y, i
; GFX11-NEXT: s_waitcnt lgkmcnt(0)
; GFX11-NEXT: v_mov_b32_e32 v0, s1
; GFX11-NEXT: s_bitcmp1_b32 s6, 0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cselect_b32 vcc_lo, -1, 0
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_3) | instid1(VALU_DEP_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: v_cndmask_b32_e32 v0, s3, v0, vcc_lo
; GFX11-NEXT: s_and_b32 s6, vcc_lo, exec_lo
; GFX11-NEXT: s_cselect_b32 s1, s1, s3
; GFX11-NEXT: s_cselect_b32 s0, s0, s2
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_cndmask_b32_e64 v1, s1, -v0, vcc_lo
; GFX11-NEXT: v_mov_b32_e32 v0, s0
; GFX11-NEXT: global_store_b64 v2, v[0:1], s[4:5]
diff --git a/llvm/test/CodeGen/AMDGPU/fold-gep-offset.ll b/llvm/test/CodeGen/AMDGPU/fold-gep-offset.ll
index 56415ef44c527e..d47ee3c8c9156d 100644
--- a/llvm/test/CodeGen/AMDGPU/fold-gep-offset.ll
+++ b/llvm/test/CodeGen/AMDGPU/fold-gep-offset.ll
@@ -70,10 +70,10 @@ define i32 @flat_offset_maybe_oob(ptr %p, i32 %i) {
; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_lshlrev_b64 v[2:3], 2, v[2:3]
; GFX11-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, v1, v3, vcc_lo
; GFX11-NEXT: v_add_co_u32 v0, vcc_lo, v0, 12
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-NEXT: flat_load_b32 v0, v[0:1]
; GFX11-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
@@ -262,7 +262,7 @@ define i32 @flat_offset_inbounds(ptr %p, i32 %i) {
; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_lshlrev_b64 v[2:3], 2, v[2:3]
; GFX11-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, v1, v3, vcc_lo
; GFX11-NEXT: flat_load_b32 v0, v[0:1] offset:12
; GFX11-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
@@ -349,10 +349,10 @@ define void @flat_offset_inbounds_wide(ptr %p, ptr %pout, i32 %i) {
; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-SDAG-NEXT: v_lshlrev_b64 v[4:5], 2, v[4:5]
; GFX11-SDAG-NEXT: v_add_co_u32 v0, vcc_lo, v0, v4
-; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-SDAG-NEXT: v_add_co_ci_u32_e64 v1, null, v1, v5, vcc_lo
; GFX11-SDAG-NEXT: v_add_co_u32 v4, vcc_lo, v0, 28
-; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX11-SDAG-NEXT: v_add_co_ci_u32_e64 v5, null, 0, v1, vcc_lo
; GFX11-SDAG-NEXT: s_clause 0x1
; GFX11-SDAG-NEXT: flat_load_b32 v8, v[4:5]
@@ -444,7 +444,7 @@ define void @flat_offset_inbounds_wide(ptr %p, ptr %pout, i32 %i) {
; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-GISEL-NEXT: v_lshlrev_b64 v[4:5], 2, v[4:5]
; GFX11-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, v0, v4
-; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX11-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, v1, v5, vcc_lo
; GFX11-GISEL-NEXT: s_clause 0x1
; GFX11-GISEL-NEXT: flat_load_b128 v[4:7], v[0:1] offset:12
@@ -669,10 +669,10 @@ define void @flat_offset_inbounds_very_wide(ptr %p, ptr %pout, i32 %i) {
; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-SDAG-NEXT: v_lshlrev_b64 v[4:5], 2, v[4:5]
; GFX11-SDAG-NEXT: v_add_co_u32 v0, vcc_lo, v0, v4
-; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-SDAG-NEXT: v_add_co_ci_u32_e64 v1, null, v1, v5, vcc_lo
; GFX11-SDAG-NEXT: v_add_co_u32 v36, vcc_lo, v0, 28
-; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX11-SDAG-NEXT: v_add_co_ci_u32_e64 v37, null, 0, v1, vcc_lo
; GFX11-SDAG-NEXT: s_clause 0x7
; GFX11-SDAG-NEXT: flat_load_b128 v[4:7], v[36:37] offset:80
@@ -686,7 +686,6 @@ define void @flat_offset_inbounds_very_wide(ptr %p, ptr %pout, i32 %i) {
; GFX11-SDAG-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
; GFX11-SDAG-NEXT: flat_load_b128 v[35:38], v[36:37] offset:48
; GFX11-SDAG-NEXT: v_add_co_u32 v0, vcc_lo, v2, 48
-; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-SDAG-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v3, vcc_lo
; GFX11-SDAG-NEXT: v_add_co_u32 v48, vcc_lo, 0x88, v2
; GFX11-SDAG-NEXT: v_add_co_ci_u32_e64 v49, null, 0, v3, vcc_lo
@@ -858,7 +857,7 @@ define void @flat_offset_inbounds_very_wide(ptr %p, ptr %pout, i32 %i) {
; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-GISEL-NEXT: v_lshlrev_b64 v[4:5], 2, v[4:5]
; GFX11-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, v0, v4
-; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX11-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, v1, v5, vcc_lo
; GFX11-GISEL-NEXT: s_clause 0x8
; GFX11-GISEL-NEXT: flat_load_b128 v[4:7], v[0:1] offset:12
diff --git a/llvm/test/CodeGen/AMDGPU/fold-int-pow2-with-fmul-or-fdiv.ll b/llvm/test/CodeGen/AMDGPU/fold-int-pow2-with-fmul-or-fdiv.ll
index 0507e926ec6fe6..189464482e6dc7 100644
--- a/llvm/test/CodeGen/AMDGPU/fold-int-pow2-with-fmul-or-fdiv.ll
+++ b/llvm/test/CodeGen/AMDGPU/fold-int-pow2-with-fmul-or-fdiv.ll
@@ -1248,7 +1248,7 @@ define <2 x double> @fdiv_pow_shl_cnt_vec(<2 x i64> %cnt) nounwind {
; GFX11-NEXT: v_lshlrev_b32_e32 v1, 20, v0
; GFX11-NEXT: v_lshlrev_b32_e32 v3, 20, v2
; GFX11-NEXT: v_sub_co_u32 v0, vcc_lo, 0, 0
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_1) | instid1(VALU_DEP_4)
; GFX11-NEXT: v_sub_co_ci_u32_e64 v1, null, 0x3ff00000, v1, vcc_lo
; GFX11-NEXT: v_sub_co_u32 v2, vcc_lo, 0, 0
; GFX11-NEXT: v_sub_co_ci_u32_e64 v3, null, 0x3ff00000, v3, vcc_lo
@@ -1812,7 +1812,7 @@ define double @fdiv_pow_shl_cnt32_to_dbl_okay(i32 %cnt) nounwind {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: v_lshlrev_b32_e32 v0, 20, v0
; GFX11-NEXT: v_sub_co_u32 v1, vcc_lo, 0, 0
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX11-NEXT: v_sub_co_ci_u32_e64 v1, null, 0x36a00000, v0, vcc_lo
; GFX11-NEXT: v_mov_b32_e32 v0, 0
; GFX11-NEXT: s_setpc_b64 s[30:31]
diff --git a/llvm/test/CodeGen/AMDGPU/fp_to_sint.ll b/llvm/test/CodeGen/AMDGPU/fp_to_sint.ll
index 2618c670b8c23a..23477a917b82dd 100644
--- a/llvm/test/CodeGen/AMDGPU/fp_to_sint.ll
+++ b/llvm/test/CodeGen/AMDGPU/fp_to_sint.ll
@@ -345,7 +345,7 @@ define amdgpu_kernel void @fp_to_sint_i64 (ptr addrspace(1) %out, float %in) {
; GFX11-SDAG-NEXT: v_xor_b32_e32 v1, v1, v3
; GFX11-SDAG-NEXT: v_mov_b32_e32 v2, 0
; GFX11-SDAG-NEXT: v_xor_b32_e32 v0, v0, v3
-; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_4)
; GFX11-SDAG-NEXT: v_sub_co_u32 v0, vcc_lo, v0, v3
; GFX11-SDAG-NEXT: v_sub_co_ci_u32_e64 v1, null, v1, v3, vcc_lo
; GFX11-SDAG-NEXT: s_waitcnt lgkmcnt(0)
@@ -532,11 +532,11 @@ define amdgpu_kernel void @fp_to_sint_v2i64(ptr addrspace(1) %out, <2 x float> %
; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
; GFX11-SDAG-NEXT: v_xor_b32_e32 v5, v5, v1
; GFX11-SDAG-NEXT: v_xor_b32_e32 v8, v3, v1
-; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_1) | instid1(VALU_DEP_4)
; GFX11-SDAG-NEXT: v_sub_co_u32 v2, vcc_lo, v4, v0
; GFX11-SDAG-NEXT: v_sub_co_ci_u32_e64 v3, null, v7, v0, vcc_lo
-; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-SDAG-NEXT: v_sub_co_u32 v0, vcc_lo, v5, v1
+; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_4)
; GFX11-SDAG-NEXT: v_sub_co_ci_u32_e64 v1, null, v8, v1, vcc_lo
; GFX11-SDAG-NEXT: global_store_b128 v6, v[0:3], s[0:1]
; GFX11-SDAG-NEXT: s_endpgm
@@ -835,12 +835,10 @@ define amdgpu_kernel void @fp_to_sint_v4i64(ptr addrspace(1) %out, <4 x float> %
; GFX11-SDAG-NEXT: v_xor_b32_e32 v1, v1, v9
; GFX11-SDAG-NEXT: v_sub_co_ci_u32_e64 v3, null, v11, v5, vcc_lo
; GFX11-SDAG-NEXT: v_sub_co_u32 v6, vcc_lo, v6, v10
-; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-SDAG-NEXT: v_sub_co_ci_u32_e64 v7, null, v4, v10, vcc_lo
; GFX11-SDAG-NEXT: v_sub_co_u32 v4, vcc_lo, v15, v12
; GFX11-SDAG-NEXT: v_sub_co_ci_u32_e64 v5, null, v14, v12, vcc_lo
; GFX11-SDAG-NEXT: v_sub_co_u32 v0, vcc_lo, v1, v9
-; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-SDAG-NEXT: v_sub_co_ci_u32_e64 v1, null, v13, v9, vcc_lo
; GFX11-SDAG-NEXT: s_clause 0x1
; GFX11-SDAG-NEXT: global_store_b128 v8, v[4:7], s[4:5] offset:16
diff --git a/llvm/test/CodeGen/AMDGPU/fptrunc.f16.ll b/llvm/test/CodeGen/AMDGPU/fptrunc.f16.ll
index 36b3ea74920781..0e6e0b5d6e2023 100644
--- a/llvm/test/CodeGen/AMDGPU/fptrunc.f16.ll
+++ b/llvm/test/CodeGen/AMDGPU/fptrunc.f16.ll
@@ -216,6 +216,7 @@ define amdgpu_kernel void @fptrunc_f32_to_f16(
; GFX1250-SDAG-TRUE16-NEXT: s_mov_b32 s5, s1
; GFX1250-SDAG-TRUE16-NEXT: s_wait_loadcnt 0x0
; GFX1250-SDAG-TRUE16-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 2, 2), 0 ; msbs: dst=0 src0=0 src1=0 src2=0
+; GFX1250-SDAG-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1250-SDAG-TRUE16-NEXT: v_cvt_f16_f32_e32 v0.l, v0
; GFX1250-SDAG-TRUE16-NEXT: buffer_store_b16 v0, off, s[4:7], null
; GFX1250-SDAG-TRUE16-NEXT: s_endpgm
@@ -239,6 +240,7 @@ define amdgpu_kernel void @fptrunc_f32_to_f16(
; GFX1250-SDAG-FAKE16-NEXT: s_mov_b32 s5, s1
; GFX1250-SDAG-FAKE16-NEXT: s_wait_loadcnt 0x0
; GFX1250-SDAG-FAKE16-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 2, 2), 0 ; msbs: dst=0 src0=0 src1=0 src2=0
+; GFX1250-SDAG-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1250-SDAG-FAKE16-NEXT: v_cvt_f16_f32_e32 v0, v0
; GFX1250-SDAG-FAKE16-NEXT: buffer_store_b16 v0, off, s[4:7], null
; GFX1250-SDAG-FAKE16-NEXT: s_endpgm
@@ -256,8 +258,8 @@ define amdgpu_kernel void @fptrunc_f32_to_f16(
; GFX1250-GISEL-TRUE16-NEXT: s_mov_b32 s3, 0x31016000
; GFX1250-GISEL-TRUE16-NEXT: s_wait_kmcnt 0x0
; GFX1250-GISEL-TRUE16-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 2, 2), 0 ; msbs: dst=0 src0=0 src1=0 src2=0
+; GFX1250-GISEL-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_3)
; GFX1250-GISEL-TRUE16-NEXT: s_cvt_f16_f32 s2, s2
-; GFX1250-GISEL-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_3)
; GFX1250-GISEL-TRUE16-NEXT: v_mov_b16_e32 v0.l, s2
; GFX1250-GISEL-TRUE16-NEXT: s_mov_b32 s2, -1
; GFX1250-GISEL-TRUE16-NEXT: buffer_store_b16 v0, off, s[0:3], null
@@ -276,8 +278,8 @@ define amdgpu_kernel void @fptrunc_f32_to_f16(
; GFX1250-GISEL-FAKE16-NEXT: s_mov_b32 s3, 0x31016000
; GFX1250-GISEL-FAKE16-NEXT: s_wait_kmcnt 0x0
; GFX1250-GISEL-FAKE16-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 2, 2), 0 ; msbs: dst=0 src0=0 src1=0 src2=0
+; GFX1250-GISEL-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_3)
; GFX1250-GISEL-FAKE16-NEXT: s_cvt_f16_f32 s2, s2
-; GFX1250-GISEL-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_3)
; GFX1250-GISEL-FAKE16-NEXT: v_mov_b32_e32 v0, s2
; GFX1250-GISEL-FAKE16-NEXT: s_mov_b32 s2, -1
; GFX1250-GISEL-FAKE16-NEXT: buffer_store_b16 v0, off, s[0:3], null
@@ -491,6 +493,7 @@ define amdgpu_kernel void @fptrunc_f32_to_f16_afn(ptr addrspace(1) %r,
; GFX1250-SDAG-TRUE16-NEXT: s_mov_b32 s5, s1
; GFX1250-SDAG-TRUE16-NEXT: s_wait_loadcnt 0x0
; GFX1250-SDAG-TRUE16-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 2, 2), 0 ; msbs: dst=0 src0=0 src1=0 src2=0
+; GFX1250-SDAG-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1250-SDAG-TRUE16-NEXT: v_cvt_f16_f32_e32 v0.l, v0
; GFX1250-SDAG-TRUE16-NEXT: buffer_store_b16 v0, off, s[4:7], null
; GFX1250-SDAG-TRUE16-NEXT: s_endpgm
@@ -514,6 +517,7 @@ define amdgpu_kernel void @fptrunc_f32_to_f16_afn(ptr addrspace(1) %r,
; GFX1250-SDAG-FAKE16-NEXT: s_mov_b32 s5, s1
; GFX1250-SDAG-FAKE16-NEXT: s_wait_loadcnt 0x0
; GFX1250-SDAG-FAKE16-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 2, 2), 0 ; msbs: dst=0 src0=0 src1=0 src2=0
+; GFX1250-SDAG-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1250-SDAG-FAKE16-NEXT: v_cvt_f16_f32_e32 v0, v0
; GFX1250-SDAG-FAKE16-NEXT: buffer_store_b16 v0, off, s[4:7], null
; GFX1250-SDAG-FAKE16-NEXT: s_endpgm
@@ -531,8 +535,8 @@ define amdgpu_kernel void @fptrunc_f32_to_f16_afn(ptr addrspace(1) %r,
; GFX1250-GISEL-TRUE16-NEXT: s_mov_b32 s3, 0x31016000
; GFX1250-GISEL-TRUE16-NEXT: s_wait_kmcnt 0x0
; GFX1250-GISEL-TRUE16-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 2, 2), 0 ; msbs: dst=0 src0=0 src1=0 src2=0
+; GFX1250-GISEL-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_3)
; GFX1250-GISEL-TRUE16-NEXT: s_cvt_f16_f32 s2, s2
-; GFX1250-GISEL-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_3)
; GFX1250-GISEL-TRUE16-NEXT: v_mov_b16_e32 v0.l, s2
; GFX1250-GISEL-TRUE16-NEXT: s_mov_b32 s2, -1
; GFX1250-GISEL-TRUE16-NEXT: buffer_store_b16 v0, off, s[0:3], null
@@ -551,8 +555,8 @@ define amdgpu_kernel void @fptrunc_f32_to_f16_afn(ptr addrspace(1) %r,
; GFX1250-GISEL-FAKE16-NEXT: s_mov_b32 s3, 0x31016000
; GFX1250-GISEL-FAKE16-NEXT: s_wait_kmcnt 0x0
; GFX1250-GISEL-FAKE16-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 2, 2), 0 ; msbs: dst=0 src0=0 src1=0 src2=0
+; GFX1250-GISEL-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_3)
; GFX1250-GISEL-FAKE16-NEXT: s_cvt_f16_f32 s2, s2
-; GFX1250-GISEL-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_3)
; GFX1250-GISEL-FAKE16-NEXT: v_mov_b32_e32 v0, s2
; GFX1250-GISEL-FAKE16-NEXT: s_mov_b32 s2, -1
; GFX1250-GISEL-FAKE16-NEXT: buffer_store_b16 v0, off, s[0:3], null
@@ -1045,24 +1049,26 @@ define amdgpu_kernel void @fptrunc_f64_to_f16(
; GFX11-SDAG-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-SDAG-TRUE16-NEXT: s_lshr_b32 s9, s5, s8
; GFX11-SDAG-TRUE16-NEXT: s_lshl_b32 s8, s9, s8
-; GFX11-SDAG-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_4) | instid1(SALU_CYCLE_1)
+; GFX11-SDAG-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-SDAG-TRUE16-NEXT: s_cmp_lg_u32 s8, s5
; GFX11-SDAG-TRUE16-NEXT: s_cselect_b32 s5, 1, 0
; GFX11-SDAG-TRUE16-NEXT: s_addk_i32 s4, 0xfc10
; GFX11-SDAG-TRUE16-NEXT: s_or_b32 s5, s9, s5
; GFX11-SDAG-TRUE16-NEXT: s_lshl_b32 s8, s4, 12
+; GFX11-SDAG-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX11-SDAG-TRUE16-NEXT: s_or_b32 s8, s3, s8
; GFX11-SDAG-TRUE16-NEXT: s_cmp_lt_i32 s4, 1
; GFX11-SDAG-TRUE16-NEXT: s_cselect_b32 s5, s5, s8
; GFX11-SDAG-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-SDAG-TRUE16-NEXT: s_and_b32 s8, s5, 7
; GFX11-SDAG-TRUE16-NEXT: s_cmp_gt_i32 s8, 5
+; GFX11-SDAG-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX11-SDAG-TRUE16-NEXT: s_cselect_b32 s9, 1, 0
; GFX11-SDAG-TRUE16-NEXT: s_cmp_eq_u32 s8, 3
; GFX11-SDAG-TRUE16-NEXT: s_cselect_b32 s8, 1, 0
; GFX11-SDAG-TRUE16-NEXT: s_lshr_b32 s5, s5, 2
; GFX11-SDAG-TRUE16-NEXT: s_or_b32 s8, s8, s9
-; GFX11-SDAG-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-SDAG-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_4) | instid1(SALU_CYCLE_1)
; GFX11-SDAG-TRUE16-NEXT: s_add_i32 s5, s5, s8
; GFX11-SDAG-TRUE16-NEXT: s_cmp_lt_i32 s4, 31
; GFX11-SDAG-TRUE16-NEXT: s_movk_i32 s8, 0x7e00
@@ -1112,24 +1118,26 @@ define amdgpu_kernel void @fptrunc_f64_to_f16(
; GFX11-SDAG-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-SDAG-FAKE16-NEXT: s_lshr_b32 s9, s5, s8
; GFX11-SDAG-FAKE16-NEXT: s_lshl_b32 s8, s9, s8
-; GFX11-SDAG-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_4) | instid1(SALU_CYCLE_1)
+; GFX11-SDAG-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-SDAG-FAKE16-NEXT: s_cmp_lg_u32 s8, s5
; GFX11-SDAG-FAKE16-NEXT: s_cselect_b32 s5, 1, 0
; GFX11-SDAG-FAKE16-NEXT: s_addk_i32 s4, 0xfc10
; GFX11-SDAG-FAKE16-NEXT: s_or_b32 s5, s9, s5
; GFX11-SDAG-FAKE16-NEXT: s_lshl_b32 s8, s4, 12
+; GFX11-SDAG-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX11-SDAG-FAKE16-NEXT: s_or_b32 s8, s3, s8
; GFX11-SDAG-FAKE16-NEXT: s_cmp_lt_i32 s4, 1
; GFX11-SDAG-FAKE16-NEXT: s_cselect_b32 s5, s5, s8
; GFX11-SDAG-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-SDAG-FAKE16-NEXT: s_and_b32 s8, s5, 7
; GFX11-SDAG-FAKE16-NEXT: s_cmp_gt_i32 s8, 5
+; GFX11-SDAG-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX11-SDAG-FAKE16-NEXT: s_cselect_b32 s9, 1, 0
; GFX11-SDAG-FAKE16-NEXT: s_cmp_eq_u32 s8, 3
; GFX11-SDAG-FAKE16-NEXT: s_cselect_b32 s8, 1, 0
; GFX11-SDAG-FAKE16-NEXT: s_lshr_b32 s5, s5, 2
; GFX11-SDAG-FAKE16-NEXT: s_or_b32 s8, s8, s9
-; GFX11-SDAG-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-SDAG-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_4) | instid1(SALU_CYCLE_1)
; GFX11-SDAG-FAKE16-NEXT: s_add_i32 s5, s5, s8
; GFX11-SDAG-FAKE16-NEXT: s_cmp_lt_i32 s4, 31
; GFX11-SDAG-FAKE16-NEXT: s_movk_i32 s8, 0x7e00
@@ -1175,24 +1183,27 @@ define amdgpu_kernel void @fptrunc_f64_to_f16(
; GFX11-GISEL-TRUE16-NEXT: s_lshl_b32 s6, s9, s6
; GFX11-GISEL-TRUE16-NEXT: s_or_b32 s5, s5, 0x7c00
; GFX11-GISEL-TRUE16-NEXT: s_cmp_lg_u32 s6, s8
+; GFX11-GISEL-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-GISEL-TRUE16-NEXT: s_cselect_b32 s6, 1, 0
-; GFX11-GISEL-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
; GFX11-GISEL-TRUE16-NEXT: s_or_b32 s6, s9, s6
; GFX11-GISEL-TRUE16-NEXT: s_cmp_lt_i32 s4, 1
+; GFX11-GISEL-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-GISEL-TRUE16-NEXT: s_cselect_b32 s2, s6, s2
; GFX11-GISEL-TRUE16-NEXT: s_and_b32 s6, s2, 7
; GFX11-GISEL-TRUE16-NEXT: s_lshr_b32 s2, s2, 2
; GFX11-GISEL-TRUE16-NEXT: s_cmp_eq_u32 s6, 3
+; GFX11-GISEL-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX11-GISEL-TRUE16-NEXT: s_cselect_b32 s7, 1, 0
; GFX11-GISEL-TRUE16-NEXT: s_cmp_gt_i32 s6, 5
; GFX11-GISEL-TRUE16-NEXT: s_cselect_b32 s6, 1, 0
; GFX11-GISEL-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-GISEL-TRUE16-NEXT: s_or_b32 s6, s7, s6
; GFX11-GISEL-TRUE16-NEXT: s_cmp_lg_u32 s6, 0
+; GFX11-GISEL-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-GISEL-TRUE16-NEXT: s_cselect_b32 s6, 1, 0
-; GFX11-GISEL-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-GISEL-TRUE16-NEXT: s_add_i32 s2, s2, s6
; GFX11-GISEL-TRUE16-NEXT: s_cmp_gt_i32 s4, 30
+; GFX11-GISEL-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX11-GISEL-TRUE16-NEXT: s_cselect_b32 s2, 0x7c00, s2
; GFX11-GISEL-TRUE16-NEXT: s_cmpk_eq_i32 s4, 0x40f
; GFX11-GISEL-TRUE16-NEXT: s_cselect_b32 s2, s5, s2
@@ -1233,24 +1244,27 @@ define amdgpu_kernel void @fptrunc_f64_to_f16(
; GFX11-GISEL-FAKE16-NEXT: s_lshl_b32 s6, s9, s6
; GFX11-GISEL-FAKE16-NEXT: s_or_b32 s5, s5, 0x7c00
; GFX11-GISEL-FAKE16-NEXT: s_cmp_lg_u32 s6, s8
+; GFX11-GISEL-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-GISEL-FAKE16-NEXT: s_cselect_b32 s6, 1, 0
-; GFX11-GISEL-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
; GFX11-GISEL-FAKE16-NEXT: s_or_b32 s6, s9, s6
; GFX11-GISEL-FAKE16-NEXT: s_cmp_lt_i32 s4, 1
+; GFX11-GISEL-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-GISEL-FAKE16-NEXT: s_cselect_b32 s2, s6, s2
; GFX11-GISEL-FAKE16-NEXT: s_and_b32 s6, s2, 7
; GFX11-GISEL-FAKE16-NEXT: s_lshr_b32 s2, s2, 2
; GFX11-GISEL-FAKE16-NEXT: s_cmp_eq_u32 s6, 3
+; GFX11-GISEL-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX11-GISEL-FAKE16-NEXT: s_cselect_b32 s7, 1, 0
; GFX11-GISEL-FAKE16-NEXT: s_cmp_gt_i32 s6, 5
; GFX11-GISEL-FAKE16-NEXT: s_cselect_b32 s6, 1, 0
; GFX11-GISEL-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-GISEL-FAKE16-NEXT: s_or_b32 s6, s7, s6
; GFX11-GISEL-FAKE16-NEXT: s_cmp_lg_u32 s6, 0
+; GFX11-GISEL-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-GISEL-FAKE16-NEXT: s_cselect_b32 s6, 1, 0
-; GFX11-GISEL-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-GISEL-FAKE16-NEXT: s_add_i32 s2, s2, s6
; GFX11-GISEL-FAKE16-NEXT: s_cmp_gt_i32 s4, 30
+; GFX11-GISEL-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX11-GISEL-FAKE16-NEXT: s_cselect_b32 s2, 0x7c00, s2
; GFX11-GISEL-FAKE16-NEXT: s_cmpk_eq_i32 s4, 0x40f
; GFX11-GISEL-FAKE16-NEXT: s_cselect_b32 s2, s5, s2
@@ -1299,24 +1313,26 @@ define amdgpu_kernel void @fptrunc_f64_to_f16(
; GFX1250-SDAG-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1250-SDAG-TRUE16-NEXT: s_lshr_b32 s9, s5, s8
; GFX1250-SDAG-TRUE16-NEXT: s_lshl_b32 s8, s9, s8
-; GFX1250-SDAG-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_4) | instid1(SALU_CYCLE_1)
+; GFX1250-SDAG-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1250-SDAG-TRUE16-NEXT: s_cmp_lg_u32 s8, s5
; GFX1250-SDAG-TRUE16-NEXT: s_cselect_b32 s5, 1, 0
; GFX1250-SDAG-TRUE16-NEXT: s_addk_co_i32 s4, 0xfc10
; GFX1250-SDAG-TRUE16-NEXT: s_or_b32 s5, s9, s5
; GFX1250-SDAG-TRUE16-NEXT: s_lshl_b32 s8, s4, 12
+; GFX1250-SDAG-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX1250-SDAG-TRUE16-NEXT: s_or_b32 s8, s3, s8
; GFX1250-SDAG-TRUE16-NEXT: s_cmp_lt_i32 s4, 1
; GFX1250-SDAG-TRUE16-NEXT: s_cselect_b32 s5, s5, s8
; GFX1250-SDAG-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1250-SDAG-TRUE16-NEXT: s_and_b32 s8, s5, 7
; GFX1250-SDAG-TRUE16-NEXT: s_cmp_gt_i32 s8, 5
+; GFX1250-SDAG-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX1250-SDAG-TRUE16-NEXT: s_cselect_b32 s9, 1, 0
; GFX1250-SDAG-TRUE16-NEXT: s_cmp_eq_u32 s8, 3
; GFX1250-SDAG-TRUE16-NEXT: s_cselect_b32 s8, 1, 0
; GFX1250-SDAG-TRUE16-NEXT: s_lshr_b32 s5, s5, 2
; GFX1250-SDAG-TRUE16-NEXT: s_or_b32 s8, s8, s9
-; GFX1250-SDAG-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1250-SDAG-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_4) | instid1(SALU_CYCLE_1)
; GFX1250-SDAG-TRUE16-NEXT: s_add_co_i32 s5, s5, s8
; GFX1250-SDAG-TRUE16-NEXT: s_cmp_lt_i32 s4, 31
; GFX1250-SDAG-TRUE16-NEXT: s_movk_i32 s8, 0x7e00
@@ -1370,24 +1386,26 @@ define amdgpu_kernel void @fptrunc_f64_to_f16(
; GFX1250-SDAG-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1250-SDAG-FAKE16-NEXT: s_lshr_b32 s9, s5, s8
; GFX1250-SDAG-FAKE16-NEXT: s_lshl_b32 s8, s9, s8
-; GFX1250-SDAG-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_4) | instid1(SALU_CYCLE_1)
+; GFX1250-SDAG-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1250-SDAG-FAKE16-NEXT: s_cmp_lg_u32 s8, s5
; GFX1250-SDAG-FAKE16-NEXT: s_cselect_b32 s5, 1, 0
; GFX1250-SDAG-FAKE16-NEXT: s_addk_co_i32 s4, 0xfc10
; GFX1250-SDAG-FAKE16-NEXT: s_or_b32 s5, s9, s5
; GFX1250-SDAG-FAKE16-NEXT: s_lshl_b32 s8, s4, 12
+; GFX1250-SDAG-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX1250-SDAG-FAKE16-NEXT: s_or_b32 s8, s3, s8
; GFX1250-SDAG-FAKE16-NEXT: s_cmp_lt_i32 s4, 1
; GFX1250-SDAG-FAKE16-NEXT: s_cselect_b32 s5, s5, s8
; GFX1250-SDAG-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1250-SDAG-FAKE16-NEXT: s_and_b32 s8, s5, 7
; GFX1250-SDAG-FAKE16-NEXT: s_cmp_gt_i32 s8, 5
+; GFX1250-SDAG-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX1250-SDAG-FAKE16-NEXT: s_cselect_b32 s9, 1, 0
; GFX1250-SDAG-FAKE16-NEXT: s_cmp_eq_u32 s8, 3
; GFX1250-SDAG-FAKE16-NEXT: s_cselect_b32 s8, 1, 0
; GFX1250-SDAG-FAKE16-NEXT: s_lshr_b32 s5, s5, 2
; GFX1250-SDAG-FAKE16-NEXT: s_or_b32 s8, s8, s9
-; GFX1250-SDAG-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1250-SDAG-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_4) | instid1(SALU_CYCLE_1)
; GFX1250-SDAG-FAKE16-NEXT: s_add_co_i32 s5, s5, s8
; GFX1250-SDAG-FAKE16-NEXT: s_cmp_lt_i32 s4, 31
; GFX1250-SDAG-FAKE16-NEXT: s_movk_i32 s8, 0x7e00
@@ -1437,24 +1455,27 @@ define amdgpu_kernel void @fptrunc_f64_to_f16(
; GFX1250-GISEL-TRUE16-NEXT: s_lshl_b32 s6, s9, s6
; GFX1250-GISEL-TRUE16-NEXT: s_or_b32 s5, s5, 0x7c00
; GFX1250-GISEL-TRUE16-NEXT: s_cmp_lg_u32 s6, s8
+; GFX1250-GISEL-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1250-GISEL-TRUE16-NEXT: s_cselect_b32 s6, 1, 0
-; GFX1250-GISEL-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
; GFX1250-GISEL-TRUE16-NEXT: s_or_b32 s6, s9, s6
; GFX1250-GISEL-TRUE16-NEXT: s_cmp_lt_i32 s4, 1
+; GFX1250-GISEL-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1250-GISEL-TRUE16-NEXT: s_cselect_b32 s2, s6, s2
; GFX1250-GISEL-TRUE16-NEXT: s_and_b32 s6, s2, 7
; GFX1250-GISEL-TRUE16-NEXT: s_lshr_b32 s2, s2, 2
; GFX1250-GISEL-TRUE16-NEXT: s_cmp_eq_u32 s6, 3
+; GFX1250-GISEL-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX1250-GISEL-TRUE16-NEXT: s_cselect_b32 s7, 1, 0
; GFX1250-GISEL-TRUE16-NEXT: s_cmp_gt_i32 s6, 5
; GFX1250-GISEL-TRUE16-NEXT: s_cselect_b32 s6, 1, 0
; GFX1250-GISEL-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1250-GISEL-TRUE16-NEXT: s_or_b32 s6, s7, s6
; GFX1250-GISEL-TRUE16-NEXT: s_cmp_lg_u32 s6, 0
+; GFX1250-GISEL-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1250-GISEL-TRUE16-NEXT: s_cselect_b32 s6, 1, 0
-; GFX1250-GISEL-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1250-GISEL-TRUE16-NEXT: s_add_co_i32 s2, s2, s6
; GFX1250-GISEL-TRUE16-NEXT: s_cmp_gt_i32 s4, 30
+; GFX1250-GISEL-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX1250-GISEL-TRUE16-NEXT: s_cselect_b32 s2, 0x7c00, s2
; GFX1250-GISEL-TRUE16-NEXT: s_cmp_eq_u32 s4, 0x40f
; GFX1250-GISEL-TRUE16-NEXT: s_cselect_b32 s2, s5, s2
@@ -1499,24 +1520,27 @@ define amdgpu_kernel void @fptrunc_f64_to_f16(
; GFX1250-GISEL-FAKE16-NEXT: s_lshl_b32 s6, s9, s6
; GFX1250-GISEL-FAKE16-NEXT: s_or_b32 s5, s5, 0x7c00
; GFX1250-GISEL-FAKE16-NEXT: s_cmp_lg_u32 s6, s8
+; GFX1250-GISEL-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1250-GISEL-FAKE16-NEXT: s_cselect_b32 s6, 1, 0
-; GFX1250-GISEL-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
; GFX1250-GISEL-FAKE16-NEXT: s_or_b32 s6, s9, s6
; GFX1250-GISEL-FAKE16-NEXT: s_cmp_lt_i32 s4, 1
+; GFX1250-GISEL-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1250-GISEL-FAKE16-NEXT: s_cselect_b32 s2, s6, s2
; GFX1250-GISEL-FAKE16-NEXT: s_and_b32 s6, s2, 7
; GFX1250-GISEL-FAKE16-NEXT: s_lshr_b32 s2, s2, 2
; GFX1250-GISEL-FAKE16-NEXT: s_cmp_eq_u32 s6, 3
+; GFX1250-GISEL-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX1250-GISEL-FAKE16-NEXT: s_cselect_b32 s7, 1, 0
; GFX1250-GISEL-FAKE16-NEXT: s_cmp_gt_i32 s6, 5
; GFX1250-GISEL-FAKE16-NEXT: s_cselect_b32 s6, 1, 0
; GFX1250-GISEL-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1250-GISEL-FAKE16-NEXT: s_or_b32 s6, s7, s6
; GFX1250-GISEL-FAKE16-NEXT: s_cmp_lg_u32 s6, 0
+; GFX1250-GISEL-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1250-GISEL-FAKE16-NEXT: s_cselect_b32 s6, 1, 0
-; GFX1250-GISEL-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1250-GISEL-FAKE16-NEXT: s_add_co_i32 s2, s2, s6
; GFX1250-GISEL-FAKE16-NEXT: s_cmp_gt_i32 s4, 30
+; GFX1250-GISEL-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX1250-GISEL-FAKE16-NEXT: s_cselect_b32 s2, 0x7c00, s2
; GFX1250-GISEL-FAKE16-NEXT: s_cmp_eq_u32 s4, 0x40f
; GFX1250-GISEL-FAKE16-NEXT: s_cselect_b32 s2, s5, s2
@@ -1755,7 +1779,7 @@ define amdgpu_kernel void @fptrunc_f64_to_f16_afn(
; GFX1250-SDAG-TRUE16-NEXT: s_wait_loadcnt 0x0
; GFX1250-SDAG-TRUE16-NEXT: v_cvt_f32_f64_e32 v0, v[0:1]
; GFX1250-SDAG-TRUE16-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 2, 2), 0 ; msbs: dst=0 src0=0 src1=0 src2=0
-; GFX1250-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1250-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX1250-SDAG-TRUE16-NEXT: v_cvt_f16_f32_e32 v0.l, v0
; GFX1250-SDAG-TRUE16-NEXT: buffer_store_b16 v0, off, s[4:7], null
; GFX1250-SDAG-TRUE16-NEXT: s_endpgm
@@ -1780,7 +1804,7 @@ define amdgpu_kernel void @fptrunc_f64_to_f16_afn(
; GFX1250-SDAG-FAKE16-NEXT: s_wait_loadcnt 0x0
; GFX1250-SDAG-FAKE16-NEXT: v_cvt_f32_f64_e32 v0, v[0:1]
; GFX1250-SDAG-FAKE16-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 2, 2), 0 ; msbs: dst=0 src0=0 src1=0 src2=0
-; GFX1250-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1250-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX1250-SDAG-FAKE16-NEXT: v_cvt_f16_f32_e32 v0, v0
; GFX1250-SDAG-FAKE16-NEXT: buffer_store_b16 v0, off, s[4:7], null
; GFX1250-SDAG-FAKE16-NEXT: s_endpgm
@@ -1797,10 +1821,11 @@ define amdgpu_kernel void @fptrunc_f64_to_f16_afn(
; GFX1250-GISEL-TRUE16-NEXT: s_wait_kmcnt 0x0
; GFX1250-GISEL-TRUE16-NEXT: v_cvt_f32_f64_e32 v0, s[2:3]
; GFX1250-GISEL-TRUE16-NEXT: s_mov_b32 s3, 0x31016000
-; GFX1250-GISEL-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_3)
+; GFX1250-GISEL-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX1250-GISEL-TRUE16-NEXT: v_readfirstlane_b32 s2, v0
; GFX1250-GISEL-TRUE16-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 2, 2), 0 ; msbs: dst=0 src0=0 src1=0 src2=0
; GFX1250-GISEL-TRUE16-NEXT: s_cvt_f16_f32 s2, s2
+; GFX1250-GISEL-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_3)
; GFX1250-GISEL-TRUE16-NEXT: v_mov_b16_e32 v0.l, s2
; GFX1250-GISEL-TRUE16-NEXT: s_mov_b32 s2, -1
; GFX1250-GISEL-TRUE16-NEXT: buffer_store_b16 v0, off, s[0:3], null
@@ -1818,10 +1843,11 @@ define amdgpu_kernel void @fptrunc_f64_to_f16_afn(
; GFX1250-GISEL-FAKE16-NEXT: s_wait_kmcnt 0x0
; GFX1250-GISEL-FAKE16-NEXT: v_cvt_f32_f64_e32 v0, s[2:3]
; GFX1250-GISEL-FAKE16-NEXT: s_mov_b32 s3, 0x31016000
-; GFX1250-GISEL-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_3)
+; GFX1250-GISEL-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX1250-GISEL-FAKE16-NEXT: v_readfirstlane_b32 s2, v0
; GFX1250-GISEL-FAKE16-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 2, 2), 0 ; msbs: dst=0 src0=0 src1=0 src2=0
; GFX1250-GISEL-FAKE16-NEXT: s_cvt_f16_f32 s2, s2
+; GFX1250-GISEL-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_3)
; GFX1250-GISEL-FAKE16-NEXT: v_mov_b32_e32 v0, s2
; GFX1250-GISEL-FAKE16-NEXT: s_mov_b32 s2, -1
; GFX1250-GISEL-FAKE16-NEXT: buffer_store_b16 v0, off, s[0:3], null
@@ -2972,18 +2998,20 @@ define amdgpu_kernel void @fptrunc_v2f64_to_v2f16(
; GFX11-SDAG-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-SDAG-TRUE16-NEXT: s_lshr_b32 s9, s5, s8
; GFX11-SDAG-TRUE16-NEXT: s_lshl_b32 s8, s9, s8
-; GFX11-SDAG-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_4) | instid1(SALU_CYCLE_1)
+; GFX11-SDAG-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-SDAG-TRUE16-NEXT: s_cmp_lg_u32 s8, s5
; GFX11-SDAG-TRUE16-NEXT: s_cselect_b32 s5, 1, 0
; GFX11-SDAG-TRUE16-NEXT: s_addk_i32 s4, 0xfc10
; GFX11-SDAG-TRUE16-NEXT: s_or_b32 s5, s9, s5
; GFX11-SDAG-TRUE16-NEXT: s_lshl_b32 s8, s4, 12
+; GFX11-SDAG-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX11-SDAG-TRUE16-NEXT: s_or_b32 s8, s3, s8
; GFX11-SDAG-TRUE16-NEXT: s_cmp_lt_i32 s4, 1
; GFX11-SDAG-TRUE16-NEXT: s_cselect_b32 s5, s5, s8
; GFX11-SDAG-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-SDAG-TRUE16-NEXT: s_and_b32 s8, s5, 7
; GFX11-SDAG-TRUE16-NEXT: s_cmp_gt_i32 s8, 5
+; GFX11-SDAG-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX11-SDAG-TRUE16-NEXT: s_cselect_b32 s9, 1, 0
; GFX11-SDAG-TRUE16-NEXT: s_cmp_eq_u32 s8, 3
; GFX11-SDAG-TRUE16-NEXT: s_cselect_b32 s8, 1, 0
@@ -3020,39 +3048,41 @@ define amdgpu_kernel void @fptrunc_v2f64_to_v2f16(
; GFX11-SDAG-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-SDAG-TRUE16-NEXT: s_lshr_b32 s11, s9, s10
; GFX11-SDAG-TRUE16-NEXT: s_lshl_b32 s10, s11, s10
-; GFX11-SDAG-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_4) | instid1(SALU_CYCLE_1)
+; GFX11-SDAG-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-SDAG-TRUE16-NEXT: s_cmp_lg_u32 s10, s9
; GFX11-SDAG-TRUE16-NEXT: s_cselect_b32 s9, 1, 0
; GFX11-SDAG-TRUE16-NEXT: s_addk_i32 s5, 0xfc10
; GFX11-SDAG-TRUE16-NEXT: s_or_b32 s9, s11, s9
; GFX11-SDAG-TRUE16-NEXT: s_lshl_b32 s10, s5, 12
+; GFX11-SDAG-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX11-SDAG-TRUE16-NEXT: s_or_b32 s10, s4, s10
; GFX11-SDAG-TRUE16-NEXT: s_cmp_lt_i32 s5, 1
; GFX11-SDAG-TRUE16-NEXT: s_cselect_b32 s9, s9, s10
; GFX11-SDAG-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-SDAG-TRUE16-NEXT: s_and_b32 s10, s9, 7
; GFX11-SDAG-TRUE16-NEXT: s_cmp_gt_i32 s10, 5
+; GFX11-SDAG-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX11-SDAG-TRUE16-NEXT: s_cselect_b32 s11, 1, 0
; GFX11-SDAG-TRUE16-NEXT: s_cmp_eq_u32 s10, 3
; GFX11-SDAG-TRUE16-NEXT: s_cselect_b32 s10, 1, 0
; GFX11-SDAG-TRUE16-NEXT: s_lshr_b32 s9, s9, 2
; GFX11-SDAG-TRUE16-NEXT: s_or_b32 s10, s10, s11
-; GFX11-SDAG-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-SDAG-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX11-SDAG-TRUE16-NEXT: s_add_i32 s9, s9, s10
; GFX11-SDAG-TRUE16-NEXT: s_cmp_lt_i32 s5, 31
; GFX11-SDAG-TRUE16-NEXT: s_cselect_b32 s9, s9, 0x7c00
; GFX11-SDAG-TRUE16-NEXT: s_cmp_lg_u32 s4, 0
+; GFX11-SDAG-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_4) | instid1(SALU_CYCLE_1)
; GFX11-SDAG-TRUE16-NEXT: s_cselect_b32 s4, s8, 0x7c00
; GFX11-SDAG-TRUE16-NEXT: s_cmpk_eq_i32 s5, 0x40f
; GFX11-SDAG-TRUE16-NEXT: s_mov_b32 s5, s1
; GFX11-SDAG-TRUE16-NEXT: s_cselect_b32 s4, s4, s9
; GFX11-SDAG-TRUE16-NEXT: s_lshr_b32 s3, s3, 16
-; GFX11-SDAG-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-SDAG-TRUE16-NEXT: s_and_b32 s3, s3, 0x8000
+; GFX11-SDAG-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
; GFX11-SDAG-TRUE16-NEXT: s_or_b32 s3, s3, s4
; GFX11-SDAG-TRUE16-NEXT: s_mov_b32 s4, s0
; GFX11-SDAG-TRUE16-NEXT: s_pack_ll_b32_b16 s2, s3, s2
-; GFX11-SDAG-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-SDAG-TRUE16-NEXT: v_mov_b32_e32 v0, s2
; GFX11-SDAG-TRUE16-NEXT: buffer_store_b32 v0, off, s[4:7], 0
; GFX11-SDAG-TRUE16-NEXT: s_endpgm
@@ -3088,18 +3118,20 @@ define amdgpu_kernel void @fptrunc_v2f64_to_v2f16(
; GFX11-SDAG-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-SDAG-FAKE16-NEXT: s_lshr_b32 s9, s5, s8
; GFX11-SDAG-FAKE16-NEXT: s_lshl_b32 s8, s9, s8
-; GFX11-SDAG-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_4) | instid1(SALU_CYCLE_1)
+; GFX11-SDAG-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-SDAG-FAKE16-NEXT: s_cmp_lg_u32 s8, s5
; GFX11-SDAG-FAKE16-NEXT: s_cselect_b32 s5, 1, 0
; GFX11-SDAG-FAKE16-NEXT: s_addk_i32 s4, 0xfc10
; GFX11-SDAG-FAKE16-NEXT: s_or_b32 s5, s9, s5
; GFX11-SDAG-FAKE16-NEXT: s_lshl_b32 s8, s4, 12
+; GFX11-SDAG-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX11-SDAG-FAKE16-NEXT: s_or_b32 s8, s3, s8
; GFX11-SDAG-FAKE16-NEXT: s_cmp_lt_i32 s4, 1
; GFX11-SDAG-FAKE16-NEXT: s_cselect_b32 s5, s5, s8
; GFX11-SDAG-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-SDAG-FAKE16-NEXT: s_and_b32 s8, s5, 7
; GFX11-SDAG-FAKE16-NEXT: s_cmp_gt_i32 s8, 5
+; GFX11-SDAG-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX11-SDAG-FAKE16-NEXT: s_cselect_b32 s9, 1, 0
; GFX11-SDAG-FAKE16-NEXT: s_cmp_eq_u32 s8, 3
; GFX11-SDAG-FAKE16-NEXT: s_cselect_b32 s8, 1, 0
@@ -3136,39 +3168,41 @@ define amdgpu_kernel void @fptrunc_v2f64_to_v2f16(
; GFX11-SDAG-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-SDAG-FAKE16-NEXT: s_lshr_b32 s11, s9, s10
; GFX11-SDAG-FAKE16-NEXT: s_lshl_b32 s10, s11, s10
-; GFX11-SDAG-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_4) | instid1(SALU_CYCLE_1)
+; GFX11-SDAG-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-SDAG-FAKE16-NEXT: s_cmp_lg_u32 s10, s9
; GFX11-SDAG-FAKE16-NEXT: s_cselect_b32 s9, 1, 0
; GFX11-SDAG-FAKE16-NEXT: s_addk_i32 s5, 0xfc10
; GFX11-SDAG-FAKE16-NEXT: s_or_b32 s9, s11, s9
; GFX11-SDAG-FAKE16-NEXT: s_lshl_b32 s10, s5, 12
+; GFX11-SDAG-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX11-SDAG-FAKE16-NEXT: s_or_b32 s10, s4, s10
; GFX11-SDAG-FAKE16-NEXT: s_cmp_lt_i32 s5, 1
; GFX11-SDAG-FAKE16-NEXT: s_cselect_b32 s9, s9, s10
; GFX11-SDAG-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-SDAG-FAKE16-NEXT: s_and_b32 s10, s9, 7
; GFX11-SDAG-FAKE16-NEXT: s_cmp_gt_i32 s10, 5
+; GFX11-SDAG-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX11-SDAG-FAKE16-NEXT: s_cselect_b32 s11, 1, 0
; GFX11-SDAG-FAKE16-NEXT: s_cmp_eq_u32 s10, 3
; GFX11-SDAG-FAKE16-NEXT: s_cselect_b32 s10, 1, 0
; GFX11-SDAG-FAKE16-NEXT: s_lshr_b32 s9, s9, 2
; GFX11-SDAG-FAKE16-NEXT: s_or_b32 s10, s10, s11
-; GFX11-SDAG-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-SDAG-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX11-SDAG-FAKE16-NEXT: s_add_i32 s9, s9, s10
; GFX11-SDAG-FAKE16-NEXT: s_cmp_lt_i32 s5, 31
; GFX11-SDAG-FAKE16-NEXT: s_cselect_b32 s9, s9, 0x7c00
; GFX11-SDAG-FAKE16-NEXT: s_cmp_lg_u32 s4, 0
+; GFX11-SDAG-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_4) | instid1(SALU_CYCLE_1)
; GFX11-SDAG-FAKE16-NEXT: s_cselect_b32 s4, s8, 0x7c00
; GFX11-SDAG-FAKE16-NEXT: s_cmpk_eq_i32 s5, 0x40f
; GFX11-SDAG-FAKE16-NEXT: s_mov_b32 s5, s1
; GFX11-SDAG-FAKE16-NEXT: s_cselect_b32 s4, s4, s9
; GFX11-SDAG-FAKE16-NEXT: s_lshr_b32 s3, s3, 16
-; GFX11-SDAG-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-SDAG-FAKE16-NEXT: s_and_b32 s3, s3, 0x8000
+; GFX11-SDAG-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
; GFX11-SDAG-FAKE16-NEXT: s_or_b32 s3, s3, s4
; GFX11-SDAG-FAKE16-NEXT: s_mov_b32 s4, s0
; GFX11-SDAG-FAKE16-NEXT: s_pack_ll_b32_b16 s2, s3, s2
-; GFX11-SDAG-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-SDAG-FAKE16-NEXT: v_mov_b32_e32 v0, s2
; GFX11-SDAG-FAKE16-NEXT: buffer_store_b32 v0, off, s[4:7], 0
; GFX11-SDAG-FAKE16-NEXT: s_endpgm
@@ -3200,24 +3234,27 @@ define amdgpu_kernel void @fptrunc_v2f64_to_v2f16(
; GFX11-GISEL-TRUE16-NEXT: s_lshl_b32 s8, s11, s8
; GFX11-GISEL-TRUE16-NEXT: s_or_b32 s4, s4, 0x7c00
; GFX11-GISEL-TRUE16-NEXT: s_cmp_lg_u32 s8, s10
+; GFX11-GISEL-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-GISEL-TRUE16-NEXT: s_cselect_b32 s8, 1, 0
-; GFX11-GISEL-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
; GFX11-GISEL-TRUE16-NEXT: s_or_b32 s8, s11, s8
; GFX11-GISEL-TRUE16-NEXT: s_cmp_lt_i32 s2, 1
+; GFX11-GISEL-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-GISEL-TRUE16-NEXT: s_cselect_b32 s3, s8, s3
; GFX11-GISEL-TRUE16-NEXT: s_and_b32 s8, s3, 7
; GFX11-GISEL-TRUE16-NEXT: s_lshr_b32 s3, s3, 2
; GFX11-GISEL-TRUE16-NEXT: s_cmp_eq_u32 s8, 3
+; GFX11-GISEL-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX11-GISEL-TRUE16-NEXT: s_cselect_b32 s9, 1, 0
; GFX11-GISEL-TRUE16-NEXT: s_cmp_gt_i32 s8, 5
; GFX11-GISEL-TRUE16-NEXT: s_cselect_b32 s8, 1, 0
; GFX11-GISEL-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-GISEL-TRUE16-NEXT: s_or_b32 s8, s9, s8
; GFX11-GISEL-TRUE16-NEXT: s_cmp_lg_u32 s8, 0
+; GFX11-GISEL-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-GISEL-TRUE16-NEXT: s_cselect_b32 s8, 1, 0
-; GFX11-GISEL-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-GISEL-TRUE16-NEXT: s_add_i32 s3, s3, s8
; GFX11-GISEL-TRUE16-NEXT: s_cmp_gt_i32 s2, 30
+; GFX11-GISEL-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX11-GISEL-TRUE16-NEXT: s_cselect_b32 s3, 0x7c00, s3
; GFX11-GISEL-TRUE16-NEXT: s_cmpk_eq_i32 s2, 0x40f
; GFX11-GISEL-TRUE16-NEXT: s_cselect_b32 s2, s4, s3
@@ -3245,24 +3282,27 @@ define amdgpu_kernel void @fptrunc_v2f64_to_v2f16(
; GFX11-GISEL-TRUE16-NEXT: s_lshl_b32 s6, s10, s6
; GFX11-GISEL-TRUE16-NEXT: s_or_b32 s5, s5, 0x7c00
; GFX11-GISEL-TRUE16-NEXT: s_cmp_lg_u32 s6, s9
+; GFX11-GISEL-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-GISEL-TRUE16-NEXT: s_cselect_b32 s6, 1, 0
-; GFX11-GISEL-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
; GFX11-GISEL-TRUE16-NEXT: s_or_b32 s6, s10, s6
; GFX11-GISEL-TRUE16-NEXT: s_cmp_lt_i32 s4, 1
+; GFX11-GISEL-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-GISEL-TRUE16-NEXT: s_cselect_b32 s3, s6, s3
; GFX11-GISEL-TRUE16-NEXT: s_and_b32 s6, s3, 7
; GFX11-GISEL-TRUE16-NEXT: s_lshr_b32 s3, s3, 2
; GFX11-GISEL-TRUE16-NEXT: s_cmp_eq_u32 s6, 3
+; GFX11-GISEL-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX11-GISEL-TRUE16-NEXT: s_cselect_b32 s8, 1, 0
; GFX11-GISEL-TRUE16-NEXT: s_cmp_gt_i32 s6, 5
; GFX11-GISEL-TRUE16-NEXT: s_cselect_b32 s6, 1, 0
; GFX11-GISEL-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-GISEL-TRUE16-NEXT: s_or_b32 s6, s8, s6
; GFX11-GISEL-TRUE16-NEXT: s_cmp_lg_u32 s6, 0
+; GFX11-GISEL-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-GISEL-TRUE16-NEXT: s_cselect_b32 s6, 1, 0
-; GFX11-GISEL-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-GISEL-TRUE16-NEXT: s_add_i32 s3, s3, s6
; GFX11-GISEL-TRUE16-NEXT: s_cmp_gt_i32 s4, 30
+; GFX11-GISEL-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX11-GISEL-TRUE16-NEXT: s_cselect_b32 s3, 0x7c00, s3
; GFX11-GISEL-TRUE16-NEXT: s_cmpk_eq_i32 s4, 0x40f
; GFX11-GISEL-TRUE16-NEXT: s_cselect_b32 s3, s5, s3
@@ -3305,24 +3345,27 @@ define amdgpu_kernel void @fptrunc_v2f64_to_v2f16(
; GFX11-GISEL-FAKE16-NEXT: s_lshl_b32 s8, s11, s8
; GFX11-GISEL-FAKE16-NEXT: s_or_b32 s4, s4, 0x7c00
; GFX11-GISEL-FAKE16-NEXT: s_cmp_lg_u32 s8, s10
+; GFX11-GISEL-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-GISEL-FAKE16-NEXT: s_cselect_b32 s8, 1, 0
-; GFX11-GISEL-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
; GFX11-GISEL-FAKE16-NEXT: s_or_b32 s8, s11, s8
; GFX11-GISEL-FAKE16-NEXT: s_cmp_lt_i32 s2, 1
+; GFX11-GISEL-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-GISEL-FAKE16-NEXT: s_cselect_b32 s3, s8, s3
; GFX11-GISEL-FAKE16-NEXT: s_and_b32 s8, s3, 7
; GFX11-GISEL-FAKE16-NEXT: s_lshr_b32 s3, s3, 2
; GFX11-GISEL-FAKE16-NEXT: s_cmp_eq_u32 s8, 3
+; GFX11-GISEL-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX11-GISEL-FAKE16-NEXT: s_cselect_b32 s9, 1, 0
; GFX11-GISEL-FAKE16-NEXT: s_cmp_gt_i32 s8, 5
; GFX11-GISEL-FAKE16-NEXT: s_cselect_b32 s8, 1, 0
; GFX11-GISEL-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-GISEL-FAKE16-NEXT: s_or_b32 s8, s9, s8
; GFX11-GISEL-FAKE16-NEXT: s_cmp_lg_u32 s8, 0
+; GFX11-GISEL-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-GISEL-FAKE16-NEXT: s_cselect_b32 s8, 1, 0
-; GFX11-GISEL-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-GISEL-FAKE16-NEXT: s_add_i32 s3, s3, s8
; GFX11-GISEL-FAKE16-NEXT: s_cmp_gt_i32 s2, 30
+; GFX11-GISEL-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX11-GISEL-FAKE16-NEXT: s_cselect_b32 s3, 0x7c00, s3
; GFX11-GISEL-FAKE16-NEXT: s_cmpk_eq_i32 s2, 0x40f
; GFX11-GISEL-FAKE16-NEXT: s_cselect_b32 s2, s4, s3
@@ -3350,24 +3393,27 @@ define amdgpu_kernel void @fptrunc_v2f64_to_v2f16(
; GFX11-GISEL-FAKE16-NEXT: s_lshl_b32 s6, s10, s6
; GFX11-GISEL-FAKE16-NEXT: s_or_b32 s5, s5, 0x7c00
; GFX11-GISEL-FAKE16-NEXT: s_cmp_lg_u32 s6, s9
+; GFX11-GISEL-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-GISEL-FAKE16-NEXT: s_cselect_b32 s6, 1, 0
-; GFX11-GISEL-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
; GFX11-GISEL-FAKE16-NEXT: s_or_b32 s6, s10, s6
; GFX11-GISEL-FAKE16-NEXT: s_cmp_lt_i32 s4, 1
+; GFX11-GISEL-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-GISEL-FAKE16-NEXT: s_cselect_b32 s3, s6, s3
; GFX11-GISEL-FAKE16-NEXT: s_and_b32 s6, s3, 7
; GFX11-GISEL-FAKE16-NEXT: s_lshr_b32 s3, s3, 2
; GFX11-GISEL-FAKE16-NEXT: s_cmp_eq_u32 s6, 3
+; GFX11-GISEL-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX11-GISEL-FAKE16-NEXT: s_cselect_b32 s8, 1, 0
; GFX11-GISEL-FAKE16-NEXT: s_cmp_gt_i32 s6, 5
; GFX11-GISEL-FAKE16-NEXT: s_cselect_b32 s6, 1, 0
; GFX11-GISEL-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-GISEL-FAKE16-NEXT: s_or_b32 s6, s8, s6
; GFX11-GISEL-FAKE16-NEXT: s_cmp_lg_u32 s6, 0
+; GFX11-GISEL-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-GISEL-FAKE16-NEXT: s_cselect_b32 s6, 1, 0
-; GFX11-GISEL-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-GISEL-FAKE16-NEXT: s_add_i32 s3, s3, s6
; GFX11-GISEL-FAKE16-NEXT: s_cmp_gt_i32 s4, 30
+; GFX11-GISEL-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX11-GISEL-FAKE16-NEXT: s_cselect_b32 s3, 0x7c00, s3
; GFX11-GISEL-FAKE16-NEXT: s_cmpk_eq_i32 s4, 0x40f
; GFX11-GISEL-FAKE16-NEXT: s_cselect_b32 s3, s5, s3
@@ -3418,18 +3464,20 @@ define amdgpu_kernel void @fptrunc_v2f64_to_v2f16(
; GFX1250-SDAG-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1250-SDAG-TRUE16-NEXT: s_lshr_b32 s9, s5, s8
; GFX1250-SDAG-TRUE16-NEXT: s_lshl_b32 s8, s9, s8
-; GFX1250-SDAG-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_4) | instid1(SALU_CYCLE_1)
+; GFX1250-SDAG-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1250-SDAG-TRUE16-NEXT: s_cmp_lg_u32 s8, s5
; GFX1250-SDAG-TRUE16-NEXT: s_cselect_b32 s5, 1, 0
; GFX1250-SDAG-TRUE16-NEXT: s_addk_co_i32 s4, 0xfc10
; GFX1250-SDAG-TRUE16-NEXT: s_or_b32 s5, s9, s5
; GFX1250-SDAG-TRUE16-NEXT: s_lshl_b32 s8, s4, 12
+; GFX1250-SDAG-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX1250-SDAG-TRUE16-NEXT: s_or_b32 s8, s3, s8
; GFX1250-SDAG-TRUE16-NEXT: s_cmp_lt_i32 s4, 1
; GFX1250-SDAG-TRUE16-NEXT: s_cselect_b32 s5, s5, s8
; GFX1250-SDAG-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1250-SDAG-TRUE16-NEXT: s_and_b32 s8, s5, 7
; GFX1250-SDAG-TRUE16-NEXT: s_cmp_gt_i32 s8, 5
+; GFX1250-SDAG-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX1250-SDAG-TRUE16-NEXT: s_cselect_b32 s9, 1, 0
; GFX1250-SDAG-TRUE16-NEXT: s_cmp_eq_u32 s8, 3
; GFX1250-SDAG-TRUE16-NEXT: s_cselect_b32 s8, 1, 0
@@ -3466,39 +3514,41 @@ define amdgpu_kernel void @fptrunc_v2f64_to_v2f16(
; GFX1250-SDAG-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1250-SDAG-TRUE16-NEXT: s_lshr_b32 s11, s9, s10
; GFX1250-SDAG-TRUE16-NEXT: s_lshl_b32 s10, s11, s10
-; GFX1250-SDAG-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_4) | instid1(SALU_CYCLE_1)
+; GFX1250-SDAG-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1250-SDAG-TRUE16-NEXT: s_cmp_lg_u32 s10, s9
; GFX1250-SDAG-TRUE16-NEXT: s_cselect_b32 s9, 1, 0
; GFX1250-SDAG-TRUE16-NEXT: s_addk_co_i32 s5, 0xfc10
; GFX1250-SDAG-TRUE16-NEXT: s_or_b32 s9, s11, s9
; GFX1250-SDAG-TRUE16-NEXT: s_lshl_b32 s10, s5, 12
+; GFX1250-SDAG-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX1250-SDAG-TRUE16-NEXT: s_or_b32 s10, s4, s10
; GFX1250-SDAG-TRUE16-NEXT: s_cmp_lt_i32 s5, 1
; GFX1250-SDAG-TRUE16-NEXT: s_cselect_b32 s9, s9, s10
; GFX1250-SDAG-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1250-SDAG-TRUE16-NEXT: s_and_b32 s10, s9, 7
; GFX1250-SDAG-TRUE16-NEXT: s_cmp_gt_i32 s10, 5
+; GFX1250-SDAG-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX1250-SDAG-TRUE16-NEXT: s_cselect_b32 s11, 1, 0
; GFX1250-SDAG-TRUE16-NEXT: s_cmp_eq_u32 s10, 3
; GFX1250-SDAG-TRUE16-NEXT: s_cselect_b32 s10, 1, 0
; GFX1250-SDAG-TRUE16-NEXT: s_lshr_b32 s9, s9, 2
; GFX1250-SDAG-TRUE16-NEXT: s_or_b32 s10, s10, s11
-; GFX1250-SDAG-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1250-SDAG-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX1250-SDAG-TRUE16-NEXT: s_add_co_i32 s9, s9, s10
; GFX1250-SDAG-TRUE16-NEXT: s_cmp_lt_i32 s5, 31
; GFX1250-SDAG-TRUE16-NEXT: s_cselect_b32 s9, s9, 0x7c00
; GFX1250-SDAG-TRUE16-NEXT: s_cmp_lg_u32 s4, 0
+; GFX1250-SDAG-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_4) | instid1(SALU_CYCLE_1)
; GFX1250-SDAG-TRUE16-NEXT: s_cselect_b32 s4, s8, 0x7c00
; GFX1250-SDAG-TRUE16-NEXT: s_cmp_eq_u32 s5, 0x40f
; GFX1250-SDAG-TRUE16-NEXT: s_mov_b32 s5, s1
; GFX1250-SDAG-TRUE16-NEXT: s_cselect_b32 s4, s4, s9
; GFX1250-SDAG-TRUE16-NEXT: s_lshr_b32 s3, s3, 16
-; GFX1250-SDAG-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1250-SDAG-TRUE16-NEXT: s_and_b32 s3, s3, 0x8000
+; GFX1250-SDAG-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
; GFX1250-SDAG-TRUE16-NEXT: s_or_b32 s3, s3, s4
; GFX1250-SDAG-TRUE16-NEXT: s_mov_b32 s4, s0
; GFX1250-SDAG-TRUE16-NEXT: s_pack_ll_b32_b16 s2, s3, s2
-; GFX1250-SDAG-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1250-SDAG-TRUE16-NEXT: v_mov_b32_e32 v0, s2
; GFX1250-SDAG-TRUE16-NEXT: buffer_store_b32 v0, off, s[4:7], null
; GFX1250-SDAG-TRUE16-NEXT: s_endpgm
@@ -3538,18 +3588,20 @@ define amdgpu_kernel void @fptrunc_v2f64_to_v2f16(
; GFX1250-SDAG-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1250-SDAG-FAKE16-NEXT: s_lshr_b32 s9, s5, s8
; GFX1250-SDAG-FAKE16-NEXT: s_lshl_b32 s8, s9, s8
-; GFX1250-SDAG-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_4) | instid1(SALU_CYCLE_1)
+; GFX1250-SDAG-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1250-SDAG-FAKE16-NEXT: s_cmp_lg_u32 s8, s5
; GFX1250-SDAG-FAKE16-NEXT: s_cselect_b32 s5, 1, 0
; GFX1250-SDAG-FAKE16-NEXT: s_addk_co_i32 s4, 0xfc10
; GFX1250-SDAG-FAKE16-NEXT: s_or_b32 s5, s9, s5
; GFX1250-SDAG-FAKE16-NEXT: s_lshl_b32 s8, s4, 12
+; GFX1250-SDAG-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX1250-SDAG-FAKE16-NEXT: s_or_b32 s8, s3, s8
; GFX1250-SDAG-FAKE16-NEXT: s_cmp_lt_i32 s4, 1
; GFX1250-SDAG-FAKE16-NEXT: s_cselect_b32 s5, s5, s8
; GFX1250-SDAG-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1250-SDAG-FAKE16-NEXT: s_and_b32 s8, s5, 7
; GFX1250-SDAG-FAKE16-NEXT: s_cmp_gt_i32 s8, 5
+; GFX1250-SDAG-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX1250-SDAG-FAKE16-NEXT: s_cselect_b32 s9, 1, 0
; GFX1250-SDAG-FAKE16-NEXT: s_cmp_eq_u32 s8, 3
; GFX1250-SDAG-FAKE16-NEXT: s_cselect_b32 s8, 1, 0
@@ -3586,39 +3638,41 @@ define amdgpu_kernel void @fptrunc_v2f64_to_v2f16(
; GFX1250-SDAG-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1250-SDAG-FAKE16-NEXT: s_lshr_b32 s11, s9, s10
; GFX1250-SDAG-FAKE16-NEXT: s_lshl_b32 s10, s11, s10
-; GFX1250-SDAG-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_4) | instid1(SALU_CYCLE_1)
+; GFX1250-SDAG-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1250-SDAG-FAKE16-NEXT: s_cmp_lg_u32 s10, s9
; GFX1250-SDAG-FAKE16-NEXT: s_cselect_b32 s9, 1, 0
; GFX1250-SDAG-FAKE16-NEXT: s_addk_co_i32 s5, 0xfc10
; GFX1250-SDAG-FAKE16-NEXT: s_or_b32 s9, s11, s9
; GFX1250-SDAG-FAKE16-NEXT: s_lshl_b32 s10, s5, 12
+; GFX1250-SDAG-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX1250-SDAG-FAKE16-NEXT: s_or_b32 s10, s4, s10
; GFX1250-SDAG-FAKE16-NEXT: s_cmp_lt_i32 s5, 1
; GFX1250-SDAG-FAKE16-NEXT: s_cselect_b32 s9, s9, s10
; GFX1250-SDAG-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1250-SDAG-FAKE16-NEXT: s_and_b32 s10, s9, 7
; GFX1250-SDAG-FAKE16-NEXT: s_cmp_gt_i32 s10, 5
+; GFX1250-SDAG-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX1250-SDAG-FAKE16-NEXT: s_cselect_b32 s11, 1, 0
; GFX1250-SDAG-FAKE16-NEXT: s_cmp_eq_u32 s10, 3
; GFX1250-SDAG-FAKE16-NEXT: s_cselect_b32 s10, 1, 0
; GFX1250-SDAG-FAKE16-NEXT: s_lshr_b32 s9, s9, 2
; GFX1250-SDAG-FAKE16-NEXT: s_or_b32 s10, s10, s11
-; GFX1250-SDAG-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1250-SDAG-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX1250-SDAG-FAKE16-NEXT: s_add_co_i32 s9, s9, s10
; GFX1250-SDAG-FAKE16-NEXT: s_cmp_lt_i32 s5, 31
; GFX1250-SDAG-FAKE16-NEXT: s_cselect_b32 s9, s9, 0x7c00
; GFX1250-SDAG-FAKE16-NEXT: s_cmp_lg_u32 s4, 0
+; GFX1250-SDAG-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_4) | instid1(SALU_CYCLE_1)
; GFX1250-SDAG-FAKE16-NEXT: s_cselect_b32 s4, s8, 0x7c00
; GFX1250-SDAG-FAKE16-NEXT: s_cmp_eq_u32 s5, 0x40f
; GFX1250-SDAG-FAKE16-NEXT: s_mov_b32 s5, s1
; GFX1250-SDAG-FAKE16-NEXT: s_cselect_b32 s4, s4, s9
; GFX1250-SDAG-FAKE16-NEXT: s_lshr_b32 s3, s3, 16
-; GFX1250-SDAG-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1250-SDAG-FAKE16-NEXT: s_and_b32 s3, s3, 0x8000
+; GFX1250-SDAG-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
; GFX1250-SDAG-FAKE16-NEXT: s_or_b32 s3, s3, s4
; GFX1250-SDAG-FAKE16-NEXT: s_mov_b32 s4, s0
; GFX1250-SDAG-FAKE16-NEXT: s_pack_ll_b32_b16 s2, s3, s2
-; GFX1250-SDAG-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1250-SDAG-FAKE16-NEXT: v_mov_b32_e32 v0, s2
; GFX1250-SDAG-FAKE16-NEXT: buffer_store_b32 v0, off, s[4:7], null
; GFX1250-SDAG-FAKE16-NEXT: s_endpgm
@@ -3654,24 +3708,27 @@ define amdgpu_kernel void @fptrunc_v2f64_to_v2f16(
; GFX1250-GISEL-TRUE16-NEXT: s_lshl_b32 s8, s11, s8
; GFX1250-GISEL-TRUE16-NEXT: s_or_b32 s4, s4, 0x7c00
; GFX1250-GISEL-TRUE16-NEXT: s_cmp_lg_u32 s8, s10
+; GFX1250-GISEL-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1250-GISEL-TRUE16-NEXT: s_cselect_b32 s8, 1, 0
-; GFX1250-GISEL-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
; GFX1250-GISEL-TRUE16-NEXT: s_or_b32 s8, s11, s8
; GFX1250-GISEL-TRUE16-NEXT: s_cmp_lt_i32 s2, 1
+; GFX1250-GISEL-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1250-GISEL-TRUE16-NEXT: s_cselect_b32 s3, s8, s3
; GFX1250-GISEL-TRUE16-NEXT: s_and_b32 s8, s3, 7
; GFX1250-GISEL-TRUE16-NEXT: s_lshr_b32 s3, s3, 2
; GFX1250-GISEL-TRUE16-NEXT: s_cmp_eq_u32 s8, 3
+; GFX1250-GISEL-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX1250-GISEL-TRUE16-NEXT: s_cselect_b32 s9, 1, 0
; GFX1250-GISEL-TRUE16-NEXT: s_cmp_gt_i32 s8, 5
; GFX1250-GISEL-TRUE16-NEXT: s_cselect_b32 s8, 1, 0
; GFX1250-GISEL-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1250-GISEL-TRUE16-NEXT: s_or_b32 s8, s9, s8
; GFX1250-GISEL-TRUE16-NEXT: s_cmp_lg_u32 s8, 0
+; GFX1250-GISEL-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1250-GISEL-TRUE16-NEXT: s_cselect_b32 s8, 1, 0
-; GFX1250-GISEL-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1250-GISEL-TRUE16-NEXT: s_add_co_i32 s3, s3, s8
; GFX1250-GISEL-TRUE16-NEXT: s_cmp_gt_i32 s2, 30
+; GFX1250-GISEL-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX1250-GISEL-TRUE16-NEXT: s_cselect_b32 s3, 0x7c00, s3
; GFX1250-GISEL-TRUE16-NEXT: s_cmp_eq_u32 s2, 0x40f
; GFX1250-GISEL-TRUE16-NEXT: s_cselect_b32 s2, s4, s3
@@ -3699,24 +3756,27 @@ define amdgpu_kernel void @fptrunc_v2f64_to_v2f16(
; GFX1250-GISEL-TRUE16-NEXT: s_lshl_b32 s6, s10, s6
; GFX1250-GISEL-TRUE16-NEXT: s_or_b32 s5, s5, 0x7c00
; GFX1250-GISEL-TRUE16-NEXT: s_cmp_lg_u32 s6, s9
+; GFX1250-GISEL-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1250-GISEL-TRUE16-NEXT: s_cselect_b32 s6, 1, 0
-; GFX1250-GISEL-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
; GFX1250-GISEL-TRUE16-NEXT: s_or_b32 s6, s10, s6
; GFX1250-GISEL-TRUE16-NEXT: s_cmp_lt_i32 s4, 1
+; GFX1250-GISEL-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1250-GISEL-TRUE16-NEXT: s_cselect_b32 s3, s6, s3
; GFX1250-GISEL-TRUE16-NEXT: s_and_b32 s6, s3, 7
; GFX1250-GISEL-TRUE16-NEXT: s_lshr_b32 s3, s3, 2
; GFX1250-GISEL-TRUE16-NEXT: s_cmp_eq_u32 s6, 3
+; GFX1250-GISEL-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX1250-GISEL-TRUE16-NEXT: s_cselect_b32 s8, 1, 0
; GFX1250-GISEL-TRUE16-NEXT: s_cmp_gt_i32 s6, 5
; GFX1250-GISEL-TRUE16-NEXT: s_cselect_b32 s6, 1, 0
; GFX1250-GISEL-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1250-GISEL-TRUE16-NEXT: s_or_b32 s6, s8, s6
; GFX1250-GISEL-TRUE16-NEXT: s_cmp_lg_u32 s6, 0
+; GFX1250-GISEL-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1250-GISEL-TRUE16-NEXT: s_cselect_b32 s6, 1, 0
-; GFX1250-GISEL-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1250-GISEL-TRUE16-NEXT: s_add_co_i32 s3, s3, s6
; GFX1250-GISEL-TRUE16-NEXT: s_cmp_gt_i32 s4, 30
+; GFX1250-GISEL-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX1250-GISEL-TRUE16-NEXT: s_cselect_b32 s3, 0x7c00, s3
; GFX1250-GISEL-TRUE16-NEXT: s_cmp_eq_u32 s4, 0x40f
; GFX1250-GISEL-TRUE16-NEXT: s_cselect_b32 s3, s5, s3
@@ -3763,24 +3823,27 @@ define amdgpu_kernel void @fptrunc_v2f64_to_v2f16(
; GFX1250-GISEL-FAKE16-NEXT: s_lshl_b32 s8, s11, s8
; GFX1250-GISEL-FAKE16-NEXT: s_or_b32 s4, s4, 0x7c00
; GFX1250-GISEL-FAKE16-NEXT: s_cmp_lg_u32 s8, s10
+; GFX1250-GISEL-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1250-GISEL-FAKE16-NEXT: s_cselect_b32 s8, 1, 0
-; GFX1250-GISEL-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
; GFX1250-GISEL-FAKE16-NEXT: s_or_b32 s8, s11, s8
; GFX1250-GISEL-FAKE16-NEXT: s_cmp_lt_i32 s2, 1
+; GFX1250-GISEL-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1250-GISEL-FAKE16-NEXT: s_cselect_b32 s3, s8, s3
; GFX1250-GISEL-FAKE16-NEXT: s_and_b32 s8, s3, 7
; GFX1250-GISEL-FAKE16-NEXT: s_lshr_b32 s3, s3, 2
; GFX1250-GISEL-FAKE16-NEXT: s_cmp_eq_u32 s8, 3
+; GFX1250-GISEL-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX1250-GISEL-FAKE16-NEXT: s_cselect_b32 s9, 1, 0
; GFX1250-GISEL-FAKE16-NEXT: s_cmp_gt_i32 s8, 5
; GFX1250-GISEL-FAKE16-NEXT: s_cselect_b32 s8, 1, 0
; GFX1250-GISEL-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1250-GISEL-FAKE16-NEXT: s_or_b32 s8, s9, s8
; GFX1250-GISEL-FAKE16-NEXT: s_cmp_lg_u32 s8, 0
+; GFX1250-GISEL-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1250-GISEL-FAKE16-NEXT: s_cselect_b32 s8, 1, 0
-; GFX1250-GISEL-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1250-GISEL-FAKE16-NEXT: s_add_co_i32 s3, s3, s8
; GFX1250-GISEL-FAKE16-NEXT: s_cmp_gt_i32 s2, 30
+; GFX1250-GISEL-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX1250-GISEL-FAKE16-NEXT: s_cselect_b32 s3, 0x7c00, s3
; GFX1250-GISEL-FAKE16-NEXT: s_cmp_eq_u32 s2, 0x40f
; GFX1250-GISEL-FAKE16-NEXT: s_cselect_b32 s2, s4, s3
@@ -3808,24 +3871,27 @@ define amdgpu_kernel void @fptrunc_v2f64_to_v2f16(
; GFX1250-GISEL-FAKE16-NEXT: s_lshl_b32 s6, s10, s6
; GFX1250-GISEL-FAKE16-NEXT: s_or_b32 s5, s5, 0x7c00
; GFX1250-GISEL-FAKE16-NEXT: s_cmp_lg_u32 s6, s9
+; GFX1250-GISEL-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1250-GISEL-FAKE16-NEXT: s_cselect_b32 s6, 1, 0
-; GFX1250-GISEL-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
; GFX1250-GISEL-FAKE16-NEXT: s_or_b32 s6, s10, s6
; GFX1250-GISEL-FAKE16-NEXT: s_cmp_lt_i32 s4, 1
+; GFX1250-GISEL-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1250-GISEL-FAKE16-NEXT: s_cselect_b32 s3, s6, s3
; GFX1250-GISEL-FAKE16-NEXT: s_and_b32 s6, s3, 7
; GFX1250-GISEL-FAKE16-NEXT: s_lshr_b32 s3, s3, 2
; GFX1250-GISEL-FAKE16-NEXT: s_cmp_eq_u32 s6, 3
+; GFX1250-GISEL-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX1250-GISEL-FAKE16-NEXT: s_cselect_b32 s8, 1, 0
; GFX1250-GISEL-FAKE16-NEXT: s_cmp_gt_i32 s6, 5
; GFX1250-GISEL-FAKE16-NEXT: s_cselect_b32 s6, 1, 0
; GFX1250-GISEL-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1250-GISEL-FAKE16-NEXT: s_or_b32 s6, s8, s6
; GFX1250-GISEL-FAKE16-NEXT: s_cmp_lg_u32 s6, 0
+; GFX1250-GISEL-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1250-GISEL-FAKE16-NEXT: s_cselect_b32 s6, 1, 0
-; GFX1250-GISEL-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1250-GISEL-FAKE16-NEXT: s_add_co_i32 s3, s3, s6
; GFX1250-GISEL-FAKE16-NEXT: s_cmp_gt_i32 s4, 30
+; GFX1250-GISEL-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX1250-GISEL-FAKE16-NEXT: s_cselect_b32 s3, 0x7c00, s3
; GFX1250-GISEL-FAKE16-NEXT: s_cmp_eq_u32 s4, 0x40f
; GFX1250-GISEL-FAKE16-NEXT: s_cselect_b32 s3, s5, s3
@@ -4175,9 +4241,9 @@ define amdgpu_kernel void @fptrunc_v2f64_to_v2f16_afn(
; GFX1250-GISEL-TRUE16-NEXT: v_readfirstlane_b32 s2, v0
; GFX1250-GISEL-TRUE16-NEXT: v_readfirstlane_b32 s3, v1
; GFX1250-GISEL-TRUE16-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 2, 2), 0 ; msbs: dst=0 src0=0 src1=0 src2=0
+; GFX1250-GISEL-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_3)
; GFX1250-GISEL-TRUE16-NEXT: s_cvt_f16_f32 s2, s2
; GFX1250-GISEL-TRUE16-NEXT: s_cvt_f16_f32 s3, s3
-; GFX1250-GISEL-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_3)
; GFX1250-GISEL-TRUE16-NEXT: s_pack_ll_b32_b16 s2, s2, s3
; GFX1250-GISEL-TRUE16-NEXT: s_mov_b32 s3, 0x31016000
; GFX1250-GISEL-TRUE16-NEXT: v_mov_b32_e32 v0, s2
@@ -4201,9 +4267,9 @@ define amdgpu_kernel void @fptrunc_v2f64_to_v2f16_afn(
; GFX1250-GISEL-FAKE16-NEXT: v_readfirstlane_b32 s2, v0
; GFX1250-GISEL-FAKE16-NEXT: v_readfirstlane_b32 s3, v1
; GFX1250-GISEL-FAKE16-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 2, 2), 0 ; msbs: dst=0 src0=0 src1=0 src2=0
+; GFX1250-GISEL-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_3)
; GFX1250-GISEL-FAKE16-NEXT: s_cvt_f16_f32 s2, s2
; GFX1250-GISEL-FAKE16-NEXT: s_cvt_f16_f32 s3, s3
-; GFX1250-GISEL-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_3)
; GFX1250-GISEL-FAKE16-NEXT: s_pack_ll_b32_b16 s2, s2, s3
; GFX1250-GISEL-FAKE16-NEXT: s_mov_b32 s3, 0x31016000
; GFX1250-GISEL-FAKE16-NEXT: v_mov_b32_e32 v0, s2
@@ -4420,7 +4486,7 @@ define amdgpu_kernel void @fneg_fptrunc_f32_to_f16(
; GFX1250-SDAG-TRUE16-NEXT: s_wait_loadcnt 0x0
; GFX1250-SDAG-TRUE16-NEXT: v_xor_b32_e32 v0, 0x80000000, v0
; GFX1250-SDAG-TRUE16-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 2, 2), 0 ; msbs: dst=0 src0=0 src1=0 src2=0
-; GFX1250-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1250-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX1250-SDAG-TRUE16-NEXT: v_cvt_f16_f32_e32 v0.l, v0
; GFX1250-SDAG-TRUE16-NEXT: buffer_store_b16 v0, off, s[4:7], null
; GFX1250-SDAG-TRUE16-NEXT: s_endpgm
@@ -4445,7 +4511,7 @@ define amdgpu_kernel void @fneg_fptrunc_f32_to_f16(
; GFX1250-SDAG-FAKE16-NEXT: s_wait_loadcnt 0x0
; GFX1250-SDAG-FAKE16-NEXT: v_xor_b32_e32 v0, 0x80000000, v0
; GFX1250-SDAG-FAKE16-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 2, 2), 0 ; msbs: dst=0 src0=0 src1=0 src2=0
-; GFX1250-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1250-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX1250-SDAG-FAKE16-NEXT: v_cvt_f16_f32_e32 v0, v0
; GFX1250-SDAG-FAKE16-NEXT: buffer_store_b16 v0, off, s[4:7], null
; GFX1250-SDAG-FAKE16-NEXT: s_endpgm
@@ -4464,8 +4530,8 @@ define amdgpu_kernel void @fneg_fptrunc_f32_to_f16(
; GFX1250-GISEL-TRUE16-NEXT: s_wait_kmcnt 0x0
; GFX1250-GISEL-TRUE16-NEXT: s_xor_b32 s2, s2, 0x80000000
; GFX1250-GISEL-TRUE16-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 2, 2), 0 ; msbs: dst=0 src0=0 src1=0 src2=0
+; GFX1250-GISEL-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_3)
; GFX1250-GISEL-TRUE16-NEXT: s_cvt_f16_f32 s2, s2
-; GFX1250-GISEL-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_3)
; GFX1250-GISEL-TRUE16-NEXT: v_mov_b16_e32 v0.l, s2
; GFX1250-GISEL-TRUE16-NEXT: s_mov_b32 s2, -1
; GFX1250-GISEL-TRUE16-NEXT: buffer_store_b16 v0, off, s[0:3], null
@@ -4485,8 +4551,8 @@ define amdgpu_kernel void @fneg_fptrunc_f32_to_f16(
; GFX1250-GISEL-FAKE16-NEXT: s_wait_kmcnt 0x0
; GFX1250-GISEL-FAKE16-NEXT: s_xor_b32 s2, s2, 0x80000000
; GFX1250-GISEL-FAKE16-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 2, 2), 0 ; msbs: dst=0 src0=0 src1=0 src2=0
+; GFX1250-GISEL-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_3)
; GFX1250-GISEL-FAKE16-NEXT: s_cvt_f16_f32 s2, s2
-; GFX1250-GISEL-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_3)
; GFX1250-GISEL-FAKE16-NEXT: v_mov_b32_e32 v0, s2
; GFX1250-GISEL-FAKE16-NEXT: s_mov_b32 s2, -1
; GFX1250-GISEL-FAKE16-NEXT: buffer_store_b16 v0, off, s[0:3], null
@@ -4702,7 +4768,7 @@ define amdgpu_kernel void @fabs_fptrunc_f32_to_f16(
; GFX1250-SDAG-TRUE16-NEXT: s_wait_loadcnt 0x0
; GFX1250-SDAG-TRUE16-NEXT: v_and_b32_e32 v0, 0x7fffffff, v0
; GFX1250-SDAG-TRUE16-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 2, 2), 0 ; msbs: dst=0 src0=0 src1=0 src2=0
-; GFX1250-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1250-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX1250-SDAG-TRUE16-NEXT: v_cvt_f16_f32_e32 v0.l, v0
; GFX1250-SDAG-TRUE16-NEXT: buffer_store_b16 v0, off, s[4:7], null
; GFX1250-SDAG-TRUE16-NEXT: s_endpgm
@@ -4727,7 +4793,7 @@ define amdgpu_kernel void @fabs_fptrunc_f32_to_f16(
; GFX1250-SDAG-FAKE16-NEXT: s_wait_loadcnt 0x0
; GFX1250-SDAG-FAKE16-NEXT: v_and_b32_e32 v0, 0x7fffffff, v0
; GFX1250-SDAG-FAKE16-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 2, 2), 0 ; msbs: dst=0 src0=0 src1=0 src2=0
-; GFX1250-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1250-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX1250-SDAG-FAKE16-NEXT: v_cvt_f16_f32_e32 v0, v0
; GFX1250-SDAG-FAKE16-NEXT: buffer_store_b16 v0, off, s[4:7], null
; GFX1250-SDAG-FAKE16-NEXT: s_endpgm
@@ -4746,8 +4812,8 @@ define amdgpu_kernel void @fabs_fptrunc_f32_to_f16(
; GFX1250-GISEL-TRUE16-NEXT: s_wait_kmcnt 0x0
; GFX1250-GISEL-TRUE16-NEXT: s_bitset0_b32 s2, 31
; GFX1250-GISEL-TRUE16-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 2, 2), 0 ; msbs: dst=0 src0=0 src1=0 src2=0
+; GFX1250-GISEL-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_3)
; GFX1250-GISEL-TRUE16-NEXT: s_cvt_f16_f32 s2, s2
-; GFX1250-GISEL-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_3)
; GFX1250-GISEL-TRUE16-NEXT: v_mov_b16_e32 v0.l, s2
; GFX1250-GISEL-TRUE16-NEXT: s_mov_b32 s2, -1
; GFX1250-GISEL-TRUE16-NEXT: buffer_store_b16 v0, off, s[0:3], null
@@ -4767,8 +4833,8 @@ define amdgpu_kernel void @fabs_fptrunc_f32_to_f16(
; GFX1250-GISEL-FAKE16-NEXT: s_wait_kmcnt 0x0
; GFX1250-GISEL-FAKE16-NEXT: s_bitset0_b32 s2, 31
; GFX1250-GISEL-FAKE16-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 2, 2), 0 ; msbs: dst=0 src0=0 src1=0 src2=0
+; GFX1250-GISEL-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_3)
; GFX1250-GISEL-FAKE16-NEXT: s_cvt_f16_f32 s2, s2
-; GFX1250-GISEL-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_3)
; GFX1250-GISEL-FAKE16-NEXT: v_mov_b32_e32 v0, s2
; GFX1250-GISEL-FAKE16-NEXT: s_mov_b32 s2, -1
; GFX1250-GISEL-FAKE16-NEXT: buffer_store_b16 v0, off, s[0:3], null
@@ -4984,7 +5050,7 @@ define amdgpu_kernel void @fneg_fabs_fptrunc_f32_to_f16(
; GFX1250-SDAG-TRUE16-NEXT: s_wait_loadcnt 0x0
; GFX1250-SDAG-TRUE16-NEXT: v_or_b32_e32 v0, 0x80000000, v0
; GFX1250-SDAG-TRUE16-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 2, 2), 0 ; msbs: dst=0 src0=0 src1=0 src2=0
-; GFX1250-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1250-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX1250-SDAG-TRUE16-NEXT: v_cvt_f16_f32_e32 v0.l, v0
; GFX1250-SDAG-TRUE16-NEXT: buffer_store_b16 v0, off, s[4:7], null
; GFX1250-SDAG-TRUE16-NEXT: s_endpgm
@@ -5009,7 +5075,7 @@ define amdgpu_kernel void @fneg_fabs_fptrunc_f32_to_f16(
; GFX1250-SDAG-FAKE16-NEXT: s_wait_loadcnt 0x0
; GFX1250-SDAG-FAKE16-NEXT: v_or_b32_e32 v0, 0x80000000, v0
; GFX1250-SDAG-FAKE16-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 2, 2), 0 ; msbs: dst=0 src0=0 src1=0 src2=0
-; GFX1250-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1250-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX1250-SDAG-FAKE16-NEXT: v_cvt_f16_f32_e32 v0, v0
; GFX1250-SDAG-FAKE16-NEXT: buffer_store_b16 v0, off, s[4:7], null
; GFX1250-SDAG-FAKE16-NEXT: s_endpgm
@@ -5028,8 +5094,8 @@ define amdgpu_kernel void @fneg_fabs_fptrunc_f32_to_f16(
; GFX1250-GISEL-TRUE16-NEXT: s_wait_kmcnt 0x0
; GFX1250-GISEL-TRUE16-NEXT: s_bitset1_b32 s2, 31
; GFX1250-GISEL-TRUE16-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 2, 2), 0 ; msbs: dst=0 src0=0 src1=0 src2=0
+; GFX1250-GISEL-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_3)
; GFX1250-GISEL-TRUE16-NEXT: s_cvt_f16_f32 s2, s2
-; GFX1250-GISEL-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_3)
; GFX1250-GISEL-TRUE16-NEXT: v_mov_b16_e32 v0.l, s2
; GFX1250-GISEL-TRUE16-NEXT: s_mov_b32 s2, -1
; GFX1250-GISEL-TRUE16-NEXT: buffer_store_b16 v0, off, s[0:3], null
@@ -5049,8 +5115,8 @@ define amdgpu_kernel void @fneg_fabs_fptrunc_f32_to_f16(
; GFX1250-GISEL-FAKE16-NEXT: s_wait_kmcnt 0x0
; GFX1250-GISEL-FAKE16-NEXT: s_bitset1_b32 s2, 31
; GFX1250-GISEL-FAKE16-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 2, 2), 0 ; msbs: dst=0 src0=0 src1=0 src2=0
+; GFX1250-GISEL-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_3)
; GFX1250-GISEL-FAKE16-NEXT: s_cvt_f16_f32 s2, s2
-; GFX1250-GISEL-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_3)
; GFX1250-GISEL-FAKE16-NEXT: v_mov_b32_e32 v0, s2
; GFX1250-GISEL-FAKE16-NEXT: s_mov_b32 s2, -1
; GFX1250-GISEL-FAKE16-NEXT: buffer_store_b16 v0, off, s[0:3], null
@@ -5294,8 +5360,8 @@ define amdgpu_kernel void @fptrunc_f32_to_f16_zext_i32(
; GFX1250-SDAG-TRUE16-NEXT: s_mov_b32 s5, s1
; GFX1250-SDAG-TRUE16-NEXT: s_wait_loadcnt 0x0
; GFX1250-SDAG-TRUE16-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 2, 2), 0 ; msbs: dst=0 src0=0 src1=0 src2=0
+; GFX1250-SDAG-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1250-SDAG-TRUE16-NEXT: v_cvt_f16_f32_e32 v0.l, v0
-; GFX1250-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1250-SDAG-TRUE16-NEXT: v_and_b32_e32 v0, 0xffff, v0
; GFX1250-SDAG-TRUE16-NEXT: buffer_store_b32 v0, off, s[4:7], null
; GFX1250-SDAG-TRUE16-NEXT: s_endpgm
@@ -5319,8 +5385,8 @@ define amdgpu_kernel void @fptrunc_f32_to_f16_zext_i32(
; GFX1250-SDAG-FAKE16-NEXT: s_mov_b32 s5, s1
; GFX1250-SDAG-FAKE16-NEXT: s_wait_loadcnt 0x0
; GFX1250-SDAG-FAKE16-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 2, 2), 0 ; msbs: dst=0 src0=0 src1=0 src2=0
+; GFX1250-SDAG-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1250-SDAG-FAKE16-NEXT: v_cvt_f16_f32_e32 v0, v0
-; GFX1250-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1250-SDAG-FAKE16-NEXT: v_and_b32_e32 v0, 0xffff, v0
; GFX1250-SDAG-FAKE16-NEXT: buffer_store_b32 v0, off, s[4:7], null
; GFX1250-SDAG-FAKE16-NEXT: s_endpgm
@@ -5338,9 +5404,10 @@ define amdgpu_kernel void @fptrunc_f32_to_f16_zext_i32(
; GFX1250-GISEL-TRUE16-NEXT: s_mov_b32 s3, 0x31016000
; GFX1250-GISEL-TRUE16-NEXT: s_wait_kmcnt 0x0
; GFX1250-GISEL-TRUE16-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 2, 2), 0 ; msbs: dst=0 src0=0 src1=0 src2=0
+; GFX1250-GISEL-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_3)
; GFX1250-GISEL-TRUE16-NEXT: s_cvt_f16_f32 s2, s2
-; GFX1250-GISEL-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_3) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1250-GISEL-TRUE16-NEXT: s_and_b32 s2, 0xffff, s2
+; GFX1250-GISEL-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1250-GISEL-TRUE16-NEXT: v_mov_b32_e32 v0, s2
; GFX1250-GISEL-TRUE16-NEXT: s_mov_b32 s2, -1
; GFX1250-GISEL-TRUE16-NEXT: buffer_store_b32 v0, off, s[0:3], null
@@ -5359,9 +5426,10 @@ define amdgpu_kernel void @fptrunc_f32_to_f16_zext_i32(
; GFX1250-GISEL-FAKE16-NEXT: s_mov_b32 s3, 0x31016000
; GFX1250-GISEL-FAKE16-NEXT: s_wait_kmcnt 0x0
; GFX1250-GISEL-FAKE16-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 2, 2), 0 ; msbs: dst=0 src0=0 src1=0 src2=0
+; GFX1250-GISEL-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_3)
; GFX1250-GISEL-FAKE16-NEXT: s_cvt_f16_f32 s2, s2
-; GFX1250-GISEL-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_3) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1250-GISEL-FAKE16-NEXT: s_and_b32 s2, s2, 0xffff
+; GFX1250-GISEL-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1250-GISEL-FAKE16-NEXT: v_mov_b32_e32 v0, s2
; GFX1250-GISEL-FAKE16-NEXT: s_mov_b32 s2, -1
; GFX1250-GISEL-FAKE16-NEXT: buffer_store_b32 v0, off, s[0:3], null
@@ -5606,8 +5674,9 @@ define amdgpu_kernel void @fptrunc_fabs_f32_to_f16_zext_i32(
; GFX1250-SDAG-TRUE16-NEXT: s_wait_loadcnt 0x0
; GFX1250-SDAG-TRUE16-NEXT: v_and_b32_e32 v0, 0x7fffffff, v0
; GFX1250-SDAG-TRUE16-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 2, 2), 0 ; msbs: dst=0 src0=0 src1=0 src2=0
-; GFX1250-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX1250-SDAG-TRUE16-NEXT: v_cvt_f16_f32_e32 v0.l, v0
+; GFX1250-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1250-SDAG-TRUE16-NEXT: v_and_b32_e32 v0, 0xffff, v0
; GFX1250-SDAG-TRUE16-NEXT: buffer_store_b32 v0, off, s[4:7], null
; GFX1250-SDAG-TRUE16-NEXT: s_endpgm
@@ -5632,8 +5701,9 @@ define amdgpu_kernel void @fptrunc_fabs_f32_to_f16_zext_i32(
; GFX1250-SDAG-FAKE16-NEXT: s_wait_loadcnt 0x0
; GFX1250-SDAG-FAKE16-NEXT: v_and_b32_e32 v0, 0x7fffffff, v0
; GFX1250-SDAG-FAKE16-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 2, 2), 0 ; msbs: dst=0 src0=0 src1=0 src2=0
-; GFX1250-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX1250-SDAG-FAKE16-NEXT: v_cvt_f16_f32_e32 v0, v0
+; GFX1250-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1250-SDAG-FAKE16-NEXT: v_and_b32_e32 v0, 0xffff, v0
; GFX1250-SDAG-FAKE16-NEXT: buffer_store_b32 v0, off, s[4:7], null
; GFX1250-SDAG-FAKE16-NEXT: s_endpgm
@@ -5652,9 +5722,10 @@ define amdgpu_kernel void @fptrunc_fabs_f32_to_f16_zext_i32(
; GFX1250-GISEL-TRUE16-NEXT: s_wait_kmcnt 0x0
; GFX1250-GISEL-TRUE16-NEXT: s_bitset0_b32 s2, 31
; GFX1250-GISEL-TRUE16-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 2, 2), 0 ; msbs: dst=0 src0=0 src1=0 src2=0
+; GFX1250-GISEL-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_3)
; GFX1250-GISEL-TRUE16-NEXT: s_cvt_f16_f32 s2, s2
-; GFX1250-GISEL-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_3) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1250-GISEL-TRUE16-NEXT: s_and_b32 s2, 0xffff, s2
+; GFX1250-GISEL-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1250-GISEL-TRUE16-NEXT: v_mov_b32_e32 v0, s2
; GFX1250-GISEL-TRUE16-NEXT: s_mov_b32 s2, -1
; GFX1250-GISEL-TRUE16-NEXT: buffer_store_b32 v0, off, s[0:3], null
@@ -5674,9 +5745,10 @@ define amdgpu_kernel void @fptrunc_fabs_f32_to_f16_zext_i32(
; GFX1250-GISEL-FAKE16-NEXT: s_wait_kmcnt 0x0
; GFX1250-GISEL-FAKE16-NEXT: s_bitset0_b32 s2, 31
; GFX1250-GISEL-FAKE16-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 2, 2), 0 ; msbs: dst=0 src0=0 src1=0 src2=0
+; GFX1250-GISEL-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_3)
; GFX1250-GISEL-FAKE16-NEXT: s_cvt_f16_f32 s2, s2
-; GFX1250-GISEL-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_3) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1250-GISEL-FAKE16-NEXT: s_and_b32 s2, s2, 0xffff
+; GFX1250-GISEL-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1250-GISEL-FAKE16-NEXT: v_mov_b32_e32 v0, s2
; GFX1250-GISEL-FAKE16-NEXT: s_mov_b32 s2, -1
; GFX1250-GISEL-FAKE16-NEXT: buffer_store_b32 v0, off, s[0:3], null
@@ -5925,8 +5997,8 @@ define amdgpu_kernel void @fptrunc_f32_to_f16_sext_i32(
; GFX1250-SDAG-TRUE16-NEXT: s_mov_b32 s5, s1
; GFX1250-SDAG-TRUE16-NEXT: s_wait_loadcnt 0x0
; GFX1250-SDAG-TRUE16-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 2, 2), 0 ; msbs: dst=0 src0=0 src1=0 src2=0
+; GFX1250-SDAG-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1250-SDAG-TRUE16-NEXT: v_cvt_f16_f32_e32 v0.l, v0
-; GFX1250-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1250-SDAG-TRUE16-NEXT: v_bfe_i32 v0, v0, 0, 16
; GFX1250-SDAG-TRUE16-NEXT: buffer_store_b32 v0, off, s[4:7], null
; GFX1250-SDAG-TRUE16-NEXT: s_endpgm
@@ -5950,8 +6022,8 @@ define amdgpu_kernel void @fptrunc_f32_to_f16_sext_i32(
; GFX1250-SDAG-FAKE16-NEXT: s_mov_b32 s5, s1
; GFX1250-SDAG-FAKE16-NEXT: s_wait_loadcnt 0x0
; GFX1250-SDAG-FAKE16-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 2, 2), 0 ; msbs: dst=0 src0=0 src1=0 src2=0
+; GFX1250-SDAG-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1250-SDAG-FAKE16-NEXT: v_cvt_f16_f32_e32 v0, v0
-; GFX1250-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1250-SDAG-FAKE16-NEXT: v_bfe_i32 v0, v0, 0, 16
; GFX1250-SDAG-FAKE16-NEXT: buffer_store_b32 v0, off, s[4:7], null
; GFX1250-SDAG-FAKE16-NEXT: s_endpgm
@@ -5969,9 +6041,10 @@ define amdgpu_kernel void @fptrunc_f32_to_f16_sext_i32(
; GFX1250-GISEL-TRUE16-NEXT: s_mov_b32 s3, 0x31016000
; GFX1250-GISEL-TRUE16-NEXT: s_wait_kmcnt 0x0
; GFX1250-GISEL-TRUE16-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 2, 2), 0 ; msbs: dst=0 src0=0 src1=0 src2=0
+; GFX1250-GISEL-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_3)
; GFX1250-GISEL-TRUE16-NEXT: s_cvt_f16_f32 s2, s2
-; GFX1250-GISEL-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_3) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1250-GISEL-TRUE16-NEXT: s_sext_i32_i16 s2, s2
+; GFX1250-GISEL-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1250-GISEL-TRUE16-NEXT: v_mov_b32_e32 v0, s2
; GFX1250-GISEL-TRUE16-NEXT: s_mov_b32 s2, -1
; GFX1250-GISEL-TRUE16-NEXT: buffer_store_b32 v0, off, s[0:3], null
@@ -5990,9 +6063,10 @@ define amdgpu_kernel void @fptrunc_f32_to_f16_sext_i32(
; GFX1250-GISEL-FAKE16-NEXT: s_mov_b32 s3, 0x31016000
; GFX1250-GISEL-FAKE16-NEXT: s_wait_kmcnt 0x0
; GFX1250-GISEL-FAKE16-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 2, 2), 0 ; msbs: dst=0 src0=0 src1=0 src2=0
+; GFX1250-GISEL-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_3)
; GFX1250-GISEL-FAKE16-NEXT: s_cvt_f16_f32 s2, s2
-; GFX1250-GISEL-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_3) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1250-GISEL-FAKE16-NEXT: s_sext_i32_i16 s2, s2
+; GFX1250-GISEL-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1250-GISEL-FAKE16-NEXT: v_mov_b32_e32 v0, s2
; GFX1250-GISEL-FAKE16-NEXT: s_mov_b32 s2, -1
; GFX1250-GISEL-FAKE16-NEXT: buffer_store_b32 v0, off, s[0:3], null
diff --git a/llvm/test/CodeGen/AMDGPU/fptrunc.ll b/llvm/test/CodeGen/AMDGPU/fptrunc.ll
index 685fc615812dce..9d8660ad89ac2e 100644
--- a/llvm/test/CodeGen/AMDGPU/fptrunc.ll
+++ b/llvm/test/CodeGen/AMDGPU/fptrunc.ll
@@ -443,24 +443,26 @@ define amdgpu_kernel void @fptrunc_f64_to_f16(ptr addrspace(1) %out, double %in)
; GFX11-SDAG-NEXT: v_readfirstlane_b32 s6, v0
; GFX11-SDAG-NEXT: s_lshr_b32 s7, s4, s6
; GFX11-SDAG-NEXT: s_lshl_b32 s6, s7, s6
-; GFX11-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_4) | instid1(SALU_CYCLE_1)
+; GFX11-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-SDAG-NEXT: s_cmp_lg_u32 s6, s4
; GFX11-SDAG-NEXT: s_cselect_b32 s4, 1, 0
; GFX11-SDAG-NEXT: s_addk_i32 s5, 0xfc10
; GFX11-SDAG-NEXT: s_or_b32 s4, s7, s4
; GFX11-SDAG-NEXT: s_lshl_b32 s6, s5, 12
+; GFX11-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX11-SDAG-NEXT: s_or_b32 s6, s2, s6
; GFX11-SDAG-NEXT: s_cmp_lt_i32 s5, 1
; GFX11-SDAG-NEXT: s_cselect_b32 s4, s4, s6
; GFX11-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-SDAG-NEXT: s_and_b32 s6, s4, 7
; GFX11-SDAG-NEXT: s_cmp_gt_i32 s6, 5
+; GFX11-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX11-SDAG-NEXT: s_cselect_b32 s7, 1, 0
; GFX11-SDAG-NEXT: s_cmp_eq_u32 s6, 3
; GFX11-SDAG-NEXT: s_cselect_b32 s6, 1, 0
; GFX11-SDAG-NEXT: s_lshr_b32 s4, s4, 2
; GFX11-SDAG-NEXT: s_or_b32 s6, s6, s7
-; GFX11-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_4) | instid1(SALU_CYCLE_1)
; GFX11-SDAG-NEXT: s_add_i32 s4, s4, s6
; GFX11-SDAG-NEXT: s_cmp_lt_i32 s5, 31
; GFX11-SDAG-NEXT: s_movk_i32 s6, 0x7e00
@@ -468,10 +470,11 @@ define amdgpu_kernel void @fptrunc_f64_to_f16(ptr addrspace(1) %out, double %in)
; GFX11-SDAG-NEXT: s_cmp_lg_u32 s2, 0
; GFX11-SDAG-NEXT: s_cselect_b32 s2, s6, 0x7c00
; GFX11-SDAG-NEXT: s_cmpk_eq_i32 s5, 0x40f
+; GFX11-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX11-SDAG-NEXT: s_cselect_b32 s2, s2, s4
; GFX11-SDAG-NEXT: s_lshr_b32 s3, s3, 16
-; GFX11-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-SDAG-NEXT: s_and_b32 s3, s3, 0x8000
+; GFX11-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-SDAG-NEXT: s_or_b32 s2, s3, s2
; GFX11-SDAG-NEXT: s_mov_b32 s3, 0x31016000
; GFX11-SDAG-NEXT: v_mov_b32_e32 v0, s2
@@ -504,24 +507,27 @@ define amdgpu_kernel void @fptrunc_f64_to_f16(ptr addrspace(1) %out, double %in)
; GFX11-SAFE-GISEL-NEXT: s_lshl_b32 s6, s9, s6
; GFX11-SAFE-GISEL-NEXT: s_or_b32 s5, s5, 0x7c00
; GFX11-SAFE-GISEL-NEXT: s_cmp_lg_u32 s6, s8
+; GFX11-SAFE-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-SAFE-GISEL-NEXT: s_cselect_b32 s6, 1, 0
-; GFX11-SAFE-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
; GFX11-SAFE-GISEL-NEXT: s_or_b32 s6, s9, s6
; GFX11-SAFE-GISEL-NEXT: s_cmp_lt_i32 s4, 1
+; GFX11-SAFE-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-SAFE-GISEL-NEXT: s_cselect_b32 s2, s6, s2
; GFX11-SAFE-GISEL-NEXT: s_and_b32 s6, s2, 7
; GFX11-SAFE-GISEL-NEXT: s_lshr_b32 s2, s2, 2
; GFX11-SAFE-GISEL-NEXT: s_cmp_eq_u32 s6, 3
+; GFX11-SAFE-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX11-SAFE-GISEL-NEXT: s_cselect_b32 s7, 1, 0
; GFX11-SAFE-GISEL-NEXT: s_cmp_gt_i32 s6, 5
; GFX11-SAFE-GISEL-NEXT: s_cselect_b32 s6, 1, 0
; GFX11-SAFE-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-SAFE-GISEL-NEXT: s_or_b32 s6, s7, s6
; GFX11-SAFE-GISEL-NEXT: s_cmp_lg_u32 s6, 0
+; GFX11-SAFE-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-SAFE-GISEL-NEXT: s_cselect_b32 s6, 1, 0
-; GFX11-SAFE-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-SAFE-GISEL-NEXT: s_add_i32 s2, s2, s6
; GFX11-SAFE-GISEL-NEXT: s_cmp_gt_i32 s4, 30
+; GFX11-SAFE-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX11-SAFE-GISEL-NEXT: s_cselect_b32 s2, 0x7c00, s2
; GFX11-SAFE-GISEL-NEXT: s_cmpk_eq_i32 s4, 0x40f
; GFX11-SAFE-GISEL-NEXT: s_cselect_b32 s2, s5, s2
@@ -560,24 +566,27 @@ define amdgpu_kernel void @fptrunc_f64_to_f16(ptr addrspace(1) %out, double %in)
; GFX11-UNSAFE-GISEL-TRUE16-NEXT: s_lshl_b32 s6, s9, s6
; GFX11-UNSAFE-GISEL-TRUE16-NEXT: s_or_b32 s5, s5, 0x7c00
; GFX11-UNSAFE-GISEL-TRUE16-NEXT: s_cmp_lg_u32 s6, s8
+; GFX11-UNSAFE-GISEL-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-UNSAFE-GISEL-TRUE16-NEXT: s_cselect_b32 s6, 1, 0
-; GFX11-UNSAFE-GISEL-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
; GFX11-UNSAFE-GISEL-TRUE16-NEXT: s_or_b32 s6, s9, s6
; GFX11-UNSAFE-GISEL-TRUE16-NEXT: s_cmp_lt_i32 s4, 1
+; GFX11-UNSAFE-GISEL-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-UNSAFE-GISEL-TRUE16-NEXT: s_cselect_b32 s2, s6, s2
; GFX11-UNSAFE-GISEL-TRUE16-NEXT: s_and_b32 s6, s2, 7
; GFX11-UNSAFE-GISEL-TRUE16-NEXT: s_lshr_b32 s2, s2, 2
; GFX11-UNSAFE-GISEL-TRUE16-NEXT: s_cmp_eq_u32 s6, 3
+; GFX11-UNSAFE-GISEL-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX11-UNSAFE-GISEL-TRUE16-NEXT: s_cselect_b32 s7, 1, 0
; GFX11-UNSAFE-GISEL-TRUE16-NEXT: s_cmp_gt_i32 s6, 5
; GFX11-UNSAFE-GISEL-TRUE16-NEXT: s_cselect_b32 s6, 1, 0
; GFX11-UNSAFE-GISEL-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-UNSAFE-GISEL-TRUE16-NEXT: s_or_b32 s6, s7, s6
; GFX11-UNSAFE-GISEL-TRUE16-NEXT: s_cmp_lg_u32 s6, 0
+; GFX11-UNSAFE-GISEL-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-UNSAFE-GISEL-TRUE16-NEXT: s_cselect_b32 s6, 1, 0
-; GFX11-UNSAFE-GISEL-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-UNSAFE-GISEL-TRUE16-NEXT: s_add_i32 s2, s2, s6
; GFX11-UNSAFE-GISEL-TRUE16-NEXT: s_cmp_gt_i32 s4, 30
+; GFX11-UNSAFE-GISEL-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX11-UNSAFE-GISEL-TRUE16-NEXT: s_cselect_b32 s2, 0x7c00, s2
; GFX11-UNSAFE-GISEL-TRUE16-NEXT: s_cmpk_eq_i32 s4, 0x40f
; GFX11-UNSAFE-GISEL-TRUE16-NEXT: s_cselect_b32 s2, s5, s2
@@ -616,24 +625,27 @@ define amdgpu_kernel void @fptrunc_f64_to_f16(ptr addrspace(1) %out, double %in)
; GFX11-UNSAFE-GISEL-FAKE16-NEXT: s_lshl_b32 s6, s9, s6
; GFX11-UNSAFE-GISEL-FAKE16-NEXT: s_or_b32 s5, s5, 0x7c00
; GFX11-UNSAFE-GISEL-FAKE16-NEXT: s_cmp_lg_u32 s6, s8
+; GFX11-UNSAFE-GISEL-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-UNSAFE-GISEL-FAKE16-NEXT: s_cselect_b32 s6, 1, 0
-; GFX11-UNSAFE-GISEL-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
; GFX11-UNSAFE-GISEL-FAKE16-NEXT: s_or_b32 s6, s9, s6
; GFX11-UNSAFE-GISEL-FAKE16-NEXT: s_cmp_lt_i32 s4, 1
+; GFX11-UNSAFE-GISEL-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-UNSAFE-GISEL-FAKE16-NEXT: s_cselect_b32 s2, s6, s2
; GFX11-UNSAFE-GISEL-FAKE16-NEXT: s_and_b32 s6, s2, 7
; GFX11-UNSAFE-GISEL-FAKE16-NEXT: s_lshr_b32 s2, s2, 2
; GFX11-UNSAFE-GISEL-FAKE16-NEXT: s_cmp_eq_u32 s6, 3
+; GFX11-UNSAFE-GISEL-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX11-UNSAFE-GISEL-FAKE16-NEXT: s_cselect_b32 s7, 1, 0
; GFX11-UNSAFE-GISEL-FAKE16-NEXT: s_cmp_gt_i32 s6, 5
; GFX11-UNSAFE-GISEL-FAKE16-NEXT: s_cselect_b32 s6, 1, 0
; GFX11-UNSAFE-GISEL-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-UNSAFE-GISEL-FAKE16-NEXT: s_or_b32 s6, s7, s6
; GFX11-UNSAFE-GISEL-FAKE16-NEXT: s_cmp_lg_u32 s6, 0
+; GFX11-UNSAFE-GISEL-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-UNSAFE-GISEL-FAKE16-NEXT: s_cselect_b32 s6, 1, 0
-; GFX11-UNSAFE-GISEL-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-UNSAFE-GISEL-FAKE16-NEXT: s_add_i32 s2, s2, s6
; GFX11-UNSAFE-GISEL-FAKE16-NEXT: s_cmp_gt_i32 s4, 30
+; GFX11-UNSAFE-GISEL-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX11-UNSAFE-GISEL-FAKE16-NEXT: s_cselect_b32 s2, 0x7c00, s2
; GFX11-UNSAFE-GISEL-FAKE16-NEXT: s_cmpk_eq_i32 s4, 0x40f
; GFX11-UNSAFE-GISEL-FAKE16-NEXT: s_cselect_b32 s2, s5, s2
diff --git a/llvm/test/CodeGen/AMDGPU/fract-match.ll b/llvm/test/CodeGen/AMDGPU/fract-match.ll
index b40763b4ca5f4a..0dd8e1d301b2f6 100644
--- a/llvm/test/CodeGen/AMDGPU/fract-match.ll
+++ b/llvm/test/CodeGen/AMDGPU/fract-match.ll
@@ -66,6 +66,7 @@ define float @safe_math_fract_f32(float %x, ptr addrspace(1) writeonly captures(
; GFX6-NEXT: buffer_store_dword v3, v[1:2], s[4:7], 0 addr64
; GFX6-NEXT: s_waitcnt vmcnt(0) expcnt(0)
; GFX6-NEXT: s_setpc_b64 s[30:31]
+;
; GFX7-LABEL: safe_math_fract_f32:
; GFX7: ; %bb.0: ; %entry
; GFX7-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
@@ -81,6 +82,7 @@ define float @safe_math_fract_f32(float %x, ptr addrspace(1) writeonly captures(
; GFX7-NEXT: buffer_store_dword v3, v[1:2], s[4:7], 0 addr64
; GFX7-NEXT: s_waitcnt vmcnt(0)
; GFX7-NEXT: s_setpc_b64 s[30:31]
+;
; GFX8-LABEL: safe_math_fract_f32:
; GFX8: ; %bb.0: ; %entry
; GFX8-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
@@ -92,6 +94,7 @@ define float @safe_math_fract_f32(float %x, ptr addrspace(1) writeonly captures(
; GFX8-NEXT: global_store_dword v[1:2], v3, off
; GFX8-NEXT: s_waitcnt vmcnt(0)
; GFX8-NEXT: s_setpc_b64 s[30:31]
+;
; GFX11-LABEL: safe_math_fract_f32:
; GFX11: ; %bb.0: ; %entry
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
@@ -102,6 +105,7 @@ define float @safe_math_fract_f32(float %x, ptr addrspace(1) writeonly captures(
; GFX11-NEXT: v_cndmask_b32_e32 v0, 0, v3, vcc_lo
; GFX11-NEXT: global_store_b32 v[1:2], v4, off
; GFX11-NEXT: s_setpc_b64 s[30:31]
+;
; GFX12-LABEL: safe_math_fract_f32:
; GFX12: ; %bb.0: ; %entry
; GFX12-NEXT: s_wait_loadcnt_dscnt 0x0
@@ -248,7 +252,6 @@ define <3 x float> @safe_math_fract_v3f32(<3 x float> %x, ptr addrspace(1) write
; GFX6-IR-NEXT: [[COND6:%.*]] = select <3 x i1> [[CMPINF]], <3 x float> <float 0.000000e+00, float poison, float 0.000000e+00>, <3 x float> [[COND]]
; GFX6-IR-NEXT: store <3 x float> [[FLOOR]], ptr addrspace(1) [[IP]], align 4
; GFX6-IR-NEXT: ret <3 x float> [[COND6]]
-;
; IR-FRACT-LABEL: define <3 x float> @safe_math_fract_v3f32(
; IR-FRACT-SAME: <3 x float> [[X:%.*]], ptr addrspace(1) writeonly captures(none) [[IP:%.*]]) {
; IR-FRACT-NEXT: [[FLOOR:%.*]] = tail call <3 x float> @llvm.floor.v3f32(<3 x float> [[X]])
@@ -266,7 +269,6 @@ define <3 x float> @safe_math_fract_v3f32(<3 x float> %x, ptr addrspace(1) write
; IR-FRACT-NEXT: [[COND6:%.*]] = select <3 x i1> [[CMPINF]], <3 x float> <float 0.000000e+00, float poison, float 0.000000e+00>, <3 x float> [[COND]]
; IR-FRACT-NEXT: store <3 x float> [[FLOOR]], ptr addrspace(1) [[IP]], align 4
; IR-FRACT-NEXT: ret <3 x float> [[COND6]]
-;
%floor = tail call <3 x float> @llvm.floor.v3f32(<3 x float> %x)
%sub = fsub <3 x float> %x, %floor
%min = tail call <3 x float> @llvm.minnum.v3f32(<3 x float> %sub, <3 x float> <float 0x3FEFFFFFE0000000, float poison, float 0x3FEFFFFFE0000000>)
@@ -316,6 +318,7 @@ define <2 x float> @safe_math_fract_v2f32_const_splat_poison(<2 x float> %x, ptr
; GFX6-NEXT: buffer_store_dwordx2 v[4:5], v[2:3], s[4:7], 0 addr64
; GFX6-NEXT: s_waitcnt vmcnt(0) expcnt(0)
; GFX6-NEXT: s_setpc_b64 s[30:31]
+;
; GFX7-LABEL: safe_math_fract_v2f32_const_splat_poison:
; GFX7: ; %bb.0:
; GFX7-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
@@ -339,6 +342,7 @@ define <2 x float> @safe_math_fract_v2f32_const_splat_poison(<2 x float> %x, ptr
; GFX7-NEXT: buffer_store_dwordx2 v[4:5], v[2:3], s[4:7], 0 addr64
; GFX7-NEXT: s_waitcnt vmcnt(0)
; GFX7-NEXT: s_setpc_b64 s[30:31]
+;
; GFX8-LABEL: safe_math_fract_v2f32_const_splat_poison:
; GFX8: ; %bb.0:
; GFX8-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
@@ -358,6 +362,7 @@ define <2 x float> @safe_math_fract_v2f32_const_splat_poison(<2 x float> %x, ptr
; GFX8-NEXT: global_store_dwordx2 v[2:3], v[4:5], off
; GFX8-NEXT: s_waitcnt vmcnt(0)
; GFX8-NEXT: s_setpc_b64 s[30:31]
+;
; GFX11-LABEL: safe_math_fract_v2f32_const_splat_poison:
; GFX11: ; %bb.0:
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
@@ -376,6 +381,7 @@ define <2 x float> @safe_math_fract_v2f32_const_splat_poison(<2 x float> %x, ptr
; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_cndmask_b32_e64 v1, v7, 0, s0
; GFX11-NEXT: s_setpc_b64 s[30:31]
+;
; GFX12-LABEL: safe_math_fract_v2f32_const_splat_poison:
; GFX12: ; %bb.0:
; GFX12-NEXT: s_wait_loadcnt_dscnt 0x0
@@ -458,6 +464,7 @@ define float @safe_math_fract_f32_swap(float %x, ptr addrspace(1) writeonly capt
; GFX6-NEXT: buffer_store_dword v3, v[1:2], s[4:7], 0 addr64
; GFX6-NEXT: s_waitcnt vmcnt(0) expcnt(0)
; GFX6-NEXT: s_setpc_b64 s[30:31]
+;
; GFX7-LABEL: safe_math_fract_f32_swap:
; GFX7: ; %bb.0: ; %entry
; GFX7-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
@@ -473,6 +480,7 @@ define float @safe_math_fract_f32_swap(float %x, ptr addrspace(1) writeonly capt
; GFX7-NEXT: buffer_store_dword v3, v[1:2], s[4:7], 0 addr64
; GFX7-NEXT: s_waitcnt vmcnt(0)
; GFX7-NEXT: s_setpc_b64 s[30:31]
+;
; GFX8-LABEL: safe_math_fract_f32_swap:
; GFX8: ; %bb.0: ; %entry
; GFX8-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
@@ -484,6 +492,7 @@ define float @safe_math_fract_f32_swap(float %x, ptr addrspace(1) writeonly capt
; GFX8-NEXT: global_store_dword v[1:2], v3, off
; GFX8-NEXT: s_waitcnt vmcnt(0)
; GFX8-NEXT: s_setpc_b64 s[30:31]
+;
; GFX11-LABEL: safe_math_fract_f32_swap:
; GFX11: ; %bb.0: ; %entry
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
@@ -494,6 +503,7 @@ define float @safe_math_fract_f32_swap(float %x, ptr addrspace(1) writeonly capt
; GFX11-NEXT: v_cndmask_b32_e32 v0, 0, v3, vcc_lo
; GFX11-NEXT: global_store_b32 v[1:2], v4, off
; GFX11-NEXT: s_setpc_b64 s[30:31]
+;
; GFX12-LABEL: safe_math_fract_f32_swap:
; GFX12: ; %bb.0: ; %entry
; GFX12-NEXT: s_wait_loadcnt_dscnt 0x0
@@ -557,6 +567,7 @@ define float @safe_math_fract_f32_noinf_check(float %x, ptr addrspace(1) writeon
; GFX6-NEXT: buffer_store_dword v3, v[1:2], s[4:7], 0 addr64
; GFX6-NEXT: s_waitcnt vmcnt(0) expcnt(0)
; GFX6-NEXT: s_setpc_b64 s[30:31]
+;
; GFX7-LABEL: safe_math_fract_f32_noinf_check:
; GFX7: ; %bb.0: ; %entry
; GFX7-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
@@ -569,6 +580,7 @@ define float @safe_math_fract_f32_noinf_check(float %x, ptr addrspace(1) writeon
; GFX7-NEXT: buffer_store_dword v3, v[1:2], s[4:7], 0 addr64
; GFX7-NEXT: s_waitcnt vmcnt(0)
; GFX7-NEXT: s_setpc_b64 s[30:31]
+;
; GFX8-LABEL: safe_math_fract_f32_noinf_check:
; GFX8: ; %bb.0: ; %entry
; GFX8-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
@@ -577,6 +589,7 @@ define float @safe_math_fract_f32_noinf_check(float %x, ptr addrspace(1) writeon
; GFX8-NEXT: global_store_dword v[1:2], v3, off
; GFX8-NEXT: s_waitcnt vmcnt(0)
; GFX8-NEXT: s_setpc_b64 s[30:31]
+;
; GFX11-LABEL: safe_math_fract_f32_noinf_check:
; GFX11: ; %bb.0: ; %entry
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
@@ -584,6 +597,7 @@ define float @safe_math_fract_f32_noinf_check(float %x, ptr addrspace(1) writeon
; GFX11-NEXT: v_fract_f32_e32 v0, v0
; GFX11-NEXT: global_store_b32 v[1:2], v3, off
; GFX11-NEXT: s_setpc_b64 s[30:31]
+;
; GFX12-LABEL: safe_math_fract_f32_noinf_check:
; GFX12: ; %bb.0: ; %entry
; GFX12-NEXT: s_wait_loadcnt_dscnt 0x0
@@ -635,6 +649,7 @@ define float @no_nan_check_math_fract_f32(float %x, ptr addrspace(1) writeonly c
; GFX6-NEXT: buffer_store_dword v3, v[1:2], s[4:7], 0 addr64
; GFX6-NEXT: s_waitcnt vmcnt(0) expcnt(0)
; GFX6-NEXT: s_setpc_b64 s[30:31]
+;
; GFX7-LABEL: no_nan_check_math_fract_f32:
; GFX7: ; %bb.0: ; %entry
; GFX7-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
@@ -651,6 +666,7 @@ define float @no_nan_check_math_fract_f32(float %x, ptr addrspace(1) writeonly c
; GFX7-NEXT: buffer_store_dword v3, v[1:2], s[4:7], 0 addr64
; GFX7-NEXT: s_waitcnt vmcnt(0)
; GFX7-NEXT: s_setpc_b64 s[30:31]
+;
; GFX8-LABEL: no_nan_check_math_fract_f32:
; GFX8: ; %bb.0: ; %entry
; GFX8-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
@@ -663,6 +679,7 @@ define float @no_nan_check_math_fract_f32(float %x, ptr addrspace(1) writeonly c
; GFX8-NEXT: global_store_dword v[1:2], v3, off
; GFX8-NEXT: s_waitcnt vmcnt(0)
; GFX8-NEXT: s_setpc_b64 s[30:31]
+;
; GFX11-LABEL: no_nan_check_math_fract_f32:
; GFX11: ; %bb.0: ; %entry
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
@@ -674,6 +691,7 @@ define float @no_nan_check_math_fract_f32(float %x, ptr addrspace(1) writeonly c
; GFX11-NEXT: v_min_f32_e32 v4, 0x3f7fffff, v4
; GFX11-NEXT: v_cndmask_b32_e32 v0, 0, v4, vcc_lo
; GFX11-NEXT: s_setpc_b64 s[30:31]
+;
; GFX12-LABEL: no_nan_check_math_fract_f32:
; GFX12: ; %bb.0: ; %entry
; GFX12-NEXT: s_wait_loadcnt_dscnt 0x0
@@ -723,21 +741,25 @@ define float @basic_fract_f32_nonans(float nofpclass(nan) %x) {
; GFX6-NEXT: v_sub_f32_e32 v0, v0, v1
; GFX6-NEXT: v_min_f32_e32 v0, 0x3f7fffff, v0
; GFX6-NEXT: s_setpc_b64 s[30:31]
+;
; GFX7-LABEL: basic_fract_f32_nonans:
; GFX7: ; %bb.0: ; %entry
; GFX7-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX7-NEXT: v_fract_f32_e32 v0, v0
; GFX7-NEXT: s_setpc_b64 s[30:31]
+;
; GFX8-LABEL: basic_fract_f32_nonans:
; GFX8: ; %bb.0: ; %entry
; GFX8-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX8-NEXT: v_fract_f32_e32 v0, v0
; GFX8-NEXT: s_setpc_b64 s[30:31]
+;
; GFX11-LABEL: basic_fract_f32_nonans:
; GFX11: ; %bb.0: ; %entry
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: v_fract_f32_e32 v0, v0
; GFX11-NEXT: s_setpc_b64 s[30:31]
+;
; GFX12-LABEL: basic_fract_f32_nonans:
; GFX12: ; %bb.0: ; %entry
; GFX12-NEXT: s_wait_loadcnt_dscnt 0x0
@@ -770,6 +792,7 @@ define float @basic_fract_f32_flags_minnum(float %x) {
; GFX6-NEXT: v_sub_f32_e32 v0, v0, v1
; GFX6-NEXT: v_min_f32_e32 v0, 0x3f7fffff, v0
; GFX6-NEXT: s_setpc_b64 s[30:31]
+;
; GFX7-LABEL: basic_fract_f32_flags_minnum:
; GFX7: ; %bb.0: ; %entry
; GFX7-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
@@ -777,6 +800,7 @@ define float @basic_fract_f32_flags_minnum(float %x) {
; GFX7-NEXT: v_sub_f32_e32 v0, v0, v1
; GFX7-NEXT: v_min_f32_e32 v0, 0x3f7fffff, v0
; GFX7-NEXT: s_setpc_b64 s[30:31]
+;
; GFX8-LABEL: basic_fract_f32_flags_minnum:
; GFX8: ; %bb.0: ; %entry
; GFX8-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
@@ -784,6 +808,7 @@ define float @basic_fract_f32_flags_minnum(float %x) {
; GFX8-NEXT: v_sub_f32_e32 v0, v0, v1
; GFX8-NEXT: v_min_f32_e32 v0, 0x3f7fffff, v0
; GFX8-NEXT: s_setpc_b64 s[30:31]
+;
; GFX11-LABEL: basic_fract_f32_flags_minnum:
; GFX11: ; %bb.0: ; %entry
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
@@ -792,6 +817,7 @@ define float @basic_fract_f32_flags_minnum(float %x) {
; GFX11-NEXT: v_sub_f32_e32 v0, v0, v1
; GFX11-NEXT: v_min_f32_e32 v0, 0x3f7fffff, v0
; GFX11-NEXT: s_setpc_b64 s[30:31]
+;
; GFX12-LABEL: basic_fract_f32_flags_minnum:
; GFX12: ; %bb.0: ; %entry
; GFX12-NEXT: s_wait_loadcnt_dscnt 0x0
@@ -833,21 +859,25 @@ define float @basic_fract_f32_flags_fsub(float nofpclass(nan) %x) {
; GFX6-NEXT: v_sub_f32_e32 v0, v0, v1
; GFX6-NEXT: v_min_f32_e32 v0, 0x3f7fffff, v0
; GFX6-NEXT: s_setpc_b64 s[30:31]
+;
; GFX7-LABEL: basic_fract_f32_flags_fsub:
; GFX7: ; %bb.0: ; %entry
; GFX7-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX7-NEXT: v_fract_f32_e32 v0, v0
; GFX7-NEXT: s_setpc_b64 s[30:31]
+;
; GFX8-LABEL: basic_fract_f32_flags_fsub:
; GFX8: ; %bb.0: ; %entry
; GFX8-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX8-NEXT: v_fract_f32_e32 v0, v0
; GFX8-NEXT: s_setpc_b64 s[30:31]
+;
; GFX11-LABEL: basic_fract_f32_flags_fsub:
; GFX11: ; %bb.0: ; %entry
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: v_fract_f32_e32 v0, v0
; GFX11-NEXT: s_setpc_b64 s[30:31]
+;
; GFX12-LABEL: basic_fract_f32_flags_fsub:
; GFX12: ; %bb.0: ; %entry
; GFX12-NEXT: s_wait_loadcnt_dscnt 0x0
@@ -894,24 +924,28 @@ define <2 x float> @basic_fract_v2f32_nonans(<2 x float> nofpclass(nan) %x) {
; GFX6-NEXT: v_min_f32_e32 v0, 0x3f7fffff, v0
; GFX6-NEXT: v_min_f32_e32 v1, 0x3f7fffff, v1
; GFX6-NEXT: s_setpc_b64 s[30:31]
+;
; GFX7-LABEL: basic_fract_v2f32_nonans:
; GFX7: ; %bb.0: ; %entry
; GFX7-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX7-NEXT: v_fract_f32_e32 v0, v0
; GFX7-NEXT: v_fract_f32_e32 v1, v1
; GFX7-NEXT: s_setpc_b64 s[30:31]
+;
; GFX8-LABEL: basic_fract_v2f32_nonans:
; GFX8: ; %bb.0: ; %entry
; GFX8-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX8-NEXT: v_fract_f32_e32 v0, v0
; GFX8-NEXT: v_fract_f32_e32 v1, v1
; GFX8-NEXT: s_setpc_b64 s[30:31]
+;
; GFX11-LABEL: basic_fract_v2f32_nonans:
; GFX11: ; %bb.0: ; %entry
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: v_fract_f32_e32 v0, v0
; GFX11-NEXT: v_fract_f32_e32 v1, v1
; GFX11-NEXT: s_setpc_b64 s[30:31]
+;
; GFX12-LABEL: basic_fract_v2f32_nonans:
; GFX12: ; %bb.0: ; %entry
; GFX12-NEXT: s_wait_loadcnt_dscnt 0x0
@@ -961,6 +995,7 @@ define float @basic_fract_f32_multi_use_fsub_nonans(float nofpclass(nan) %x, ptr
; GFX6-NEXT: buffer_store_dword v3, v[1:2], s[4:7], 0 addr64
; GFX6-NEXT: s_waitcnt vmcnt(0) expcnt(0)
; GFX6-NEXT: s_setpc_b64 s[30:31]
+;
; GFX7-LABEL: basic_fract_f32_multi_use_fsub_nonans:
; GFX7: ; %bb.0: ; %entry
; GFX7-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
@@ -974,6 +1009,7 @@ define float @basic_fract_f32_multi_use_fsub_nonans(float nofpclass(nan) %x, ptr
; GFX7-NEXT: buffer_store_dword v3, v[1:2], s[4:7], 0 addr64
; GFX7-NEXT: s_waitcnt vmcnt(0)
; GFX7-NEXT: s_setpc_b64 s[30:31]
+;
; GFX8-LABEL: basic_fract_f32_multi_use_fsub_nonans:
; GFX8: ; %bb.0: ; %entry
; GFX8-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
@@ -983,6 +1019,7 @@ define float @basic_fract_f32_multi_use_fsub_nonans(float nofpclass(nan) %x, ptr
; GFX8-NEXT: global_store_dword v[1:2], v3, off
; GFX8-NEXT: s_waitcnt vmcnt(0)
; GFX8-NEXT: s_setpc_b64 s[30:31]
+;
; GFX11-LABEL: basic_fract_f32_multi_use_fsub_nonans:
; GFX11: ; %bb.0: ; %entry
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
@@ -992,6 +1029,7 @@ define float @basic_fract_f32_multi_use_fsub_nonans(float nofpclass(nan) %x, ptr
; GFX11-NEXT: v_fract_f32_e32 v0, v0
; GFX11-NEXT: global_store_b32 v[1:2], v3, off
; GFX11-NEXT: s_setpc_b64 s[30:31]
+;
; GFX12-LABEL: basic_fract_f32_multi_use_fsub_nonans:
; GFX12: ; %bb.0: ; %entry
; GFX12-NEXT: s_wait_loadcnt_dscnt 0x0
@@ -1035,21 +1073,25 @@ define float @nnan_minnum_fract_f32(float %x) {
; GFX6-NEXT: v_sub_f32_e32 v0, v0, v1
; GFX6-NEXT: v_min_f32_e32 v0, 0x3f7fffff, v0
; GFX6-NEXT: s_setpc_b64 s[30:31]
+;
; GFX7-LABEL: nnan_minnum_fract_f32:
; GFX7: ; %bb.0: ; %entry
; GFX7-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX7-NEXT: v_fract_f32_e32 v0, v0
; GFX7-NEXT: s_setpc_b64 s[30:31]
+;
; GFX8-LABEL: nnan_minnum_fract_f32:
; GFX8: ; %bb.0: ; %entry
; GFX8-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX8-NEXT: v_fract_f32_e32 v0, v0
; GFX8-NEXT: s_setpc_b64 s[30:31]
+;
; GFX11-LABEL: nnan_minnum_fract_f32:
; GFX11: ; %bb.0: ; %entry
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: v_fract_f32_e32 v0, v0
; GFX11-NEXT: s_setpc_b64 s[30:31]
+;
; GFX12-LABEL: nnan_minnum_fract_f32:
; GFX12: ; %bb.0: ; %entry
; GFX12-NEXT: s_wait_loadcnt_dscnt 0x0
@@ -1084,6 +1126,7 @@ define float @nnan_fsub_fract_f32(float %x) {
; GFX6-NEXT: v_sub_f32_e32 v0, v0, v1
; GFX6-NEXT: v_min_f32_e32 v0, 0x3f7fffff, v0
; GFX6-NEXT: s_setpc_b64 s[30:31]
+;
; GFX7-LABEL: nnan_fsub_fract_f32:
; GFX7: ; %bb.0: ; %entry
; GFX7-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
@@ -1091,6 +1134,7 @@ define float @nnan_fsub_fract_f32(float %x) {
; GFX7-NEXT: v_sub_f32_e32 v0, v0, v1
; GFX7-NEXT: v_min_f32_e32 v0, 0x3f7fffff, v0
; GFX7-NEXT: s_setpc_b64 s[30:31]
+;
; GFX8-LABEL: nnan_fsub_fract_f32:
; GFX8: ; %bb.0: ; %entry
; GFX8-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
@@ -1098,6 +1142,7 @@ define float @nnan_fsub_fract_f32(float %x) {
; GFX8-NEXT: v_sub_f32_e32 v0, v0, v1
; GFX8-NEXT: v_min_f32_e32 v0, 0x3f7fffff, v0
; GFX8-NEXT: s_setpc_b64 s[30:31]
+;
; GFX11-LABEL: nnan_fsub_fract_f32:
; GFX11: ; %bb.0: ; %entry
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
@@ -1106,6 +1151,7 @@ define float @nnan_fsub_fract_f32(float %x) {
; GFX11-NEXT: v_sub_f32_e32 v0, v0, v1
; GFX11-NEXT: v_min_f32_e32 v0, 0x3f7fffff, v0
; GFX11-NEXT: s_setpc_b64 s[30:31]
+;
; GFX12-LABEL: nnan_fsub_fract_f32:
; GFX12: ; %bb.0: ; %entry
; GFX12-NEXT: s_wait_loadcnt_dscnt 0x0
@@ -1141,6 +1187,7 @@ define float @nnan_floor_fract_f32(float %x) {
; GFX6-NEXT: v_sub_f32_e32 v0, v0, v1
; GFX6-NEXT: v_min_f32_e32 v0, 0x3f7fffff, v0
; GFX6-NEXT: s_setpc_b64 s[30:31]
+;
; GFX7-LABEL: nnan_floor_fract_f32:
; GFX7: ; %bb.0: ; %entry
; GFX7-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
@@ -1148,6 +1195,7 @@ define float @nnan_floor_fract_f32(float %x) {
; GFX7-NEXT: v_sub_f32_e32 v0, v0, v1
; GFX7-NEXT: v_min_f32_e32 v0, 0x3f7fffff, v0
; GFX7-NEXT: s_setpc_b64 s[30:31]
+;
; GFX8-LABEL: nnan_floor_fract_f32:
; GFX8: ; %bb.0: ; %entry
; GFX8-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
@@ -1155,6 +1203,7 @@ define float @nnan_floor_fract_f32(float %x) {
; GFX8-NEXT: v_sub_f32_e32 v0, v0, v1
; GFX8-NEXT: v_min_f32_e32 v0, 0x3f7fffff, v0
; GFX8-NEXT: s_setpc_b64 s[30:31]
+;
; GFX11-LABEL: nnan_floor_fract_f32:
; GFX11: ; %bb.0: ; %entry
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
@@ -1163,6 +1212,7 @@ define float @nnan_floor_fract_f32(float %x) {
; GFX11-NEXT: v_sub_f32_e32 v0, v0, v1
; GFX11-NEXT: v_min_f32_e32 v0, 0x3f7fffff, v0
; GFX11-NEXT: s_setpc_b64 s[30:31]
+;
; GFX12-LABEL: nnan_floor_fract_f32:
; GFX12: ; %bb.0: ; %entry
; GFX12-NEXT: s_wait_loadcnt_dscnt 0x0
@@ -1204,21 +1254,25 @@ define float @nnan_src_fract_f32(float nofpclass(nan) %x) {
; GFX6-NEXT: v_sub_f32_e32 v0, v0, v1
; GFX6-NEXT: v_min_f32_e32 v0, 0x3f7fffff, v0
; GFX6-NEXT: s_setpc_b64 s[30:31]
+;
; GFX7-LABEL: nnan_src_fract_f32:
; GFX7: ; %bb.0: ; %entry
; GFX7-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX7-NEXT: v_fract_f32_e32 v0, v0
; GFX7-NEXT: s_setpc_b64 s[30:31]
+;
; GFX8-LABEL: nnan_src_fract_f32:
; GFX8: ; %bb.0: ; %entry
; GFX8-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX8-NEXT: v_fract_f32_e32 v0, v0
; GFX8-NEXT: s_setpc_b64 s[30:31]
+;
; GFX11-LABEL: nnan_src_fract_f32:
; GFX11: ; %bb.0: ; %entry
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: v_fract_f32_e32 v0, v0
; GFX11-NEXT: s_setpc_b64 s[30:31]
+;
; GFX12-LABEL: nnan_src_fract_f32:
; GFX12: ; %bb.0: ; %entry
; GFX12-NEXT: s_wait_loadcnt_dscnt 0x0
@@ -1252,6 +1306,7 @@ define float @not_fract_f32_wrong_const(float nofpclass(nan) %x) {
; GFX6-NEXT: v_sub_f32_e32 v0, v0, v1
; GFX6-NEXT: v_min_f32_e32 v0, 0x3f7ffffe, v0
; GFX6-NEXT: s_setpc_b64 s[30:31]
+;
; GFX7-LABEL: not_fract_f32_wrong_const:
; GFX7: ; %bb.0: ; %entry
; GFX7-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
@@ -1259,6 +1314,7 @@ define float @not_fract_f32_wrong_const(float nofpclass(nan) %x) {
; GFX7-NEXT: v_sub_f32_e32 v0, v0, v1
; GFX7-NEXT: v_min_f32_e32 v0, 0x3f7ffffe, v0
; GFX7-NEXT: s_setpc_b64 s[30:31]
+;
; GFX8-LABEL: not_fract_f32_wrong_const:
; GFX8: ; %bb.0: ; %entry
; GFX8-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
@@ -1266,6 +1322,7 @@ define float @not_fract_f32_wrong_const(float nofpclass(nan) %x) {
; GFX8-NEXT: v_sub_f32_e32 v0, v0, v1
; GFX8-NEXT: v_min_f32_e32 v0, 0x3f7ffffe, v0
; GFX8-NEXT: s_setpc_b64 s[30:31]
+;
; GFX11-LABEL: not_fract_f32_wrong_const:
; GFX11: ; %bb.0: ; %entry
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
@@ -1274,6 +1331,7 @@ define float @not_fract_f32_wrong_const(float nofpclass(nan) %x) {
; GFX11-NEXT: v_sub_f32_e32 v0, v0, v1
; GFX11-NEXT: v_min_f32_e32 v0, 0x3f7ffffe, v0
; GFX11-NEXT: s_setpc_b64 s[30:31]
+;
; GFX12-LABEL: not_fract_f32_wrong_const:
; GFX12: ; %bb.0: ; %entry
; GFX12-NEXT: s_wait_loadcnt_dscnt 0x0
@@ -1310,6 +1368,7 @@ define float @not_fract_f32_swapped_fsub(float nofpclass(nan) %x) {
; GFX6-NEXT: v_sub_f32_e32 v0, v1, v0
; GFX6-NEXT: v_min_f32_e32 v0, 0x3f7fffff, v0
; GFX6-NEXT: s_setpc_b64 s[30:31]
+;
; GFX7-LABEL: not_fract_f32_swapped_fsub:
; GFX7: ; %bb.0: ; %entry
; GFX7-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
@@ -1317,6 +1376,7 @@ define float @not_fract_f32_swapped_fsub(float nofpclass(nan) %x) {
; GFX7-NEXT: v_sub_f32_e32 v0, v1, v0
; GFX7-NEXT: v_min_f32_e32 v0, 0x3f7fffff, v0
; GFX7-NEXT: s_setpc_b64 s[30:31]
+;
; GFX8-LABEL: not_fract_f32_swapped_fsub:
; GFX8: ; %bb.0: ; %entry
; GFX8-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
@@ -1324,6 +1384,7 @@ define float @not_fract_f32_swapped_fsub(float nofpclass(nan) %x) {
; GFX8-NEXT: v_sub_f32_e32 v0, v1, v0
; GFX8-NEXT: v_min_f32_e32 v0, 0x3f7fffff, v0
; GFX8-NEXT: s_setpc_b64 s[30:31]
+;
; GFX11-LABEL: not_fract_f32_swapped_fsub:
; GFX11: ; %bb.0: ; %entry
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
@@ -1332,6 +1393,7 @@ define float @not_fract_f32_swapped_fsub(float nofpclass(nan) %x) {
; GFX11-NEXT: v_sub_f32_e32 v0, v1, v0
; GFX11-NEXT: v_min_f32_e32 v0, 0x3f7fffff, v0
; GFX11-NEXT: s_setpc_b64 s[30:31]
+;
; GFX12-LABEL: not_fract_f32_swapped_fsub:
; GFX12: ; %bb.0: ; %entry
; GFX12-NEXT: s_wait_loadcnt_dscnt 0x0
@@ -1368,6 +1430,7 @@ define float @not_fract_f32_not_floor(float nofpclass(nan) %x) {
; GFX6-NEXT: v_sub_f32_e32 v0, v0, v1
; GFX6-NEXT: v_min_f32_e32 v0, 0x3f7fffff, v0
; GFX6-NEXT: s_setpc_b64 s[30:31]
+;
; GFX7-LABEL: not_fract_f32_not_floor:
; GFX7: ; %bb.0: ; %entry
; GFX7-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
@@ -1375,6 +1438,7 @@ define float @not_fract_f32_not_floor(float nofpclass(nan) %x) {
; GFX7-NEXT: v_sub_f32_e32 v0, v0, v1
; GFX7-NEXT: v_min_f32_e32 v0, 0x3f7fffff, v0
; GFX7-NEXT: s_setpc_b64 s[30:31]
+;
; GFX8-LABEL: not_fract_f32_not_floor:
; GFX8: ; %bb.0: ; %entry
; GFX8-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
@@ -1382,6 +1446,7 @@ define float @not_fract_f32_not_floor(float nofpclass(nan) %x) {
; GFX8-NEXT: v_sub_f32_e32 v0, v0, v1
; GFX8-NEXT: v_min_f32_e32 v0, 0x3f7fffff, v0
; GFX8-NEXT: s_setpc_b64 s[30:31]
+;
; GFX11-LABEL: not_fract_f32_not_floor:
; GFX11: ; %bb.0: ; %entry
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
@@ -1390,6 +1455,7 @@ define float @not_fract_f32_not_floor(float nofpclass(nan) %x) {
; GFX11-NEXT: v_sub_f32_e32 v0, v0, v1
; GFX11-NEXT: v_min_f32_e32 v0, 0x3f7fffff, v0
; GFX11-NEXT: s_setpc_b64 s[30:31]
+;
; GFX12-LABEL: not_fract_f32_not_floor:
; GFX12: ; %bb.0: ; %entry
; GFX12-NEXT: s_wait_loadcnt_dscnt 0x0
@@ -1426,6 +1492,7 @@ define float @not_fract_f32_different_floor(float %x, float %y) {
; GFX6-NEXT: v_sub_f32_e32 v0, v0, v1
; GFX6-NEXT: v_min_f32_e32 v0, 0x3f7fffff, v0
; GFX6-NEXT: s_setpc_b64 s[30:31]
+;
; GFX7-LABEL: not_fract_f32_different_floor:
; GFX7: ; %bb.0: ; %entry
; GFX7-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
@@ -1433,6 +1500,7 @@ define float @not_fract_f32_different_floor(float %x, float %y) {
; GFX7-NEXT: v_sub_f32_e32 v0, v0, v1
; GFX7-NEXT: v_min_f32_e32 v0, 0x3f7fffff, v0
; GFX7-NEXT: s_setpc_b64 s[30:31]
+;
; GFX8-LABEL: not_fract_f32_different_floor:
; GFX8: ; %bb.0: ; %entry
; GFX8-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
@@ -1440,6 +1508,7 @@ define float @not_fract_f32_different_floor(float %x, float %y) {
; GFX8-NEXT: v_sub_f32_e32 v0, v0, v1
; GFX8-NEXT: v_min_f32_e32 v0, 0x3f7fffff, v0
; GFX8-NEXT: s_setpc_b64 s[30:31]
+;
; GFX11-LABEL: not_fract_f32_different_floor:
; GFX11: ; %bb.0: ; %entry
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
@@ -1448,6 +1517,7 @@ define float @not_fract_f32_different_floor(float %x, float %y) {
; GFX11-NEXT: v_sub_f32_e32 v0, v0, v1
; GFX11-NEXT: v_min_f32_e32 v0, 0x3f7fffff, v0
; GFX11-NEXT: s_setpc_b64 s[30:31]
+;
; GFX12-LABEL: not_fract_f32_different_floor:
; GFX12: ; %bb.0: ; %entry
; GFX12-NEXT: s_wait_loadcnt_dscnt 0x0
@@ -1484,6 +1554,7 @@ define float @not_fract_f32_maxnum(float nofpclass(nan) %x) {
; GFX6-NEXT: v_sub_f32_e32 v0, v0, v1
; GFX6-NEXT: v_max_f32_e32 v0, 0x3f7fffff, v0
; GFX6-NEXT: s_setpc_b64 s[30:31]
+;
; GFX7-LABEL: not_fract_f32_maxnum:
; GFX7: ; %bb.0: ; %entry
; GFX7-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
@@ -1491,6 +1562,7 @@ define float @not_fract_f32_maxnum(float nofpclass(nan) %x) {
; GFX7-NEXT: v_sub_f32_e32 v0, v0, v1
; GFX7-NEXT: v_max_f32_e32 v0, 0x3f7fffff, v0
; GFX7-NEXT: s_setpc_b64 s[30:31]
+;
; GFX8-LABEL: not_fract_f32_maxnum:
; GFX8: ; %bb.0: ; %entry
; GFX8-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
@@ -1498,6 +1570,7 @@ define float @not_fract_f32_maxnum(float nofpclass(nan) %x) {
; GFX8-NEXT: v_sub_f32_e32 v0, v0, v1
; GFX8-NEXT: v_max_f32_e32 v0, 0x3f7fffff, v0
; GFX8-NEXT: s_setpc_b64 s[30:31]
+;
; GFX11-LABEL: not_fract_f32_maxnum:
; GFX11: ; %bb.0: ; %entry
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
@@ -1506,6 +1579,7 @@ define float @not_fract_f32_maxnum(float nofpclass(nan) %x) {
; GFX11-NEXT: v_sub_f32_e32 v0, v0, v1
; GFX11-NEXT: v_max_f32_e32 v0, 0x3f7fffff, v0
; GFX11-NEXT: s_setpc_b64 s[30:31]
+;
; GFX12-LABEL: not_fract_f32_maxnum:
; GFX12: ; %bb.0: ; %entry
; GFX12-NEXT: s_wait_loadcnt_dscnt 0x0
@@ -1540,6 +1614,7 @@ define float @fcmp_uno_check_is_nan_f32(float %x) {
; GCN: ; %bb.0: ; %entry
; GCN-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GCN-NEXT: s_setpc_b64 s[30:31]
+;
; GFX12-LABEL: fcmp_uno_check_is_nan_f32:
; GFX12: ; %bb.0: ; %entry
; GFX12-NEXT: s_wait_loadcnt_dscnt 0x0
@@ -1584,21 +1659,25 @@ define float @select_nan_fract_f32(float %x) {
; GFX6-NEXT: v_cmp_u_f32_e32 vcc, v0, v0
; GFX6-NEXT: v_cndmask_b32_e32 v0, v1, v0, vcc
; GFX6-NEXT: s_setpc_b64 s[30:31]
+;
; GFX7-LABEL: select_nan_fract_f32:
; GFX7: ; %bb.0: ; %entry
; GFX7-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX7-NEXT: v_fract_f32_e32 v0, v0
; GFX7-NEXT: s_setpc_b64 s[30:31]
+;
; GFX8-LABEL: select_nan_fract_f32:
; GFX8: ; %bb.0: ; %entry
; GFX8-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX8-NEXT: v_fract_f32_e32 v0, v0
; GFX8-NEXT: s_setpc_b64 s[30:31]
+;
; GFX11-LABEL: select_nan_fract_f32:
; GFX11: ; %bb.0: ; %entry
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: v_fract_f32_e32 v0, v0
; GFX11-NEXT: s_setpc_b64 s[30:31]
+;
; GFX12-LABEL: select_nan_fract_f32:
; GFX12: ; %bb.0: ; %entry
; GFX12-NEXT: s_wait_loadcnt_dscnt 0x0
@@ -1643,21 +1722,25 @@ define float @commuted_select_nan_fract_f32(float %x) {
; GFX6-NEXT: v_cmp_o_f32_e32 vcc, v0, v0
; GFX6-NEXT: v_cndmask_b32_e32 v0, v0, v1, vcc
; GFX6-NEXT: s_setpc_b64 s[30:31]
+;
; GFX7-LABEL: commuted_select_nan_fract_f32:
; GFX7: ; %bb.0: ; %entry
; GFX7-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX7-NEXT: v_fract_f32_e32 v0, v0
; GFX7-NEXT: s_setpc_b64 s[30:31]
+;
; GFX8-LABEL: commuted_select_nan_fract_f32:
; GFX8: ; %bb.0: ; %entry
; GFX8-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX8-NEXT: v_fract_f32_e32 v0, v0
; GFX8-NEXT: s_setpc_b64 s[30:31]
+;
; GFX11-LABEL: commuted_select_nan_fract_f32:
; GFX11: ; %bb.0: ; %entry
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: v_fract_f32_e32 v0, v0
; GFX11-NEXT: s_setpc_b64 s[30:31]
+;
; GFX12-LABEL: commuted_select_nan_fract_f32:
; GFX12: ; %bb.0: ; %entry
; GFX12-NEXT: s_wait_loadcnt_dscnt 0x0
@@ -1696,6 +1779,7 @@ define float @wrong_commuted_nan_select_f32(float %x) {
; GFX6-NEXT: v_cmp_u_f32_e32 vcc, v0, v0
; GFX6-NEXT: v_cndmask_b32_e32 v0, v0, v1, vcc
; GFX6-NEXT: s_setpc_b64 s[30:31]
+;
; GFX7-LABEL: wrong_commuted_nan_select_f32:
; GFX7: ; %bb.0: ; %entry
; GFX7-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
@@ -1705,6 +1789,7 @@ define float @wrong_commuted_nan_select_f32(float %x) {
; GFX7-NEXT: v_cmp_u_f32_e32 vcc, v0, v0
; GFX7-NEXT: v_cndmask_b32_e32 v0, v0, v1, vcc
; GFX7-NEXT: s_setpc_b64 s[30:31]
+;
; GFX8-LABEL: wrong_commuted_nan_select_f32:
; GFX8: ; %bb.0: ; %entry
; GFX8-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
@@ -1714,6 +1799,7 @@ define float @wrong_commuted_nan_select_f32(float %x) {
; GFX8-NEXT: v_cmp_u_f32_e32 vcc, v0, v0
; GFX8-NEXT: v_cndmask_b32_e32 v0, v0, v1, vcc
; GFX8-NEXT: s_setpc_b64 s[30:31]
+;
; GFX11-LABEL: wrong_commuted_nan_select_f32:
; GFX11: ; %bb.0: ; %entry
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
@@ -1725,6 +1811,7 @@ define float @wrong_commuted_nan_select_f32(float %x) {
; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_cndmask_b32_e32 v0, v0, v1, vcc_lo
; GFX11-NEXT: s_setpc_b64 s[30:31]
+;
; GFX12-LABEL: wrong_commuted_nan_select_f32:
; GFX12: ; %bb.0: ; %entry
; GFX12-NEXT: s_wait_loadcnt_dscnt 0x0
@@ -1786,6 +1873,7 @@ define half @basic_fract_f16_nonan(half nofpclass(nan) %x) {
; GFX6-NEXT: v_min_f32_e32 v0, 0x3f7fe000, v0
; GFX6-NEXT: v_cvt_f16_f32_e32 v0, v0
; GFX6-NEXT: s_setpc_b64 s[30:31]
+;
; GFX7-LABEL: basic_fract_f16_nonan:
; GFX7: ; %bb.0: ; %entry
; GFX7-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
@@ -1799,21 +1887,25 @@ define half @basic_fract_f16_nonan(half nofpclass(nan) %x) {
; GFX7-NEXT: v_min_f32_e32 v0, 0x3f7fe000, v0
; GFX7-NEXT: v_cvt_f16_f32_e32 v0, v0
; GFX7-NEXT: s_setpc_b64 s[30:31]
+;
; GFX8-LABEL: basic_fract_f16_nonan:
; GFX8: ; %bb.0: ; %entry
; GFX8-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX8-NEXT: v_fract_f16_e32 v0, v0
; GFX8-NEXT: s_setpc_b64 s[30:31]
+;
; GFX11-TRUE16-LABEL: basic_fract_f16_nonan:
; GFX11-TRUE16: ; %bb.0: ; %entry
; GFX11-TRUE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-TRUE16-NEXT: v_fract_f16_e32 v0.l, v0.l
; GFX11-TRUE16-NEXT: s_setpc_b64 s[30:31]
+;
; GFX11-FAKE16-LABEL: basic_fract_f16_nonan:
; GFX11-FAKE16: ; %bb.0: ; %entry
; GFX11-FAKE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-FAKE16-NEXT: v_fract_f16_e32 v0, v0
; GFX11-FAKE16-NEXT: s_setpc_b64 s[30:31]
+;
; GFX12-TRUE16-LABEL: basic_fract_f16_nonan:
; GFX12-TRUE16: ; %bb.0: ; %entry
; GFX12-TRUE16-NEXT: s_wait_loadcnt_dscnt 0x0
@@ -1823,6 +1915,7 @@ define half @basic_fract_f16_nonan(half nofpclass(nan) %x) {
; GFX12-TRUE16-NEXT: s_wait_kmcnt 0x0
; GFX12-TRUE16-NEXT: v_fract_f16_e32 v0.l, v0.l
; GFX12-TRUE16-NEXT: s_setpc_b64 s[30:31]
+;
; GFX12-FAKE16-LABEL: basic_fract_f16_nonan:
; GFX12-FAKE16: ; %bb.0: ; %entry
; GFX12-FAKE16-NEXT: s_wait_loadcnt_dscnt 0x0
@@ -1892,6 +1985,7 @@ define <2 x half> @basic_fract_v2f16_nonan(<2 x half> nofpclass(nan) %x) {
; GFX6-NEXT: v_lshlrev_b32_e32 v1, 16, v1
; GFX6-NEXT: v_or_b32_e32 v0, v0, v1
; GFX6-NEXT: s_setpc_b64 s[30:31]
+;
; GFX7-LABEL: basic_fract_v2f16_nonan:
; GFX7: ; %bb.0: ; %entry
; GFX7-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
@@ -1917,6 +2011,7 @@ define <2 x half> @basic_fract_v2f16_nonan(<2 x half> nofpclass(nan) %x) {
; GFX7-NEXT: v_lshlrev_b32_e32 v1, 16, v1
; GFX7-NEXT: v_or_b32_e32 v0, v0, v1
; GFX7-NEXT: s_setpc_b64 s[30:31]
+;
; GFX8-LABEL: basic_fract_v2f16_nonan:
; GFX8: ; %bb.0: ; %entry
; GFX8-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
@@ -1924,6 +2019,7 @@ define <2 x half> @basic_fract_v2f16_nonan(<2 x half> nofpclass(nan) %x) {
; GFX8-NEXT: v_fract_f16_sdwa v0, v0 dst_sel:DWORD dst_unused:UNUSED_PAD src0_sel:WORD_1
; GFX8-NEXT: v_pack_b32_f16 v0, v1, v0
; GFX8-NEXT: s_setpc_b64 s[30:31]
+;
; GFX11-TRUE16-LABEL: basic_fract_v2f16_nonan:
; GFX11-TRUE16: ; %bb.0: ; %entry
; GFX11-TRUE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
@@ -1932,6 +2028,7 @@ define <2 x half> @basic_fract_v2f16_nonan(<2 x half> nofpclass(nan) %x) {
; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX11-TRUE16-NEXT: v_fract_f16_e32 v0.h, v1.l
; GFX11-TRUE16-NEXT: s_setpc_b64 s[30:31]
+;
; GFX11-FAKE16-LABEL: basic_fract_v2f16_nonan:
; GFX11-FAKE16: ; %bb.0: ; %entry
; GFX11-FAKE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
@@ -1941,6 +2038,7 @@ define <2 x half> @basic_fract_v2f16_nonan(<2 x half> nofpclass(nan) %x) {
; GFX11-FAKE16-NEXT: v_fract_f16_e32 v1, v1
; GFX11-FAKE16-NEXT: v_pack_b32_f16 v0, v0, v1
; GFX11-FAKE16-NEXT: s_setpc_b64 s[30:31]
+;
; GFX12-TRUE16-LABEL: basic_fract_v2f16_nonan:
; GFX12-TRUE16: ; %bb.0: ; %entry
; GFX12-TRUE16-NEXT: s_wait_loadcnt_dscnt 0x0
@@ -1953,6 +2051,7 @@ define <2 x half> @basic_fract_v2f16_nonan(<2 x half> nofpclass(nan) %x) {
; GFX12-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX12-TRUE16-NEXT: v_fract_f16_e32 v0.h, v1.l
; GFX12-TRUE16-NEXT: s_setpc_b64 s[30:31]
+;
; GFX12-FAKE16-LABEL: basic_fract_v2f16_nonan:
; GFX12-FAKE16: ; %bb.0: ; %entry
; GFX12-FAKE16-NEXT: s_wait_loadcnt_dscnt 0x0
@@ -2004,21 +2103,25 @@ define double @basic_fract_f64_nanans(double nofpclass(nan) %x) {
; GFX6-NEXT: v_add_f64 v[0:1], v[0:1], -v[2:3]
; GFX6-NEXT: v_min_f64 v[0:1], v[0:1], s[4:5]
; GFX6-NEXT: s_setpc_b64 s[30:31]
+;
; GFX7-LABEL: basic_fract_f64_nanans:
; GFX7: ; %bb.0: ; %entry
; GFX7-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX7-NEXT: v_fract_f64_e32 v[0:1], v[0:1]
; GFX7-NEXT: s_setpc_b64 s[30:31]
+;
; GFX8-LABEL: basic_fract_f64_nanans:
; GFX8: ; %bb.0: ; %entry
; GFX8-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX8-NEXT: v_fract_f64_e32 v[0:1], v[0:1]
; GFX8-NEXT: s_setpc_b64 s[30:31]
+;
; GFX11-LABEL: basic_fract_f64_nanans:
; GFX11: ; %bb.0: ; %entry
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: v_fract_f64_e32 v[0:1], v[0:1]
; GFX11-NEXT: s_setpc_b64 s[30:31]
+;
; GFX12-LABEL: basic_fract_f64_nanans:
; GFX12: ; %bb.0: ; %entry
; GFX12-NEXT: s_wait_loadcnt_dscnt 0x0
@@ -2087,6 +2190,7 @@ define half @safe_math_fract_f16_noinf_check(half %x, ptr addrspace(1) writeonly
; GFX6-NEXT: v_cndmask_b32_e32 v0, v5, v0, vcc
; GFX6-NEXT: s_waitcnt vmcnt(0) expcnt(0)
; GFX6-NEXT: s_setpc_b64 s[30:31]
+;
; GFX7-LABEL: safe_math_fract_f16_noinf_check:
; GFX7: ; %bb.0: ; %entry
; GFX7-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
@@ -2108,6 +2212,7 @@ define half @safe_math_fract_f16_noinf_check(half %x, ptr addrspace(1) writeonly
; GFX7-NEXT: v_cndmask_b32_e32 v0, v5, v0, vcc
; GFX7-NEXT: s_waitcnt vmcnt(0)
; GFX7-NEXT: s_setpc_b64 s[30:31]
+;
; GFX8-LABEL: safe_math_fract_f16_noinf_check:
; GFX8: ; %bb.0: ; %entry
; GFX8-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
@@ -2116,6 +2221,7 @@ define half @safe_math_fract_f16_noinf_check(half %x, ptr addrspace(1) writeonly
; GFX8-NEXT: global_store_short v[1:2], v3, off
; GFX8-NEXT: s_waitcnt vmcnt(0)
; GFX8-NEXT: s_setpc_b64 s[30:31]
+;
; GFX11-TRUE16-LABEL: safe_math_fract_f16_noinf_check:
; GFX11-TRUE16: ; %bb.0: ; %entry
; GFX11-TRUE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
@@ -2123,6 +2229,7 @@ define half @safe_math_fract_f16_noinf_check(half %x, ptr addrspace(1) writeonly
; GFX11-TRUE16-NEXT: v_fract_f16_e32 v0.l, v0.l
; GFX11-TRUE16-NEXT: global_store_d16_hi_b16 v[1:2], v0, off
; GFX11-TRUE16-NEXT: s_setpc_b64 s[30:31]
+;
; GFX11-FAKE16-LABEL: safe_math_fract_f16_noinf_check:
; GFX11-FAKE16: ; %bb.0: ; %entry
; GFX11-FAKE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
@@ -2130,6 +2237,7 @@ define half @safe_math_fract_f16_noinf_check(half %x, ptr addrspace(1) writeonly
; GFX11-FAKE16-NEXT: v_fract_f16_e32 v0, v0
; GFX11-FAKE16-NEXT: global_store_b16 v[1:2], v3, off
; GFX11-FAKE16-NEXT: s_setpc_b64 s[30:31]
+;
; GFX12-TRUE16-LABEL: safe_math_fract_f16_noinf_check:
; GFX12-TRUE16: ; %bb.0: ; %entry
; GFX12-TRUE16-NEXT: s_wait_loadcnt_dscnt 0x0
@@ -2141,6 +2249,7 @@ define half @safe_math_fract_f16_noinf_check(half %x, ptr addrspace(1) writeonly
; GFX12-TRUE16-NEXT: v_fract_f16_e32 v0.l, v0.l
; GFX12-TRUE16-NEXT: global_store_d16_hi_b16 v[1:2], v0, off
; GFX12-TRUE16-NEXT: s_setpc_b64 s[30:31]
+;
; GFX12-FAKE16-LABEL: safe_math_fract_f16_noinf_check:
; GFX12-FAKE16: ; %bb.0: ; %entry
; GFX12-FAKE16-NEXT: s_wait_loadcnt_dscnt 0x0
@@ -2207,6 +2316,7 @@ define double @safe_math_fract_f64_noinf_check(double %x, ptr addrspace(1) write
; GFX6-NEXT: buffer_store_dwordx2 v[4:5], v[2:3], s[4:7], 0 addr64
; GFX6-NEXT: s_waitcnt vmcnt(0) expcnt(0)
; GFX6-NEXT: s_setpc_b64 s[30:31]
+;
; GFX7-LABEL: safe_math_fract_f64_noinf_check:
; GFX7: ; %bb.0: ; %entry
; GFX7-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
@@ -2219,6 +2329,7 @@ define double @safe_math_fract_f64_noinf_check(double %x, ptr addrspace(1) write
; GFX7-NEXT: buffer_store_dwordx2 v[4:5], v[2:3], s[4:7], 0 addr64
; GFX7-NEXT: s_waitcnt vmcnt(0)
; GFX7-NEXT: s_setpc_b64 s[30:31]
+;
; GFX8-LABEL: safe_math_fract_f64_noinf_check:
; GFX8: ; %bb.0: ; %entry
; GFX8-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
@@ -2227,6 +2338,7 @@ define double @safe_math_fract_f64_noinf_check(double %x, ptr addrspace(1) write
; GFX8-NEXT: global_store_dwordx2 v[2:3], v[4:5], off
; GFX8-NEXT: s_waitcnt vmcnt(0)
; GFX8-NEXT: s_setpc_b64 s[30:31]
+;
; GFX11-LABEL: safe_math_fract_f64_noinf_check:
; GFX11: ; %bb.0: ; %entry
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
@@ -2234,6 +2346,7 @@ define double @safe_math_fract_f64_noinf_check(double %x, ptr addrspace(1) write
; GFX11-NEXT: v_fract_f64_e32 v[0:1], v[0:1]
; GFX11-NEXT: global_store_b64 v[2:3], v[4:5], off
; GFX11-NEXT: s_setpc_b64 s[30:31]
+;
; GFX12-LABEL: safe_math_fract_f64_noinf_check:
; GFX12: ; %bb.0: ; %entry
; GFX12-NEXT: s_wait_loadcnt_dscnt 0x0
@@ -2281,21 +2394,25 @@ define float @select_nan_fract_f32_flags_select(float %x) {
; GFX6-NEXT: v_cmp_u_f32_e32 vcc, v0, v0
; GFX6-NEXT: v_cndmask_b32_e32 v0, v1, v0, vcc
; GFX6-NEXT: s_setpc_b64 s[30:31]
+;
; GFX7-LABEL: select_nan_fract_f32_flags_select:
; GFX7: ; %bb.0: ; %entry
; GFX7-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX7-NEXT: v_fract_f32_e32 v0, v0
; GFX7-NEXT: s_setpc_b64 s[30:31]
+;
; GFX8-LABEL: select_nan_fract_f32_flags_select:
; GFX8: ; %bb.0: ; %entry
; GFX8-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX8-NEXT: v_fract_f32_e32 v0, v0
; GFX8-NEXT: s_setpc_b64 s[30:31]
+;
; GFX11-LABEL: select_nan_fract_f32_flags_select:
; GFX11: ; %bb.0: ; %entry
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: v_fract_f32_e32 v0, v0
; GFX11-NEXT: s_setpc_b64 s[30:31]
+;
; GFX12-LABEL: select_nan_fract_f32_flags_select:
; GFX12: ; %bb.0: ; %entry
; GFX12-NEXT: s_wait_loadcnt_dscnt 0x0
@@ -2340,21 +2457,25 @@ define float @select_nan_fract_f32_flags_minnum(float %x) {
; GFX6-NEXT: v_cmp_u_f32_e32 vcc, v0, v0
; GFX6-NEXT: v_cndmask_b32_e32 v0, v1, v0, vcc
; GFX6-NEXT: s_setpc_b64 s[30:31]
+;
; GFX7-LABEL: select_nan_fract_f32_flags_minnum:
; GFX7: ; %bb.0: ; %entry
; GFX7-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX7-NEXT: v_fract_f32_e32 v0, v0
; GFX7-NEXT: s_setpc_b64 s[30:31]
+;
; GFX8-LABEL: select_nan_fract_f32_flags_minnum:
; GFX8: ; %bb.0: ; %entry
; GFX8-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX8-NEXT: v_fract_f32_e32 v0, v0
; GFX8-NEXT: s_setpc_b64 s[30:31]
+;
; GFX11-LABEL: select_nan_fract_f32_flags_minnum:
; GFX11: ; %bb.0: ; %entry
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: v_fract_f32_e32 v0, v0
; GFX11-NEXT: s_setpc_b64 s[30:31]
+;
; GFX12-LABEL: select_nan_fract_f32_flags_minnum:
; GFX12: ; %bb.0: ; %entry
; GFX12-NEXT: s_wait_loadcnt_dscnt 0x0
@@ -2429,6 +2550,7 @@ define <2 x float> @safe_math_fract_v2f32(<2 x float> %x, ptr addrspace(1) write
; GFX6-NEXT: buffer_store_dwordx2 v[4:5], v[2:3], s[4:7], 0 addr64
; GFX6-NEXT: s_waitcnt vmcnt(0) expcnt(0)
; GFX6-NEXT: s_setpc_b64 s[30:31]
+;
; GFX7-LABEL: safe_math_fract_v2f32:
; GFX7: ; %bb.0: ; %entry
; GFX7-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
@@ -2448,6 +2570,7 @@ define <2 x float> @safe_math_fract_v2f32(<2 x float> %x, ptr addrspace(1) write
; GFX7-NEXT: buffer_store_dwordx2 v[4:5], v[2:3], s[4:7], 0 addr64
; GFX7-NEXT: s_waitcnt vmcnt(0)
; GFX7-NEXT: s_setpc_b64 s[30:31]
+;
; GFX8-LABEL: safe_math_fract_v2f32:
; GFX8: ; %bb.0: ; %entry
; GFX8-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
@@ -2463,6 +2586,7 @@ define <2 x float> @safe_math_fract_v2f32(<2 x float> %x, ptr addrspace(1) write
; GFX8-NEXT: global_store_dwordx2 v[2:3], v[4:5], off
; GFX8-NEXT: s_waitcnt vmcnt(0)
; GFX8-NEXT: s_setpc_b64 s[30:31]
+;
; GFX11-LABEL: safe_math_fract_v2f32:
; GFX11: ; %bb.0: ; %entry
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
@@ -2477,6 +2601,7 @@ define <2 x float> @safe_math_fract_v2f32(<2 x float> %x, ptr addrspace(1) write
; GFX11-NEXT: global_store_b64 v[2:3], v[4:5], off
; GFX11-NEXT: v_cndmask_b32_e64 v1, v7, 0, s0
; GFX11-NEXT: s_setpc_b64 s[30:31]
+;
; GFX12-LABEL: safe_math_fract_v2f32:
; GFX12: ; %bb.0: ; %entry
; GFX12-NEXT: s_wait_loadcnt_dscnt 0x0
@@ -2565,6 +2690,7 @@ define double @safe_math_fract_f64(double %x, ptr addrspace(1) writeonly capture
; GFX6-NEXT: buffer_store_dwordx2 v[4:5], v[2:3], s[4:7], 0 addr64
; GFX6-NEXT: s_waitcnt vmcnt(0) expcnt(0)
; GFX6-NEXT: s_setpc_b64 s[30:31]
+;
; GFX7-LABEL: safe_math_fract_f64:
; GFX7: ; %bb.0: ; %entry
; GFX7-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
@@ -2582,6 +2708,7 @@ define double @safe_math_fract_f64(double %x, ptr addrspace(1) writeonly capture
; GFX7-NEXT: buffer_store_dwordx2 v[6:7], v[2:3], s[4:7], 0 addr64
; GFX7-NEXT: s_waitcnt vmcnt(0)
; GFX7-NEXT: s_setpc_b64 s[30:31]
+;
; GFX8-LABEL: safe_math_fract_f64:
; GFX8: ; %bb.0: ; %entry
; GFX8-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
@@ -2595,6 +2722,7 @@ define double @safe_math_fract_f64(double %x, ptr addrspace(1) writeonly capture
; GFX8-NEXT: global_store_dwordx2 v[2:3], v[6:7], off
; GFX8-NEXT: s_waitcnt vmcnt(0)
; GFX8-NEXT: s_setpc_b64 s[30:31]
+;
; GFX11-LABEL: safe_math_fract_f64:
; GFX11: ; %bb.0: ; %entry
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
@@ -2605,6 +2733,7 @@ define double @safe_math_fract_f64(double %x, ptr addrspace(1) writeonly capture
; GFX11-NEXT: v_dual_cndmask_b32 v0, 0, v4 :: v_dual_cndmask_b32 v1, 0, v5
; GFX11-NEXT: global_store_b64 v[2:3], v[6:7], off
; GFX11-NEXT: s_setpc_b64 s[30:31]
+;
; GFX12-LABEL: safe_math_fract_f64:
; GFX12: ; %bb.0: ; %entry
; GFX12-NEXT: s_wait_loadcnt_dscnt 0x0
@@ -2698,6 +2827,7 @@ define half @safe_math_fract_f16(half %x, ptr addrspace(1) writeonly captures(no
; GFX6-NEXT: v_cndmask_b32_e32 v0, 0, v0, vcc
; GFX6-NEXT: s_waitcnt vmcnt(0) expcnt(0)
; GFX6-NEXT: s_setpc_b64 s[30:31]
+;
; GFX7-LABEL: safe_math_fract_f16:
; GFX7: ; %bb.0: ; %entry
; GFX7-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
@@ -2723,6 +2853,7 @@ define half @safe_math_fract_f16(half %x, ptr addrspace(1) writeonly captures(no
; GFX7-NEXT: v_cndmask_b32_e32 v0, 0, v0, vcc
; GFX7-NEXT: s_waitcnt vmcnt(0)
; GFX7-NEXT: s_setpc_b64 s[30:31]
+;
; GFX8-LABEL: safe_math_fract_f16:
; GFX8: ; %bb.0: ; %entry
; GFX8-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
@@ -2734,16 +2865,18 @@ define half @safe_math_fract_f16(half %x, ptr addrspace(1) writeonly captures(no
; GFX8-NEXT: global_store_short v[1:2], v3, off
; GFX8-NEXT: s_waitcnt vmcnt(0)
; GFX8-NEXT: s_setpc_b64 s[30:31]
+;
; GFX11-TRUE16-LABEL: safe_math_fract_f16:
; GFX11-TRUE16: ; %bb.0: ; %entry
; GFX11-TRUE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-TRUE16-NEXT: v_fract_f16_e32 v0.h, v0.l
; GFX11-TRUE16-NEXT: v_cmp_neq_f16_e64 s0, 0x7c00, |v0.l|
; GFX11-TRUE16-NEXT: v_floor_f16_e32 v3.l, v0.l
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX11-TRUE16-NEXT: v_cndmask_b16 v0.l, 0, v0.h, s0
; GFX11-TRUE16-NEXT: global_store_b16 v[1:2], v3, off
; GFX11-TRUE16-NEXT: s_setpc_b64 s[30:31]
+;
; GFX11-FAKE16-LABEL: safe_math_fract_f16:
; GFX11-FAKE16: ; %bb.0: ; %entry
; GFX11-FAKE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
@@ -2754,6 +2887,7 @@ define half @safe_math_fract_f16(half %x, ptr addrspace(1) writeonly captures(no
; GFX11-FAKE16-NEXT: v_cndmask_b32_e32 v0, 0, v3, vcc_lo
; GFX11-FAKE16-NEXT: global_store_b16 v[1:2], v4, off
; GFX11-FAKE16-NEXT: s_setpc_b64 s[30:31]
+;
; GFX12-TRUE16-LABEL: safe_math_fract_f16:
; GFX12-TRUE16: ; %bb.0: ; %entry
; GFX12-TRUE16-NEXT: s_wait_loadcnt_dscnt 0x0
@@ -2765,10 +2899,11 @@ define half @safe_math_fract_f16(half %x, ptr addrspace(1) writeonly captures(no
; GFX12-TRUE16-NEXT: v_cmp_neq_f16_e64 s0, 0x7c00, |v0.l|
; GFX12-TRUE16-NEXT: v_floor_f16_e32 v3.l, v0.l
; GFX12-TRUE16-NEXT: s_wait_alu depctr_va_sdst(0)
-; GFX12-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2)
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX12-TRUE16-NEXT: v_cndmask_b16 v0.l, 0, v0.h, s0
; GFX12-TRUE16-NEXT: global_store_b16 v[1:2], v3, off
; GFX12-TRUE16-NEXT: s_setpc_b64 s[30:31]
+;
; GFX12-FAKE16-LABEL: safe_math_fract_f16:
; GFX12-FAKE16: ; %bb.0: ; %entry
; GFX12-FAKE16-NEXT: s_wait_loadcnt_dscnt 0x0
@@ -2887,6 +3022,7 @@ define <2 x half> @safe_math_fract_v2f16(<2 x half> %x, ptr addrspace(1) writeon
; GFX6-NEXT: v_or_b32_e32 v0, v4, v0
; GFX6-NEXT: s_waitcnt vmcnt(0) expcnt(0)
; GFX6-NEXT: s_setpc_b64 s[30:31]
+;
; GFX7-LABEL: safe_math_fract_v2f16:
; GFX7: ; %bb.0: ; %entry
; GFX7-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
@@ -2932,6 +3068,7 @@ define <2 x half> @safe_math_fract_v2f16(<2 x half> %x, ptr addrspace(1) writeon
; GFX7-NEXT: v_or_b32_e32 v0, v4, v0
; GFX7-NEXT: s_waitcnt vmcnt(0)
; GFX7-NEXT: s_setpc_b64 s[30:31]
+;
; GFX8-LABEL: safe_math_fract_v2f16:
; GFX8: ; %bb.0: ; %entry
; GFX8-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
@@ -2949,6 +3086,7 @@ define <2 x half> @safe_math_fract_v2f16(<2 x half> %x, ptr addrspace(1) writeon
; GFX8-NEXT: global_store_dword v[1:2], v3, off
; GFX8-NEXT: s_waitcnt vmcnt(0)
; GFX8-NEXT: s_setpc_b64 s[30:31]
+;
; GFX11-TRUE16-LABEL: safe_math_fract_v2f16:
; GFX11-TRUE16: ; %bb.0: ; %entry
; GFX11-TRUE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
@@ -2964,6 +3102,7 @@ define <2 x half> @safe_math_fract_v2f16(<2 x half> %x, ptr addrspace(1) writeon
; GFX11-TRUE16-NEXT: v_cndmask_b16 v0.h, v0.h, 0, s0
; GFX11-TRUE16-NEXT: global_store_b32 v[1:2], v5, off
; GFX11-TRUE16-NEXT: s_setpc_b64 s[30:31]
+;
; GFX11-FAKE16-LABEL: safe_math_fract_v2f16:
; GFX11-FAKE16: ; %bb.0: ; %entry
; GFX11-FAKE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
@@ -2982,6 +3121,7 @@ define <2 x half> @safe_math_fract_v2f16(<2 x half> %x, ptr addrspace(1) writeon
; GFX11-FAKE16-NEXT: global_store_b32 v[1:2], v4, off
; GFX11-FAKE16-NEXT: v_pack_b32_f16 v0, v0, v3
; GFX11-FAKE16-NEXT: s_setpc_b64 s[30:31]
+;
; GFX12-TRUE16-LABEL: safe_math_fract_v2f16:
; GFX12-TRUE16: ; %bb.0: ; %entry
; GFX12-TRUE16-NEXT: s_wait_loadcnt_dscnt 0x0
@@ -3002,6 +3142,7 @@ define <2 x half> @safe_math_fract_v2f16(<2 x half> %x, ptr addrspace(1) writeon
; GFX12-TRUE16-NEXT: v_cndmask_b16 v0.h, v0.h, 0, s0
; GFX12-TRUE16-NEXT: global_store_b32 v[1:2], v5, off
; GFX12-TRUE16-NEXT: s_setpc_b64 s[30:31]
+;
; GFX12-FAKE16-LABEL: safe_math_fract_v2f16:
; GFX12-FAKE16: ; %bb.0: ; %entry
; GFX12-FAKE16-NEXT: s_wait_loadcnt_dscnt 0x0
@@ -3113,6 +3254,7 @@ define <2 x double> @safe_math_fract_v2f64(<2 x double> %x, ptr addrspace(1) wri
; GFX6-NEXT: buffer_store_dwordx4 v[6:9], v[4:5], s[4:7], 0 addr64
; GFX6-NEXT: s_waitcnt vmcnt(0) expcnt(0)
; GFX6-NEXT: s_setpc_b64 s[30:31]
+;
; GFX7-LABEL: safe_math_fract_v2f64:
; GFX7: ; %bb.0: ; %entry
; GFX7-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
@@ -3134,6 +3276,7 @@ define <2 x double> @safe_math_fract_v2f64(<2 x double> %x, ptr addrspace(1) wri
; GFX7-NEXT: buffer_store_dwordx4 v[6:9], v[4:5], s[8:11], 0 addr64
; GFX7-NEXT: s_waitcnt vmcnt(0)
; GFX7-NEXT: s_setpc_b64 s[30:31]
+;
; GFX8-LABEL: safe_math_fract_v2f64:
; GFX8: ; %bb.0: ; %entry
; GFX8-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
@@ -3151,6 +3294,7 @@ define <2 x double> @safe_math_fract_v2f64(<2 x double> %x, ptr addrspace(1) wri
; GFX8-NEXT: global_store_dwordx4 v[4:5], v[6:9], off
; GFX8-NEXT: s_waitcnt vmcnt(0)
; GFX8-NEXT: s_setpc_b64 s[30:31]
+;
; GFX11-LABEL: safe_math_fract_v2f64:
; GFX11: ; %bb.0: ; %entry
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
@@ -3166,6 +3310,7 @@ define <2 x double> @safe_math_fract_v2f64(<2 x double> %x, ptr addrspace(1) wri
; GFX11-NEXT: v_cndmask_b32_e64 v3, v13, 0, s1
; GFX11-NEXT: global_store_b128 v[4:5], v[6:9], off
; GFX11-NEXT: s_setpc_b64 s[30:31]
+;
; GFX12-LABEL: safe_math_fract_v2f64:
; GFX12: ; %bb.0: ; %entry
; GFX12-NEXT: s_wait_loadcnt_dscnt 0x0
@@ -3246,6 +3391,7 @@ define float @safe_math_fract_f32_minimum(float %x, ptr addrspace(1) writeonly c
; GFX6-NEXT: buffer_store_dword v3, v[1:2], s[4:7], 0 addr64
; GFX6-NEXT: s_waitcnt vmcnt(0) expcnt(0)
; GFX6-NEXT: s_setpc_b64 s[30:31]
+;
; GFX7-LABEL: safe_math_fract_f32_minimum:
; GFX7: ; %bb.0: ; %entry
; GFX7-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
@@ -3261,6 +3407,7 @@ define float @safe_math_fract_f32_minimum(float %x, ptr addrspace(1) writeonly c
; GFX7-NEXT: buffer_store_dword v3, v[1:2], s[4:7], 0 addr64
; GFX7-NEXT: s_waitcnt vmcnt(0)
; GFX7-NEXT: s_setpc_b64 s[30:31]
+;
; GFX8-LABEL: safe_math_fract_f32_minimum:
; GFX8: ; %bb.0: ; %entry
; GFX8-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
@@ -3272,6 +3419,7 @@ define float @safe_math_fract_f32_minimum(float %x, ptr addrspace(1) writeonly c
; GFX8-NEXT: global_store_dword v[1:2], v3, off
; GFX8-NEXT: s_waitcnt vmcnt(0)
; GFX8-NEXT: s_setpc_b64 s[30:31]
+;
; GFX11-LABEL: safe_math_fract_f32_minimum:
; GFX11: ; %bb.0: ; %entry
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
@@ -3282,6 +3430,7 @@ define float @safe_math_fract_f32_minimum(float %x, ptr addrspace(1) writeonly c
; GFX11-NEXT: v_cndmask_b32_e32 v0, 0, v3, vcc_lo
; GFX11-NEXT: global_store_b32 v[1:2], v4, off
; GFX11-NEXT: s_setpc_b64 s[30:31]
+;
; GFX12-LABEL: safe_math_fract_f32_minimum:
; GFX12: ; %bb.0: ; %entry
; GFX12-NEXT: s_wait_loadcnt_dscnt 0x0
@@ -3357,6 +3506,7 @@ define float @safe_math_fract_f32_minimum_swap(float %x, ptr addrspace(1) writeo
; GFX6-NEXT: buffer_store_dword v3, v[1:2], s[4:7], 0 addr64
; GFX6-NEXT: s_waitcnt vmcnt(0) expcnt(0)
; GFX6-NEXT: s_setpc_b64 s[30:31]
+;
; GFX7-LABEL: safe_math_fract_f32_minimum_swap:
; GFX7: ; %bb.0: ; %entry
; GFX7-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
@@ -3372,6 +3522,7 @@ define float @safe_math_fract_f32_minimum_swap(float %x, ptr addrspace(1) writeo
; GFX7-NEXT: buffer_store_dword v3, v[1:2], s[4:7], 0 addr64
; GFX7-NEXT: s_waitcnt vmcnt(0)
; GFX7-NEXT: s_setpc_b64 s[30:31]
+;
; GFX8-LABEL: safe_math_fract_f32_minimum_swap:
; GFX8: ; %bb.0: ; %entry
; GFX8-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
@@ -3383,6 +3534,7 @@ define float @safe_math_fract_f32_minimum_swap(float %x, ptr addrspace(1) writeo
; GFX8-NEXT: global_store_dword v[1:2], v3, off
; GFX8-NEXT: s_waitcnt vmcnt(0)
; GFX8-NEXT: s_setpc_b64 s[30:31]
+;
; GFX11-LABEL: safe_math_fract_f32_minimum_swap:
; GFX11: ; %bb.0: ; %entry
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
@@ -3393,6 +3545,7 @@ define float @safe_math_fract_f32_minimum_swap(float %x, ptr addrspace(1) writeo
; GFX11-NEXT: v_cndmask_b32_e32 v0, 0, v3, vcc_lo
; GFX11-NEXT: global_store_b32 v[1:2], v4, off
; GFX11-NEXT: s_setpc_b64 s[30:31]
+;
; GFX12-LABEL: safe_math_fract_f32_minimum_swap:
; GFX12: ; %bb.0: ; %entry
; GFX12-NEXT: s_wait_loadcnt_dscnt 0x0
@@ -3465,6 +3618,7 @@ define float @safe_math_fract_f32_minimumnum(float %x, ptr addrspace(1) writeonl
; GFX6-NEXT: buffer_store_dword v3, v[1:2], s[4:7], 0 addr64
; GFX6-NEXT: s_waitcnt vmcnt(0) expcnt(0)
; GFX6-NEXT: s_setpc_b64 s[30:31]
+;
; GFX7-LABEL: safe_math_fract_f32_minimumnum:
; GFX7: ; %bb.0: ; %entry
; GFX7-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
@@ -3480,6 +3634,7 @@ define float @safe_math_fract_f32_minimumnum(float %x, ptr addrspace(1) writeonl
; GFX7-NEXT: buffer_store_dword v3, v[1:2], s[4:7], 0 addr64
; GFX7-NEXT: s_waitcnt vmcnt(0)
; GFX7-NEXT: s_setpc_b64 s[30:31]
+;
; GFX8-LABEL: safe_math_fract_f32_minimumnum:
; GFX8: ; %bb.0: ; %entry
; GFX8-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
@@ -3491,6 +3646,7 @@ define float @safe_math_fract_f32_minimumnum(float %x, ptr addrspace(1) writeonl
; GFX8-NEXT: global_store_dword v[1:2], v3, off
; GFX8-NEXT: s_waitcnt vmcnt(0)
; GFX8-NEXT: s_setpc_b64 s[30:31]
+;
; GFX11-LABEL: safe_math_fract_f32_minimumnum:
; GFX11: ; %bb.0: ; %entry
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
@@ -3501,6 +3657,7 @@ define float @safe_math_fract_f32_minimumnum(float %x, ptr addrspace(1) writeonl
; GFX11-NEXT: v_cndmask_b32_e32 v0, 0, v3, vcc_lo
; GFX11-NEXT: global_store_b32 v[1:2], v4, off
; GFX11-NEXT: s_setpc_b64 s[30:31]
+;
; GFX12-LABEL: safe_math_fract_f32_minimumnum:
; GFX12: ; %bb.0: ; %entry
; GFX12-NEXT: s_wait_loadcnt_dscnt 0x0
@@ -3573,6 +3730,7 @@ define float @safe_math_fract_f32_minimumnum_swap(float %x, ptr addrspace(1) wri
; GFX6-NEXT: buffer_store_dword v3, v[1:2], s[4:7], 0 addr64
; GFX6-NEXT: s_waitcnt vmcnt(0) expcnt(0)
; GFX6-NEXT: s_setpc_b64 s[30:31]
+;
; GFX7-LABEL: safe_math_fract_f32_minimumnum_swap:
; GFX7: ; %bb.0: ; %entry
; GFX7-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
@@ -3588,6 +3746,7 @@ define float @safe_math_fract_f32_minimumnum_swap(float %x, ptr addrspace(1) wri
; GFX7-NEXT: buffer_store_dword v3, v[1:2], s[4:7], 0 addr64
; GFX7-NEXT: s_waitcnt vmcnt(0)
; GFX7-NEXT: s_setpc_b64 s[30:31]
+;
; GFX8-LABEL: safe_math_fract_f32_minimumnum_swap:
; GFX8: ; %bb.0: ; %entry
; GFX8-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
@@ -3599,6 +3758,7 @@ define float @safe_math_fract_f32_minimumnum_swap(float %x, ptr addrspace(1) wri
; GFX8-NEXT: global_store_dword v[1:2], v3, off
; GFX8-NEXT: s_waitcnt vmcnt(0)
; GFX8-NEXT: s_setpc_b64 s[30:31]
+;
; GFX11-LABEL: safe_math_fract_f32_minimumnum_swap:
; GFX11: ; %bb.0: ; %entry
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
@@ -3609,6 +3769,7 @@ define float @safe_math_fract_f32_minimumnum_swap(float %x, ptr addrspace(1) wri
; GFX11-NEXT: v_cndmask_b32_e32 v0, 0, v3, vcc_lo
; GFX11-NEXT: global_store_b32 v[1:2], v4, off
; GFX11-NEXT: s_setpc_b64 s[30:31]
+;
; GFX12-LABEL: safe_math_fract_f32_minimumnum_swap:
; GFX12: ; %bb.0: ; %entry
; GFX12-NEXT: s_wait_loadcnt_dscnt 0x0
@@ -3659,21 +3820,25 @@ define float @basic_fract_f32_nonans_minimumnum(float nofpclass(nan) %x) {
; GFX6-NEXT: v_sub_f32_e32 v0, v0, v1
; GFX6-NEXT: v_min_f32_e32 v0, 0x3f7fffff, v0
; GFX6-NEXT: s_setpc_b64 s[30:31]
+;
; GFX7-LABEL: basic_fract_f32_nonans_minimumnum:
; GFX7: ; %bb.0: ; %entry
; GFX7-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX7-NEXT: v_fract_f32_e32 v0, v0
; GFX7-NEXT: s_setpc_b64 s[30:31]
+;
; GFX8-LABEL: basic_fract_f32_nonans_minimumnum:
; GFX8: ; %bb.0: ; %entry
; GFX8-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX8-NEXT: v_fract_f32_e32 v0, v0
; GFX8-NEXT: s_setpc_b64 s[30:31]
+;
; GFX11-LABEL: basic_fract_f32_nonans_minimumnum:
; GFX11: ; %bb.0: ; %entry
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: v_fract_f32_e32 v0, v0
; GFX11-NEXT: s_setpc_b64 s[30:31]
+;
; GFX12-LABEL: basic_fract_f32_nonans_minimumnum:
; GFX12: ; %bb.0: ; %entry
; GFX12-NEXT: s_wait_loadcnt_dscnt 0x0
@@ -3715,21 +3880,25 @@ define float @basic_fract_f32_nonans_minimum(float nofpclass(nan) %x) {
; GFX6-NEXT: v_cmp_o_f32_e32 vcc, v0, v0
; GFX6-NEXT: v_cndmask_b32_e32 v0, v2, v1, vcc
; GFX6-NEXT: s_setpc_b64 s[30:31]
+;
; GFX7-LABEL: basic_fract_f32_nonans_minimum:
; GFX7: ; %bb.0: ; %entry
; GFX7-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX7-NEXT: v_fract_f32_e32 v0, v0
; GFX7-NEXT: s_setpc_b64 s[30:31]
+;
; GFX8-LABEL: basic_fract_f32_nonans_minimum:
; GFX8: ; %bb.0: ; %entry
; GFX8-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX8-NEXT: v_fract_f32_e32 v0, v0
; GFX8-NEXT: s_setpc_b64 s[30:31]
+;
; GFX11-LABEL: basic_fract_f32_nonans_minimum:
; GFX11: ; %bb.0: ; %entry
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: v_fract_f32_e32 v0, v0
; GFX11-NEXT: s_setpc_b64 s[30:31]
+;
; GFX12-LABEL: basic_fract_f32_nonans_minimum:
; GFX12: ; %bb.0: ; %entry
; GFX12-NEXT: s_wait_loadcnt_dscnt 0x0
@@ -3768,21 +3937,25 @@ define float @nnan_minimum_fract_f32(float %x) {
; GFX6-NEXT: v_sub_f32_e32 v0, v0, v1
; GFX6-NEXT: v_min_f32_e32 v0, 0x3f7fffff, v0
; GFX6-NEXT: s_setpc_b64 s[30:31]
+;
; GFX7-LABEL: nnan_minimum_fract_f32:
; GFX7: ; %bb.0: ; %entry
; GFX7-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX7-NEXT: v_fract_f32_e32 v0, v0
; GFX7-NEXT: s_setpc_b64 s[30:31]
+;
; GFX8-LABEL: nnan_minimum_fract_f32:
; GFX8: ; %bb.0: ; %entry
; GFX8-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX8-NEXT: v_fract_f32_e32 v0, v0
; GFX8-NEXT: s_setpc_b64 s[30:31]
+;
; GFX11-LABEL: nnan_minimum_fract_f32:
; GFX11: ; %bb.0: ; %entry
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: v_fract_f32_e32 v0, v0
; GFX11-NEXT: s_setpc_b64 s[30:31]
+;
; GFX12-LABEL: nnan_minimum_fract_f32:
; GFX12: ; %bb.0: ; %entry
; GFX12-NEXT: s_wait_loadcnt_dscnt 0x0
@@ -3821,21 +3994,25 @@ define float @nnan_minimumnum_fract_f32(float %x) {
; GFX6-NEXT: v_sub_f32_e32 v0, v0, v1
; GFX6-NEXT: v_min_f32_e32 v0, 0x3f7fffff, v0
; GFX6-NEXT: s_setpc_b64 s[30:31]
+;
; GFX7-LABEL: nnan_minimumnum_fract_f32:
; GFX7: ; %bb.0: ; %entry
; GFX7-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX7-NEXT: v_fract_f32_e32 v0, v0
; GFX7-NEXT: s_setpc_b64 s[30:31]
+;
; GFX8-LABEL: nnan_minimumnum_fract_f32:
; GFX8: ; %bb.0: ; %entry
; GFX8-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX8-NEXT: v_fract_f32_e32 v0, v0
; GFX8-NEXT: s_setpc_b64 s[30:31]
+;
; GFX11-LABEL: nnan_minimumnum_fract_f32:
; GFX11: ; %bb.0: ; %entry
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: v_fract_f32_e32 v0, v0
; GFX11-NEXT: s_setpc_b64 s[30:31]
+;
; GFX12-LABEL: nnan_minimumnum_fract_f32:
; GFX12: ; %bb.0: ; %entry
; GFX12-NEXT: s_wait_loadcnt_dscnt 0x0
@@ -3868,6 +4045,7 @@ define float @basic_fract_f32_flags_minimumnum(float %x) {
; GFX6-NEXT: v_sub_f32_e32 v0, v0, v1
; GFX6-NEXT: v_min_f32_e32 v0, 0x3f7fffff, v0
; GFX6-NEXT: s_setpc_b64 s[30:31]
+;
; GFX7-LABEL: basic_fract_f32_flags_minimumnum:
; GFX7: ; %bb.0: ; %entry
; GFX7-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
@@ -3875,6 +4053,7 @@ define float @basic_fract_f32_flags_minimumnum(float %x) {
; GFX7-NEXT: v_sub_f32_e32 v0, v0, v1
; GFX7-NEXT: v_min_f32_e32 v0, 0x3f7fffff, v0
; GFX7-NEXT: s_setpc_b64 s[30:31]
+;
; GFX8-LABEL: basic_fract_f32_flags_minimumnum:
; GFX8: ; %bb.0: ; %entry
; GFX8-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
@@ -3882,6 +4061,7 @@ define float @basic_fract_f32_flags_minimumnum(float %x) {
; GFX8-NEXT: v_sub_f32_e32 v0, v0, v1
; GFX8-NEXT: v_min_f32_e32 v0, 0x3f7fffff, v0
; GFX8-NEXT: s_setpc_b64 s[30:31]
+;
; GFX11-LABEL: basic_fract_f32_flags_minimumnum:
; GFX11: ; %bb.0: ; %entry
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
@@ -3890,6 +4070,7 @@ define float @basic_fract_f32_flags_minimumnum(float %x) {
; GFX11-NEXT: v_sub_f32_e32 v0, v0, v1
; GFX11-NEXT: v_min_f32_e32 v0, 0x3f7fffff, v0
; GFX11-NEXT: s_setpc_b64 s[30:31]
+;
; GFX12-LABEL: basic_fract_f32_flags_minimumnum:
; GFX12: ; %bb.0: ; %entry
; GFX12-NEXT: s_wait_loadcnt_dscnt 0x0
@@ -3934,21 +4115,25 @@ define float @basic_fract_f32_flags_minimum(float %x) {
; GFX6-NEXT: v_cmp_o_f32_e32 vcc, v0, v0
; GFX6-NEXT: v_cndmask_b32_e32 v0, v2, v1, vcc
; GFX6-NEXT: s_setpc_b64 s[30:31]
+;
; GFX7-LABEL: basic_fract_f32_flags_minimum:
; GFX7: ; %bb.0: ; %entry
; GFX7-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX7-NEXT: v_fract_f32_e32 v0, v0
; GFX7-NEXT: s_setpc_b64 s[30:31]
+;
; GFX8-LABEL: basic_fract_f32_flags_minimum:
; GFX8: ; %bb.0: ; %entry
; GFX8-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX8-NEXT: v_fract_f32_e32 v0, v0
; GFX8-NEXT: s_setpc_b64 s[30:31]
+;
; GFX11-LABEL: basic_fract_f32_flags_minimum:
; GFX11: ; %bb.0: ; %entry
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: v_fract_f32_e32 v0, v0
; GFX11-NEXT: s_setpc_b64 s[30:31]
+;
; GFX12-LABEL: basic_fract_f32_flags_minimum:
; GFX12: ; %bb.0: ; %entry
; GFX12-NEXT: s_wait_loadcnt_dscnt 0x0
@@ -4044,7 +4229,6 @@ define double @fract_match_f64_assume_not_nan(double %x) #0 {
; GFX6-IR-NEXT: [[IS_INF:%.*]] = fcmp oeq double [[X_ABS]], +inf
; GFX6-IR-NEXT: [[RESULT:%.*]] = select i1 [[IS_INF]], double 0.000000e+00, double [[MIN]]
; GFX6-IR-NEXT: ret double [[RESULT]]
-;
; IR-FRACT-LABEL: define double @fract_match_f64_assume_not_nan(
; IR-FRACT-SAME: double [[X:%.*]]) #[[ATTR0:[0-9]+]] {
; IR-FRACT-NEXT: [[ENTRY:.*:]]
@@ -4055,7 +4239,6 @@ define double @fract_match_f64_assume_not_nan(double %x) #0 {
; IR-FRACT-NEXT: [[IS_INF:%.*]] = fcmp oeq double [[X_ABS]], +inf
; IR-FRACT-NEXT: [[RESULT:%.*]] = select i1 [[IS_INF]], double 0.000000e+00, double [[MIN]]
; IR-FRACT-NEXT: ret double [[RESULT]]
-;
entry:
%is.ord = fcmp ord double %x, 0.000000e+00
tail call void @llvm.assume(i1 %is.ord)
@@ -4104,6 +4287,7 @@ define float @safe_math_fract_f32_swapped_edge_case(float %x) #0 {
; GFX6-NEXT: v_cmp_o_f32_e32 vcc, v0, v0
; GFX6-NEXT: v_cndmask_b32_e32 v0, v0, v1, vcc
; GFX6-NEXT: s_setpc_b64 s[30:31]
+;
; GFX7-LABEL: safe_math_fract_f32_swapped_edge_case:
; GFX7: ; %bb.0: ; %entry
; GFX7-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
@@ -4112,6 +4296,7 @@ define float @safe_math_fract_f32_swapped_edge_case(float %x) #0 {
; GFX7-NEXT: v_cmp_neq_f32_e64 vcc, |v0|, s4
; GFX7-NEXT: v_cndmask_b32_e32 v0, 0, v1, vcc
; GFX7-NEXT: s_setpc_b64 s[30:31]
+;
; GFX8-LABEL: safe_math_fract_f32_swapped_edge_case:
; GFX8: ; %bb.0: ; %entry
; GFX8-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
@@ -4120,6 +4305,7 @@ define float @safe_math_fract_f32_swapped_edge_case(float %x) #0 {
; GFX8-NEXT: v_cmp_neq_f32_e64 vcc, |v0|, s4
; GFX8-NEXT: v_cndmask_b32_e32 v0, 0, v1, vcc
; GFX8-NEXT: s_setpc_b64 s[30:31]
+;
; GFX11-LABEL: safe_math_fract_f32_swapped_edge_case:
; GFX11: ; %bb.0: ; %entry
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
@@ -4128,6 +4314,7 @@ define float @safe_math_fract_f32_swapped_edge_case(float %x) #0 {
; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX11-NEXT: v_cndmask_b32_e32 v0, 0, v1, vcc_lo
; GFX11-NEXT: s_setpc_b64 s[30:31]
+;
; GFX12-LABEL: safe_math_fract_f32_swapped_edge_case:
; GFX12: ; %bb.0: ; %entry
; GFX12-NEXT: s_wait_loadcnt_dscnt 0x0
@@ -4235,7 +4422,6 @@ define float @safe_math_fract_f32_swapped_edge_case_multi_use_inner_select(float
; GFX6-IR-NEXT: [[NOT_NAN:%.*]] = fcmp ord float [[X]], 0.000000e+00
; GFX6-IR-NEXT: [[COND8:%.*]] = select i1 [[NOT_NAN]], float [[COND]], float [[X]]
; GFX6-IR-NEXT: ret float [[COND8]]
-;
; IR-FRACT-LABEL: define float @safe_math_fract_f32_swapped_edge_case_multi_use_inner_select(
; IR-FRACT-SAME: float [[X:%.*]], ptr addrspace(1) [[PTR:%.*]]) #[[ATTR0]] {
; IR-FRACT-NEXT: [[COND:%.*]] = call float @llvm.amdgcn.fract.f32(float [[X]])
@@ -4244,7 +4430,6 @@ define float @safe_math_fract_f32_swapped_edge_case_multi_use_inner_select(float
; IR-FRACT-NEXT: [[COND8:%.*]] = select i1 [[NOT_INF]], float [[COND]], float 0.000000e+00
; IR-FRACT-NEXT: store float [[COND8]], ptr addrspace(1) [[PTR]], align 4
; IR-FRACT-NEXT: ret float [[COND8]]
-;
%floor = call float @llvm.floor.f32(float %x)
%sub = fsub float %x, %floor
%min = call float @llvm.minnum.f32(float %sub, float 0x3FEFFFFFE0000000)
@@ -4311,8 +4496,8 @@ define float @safe_math_fract_f32_swapped_edge_case_multi_use_inner_select_fcmp(
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: v_fract_f32_e32 v3, v0
; GFX11-NEXT: v_cmp_neq_f32_e64 vcc_lo, 0x7f800000, |v0|
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_3)
; GFX11-NEXT: v_cndmask_b32_e64 v4, 0, 1, vcc_lo
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX11-NEXT: v_cndmask_b32_e32 v0, 0, v3, vcc_lo
; GFX11-NEXT: global_store_b8 v[1:2], v4, off
; GFX11-NEXT: s_setpc_b64 s[30:31]
@@ -4327,8 +4512,8 @@ define float @safe_math_fract_f32_swapped_edge_case_multi_use_inner_select_fcmp(
; GFX12-NEXT: v_fract_f32_e32 v3, v0
; GFX12-NEXT: v_cmp_neq_f32_e64 vcc_lo, 0x7f800000, |v0|
; GFX12-NEXT: s_wait_alu depctr_va_vcc(0)
-; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_3)
; GFX12-NEXT: v_cndmask_b32_e64 v4, 0, 1, vcc_lo
+; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX12-NEXT: v_cndmask_b32_e32 v0, 0, v3, vcc_lo
; GFX12-NEXT: global_store_b8 v[1:2], v4, off
; GFX12-NEXT: s_setpc_b64 s[30:31]
@@ -4344,7 +4529,6 @@ define float @safe_math_fract_f32_swapped_edge_case_multi_use_inner_select_fcmp(
; GFX6-IR-NEXT: [[NOT_NAN:%.*]] = fcmp ord float [[X]], 0.000000e+00
; GFX6-IR-NEXT: [[COND8:%.*]] = select i1 [[NOT_NAN]], float [[COND]], float [[X]]
; GFX6-IR-NEXT: ret float [[COND8]]
-;
; IR-FRACT-LABEL: define float @safe_math_fract_f32_swapped_edge_case_multi_use_inner_select_fcmp(
; IR-FRACT-SAME: float [[X:%.*]], ptr addrspace(1) [[PTR:%.*]]) #[[ATTR0]] {
; IR-FRACT-NEXT: [[COND:%.*]] = call float @llvm.amdgcn.fract.f32(float [[X]])
@@ -4353,7 +4537,6 @@ define float @safe_math_fract_f32_swapped_edge_case_multi_use_inner_select_fcmp(
; IR-FRACT-NEXT: store i1 [[NOT_INF]], ptr addrspace(1) [[PTR]], align 1
; IR-FRACT-NEXT: [[COND8:%.*]] = select i1 [[NOT_INF]], float [[COND]], float 0.000000e+00
; IR-FRACT-NEXT: ret float [[COND8]]
-;
%floor = call float @llvm.floor.f32(float %x)
%sub = fsub float %x, %floor
%min = call float @llvm.minnum.f32(float %sub, float 0x3FEFFFFFE0000000)
@@ -4453,7 +4636,6 @@ define float @safe_math_fract_f32_swapped_edge_case_multi_use_fabs(float %x, ptr
; GFX6-IR-NEXT: [[NOT_NAN:%.*]] = fcmp ord float [[X]], 0.000000e+00
; GFX6-IR-NEXT: [[COND8:%.*]] = select i1 [[NOT_NAN]], float [[COND]], float [[X]]
; GFX6-IR-NEXT: ret float [[COND8]]
-;
; IR-FRACT-LABEL: define float @safe_math_fract_f32_swapped_edge_case_multi_use_fabs(
; IR-FRACT-SAME: float [[X:%.*]], ptr addrspace(1) [[PTR:%.*]]) #[[ATTR0]] {
; IR-FRACT-NEXT: [[COND:%.*]] = call float @llvm.amdgcn.fract.f32(float [[X]])
@@ -4462,7 +4644,6 @@ define float @safe_math_fract_f32_swapped_edge_case_multi_use_fabs(float %x, ptr
; IR-FRACT-NEXT: [[NOT_INF:%.*]] = fcmp une float [[X_FABS]], +inf
; IR-FRACT-NEXT: [[COND8:%.*]] = select i1 [[NOT_INF]], float [[COND]], float 0.000000e+00
; IR-FRACT-NEXT: ret float [[COND8]]
-;
%floor = call float @llvm.floor.f32(float %x) #3
%sub = fsub float %x, %floor
%min = call float @llvm.minnum.f32(float %sub, float 0x3FEFFFFFE0000000)
@@ -4501,6 +4682,7 @@ define float @safe_math_fract_f32_swapped_edge_case_wrong_compared(float %x, flo
; GFX6-NEXT: v_cmp_o_f32_e32 vcc, v0, v0
; GFX6-NEXT: v_cndmask_b32_e32 v0, v0, v1, vcc
; GFX6-NEXT: s_setpc_b64 s[30:31]
+;
; GFX7-LABEL: safe_math_fract_f32_swapped_edge_case_wrong_compared:
; GFX7: ; %bb.0: ; %entry
; GFX7-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
@@ -4513,6 +4695,7 @@ define float @safe_math_fract_f32_swapped_edge_case_wrong_compared(float %x, flo
; GFX7-NEXT: v_cmp_o_f32_e32 vcc, v0, v0
; GFX7-NEXT: v_cndmask_b32_e32 v0, v0, v1, vcc
; GFX7-NEXT: s_setpc_b64 s[30:31]
+;
; GFX8-LABEL: safe_math_fract_f32_swapped_edge_case_wrong_compared:
; GFX8: ; %bb.0: ; %entry
; GFX8-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
@@ -4525,6 +4708,7 @@ define float @safe_math_fract_f32_swapped_edge_case_wrong_compared(float %x, flo
; GFX8-NEXT: v_cmp_o_f32_e32 vcc, v0, v0
; GFX8-NEXT: v_cndmask_b32_e32 v0, v0, v1, vcc
; GFX8-NEXT: s_setpc_b64 s[30:31]
+;
; GFX11-LABEL: safe_math_fract_f32_swapped_edge_case_wrong_compared:
; GFX11: ; %bb.0: ; %entry
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
@@ -4538,6 +4722,7 @@ define float @safe_math_fract_f32_swapped_edge_case_wrong_compared(float %x, flo
; GFX11-NEXT: v_cmp_o_f32_e32 vcc_lo, v0, v0
; GFX11-NEXT: v_cndmask_b32_e32 v0, v0, v1, vcc_lo
; GFX11-NEXT: s_setpc_b64 s[30:31]
+;
; GFX12-LABEL: safe_math_fract_f32_swapped_edge_case_wrong_compared:
; GFX12: ; %bb.0: ; %entry
; GFX12-NEXT: s_wait_loadcnt_dscnt 0x0
@@ -4595,6 +4780,7 @@ define float @safe_math_fract_f32_swapped_edge_case_commute_inf_check(float %x)
; GFX6-NEXT: v_cmp_o_f32_e32 vcc, v0, v0
; GFX6-NEXT: v_cndmask_b32_e32 v0, v0, v1, vcc
; GFX6-NEXT: s_setpc_b64 s[30:31]
+;
; GFX7-LABEL: safe_math_fract_f32_swapped_edge_case_commute_inf_check:
; GFX7: ; %bb.0: ; %entry
; GFX7-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
@@ -4607,6 +4793,7 @@ define float @safe_math_fract_f32_swapped_edge_case_commute_inf_check(float %x)
; GFX7-NEXT: v_cmp_o_f32_e32 vcc, v0, v0
; GFX7-NEXT: v_cndmask_b32_e32 v0, v0, v1, vcc
; GFX7-NEXT: s_setpc_b64 s[30:31]
+;
; GFX8-LABEL: safe_math_fract_f32_swapped_edge_case_commute_inf_check:
; GFX8: ; %bb.0: ; %entry
; GFX8-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
@@ -4619,6 +4806,7 @@ define float @safe_math_fract_f32_swapped_edge_case_commute_inf_check(float %x)
; GFX8-NEXT: v_cmp_o_f32_e32 vcc, v0, v0
; GFX8-NEXT: v_cndmask_b32_e32 v0, v0, v1, vcc
; GFX8-NEXT: s_setpc_b64 s[30:31]
+;
; GFX11-LABEL: safe_math_fract_f32_swapped_edge_case_commute_inf_check:
; GFX11: ; %bb.0: ; %entry
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
@@ -4632,6 +4820,7 @@ define float @safe_math_fract_f32_swapped_edge_case_commute_inf_check(float %x)
; GFX11-NEXT: v_cmp_o_f32_e32 vcc_lo, v0, v0
; GFX11-NEXT: v_cndmask_b32_e32 v0, v0, v1, vcc_lo
; GFX11-NEXT: s_setpc_b64 s[30:31]
+;
; GFX12-LABEL: safe_math_fract_f32_swapped_edge_case_commute_inf_check:
; GFX12: ; %bb.0: ; %entry
; GFX12-NEXT: s_wait_loadcnt_dscnt 0x0
@@ -4689,6 +4878,7 @@ define float @safe_math_fract_f32_swapped_edge_case_commute_nan_check(float %x)
; GFX6-NEXT: v_cmp_u_f32_e32 vcc, v0, v0
; GFX6-NEXT: v_cndmask_b32_e32 v0, v1, v0, vcc
; GFX6-NEXT: s_setpc_b64 s[30:31]
+;
; GFX7-LABEL: safe_math_fract_f32_swapped_edge_case_commute_nan_check:
; GFX7: ; %bb.0: ; %entry
; GFX7-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
@@ -4701,6 +4891,7 @@ define float @safe_math_fract_f32_swapped_edge_case_commute_nan_check(float %x)
; GFX7-NEXT: v_cmp_u_f32_e32 vcc, v0, v0
; GFX7-NEXT: v_cndmask_b32_e32 v0, v1, v0, vcc
; GFX7-NEXT: s_setpc_b64 s[30:31]
+;
; GFX8-LABEL: safe_math_fract_f32_swapped_edge_case_commute_nan_check:
; GFX8: ; %bb.0: ; %entry
; GFX8-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
@@ -4713,6 +4904,7 @@ define float @safe_math_fract_f32_swapped_edge_case_commute_nan_check(float %x)
; GFX8-NEXT: v_cmp_u_f32_e32 vcc, v0, v0
; GFX8-NEXT: v_cndmask_b32_e32 v0, v1, v0, vcc
; GFX8-NEXT: s_setpc_b64 s[30:31]
+;
; GFX11-LABEL: safe_math_fract_f32_swapped_edge_case_commute_nan_check:
; GFX11: ; %bb.0: ; %entry
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
@@ -4726,6 +4918,7 @@ define float @safe_math_fract_f32_swapped_edge_case_commute_nan_check(float %x)
; GFX11-NEXT: v_cmp_u_f32_e32 vcc_lo, v0, v0
; GFX11-NEXT: v_cndmask_b32_e32 v0, v1, v0, vcc_lo
; GFX11-NEXT: s_setpc_b64 s[30:31]
+;
; GFX12-LABEL: safe_math_fract_f32_swapped_edge_case_commute_nan_check:
; GFX12: ; %bb.0: ; %entry
; GFX12-NEXT: s_wait_loadcnt_dscnt 0x0
@@ -4781,21 +4974,25 @@ define float @safe_math_fract_f32_swapped_edge_case_cmp_neg_inf(float %x) #0 {
; GFX6-NEXT: v_cmp_o_f32_e32 vcc, v0, v0
; GFX6-NEXT: v_cndmask_b32_e32 v0, v0, v1, vcc
; GFX6-NEXT: s_setpc_b64 s[30:31]
+;
; GFX7-LABEL: safe_math_fract_f32_swapped_edge_case_cmp_neg_inf:
; GFX7: ; %bb.0: ; %entry
; GFX7-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX7-NEXT: v_fract_f32_e32 v0, v0
; GFX7-NEXT: s_setpc_b64 s[30:31]
+;
; GFX8-LABEL: safe_math_fract_f32_swapped_edge_case_cmp_neg_inf:
; GFX8: ; %bb.0: ; %entry
; GFX8-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX8-NEXT: v_fract_f32_e32 v0, v0
; GFX8-NEXT: s_setpc_b64 s[30:31]
+;
; GFX11-LABEL: safe_math_fract_f32_swapped_edge_case_cmp_neg_inf:
; GFX11: ; %bb.0: ; %entry
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: v_fract_f32_e32 v0, v0
; GFX11-NEXT: s_setpc_b64 s[30:31]
+;
; GFX12-LABEL: safe_math_fract_f32_swapped_edge_case_cmp_neg_inf:
; GFX12: ; %bb.0: ; %entry
; GFX12-NEXT: s_wait_loadcnt_dscnt 0x0
@@ -4839,6 +5036,7 @@ define float @basic_fract_f32_with_inf_check(float %x) {
; GFX6-NEXT: v_cmp_neq_f32_e64 vcc, |v0|, s4
; GFX6-NEXT: v_cndmask_b32_e32 v0, 0, v1, vcc
; GFX6-NEXT: s_setpc_b64 s[30:31]
+;
; GFX7-LABEL: basic_fract_f32_with_inf_check:
; GFX7: ; %bb.0: ; %entry
; GFX7-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
@@ -4849,6 +5047,7 @@ define float @basic_fract_f32_with_inf_check(float %x) {
; GFX7-NEXT: v_cmp_neq_f32_e64 vcc, |v0|, s4
; GFX7-NEXT: v_cndmask_b32_e32 v0, 0, v1, vcc
; GFX7-NEXT: s_setpc_b64 s[30:31]
+;
; GFX8-LABEL: basic_fract_f32_with_inf_check:
; GFX8: ; %bb.0: ; %entry
; GFX8-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
@@ -4859,6 +5058,7 @@ define float @basic_fract_f32_with_inf_check(float %x) {
; GFX8-NEXT: v_cmp_neq_f32_e64 vcc, |v0|, s4
; GFX8-NEXT: v_cndmask_b32_e32 v0, 0, v1, vcc
; GFX8-NEXT: s_setpc_b64 s[30:31]
+;
; GFX11-LABEL: basic_fract_f32_with_inf_check:
; GFX11: ; %bb.0: ; %entry
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
@@ -4870,6 +5070,7 @@ define float @basic_fract_f32_with_inf_check(float %x) {
; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_cndmask_b32_e32 v0, 0, v1, vcc_lo
; GFX11-NEXT: s_setpc_b64 s[30:31]
+;
; GFX12-LABEL: basic_fract_f32_with_inf_check:
; GFX12: ; %bb.0: ; %entry
; GFX12-NEXT: s_wait_loadcnt_dscnt 0x0
@@ -4927,6 +5128,7 @@ define float @basic_fract_f32_nonans_with_inf_check(float nofpclass(nan) %x) {
; GFX6-NEXT: v_cmp_lg_f32_e64 vcc, |v0|, s4
; GFX6-NEXT: v_cndmask_b32_e32 v0, 0, v1, vcc
; GFX6-NEXT: s_setpc_b64 s[30:31]
+;
; GFX7-LABEL: basic_fract_f32_nonans_with_inf_check:
; GFX7: ; %bb.0: ; %entry
; GFX7-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
@@ -4935,6 +5137,7 @@ define float @basic_fract_f32_nonans_with_inf_check(float nofpclass(nan) %x) {
; GFX7-NEXT: v_cmp_lg_f32_e64 vcc, |v0|, s4
; GFX7-NEXT: v_cndmask_b32_e32 v0, 0, v1, vcc
; GFX7-NEXT: s_setpc_b64 s[30:31]
+;
; GFX8-LABEL: basic_fract_f32_nonans_with_inf_check:
; GFX8: ; %bb.0: ; %entry
; GFX8-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
@@ -4943,6 +5146,7 @@ define float @basic_fract_f32_nonans_with_inf_check(float nofpclass(nan) %x) {
; GFX8-NEXT: v_cmp_lg_f32_e64 vcc, |v0|, s4
; GFX8-NEXT: v_cndmask_b32_e32 v0, 0, v1, vcc
; GFX8-NEXT: s_setpc_b64 s[30:31]
+;
; GFX11-LABEL: basic_fract_f32_nonans_with_inf_check:
; GFX11: ; %bb.0: ; %entry
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
@@ -4951,6 +5155,7 @@ define float @basic_fract_f32_nonans_with_inf_check(float nofpclass(nan) %x) {
; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX11-NEXT: v_cndmask_b32_e32 v0, 0, v1, vcc_lo
; GFX11-NEXT: s_setpc_b64 s[30:31]
+;
; GFX12-LABEL: basic_fract_f32_nonans_with_inf_check:
; GFX12: ; %bb.0: ; %entry
; GFX12-NEXT: s_wait_loadcnt_dscnt 0x0
@@ -5037,6 +5242,7 @@ define float @safe_math_fract_f32_swapped_edge_case_x_is_const() #0 {
; GFX12-NEXT: s_and_b32 s0, gv at abs32@lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_cmp_neq_f32 s0, 0x7f800000
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_3)
; GFX12-NEXT: s_cselect_b32 vcc_lo, -1, 0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: v_cndmask_b32_e32 v0, 0, v0, vcc_lo
@@ -5053,7 +5259,6 @@ define float @safe_math_fract_f32_swapped_edge_case_x_is_const() #0 {
; GFX6-IR-NEXT: [[NOT_NAN:%.*]] = fcmp ord float bitcast (i32 ptrtoint (ptr @gv to i32) to float), 0.000000e+00
; GFX6-IR-NEXT: [[COND8:%.*]] = select i1 [[NOT_NAN]], float [[COND]], float bitcast (i32 ptrtoint (ptr @gv to i32) to float)
; GFX6-IR-NEXT: ret float [[COND8]]
-;
; IR-FRACT-LABEL: define float @safe_math_fract_f32_swapped_edge_case_x_is_const(
; IR-FRACT-SAME: ) #[[ATTR0]] {
; IR-FRACT-NEXT: [[ENTRY:.*:]]
@@ -5062,7 +5267,6 @@ define float @safe_math_fract_f32_swapped_edge_case_x_is_const() #0 {
; IR-FRACT-NEXT: [[NOT_INF:%.*]] = fcmp une float [[X_FABS]], +inf
; IR-FRACT-NEXT: [[COND8:%.*]] = select i1 [[NOT_INF]], float [[COND]], float 0.000000e+00
; IR-FRACT-NEXT: ret float [[COND8]]
-;
entry:
%floor = call float @llvm.floor.f32(float bitcast (i32 ptrtoint (ptr @gv to i32) to float))
%sub = fsub float bitcast (i32 ptrtoint (ptr @gv to i32) to float), %floor
@@ -5236,7 +5440,6 @@ define float @safe_math_fract_f32_swapped_edge_case_split_block(float %x, i1 %co
; GFX6-IR-NEXT: ret float [[COND8]]
; GFX6-IR: [[RET]]:
; GFX6-IR-NEXT: ret float [[MIN]]
-;
; IR-FRACT-LABEL: define float @safe_math_fract_f32_swapped_edge_case_split_block(
; IR-FRACT-SAME: float [[X:%.*]], i1 [[COND:%.*]]) #[[ATTR0]] {
; IR-FRACT-NEXT: [[FLOOR:%.*]] = call float @llvm.floor.f32(float [[X]])
@@ -5251,7 +5454,6 @@ define float @safe_math_fract_f32_swapped_edge_case_split_block(float %x, i1 %co
; IR-FRACT-NEXT: ret float [[COND8]]
; IR-FRACT: [[RET]]:
; IR-FRACT-NEXT: ret float [[MIN]]
-;
%floor = call float @llvm.floor.f32(float %x)
%sub = fsub float %x, %floor
%min = call float @llvm.minnum.f32(float %sub, float 0x3FEFFFFFE0000000)
@@ -5323,7 +5525,7 @@ define <3 x float> @safe_math_fract_f32_swapped_edge_case_vector(<3 x float> %x)
; GFX11-NEXT: v_cmp_class_f32_e64 vcc_lo, v0, 0x1fb
; GFX11-NEXT: v_fract_f32_e32 v4, v2
; GFX11-NEXT: v_fract_f32_e32 v1, v1
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(SKIP_1) | instid1(VALU_DEP_4)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_cndmask_b32_e32 v0, 0, v3, vcc_lo
; GFX11-NEXT: v_cmp_class_f32_e64 vcc_lo, v2, 0x1fb
; GFX11-NEXT: v_cndmask_b32_e32 v2, 0, v4, vcc_lo
@@ -5341,6 +5543,7 @@ define <3 x float> @safe_math_fract_f32_swapped_edge_case_vector(<3 x float> %x)
; GFX12-NEXT: v_fract_f32_e32 v4, v2
; GFX12-NEXT: v_fract_f32_e32 v1, v1
; GFX12-NEXT: s_wait_alu depctr_va_vcc(0)
+; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_2) | instid1(VALU_DEP_1)
; GFX12-NEXT: v_cndmask_b32_e32 v0, 0, v3, vcc_lo
; GFX12-NEXT: v_cmp_class_f32_e64 vcc_lo, v2, 0x1fb
; GFX12-NEXT: s_wait_alu depctr_va_vcc(0)
@@ -5358,7 +5561,6 @@ define <3 x float> @safe_math_fract_f32_swapped_edge_case_vector(<3 x float> %x)
; GFX6-IR-NEXT: [[NOT_NAN:%.*]] = fcmp ord <3 x float> [[X]], <float 0.000000e+00, float poison, float 0.000000e+00>
; GFX6-IR-NEXT: [[COND8:%.*]] = select <3 x i1> [[NOT_NAN]], <3 x float> [[COND]], <3 x float> [[X]]
; GFX6-IR-NEXT: ret <3 x float> [[COND8]]
-;
; IR-FRACT-LABEL: define <3 x float> @safe_math_fract_f32_swapped_edge_case_vector(
; IR-FRACT-SAME: <3 x float> [[X:%.*]]) #[[ATTR0]] {
; IR-FRACT-NEXT: [[ENTRY:.*:]]
@@ -5375,7 +5577,6 @@ define <3 x float> @safe_math_fract_f32_swapped_edge_case_vector(<3 x float> %x)
; IR-FRACT-NEXT: [[NOT_INF:%.*]] = fcmp une <3 x float> [[X_FABS]], <float +inf, float poison, float +inf>
; IR-FRACT-NEXT: [[COND8:%.*]] = select <3 x i1> [[NOT_INF]], <3 x float> [[COND]], <3 x float> <float 0.000000e+00, float poison, float 0.000000e+00>
; IR-FRACT-NEXT: ret <3 x float> [[COND8]]
-;
entry:
%floor = call <3 x float> @llvm.floor.v3f32(<3 x float> %x)
%sub = fsub <3 x float> %x, %floor
@@ -5438,7 +5639,6 @@ define float @safe_math_fract_f32_swapped_edge_case_inf_check_wrong_compare(floa
; IR-NEXT: [[NOT_NAN:%.*]] = fcmp ord float [[X]], 0.000000e+00
; IR-NEXT: [[COND8:%.*]] = select i1 [[NOT_NAN]], float [[COND]], float [[X]]
; IR-NEXT: ret float [[COND8]]
-;
entry:
%floor = call float @llvm.floor.f32(float %x)
%sub = fsub float %x, %floor
@@ -5537,7 +5737,6 @@ define float @fract_pat_fcmp_oge_select(float %x, ptr addrspace(1) %iptr) #0 {
; GFX6-IR-NEXT: [[NOT_INF:%.*]] = fcmp une float [[FABS_X]], +inf
; GFX6-IR-NEXT: [[COND6:%.*]] = select i1 [[NOT_INF]], float [[COND]], float 0.000000e+00
; GFX6-IR-NEXT: ret float [[COND6]]
-;
; IR-FRACT-LABEL: define float @fract_pat_fcmp_oge_select(
; IR-FRACT-SAME: float [[X:%.*]], ptr addrspace(1) [[IPTR:%.*]]) #[[ATTR0]] {
; IR-FRACT-NEXT: [[ENTRY:.*:]]
@@ -5548,7 +5747,6 @@ define float @fract_pat_fcmp_oge_select(float %x, ptr addrspace(1) %iptr) #0 {
; IR-FRACT-NEXT: [[NOT_INF:%.*]] = fcmp une float [[FABS_X]], +inf
; IR-FRACT-NEXT: [[COND6:%.*]] = select i1 [[NOT_INF]], float [[COND]], float 0.000000e+00
; IR-FRACT-NEXT: ret float [[COND6]]
-;
entry:
%call = call float @llvm.floor.f32(float %x)
store float %call, ptr addrspace(1) %iptr, align 4
@@ -5645,7 +5843,6 @@ define float @fract_pat_fcmp_ogt_select(float %x, ptr addrspace(1) %iptr) #0 {
; GFX6-IR-NEXT: [[NOT_INF:%.*]] = fcmp une float [[FABS_X]], +inf
; GFX6-IR-NEXT: [[COND6:%.*]] = select i1 [[NOT_INF]], float [[COND]], float 0.000000e+00
; GFX6-IR-NEXT: ret float [[COND6]]
-;
; IR-FRACT-LABEL: define float @fract_pat_fcmp_ogt_select(
; IR-FRACT-SAME: float [[X:%.*]], ptr addrspace(1) [[IPTR:%.*]]) #[[ATTR0]] {
; IR-FRACT-NEXT: [[ENTRY:.*:]]
@@ -5656,7 +5853,6 @@ define float @fract_pat_fcmp_ogt_select(float %x, ptr addrspace(1) %iptr) #0 {
; IR-FRACT-NEXT: [[NOT_INF:%.*]] = fcmp une float [[FABS_X]], +inf
; IR-FRACT-NEXT: [[COND6:%.*]] = select i1 [[NOT_INF]], float [[COND]], float 0.000000e+00
; IR-FRACT-NEXT: ret float [[COND6]]
-;
entry:
%call = call float @llvm.floor.f32(float %x)
store float %call, ptr addrspace(1) %iptr, align 4
@@ -5758,7 +5954,6 @@ define float @negative_fract_pat_fcmp_olt(float %x, ptr addrspace(1) %iptr) #0 {
; IR-NEXT: [[NOT_INF:%.*]] = fcmp une float [[FABS_X]], +inf
; IR-NEXT: [[COND6:%.*]] = select i1 [[NOT_INF]], float [[COND]], float 0.000000e+00
; IR-NEXT: ret float [[COND6]]
-;
entry:
%call = call float @llvm.floor.f32(float noundef %x)
store float %call, ptr addrspace(1) %iptr, align 4
@@ -5859,7 +6054,6 @@ define float @fract_pat_fcmp_olt_not_nan_src(float nofpclass(nan) %x, ptr addrsp
; IR-NEXT: [[NOT_INF:%.*]] = fcmp une float [[FABS_X]], +inf
; IR-NEXT: [[COND6:%.*]] = select i1 [[NOT_INF]], float [[COND]], float 0.000000e+00
; IR-NEXT: ret float [[COND6]]
-;
entry:
%call = call float @llvm.floor.f32(float noundef %x)
store float %call, ptr addrspace(1) %iptr, align 4
@@ -5959,7 +6153,6 @@ define float @fract_pat_minimum(float %x, ptr addrspace(1) %iptr) #0 {
; GFX6-IR-NEXT: [[NOT_INF:%.*]] = fcmp une float [[FABS_X]], +inf
; GFX6-IR-NEXT: [[COND6:%.*]] = select i1 [[NOT_INF]], float [[MIN]], float 0.000000e+00
; GFX6-IR-NEXT: ret float [[COND6]]
-;
; IR-FRACT-LABEL: define float @fract_pat_minimum(
; IR-FRACT-SAME: float [[X:%.*]], ptr addrspace(1) [[IPTR:%.*]]) #[[ATTR0]] {
; IR-FRACT-NEXT: [[ENTRY:.*:]]
@@ -5970,7 +6163,6 @@ define float @fract_pat_minimum(float %x, ptr addrspace(1) %iptr) #0 {
; IR-FRACT-NEXT: [[NOT_INF:%.*]] = fcmp une float [[FABS_X]], +inf
; IR-FRACT-NEXT: [[COND6:%.*]] = select i1 [[NOT_INF]], float [[MIN]], float 0.000000e+00
; IR-FRACT-NEXT: ret float [[COND6]]
-;
entry:
%call = call float @llvm.floor.f32(float %x)
store float %call, ptr addrspace(1) %iptr, align 4
@@ -6025,12 +6217,10 @@ define float @core_fract_pat_fcmp_oge_select(float %x) #0 {
; GFX6-IR-NEXT: [[OGE_MIN_CONST:%.*]] = fcmp oge float [[SUB_FLOOR]], f0x3F7FFFFF
; GFX6-IR-NEXT: [[SELECT:%.*]] = select i1 [[OGE_MIN_CONST]], float f0x3F7FFFFF, float [[SUB_FLOOR]]
; GFX6-IR-NEXT: ret float [[SELECT]]
-;
; IR-FRACT-LABEL: define float @core_fract_pat_fcmp_oge_select(
; IR-FRACT-SAME: float [[X:%.*]]) #[[ATTR0]] {
; IR-FRACT-NEXT: [[SELECT:%.*]] = call float @llvm.amdgcn.fract.f32(float [[X]])
; IR-FRACT-NEXT: ret float [[SELECT]]
-;
%floor = call float @llvm.floor.f32(float %x)
%sub.floor = fsub float %x, %floor
%oge.min.const = fcmp oge float %sub.floor, 0x3FEFFFFFE0000000
@@ -6083,12 +6273,10 @@ define float @core_fract_pat_minimum(float %x) #0 {
; GFX6-IR-NEXT: [[SUB_FLOOR:%.*]] = fsub float [[X]], [[FLOOR]]
; GFX6-IR-NEXT: [[MIN:%.*]] = call float @llvm.minimum.f32(float [[SUB_FLOOR]], float f0x3F7FFFFF)
; GFX6-IR-NEXT: ret float [[MIN]]
-;
; IR-FRACT-LABEL: define float @core_fract_pat_minimum(
; IR-FRACT-SAME: float [[X:%.*]]) #[[ATTR0]] {
; IR-FRACT-NEXT: [[MIN:%.*]] = call nnan float @llvm.amdgcn.fract.f32(float [[X]])
; IR-FRACT-NEXT: ret float [[MIN]]
-;
%floor = call float @llvm.floor.f32(float %x)
%sub.floor = fsub float %x, %floor
%min = call float @llvm.minimum.f32(float %sub.floor, float 0x3FEFFFFFE0000000)
@@ -6144,7 +6332,6 @@ define <2 x float> @core_fract_pat_fcmp_oge_select_v2f32(<2 x float> %x) #0 {
; GFX6-IR-NEXT: [[OGE_MIN_CONST:%.*]] = fcmp oge <2 x float> [[SUB_FLOOR]], <float f0x3F7FFFFF, float poison>
; GFX6-IR-NEXT: [[SELECT:%.*]] = select <2 x i1> [[OGE_MIN_CONST]], <2 x float> <float f0x3F7FFFFF, float poison>, <2 x float> [[SUB_FLOOR]]
; GFX6-IR-NEXT: ret <2 x float> [[SELECT]]
-;
; IR-FRACT-LABEL: define <2 x float> @core_fract_pat_fcmp_oge_select_v2f32(
; IR-FRACT-SAME: <2 x float> [[X:%.*]]) #[[ATTR0]] {
; IR-FRACT-NEXT: [[TMP1:%.*]] = extractelement <2 x float> [[X]], i64 0
@@ -6154,7 +6341,6 @@ define <2 x float> @core_fract_pat_fcmp_oge_select_v2f32(<2 x float> %x) #0 {
; IR-FRACT-NEXT: [[TMP5:%.*]] = insertelement <2 x float> poison, float [[TMP3]], i64 0
; IR-FRACT-NEXT: [[SELECT:%.*]] = insertelement <2 x float> [[TMP5]], float [[TMP4]], i64 1
; IR-FRACT-NEXT: ret <2 x float> [[SELECT]]
-;
%floor = call <2 x float> @llvm.floor.v2f32(<2 x float> %x)
%sub.floor = fsub <2 x float> %x, %floor
%oge.min.const = fcmp oge <2 x float> %sub.floor, <float 0x3FEFFFFFE0000000, float poison>
@@ -6212,7 +6398,6 @@ define <2 x float> @core_fract_pat_minimum_v2f32(<2 x float> %x) #0 {
; GFX6-IR-NEXT: [[SUB_FLOOR:%.*]] = fsub <2 x float> [[X]], [[FLOOR]]
; GFX6-IR-NEXT: [[MIN:%.*]] = call <2 x float> @llvm.minimum.v2f32(<2 x float> [[SUB_FLOOR]], <2 x float> <float f0x3F7FFFFF, float poison>)
; GFX6-IR-NEXT: ret <2 x float> [[MIN]]
-;
; IR-FRACT-LABEL: define <2 x float> @core_fract_pat_minimum_v2f32(
; IR-FRACT-SAME: <2 x float> [[X:%.*]]) #[[ATTR0]] {
; IR-FRACT-NEXT: [[TMP1:%.*]] = extractelement <2 x float> [[X]], i64 0
@@ -6222,7 +6407,6 @@ define <2 x float> @core_fract_pat_minimum_v2f32(<2 x float> %x) #0 {
; IR-FRACT-NEXT: [[TMP5:%.*]] = insertelement <2 x float> poison, float [[TMP3]], i64 0
; IR-FRACT-NEXT: [[MIN:%.*]] = insertelement <2 x float> [[TMP5]], float [[TMP4]], i64 1
; IR-FRACT-NEXT: ret <2 x float> [[MIN]]
-;
%floor = call <2 x float> @llvm.floor.v2f32(<2 x float> %x)
%sub.floor = fsub <2 x float> %x, %floor
%min = call <2 x float> @llvm.minimum.v2f32(<2 x float> %sub.floor, <2 x float> <float 0x3FEFFFFFE0000000, float poison>)
diff --git a/llvm/test/CodeGen/AMDGPU/frem.ll b/llvm/test/CodeGen/AMDGPU/frem.ll
index 112cfabd6d4343..8cd5f28e0f7949 100644
--- a/llvm/test/CodeGen/AMDGPU/frem.ll
+++ b/llvm/test/CodeGen/AMDGPU/frem.ll
@@ -555,6 +555,7 @@ define amdgpu_kernel void @frem_f16(ptr addrspace(1) %out, ptr addrspace(1) %in1
; GFX11-TRUE16-NEXT: s_and_b32 s2, s2, exec_lo
; GFX11-TRUE16-NEXT: s_cselect_b32 s2, 1, 0
; GFX11-TRUE16-NEXT: s_cmp_lg_u32 s2, 1
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_cbranch_scc1 .LBB0_8
; GFX11-TRUE16-NEXT: ; %bb.4: ; %frem.compute
; GFX11-TRUE16-NEXT: v_frexp_mant_f32_e32 v4, v2
@@ -585,9 +586,10 @@ define amdgpu_kernel void @frem_f16(ptr addrspace(1) %out, ptr addrspace(1) %in1
; GFX11-TRUE16-NEXT: v_fmac_f32_e32 v7, v8, v6
; GFX11-TRUE16-NEXT: v_fma_f32 v3, -v5, v7, v3
; GFX11-TRUE16-NEXT: s_denorm_mode 12
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_div_fmas_f32 v3, v3, v6, v7
; GFX11-TRUE16-NEXT: v_cmp_gt_i32_e32 vcc_lo, 12, v4
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX11-TRUE16-NEXT: v_div_fixup_f32 v3, v3, v2, 1.0
; GFX11-TRUE16-NEXT: s_cbranch_vccnz .LBB0_7
; GFX11-TRUE16-NEXT: ; %bb.5: ; %frem.loop_body.preheader
@@ -654,6 +656,7 @@ define amdgpu_kernel void @frem_f16(ptr addrspace(1) %out, ptr addrspace(1) %in1
; GFX11-FAKE16-NEXT: s_and_b32 s2, s2, exec_lo
; GFX11-FAKE16-NEXT: s_cselect_b32 s2, 1, 0
; GFX11-FAKE16-NEXT: s_cmp_lg_u32 s2, 1
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_cbranch_scc1 .LBB0_9
; GFX11-FAKE16-NEXT: ; %bb.4: ; %frem.compute
; GFX11-FAKE16-NEXT: v_frexp_exp_i32_f32_e32 v5, v3
@@ -683,9 +686,10 @@ define amdgpu_kernel void @frem_f16(ptr addrspace(1) %out, ptr addrspace(1) %in1
; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_fma_f32 v10, -v7, v9, v5
; GFX11-FAKE16-NEXT: v_fmac_f32_e32 v9, v10, v8
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_fma_f32 v5, -v7, v9, v5
; GFX11-FAKE16-NEXT: s_denorm_mode 12
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_div_fmas_f32 v5, v5, v8, v9
; GFX11-FAKE16-NEXT: v_cmp_gt_i32_e32 vcc_lo, 12, v6
; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2)
@@ -737,6 +741,7 @@ define amdgpu_kernel void @frem_f16(ptr addrspace(1) %out, ptr addrspace(1) %in1
; GFX11-FAKE16-NEXT: v_cmp_nle_f16_e64 s2, 0x7c00, |v0|
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v2, 0
; GFX11-FAKE16-NEXT: s_and_b32 vcc_lo, s2, vcc_lo
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_cndmask_b32_e32 v0, 0x7e00, v4, vcc_lo
; GFX11-FAKE16-NEXT: global_store_b16 v2, v0, s[0:1]
; GFX11-FAKE16-NEXT: s_endpgm
@@ -759,12 +764,13 @@ define amdgpu_kernel void @frem_f16(ptr addrspace(1) %out, ptr addrspace(1) %in1
; GFX1150-TRUE16-NEXT: s_and_b32 s2, s1, 0x7fff
; GFX1150-TRUE16-NEXT: s_cvt_f32_f16 s1, s0
; GFX1150-TRUE16-NEXT: s_cvt_f32_f16 s0, s2
-; GFX1150-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_3)
+; GFX1150-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_3) | instskip(NEXT) | instid1(SALU_CYCLE_3)
; GFX1150-TRUE16-NEXT: s_cmp_ngt_f32 s1, s0
; GFX1150-TRUE16-NEXT: s_cbranch_scc0 .LBB0_2
; GFX1150-TRUE16-NEXT: ; %bb.1: ; %frem.else
; GFX1150-TRUE16-NEXT: s_cmp_eq_f32 s1, s0
; GFX1150-TRUE16-NEXT: v_and_b16 v0.h, 0x8000, v0.l
+; GFX1150-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_2)
; GFX1150-TRUE16-NEXT: s_cselect_b32 s2, -1, 0
; GFX1150-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX1150-TRUE16-NEXT: v_cndmask_b16 v2.l, v0.l, v0.h, s2
@@ -778,6 +784,7 @@ define amdgpu_kernel void @frem_f16(ptr addrspace(1) %out, ptr addrspace(1) %in1
; GFX1150-TRUE16-NEXT: s_and_b32 s2, s2, exec_lo
; GFX1150-TRUE16-NEXT: s_cselect_b32 s2, 1, 0
; GFX1150-TRUE16-NEXT: s_cmp_lg_u32 s2, 1
+; GFX1150-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1150-TRUE16-NEXT: s_cbranch_scc1 .LBB0_8
; GFX1150-TRUE16-NEXT: ; %bb.4: ; %frem.compute
; GFX1150-TRUE16-NEXT: v_frexp_mant_f32_e32 v3, s0
@@ -794,10 +801,11 @@ define amdgpu_kernel void @frem_f16(ptr addrspace(1) %out, ptr addrspace(1) %in1
; GFX1150-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1150-TRUE16-NEXT: v_not_b32_e32 v5, v5
; GFX1150-TRUE16-NEXT: v_rcp_f32_e32 v7, v6
-; GFX1150-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_2) | instid1(TRANS32_DEP_1)
+; GFX1150-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1150-TRUE16-NEXT: v_add_nc_u32_e32 v5, v5, v4
; GFX1150-TRUE16-NEXT: v_div_scale_f32 v4, vcc_lo, 1.0, v3, 1.0
; GFX1150-TRUE16-NEXT: s_denorm_mode 15
+; GFX1150-TRUE16-NEXT: s_delay_alu instid0(TRANS32_DEP_1) | instid1(SALU_CYCLE_1)
; GFX1150-TRUE16-NEXT: v_fma_f32 v8, -v6, v7, 1.0
; GFX1150-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1150-TRUE16-NEXT: v_fmac_f32_e32 v7, v8, v7
@@ -805,9 +813,10 @@ define amdgpu_kernel void @frem_f16(ptr addrspace(1) %out, ptr addrspace(1) %in1
; GFX1150-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1150-TRUE16-NEXT: v_fma_f32 v9, -v6, v8, v4
; GFX1150-TRUE16-NEXT: v_fmac_f32_e32 v8, v9, v7
-; GFX1150-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
+; GFX1150-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1150-TRUE16-NEXT: v_fma_f32 v4, -v6, v8, v4
; GFX1150-TRUE16-NEXT: s_denorm_mode 12
+; GFX1150-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX1150-TRUE16-NEXT: v_div_fmas_f32 v4, v4, v7, v8
; GFX1150-TRUE16-NEXT: v_cmp_gt_i32_e32 vcc_lo, 12, v5
; GFX1150-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2)
@@ -864,15 +873,16 @@ define amdgpu_kernel void @frem_f16(ptr addrspace(1) %out, ptr addrspace(1) %in1
; GFX1150-FAKE16-NEXT: s_and_b32 s2, s1, 0x7fff
; GFX1150-FAKE16-NEXT: s_cvt_f32_f16 s1, s0
; GFX1150-FAKE16-NEXT: s_cvt_f32_f16 s0, s2
-; GFX1150-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_3)
+; GFX1150-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_3) | instskip(NEXT) | instid1(SALU_CYCLE_3)
; GFX1150-FAKE16-NEXT: s_cmp_ngt_f32 s1, s0
; GFX1150-FAKE16-NEXT: s_cbranch_scc0 .LBB0_2
; GFX1150-FAKE16-NEXT: ; %bb.1: ; %frem.else
; GFX1150-FAKE16-NEXT: s_cmp_eq_f32 s1, s0
; GFX1150-FAKE16-NEXT: v_and_b32_e32 v2, 0x8000, v0
; GFX1150-FAKE16-NEXT: s_mov_b32 s2, 0
+; GFX1150-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1150-FAKE16-NEXT: s_cselect_b32 vcc_lo, -1, 0
-; GFX1150-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1150-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX1150-FAKE16-NEXT: v_cndmask_b32_e32 v2, v0, v2, vcc_lo
; GFX1150-FAKE16-NEXT: s_branch .LBB0_3
; GFX1150-FAKE16-NEXT: .LBB0_2:
@@ -883,6 +893,7 @@ define amdgpu_kernel void @frem_f16(ptr addrspace(1) %out, ptr addrspace(1) %in1
; GFX1150-FAKE16-NEXT: s_and_b32 s2, s2, exec_lo
; GFX1150-FAKE16-NEXT: s_cselect_b32 s2, 1, 0
; GFX1150-FAKE16-NEXT: s_cmp_lg_u32 s2, 1
+; GFX1150-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1150-FAKE16-NEXT: s_cbranch_scc1 .LBB0_9
; GFX1150-FAKE16-NEXT: ; %bb.4: ; %frem.compute
; GFX1150-FAKE16-NEXT: v_frexp_mant_f32_e32 v3, s0
@@ -904,19 +915,21 @@ define amdgpu_kernel void @frem_f16(ptr addrspace(1) %out, ptr addrspace(1) %in1
; GFX1150-FAKE16-NEXT: v_add_nc_u32_e32 v6, v6, v5
; GFX1150-FAKE16-NEXT: v_div_scale_f32 v5, vcc_lo, 1.0, v3, 1.0
; GFX1150-FAKE16-NEXT: s_denorm_mode 15
-; GFX1150-FAKE16-NEXT: s_delay_alu instid0(TRANS32_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1150-FAKE16-NEXT: s_delay_alu instid0(TRANS32_DEP_1) | instid1(SALU_CYCLE_1)
; GFX1150-FAKE16-NEXT: v_fma_f32 v9, -v7, v8, 1.0
-; GFX1150-FAKE16-NEXT: v_fmac_f32_e32 v8, v9, v8
; GFX1150-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1150-FAKE16-NEXT: v_fmac_f32_e32 v8, v9, v8
; GFX1150-FAKE16-NEXT: v_mul_f32_e32 v9, v5, v8
-; GFX1150-FAKE16-NEXT: v_fma_f32 v10, -v7, v9, v5
; GFX1150-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1150-FAKE16-NEXT: v_fma_f32 v10, -v7, v9, v5
; GFX1150-FAKE16-NEXT: v_fmac_f32_e32 v9, v10, v8
+; GFX1150-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1150-FAKE16-NEXT: v_fma_f32 v5, -v7, v9, v5
; GFX1150-FAKE16-NEXT: s_denorm_mode 12
-; GFX1150-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
+; GFX1150-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX1150-FAKE16-NEXT: v_div_fmas_f32 v5, v5, v8, v9
; GFX1150-FAKE16-NEXT: v_cmp_gt_i32_e32 vcc_lo, 12, v6
+; GFX1150-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX1150-FAKE16-NEXT: v_div_fixup_f32 v5, v5, v3, 1.0
; GFX1150-FAKE16-NEXT: s_cbranch_vccnz .LBB0_8
; GFX1150-FAKE16-NEXT: ; %bb.5: ; %frem.loop_body.preheader
@@ -967,7 +980,7 @@ define amdgpu_kernel void @frem_f16(ptr addrspace(1) %out, ptr addrspace(1) %in1
; GFX1150-FAKE16-NEXT: .LBB0_9: ; %Flow19
; GFX1150-FAKE16-NEXT: v_dual_mov_b32 v3, 0 :: v_dual_and_b32 v0, 0x7fff, v0
; GFX1150-FAKE16-NEXT: v_cmp_lg_f16_e32 vcc_lo, 0, v1
-; GFX1150-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2)
+; GFX1150-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX1150-FAKE16-NEXT: v_cmp_nle_f16_e64 s0, 0x7c00, v0
; GFX1150-FAKE16-NEXT: s_and_b32 vcc_lo, s0, vcc_lo
; GFX1150-FAKE16-NEXT: v_cndmask_b32_e32 v0, 0x7e00, v2, vcc_lo
@@ -992,12 +1005,13 @@ define amdgpu_kernel void @frem_f16(ptr addrspace(1) %out, ptr addrspace(1) %in1
; GFX1200-TRUE16-NEXT: s_and_b32 s2, s1, 0x7fff
; GFX1200-TRUE16-NEXT: s_cvt_f32_f16 s1, s0
; GFX1200-TRUE16-NEXT: s_cvt_f32_f16 s0, s2
-; GFX1200-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_3)
+; GFX1200-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_3) | instskip(NEXT) | instid1(SALU_CYCLE_3)
; GFX1200-TRUE16-NEXT: s_cmp_ngt_f32 s1, s0
; GFX1200-TRUE16-NEXT: s_cbranch_scc0 .LBB0_2
; GFX1200-TRUE16-NEXT: ; %bb.1: ; %frem.else
; GFX1200-TRUE16-NEXT: s_cmp_eq_f32 s1, s0
; GFX1200-TRUE16-NEXT: v_and_b16 v0.h, 0x8000, v0.l
+; GFX1200-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_2)
; GFX1200-TRUE16-NEXT: s_cselect_b32 s2, -1, 0
; GFX1200-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX1200-TRUE16-NEXT: v_cndmask_b16 v2.l, v0.l, v0.h, s2
@@ -1012,6 +1026,7 @@ define amdgpu_kernel void @frem_f16(ptr addrspace(1) %out, ptr addrspace(1) %in1
; GFX1200-TRUE16-NEXT: s_cselect_b32 s2, 1, 0
; GFX1200-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX1200-TRUE16-NEXT: s_cmp_lg_u32 s2, 1
+; GFX1200-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1200-TRUE16-NEXT: s_cbranch_scc1 .LBB0_8
; GFX1200-TRUE16-NEXT: ; %bb.4: ; %frem.compute
; GFX1200-TRUE16-NEXT: v_frexp_mant_f32_e32 v3, s0
@@ -1028,10 +1043,11 @@ define amdgpu_kernel void @frem_f16(ptr addrspace(1) %out, ptr addrspace(1) %in1
; GFX1200-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1200-TRUE16-NEXT: v_not_b32_e32 v5, v5
; GFX1200-TRUE16-NEXT: v_rcp_f32_e32 v7, v6
-; GFX1200-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_2) | instid1(TRANS32_DEP_1)
+; GFX1200-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1200-TRUE16-NEXT: v_add_nc_u32_e32 v5, v5, v4
; GFX1200-TRUE16-NEXT: v_div_scale_f32 v4, vcc_lo, 1.0, v3, 1.0
; GFX1200-TRUE16-NEXT: s_denorm_mode 15
+; GFX1200-TRUE16-NEXT: s_delay_alu instid0(TRANS32_DEP_1) | instid1(SALU_CYCLE_1)
; GFX1200-TRUE16-NEXT: v_fma_f32 v8, -v6, v7, 1.0
; GFX1200-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1200-TRUE16-NEXT: v_fmac_f32_e32 v7, v8, v7
@@ -1039,9 +1055,10 @@ define amdgpu_kernel void @frem_f16(ptr addrspace(1) %out, ptr addrspace(1) %in1
; GFX1200-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1200-TRUE16-NEXT: v_fma_f32 v9, -v6, v8, v4
; GFX1200-TRUE16-NEXT: v_fmac_f32_e32 v8, v9, v7
-; GFX1200-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
+; GFX1200-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1200-TRUE16-NEXT: v_fma_f32 v4, -v6, v8, v4
; GFX1200-TRUE16-NEXT: s_denorm_mode 12
+; GFX1200-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX1200-TRUE16-NEXT: v_div_fmas_f32 v4, v4, v7, v8
; GFX1200-TRUE16-NEXT: v_cmp_gt_i32_e32 vcc_lo, 12, v5
; GFX1200-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2)
@@ -1102,15 +1119,16 @@ define amdgpu_kernel void @frem_f16(ptr addrspace(1) %out, ptr addrspace(1) %in1
; GFX1200-FAKE16-NEXT: s_and_b32 s2, s1, 0x7fff
; GFX1200-FAKE16-NEXT: s_cvt_f32_f16 s1, s0
; GFX1200-FAKE16-NEXT: s_cvt_f32_f16 s0, s2
-; GFX1200-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_3)
+; GFX1200-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_3) | instskip(NEXT) | instid1(SALU_CYCLE_3)
; GFX1200-FAKE16-NEXT: s_cmp_ngt_f32 s1, s0
; GFX1200-FAKE16-NEXT: s_cbranch_scc0 .LBB0_2
; GFX1200-FAKE16-NEXT: ; %bb.1: ; %frem.else
; GFX1200-FAKE16-NEXT: s_cmp_eq_f32 s1, s0
; GFX1200-FAKE16-NEXT: v_and_b32_e32 v2, 0x8000, v0
; GFX1200-FAKE16-NEXT: s_mov_b32 s2, 0
+; GFX1200-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1200-FAKE16-NEXT: s_cselect_b32 vcc_lo, -1, 0
-; GFX1200-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1200-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX1200-FAKE16-NEXT: v_cndmask_b32_e32 v2, v0, v2, vcc_lo
; GFX1200-FAKE16-NEXT: s_branch .LBB0_3
; GFX1200-FAKE16-NEXT: .LBB0_2:
@@ -1121,6 +1139,7 @@ define amdgpu_kernel void @frem_f16(ptr addrspace(1) %out, ptr addrspace(1) %in1
; GFX1200-FAKE16-NEXT: s_and_b32 s2, s2, exec_lo
; GFX1200-FAKE16-NEXT: s_cselect_b32 s2, 1, 0
; GFX1200-FAKE16-NEXT: s_cmp_lg_u32 s2, 1
+; GFX1200-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1200-FAKE16-NEXT: s_cbranch_scc1 .LBB0_9
; GFX1200-FAKE16-NEXT: ; %bb.4: ; %frem.compute
; GFX1200-FAKE16-NEXT: v_frexp_mant_f32_e32 v3, s0
@@ -1142,20 +1161,21 @@ define amdgpu_kernel void @frem_f16(ptr addrspace(1) %out, ptr addrspace(1) %in1
; GFX1200-FAKE16-NEXT: v_add_nc_u32_e32 v6, v6, v5
; GFX1200-FAKE16-NEXT: v_div_scale_f32 v5, vcc_lo, 1.0, v3, 1.0
; GFX1200-FAKE16-NEXT: s_denorm_mode 15
-; GFX1200-FAKE16-NEXT: s_delay_alu instid0(TRANS32_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1200-FAKE16-NEXT: s_delay_alu instid0(TRANS32_DEP_1) | instid1(SALU_CYCLE_1)
; GFX1200-FAKE16-NEXT: v_fma_f32 v9, -v7, v8, 1.0
-; GFX1200-FAKE16-NEXT: v_fmac_f32_e32 v8, v9, v8
; GFX1200-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1200-FAKE16-NEXT: v_fmac_f32_e32 v8, v9, v8
; GFX1200-FAKE16-NEXT: v_mul_f32_e32 v9, v5, v8
-; GFX1200-FAKE16-NEXT: v_fma_f32 v10, -v7, v9, v5
; GFX1200-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1200-FAKE16-NEXT: v_fma_f32 v10, -v7, v9, v5
; GFX1200-FAKE16-NEXT: v_fmac_f32_e32 v9, v10, v8
+; GFX1200-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_2) | instid1(VALU_DEP_1)
; GFX1200-FAKE16-NEXT: v_fma_f32 v5, -v7, v9, v5
; GFX1200-FAKE16-NEXT: s_denorm_mode 12
; GFX1200-FAKE16-NEXT: s_wait_alu depctr_va_vcc(0)
-; GFX1200-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX1200-FAKE16-NEXT: v_div_fmas_f32 v5, v5, v8, v9
; GFX1200-FAKE16-NEXT: v_cmp_gt_i32_e32 vcc_lo, 12, v6
+; GFX1200-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX1200-FAKE16-NEXT: v_div_fixup_f32 v5, v5, v3, 1.0
; GFX1200-FAKE16-NEXT: s_cbranch_vccnz .LBB0_8
; GFX1200-FAKE16-NEXT: ; %bb.5: ; %frem.loop_body.preheader
@@ -2696,6 +2716,7 @@ define amdgpu_kernel void @frem_f32(ptr addrspace(1) %out, ptr addrspace(1) %in1
; GFX11-NEXT: s_waitcnt vmcnt(0)
; GFX11-NEXT: v_cmp_ngt_f32_e64 s2, |v0|, |v1|
; GFX11-NEXT: s_and_b32 vcc_lo, exec_lo, s2
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_vccz .LBB3_2
; GFX11-NEXT: ; %bb.1: ; %frem.else
; GFX11-NEXT: v_and_b32_e32 v2, 0x80000000, v0
@@ -2711,6 +2732,7 @@ define amdgpu_kernel void @frem_f32(ptr addrspace(1) %out, ptr addrspace(1) %in1
; GFX11-NEXT: s_and_b32 s2, s2, exec_lo
; GFX11-NEXT: s_cselect_b32 s2, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s2, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB3_9
; GFX11-NEXT: ; %bb.4: ; %frem.compute
; GFX11-NEXT: v_frexp_mant_f32_e64 v3, |v1|
@@ -2740,9 +2762,10 @@ define amdgpu_kernel void @frem_f32(ptr addrspace(1) %out, ptr addrspace(1) %in1
; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_fma_f32 v10, -v7, v9, v5
; GFX11-NEXT: v_fmac_f32_e32 v9, v10, v8
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_fma_f32 v5, -v7, v9, v5
; GFX11-NEXT: s_denorm_mode 12
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: v_div_fmas_f32 v5, v5, v8, v9
; GFX11-NEXT: v_cmp_gt_i32_e32 vcc_lo, 13, v6
; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2)
@@ -2792,6 +2815,7 @@ define amdgpu_kernel void @frem_f32(ptr addrspace(1) %out, ptr addrspace(1) %in1
; GFX11-NEXT: v_cmp_lg_f32_e32 vcc_lo, 0, v1
; GFX11-NEXT: v_cmp_nle_f32_e64 s2, 0x7f800000, |v0|
; GFX11-NEXT: s_and_b32 vcc_lo, s2, vcc_lo
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: v_dual_mov_b32 v3, 0 :: v_dual_cndmask_b32 v0, 0x7fc00000, v2
; GFX11-NEXT: global_store_b32 v3, v0, s[0:1]
; GFX11-NEXT: s_endpgm
@@ -2828,6 +2852,7 @@ define amdgpu_kernel void @frem_f32(ptr addrspace(1) %out, ptr addrspace(1) %in1
; GFX1150-NEXT: s_and_b32 s0, s0, exec_lo
; GFX1150-NEXT: s_cselect_b32 s0, 1, 0
; GFX1150-NEXT: s_cmp_lg_u32 s0, 1
+; GFX1150-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1150-NEXT: s_cbranch_scc1 .LBB3_9
; GFX1150-NEXT: ; %bb.4: ; %frem.compute
; GFX1150-NEXT: v_frexp_mant_f32_e64 v4, |v0|
@@ -2849,19 +2874,21 @@ define amdgpu_kernel void @frem_f32(ptr addrspace(1) %out, ptr addrspace(1) %in1
; GFX1150-NEXT: v_add_nc_u32_e32 v7, v7, v6
; GFX1150-NEXT: v_div_scale_f32 v6, vcc_lo, 1.0, v4, 1.0
; GFX1150-NEXT: s_denorm_mode 15
-; GFX1150-NEXT: s_delay_alu instid0(TRANS32_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1150-NEXT: s_delay_alu instid0(TRANS32_DEP_1) | instid1(SALU_CYCLE_1)
; GFX1150-NEXT: v_fma_f32 v10, -v8, v9, 1.0
-; GFX1150-NEXT: v_fmac_f32_e32 v9, v10, v9
; GFX1150-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1150-NEXT: v_fmac_f32_e32 v9, v10, v9
; GFX1150-NEXT: v_mul_f32_e32 v10, v6, v9
-; GFX1150-NEXT: v_fma_f32 v11, -v8, v10, v6
; GFX1150-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1150-NEXT: v_fma_f32 v11, -v8, v10, v6
; GFX1150-NEXT: v_fmac_f32_e32 v10, v11, v9
+; GFX1150-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1150-NEXT: v_fma_f32 v6, -v8, v10, v6
; GFX1150-NEXT: s_denorm_mode 12
-; GFX1150-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
+; GFX1150-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX1150-NEXT: v_div_fmas_f32 v6, v6, v9, v10
; GFX1150-NEXT: v_cmp_gt_i32_e32 vcc_lo, 13, v7
+; GFX1150-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX1150-NEXT: v_div_fixup_f32 v6, v6, v4, 1.0
; GFX1150-NEXT: s_cbranch_vccnz .LBB3_8
; GFX1150-NEXT: ; %bb.5: ; %frem.loop_body.preheader
@@ -2912,6 +2939,7 @@ define amdgpu_kernel void @frem_f32(ptr addrspace(1) %out, ptr addrspace(1) %in1
; GFX1150-NEXT: v_cmp_nle_f32_e64 s0, 0x7f800000, v1
; GFX1150-NEXT: v_mov_b32_e32 v2, 0
; GFX1150-NEXT: s_and_b32 vcc_lo, s0, vcc_lo
+; GFX1150-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1150-NEXT: v_cndmask_b32_e32 v0, 0x7fc00000, v3, vcc_lo
; GFX1150-NEXT: global_store_b32 v2, v0, s[8:9]
; GFX1150-NEXT: s_endpgm
@@ -2948,6 +2976,7 @@ define amdgpu_kernel void @frem_f32(ptr addrspace(1) %out, ptr addrspace(1) %in1
; GFX1200-NEXT: s_and_b32 s0, s0, exec_lo
; GFX1200-NEXT: s_cselect_b32 s0, 1, 0
; GFX1200-NEXT: s_cmp_lg_u32 s0, 1
+; GFX1200-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1200-NEXT: s_cbranch_scc1 .LBB3_9
; GFX1200-NEXT: ; %bb.4: ; %frem.compute
; GFX1200-NEXT: v_frexp_mant_f32_e64 v4, |v0|
@@ -2969,20 +2998,21 @@ define amdgpu_kernel void @frem_f32(ptr addrspace(1) %out, ptr addrspace(1) %in1
; GFX1200-NEXT: v_add_nc_u32_e32 v7, v7, v6
; GFX1200-NEXT: v_div_scale_f32 v6, vcc_lo, 1.0, v4, 1.0
; GFX1200-NEXT: s_denorm_mode 15
-; GFX1200-NEXT: s_delay_alu instid0(TRANS32_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1200-NEXT: s_delay_alu instid0(TRANS32_DEP_1) | instid1(SALU_CYCLE_1)
; GFX1200-NEXT: v_fma_f32 v10, -v8, v9, 1.0
-; GFX1200-NEXT: v_fmac_f32_e32 v9, v10, v9
; GFX1200-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1200-NEXT: v_fmac_f32_e32 v9, v10, v9
; GFX1200-NEXT: v_mul_f32_e32 v10, v6, v9
-; GFX1200-NEXT: v_fma_f32 v11, -v8, v10, v6
; GFX1200-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1200-NEXT: v_fma_f32 v11, -v8, v10, v6
; GFX1200-NEXT: v_fmac_f32_e32 v10, v11, v9
+; GFX1200-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_2) | instid1(VALU_DEP_1)
; GFX1200-NEXT: v_fma_f32 v6, -v8, v10, v6
; GFX1200-NEXT: s_denorm_mode 12
; GFX1200-NEXT: s_wait_alu depctr_va_vcc(0)
-; GFX1200-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX1200-NEXT: v_div_fmas_f32 v6, v6, v9, v10
; GFX1200-NEXT: v_cmp_gt_i32_e32 vcc_lo, 13, v7
+; GFX1200-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX1200-NEXT: v_div_fixup_f32 v6, v6, v4, 1.0
; GFX1200-NEXT: s_cbranch_vccnz .LBB3_8
; GFX1200-NEXT: ; %bb.5: ; %frem.loop_body.preheader
@@ -3237,11 +3267,12 @@ define amdgpu_kernel void @fast_frem_f32(ptr addrspace(1) %out, ptr addrspace(1)
; GFX11-NEXT: v_fmac_f32_e32 v6, v7, v5
; GFX11-NEXT: v_fma_f32 v3, -v4, v6, v3
; GFX11-NEXT: s_denorm_mode 12
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: v_div_fmas_f32 v3, v3, v5, v6
-; GFX11-NEXT: v_div_fixup_f32 v3, v3, v2, v1
; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX11-NEXT: v_div_fixup_f32 v3, v3, v2, v1
; GFX11-NEXT: v_trunc_f32_e32 v3, v3
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_fma_f32 v1, -v3, v2, v1
; GFX11-NEXT: global_store_b32 v0, v1, s[0:1]
; GFX11-NEXT: s_endpgm
@@ -3259,9 +3290,10 @@ define amdgpu_kernel void @fast_frem_f32(ptr addrspace(1) %out, ptr addrspace(1)
; GFX1150-NEXT: s_waitcnt vmcnt(0)
; GFX1150-NEXT: v_div_scale_f32 v4, null, v2, v2, v1
; GFX1150-NEXT: v_div_scale_f32 v3, vcc_lo, v1, v2, v1
-; GFX1150-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(TRANS32_DEP_1)
+; GFX1150-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX1150-NEXT: v_rcp_f32_e32 v5, v4
; GFX1150-NEXT: s_denorm_mode 15
+; GFX1150-NEXT: s_delay_alu instid0(TRANS32_DEP_1) | instid1(SALU_CYCLE_1)
; GFX1150-NEXT: v_fma_f32 v6, -v4, v5, 1.0
; GFX1150-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1150-NEXT: v_fmac_f32_e32 v5, v6, v5
@@ -3269,9 +3301,10 @@ define amdgpu_kernel void @fast_frem_f32(ptr addrspace(1) %out, ptr addrspace(1)
; GFX1150-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1150-NEXT: v_fma_f32 v7, -v4, v6, v3
; GFX1150-NEXT: v_fmac_f32_e32 v6, v7, v5
-; GFX1150-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
+; GFX1150-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1150-NEXT: v_fma_f32 v3, -v4, v6, v3
; GFX1150-NEXT: s_denorm_mode 12
+; GFX1150-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX1150-NEXT: v_div_fmas_f32 v3, v3, v5, v6
; GFX1150-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1150-NEXT: v_div_fixup_f32 v3, v3, v2, v1
@@ -3295,9 +3328,10 @@ define amdgpu_kernel void @fast_frem_f32(ptr addrspace(1) %out, ptr addrspace(1)
; GFX1200-NEXT: s_wait_loadcnt 0x0
; GFX1200-NEXT: v_div_scale_f32 v4, null, v2, v2, v1
; GFX1200-NEXT: v_div_scale_f32 v3, vcc_lo, v1, v2, v1
-; GFX1200-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(TRANS32_DEP_1)
+; GFX1200-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX1200-NEXT: v_rcp_f32_e32 v5, v4
; GFX1200-NEXT: s_denorm_mode 15
+; GFX1200-NEXT: s_delay_alu instid0(TRANS32_DEP_1) | instid1(SALU_CYCLE_1)
; GFX1200-NEXT: v_fma_f32 v6, -v4, v5, 1.0
; GFX1200-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1200-NEXT: v_fmac_f32_e32 v5, v6, v5
@@ -3305,9 +3339,10 @@ define amdgpu_kernel void @fast_frem_f32(ptr addrspace(1) %out, ptr addrspace(1)
; GFX1200-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1200-NEXT: v_fma_f32 v7, -v4, v6, v3
; GFX1200-NEXT: v_fmac_f32_e32 v6, v7, v5
-; GFX1200-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
+; GFX1200-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1200-NEXT: v_fma_f32 v3, -v4, v6, v3
; GFX1200-NEXT: s_denorm_mode 12
+; GFX1200-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX1200-NEXT: v_div_fmas_f32 v3, v3, v5, v6
; GFX1200-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1200-NEXT: v_div_fixup_f32 v3, v3, v2, v1
@@ -3515,11 +3550,12 @@ define amdgpu_kernel void @unsafe_frem_f32(ptr addrspace(1) %out, ptr addrspace(
; GFX11-NEXT: v_fmac_f32_e32 v6, v7, v5
; GFX11-NEXT: v_fma_f32 v3, -v4, v6, v3
; GFX11-NEXT: s_denorm_mode 12
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: v_div_fmas_f32 v3, v3, v5, v6
-; GFX11-NEXT: v_div_fixup_f32 v3, v3, v2, v1
; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX11-NEXT: v_div_fixup_f32 v3, v3, v2, v1
; GFX11-NEXT: v_trunc_f32_e32 v3, v3
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_fma_f32 v1, -v3, v2, v1
; GFX11-NEXT: global_store_b32 v0, v1, s[0:1]
; GFX11-NEXT: s_endpgm
@@ -3537,9 +3573,10 @@ define amdgpu_kernel void @unsafe_frem_f32(ptr addrspace(1) %out, ptr addrspace(
; GFX1150-NEXT: s_waitcnt vmcnt(0)
; GFX1150-NEXT: v_div_scale_f32 v4, null, v2, v2, v1
; GFX1150-NEXT: v_div_scale_f32 v3, vcc_lo, v1, v2, v1
-; GFX1150-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(TRANS32_DEP_1)
+; GFX1150-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX1150-NEXT: v_rcp_f32_e32 v5, v4
; GFX1150-NEXT: s_denorm_mode 15
+; GFX1150-NEXT: s_delay_alu instid0(TRANS32_DEP_1) | instid1(SALU_CYCLE_1)
; GFX1150-NEXT: v_fma_f32 v6, -v4, v5, 1.0
; GFX1150-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1150-NEXT: v_fmac_f32_e32 v5, v6, v5
@@ -3547,9 +3584,10 @@ define amdgpu_kernel void @unsafe_frem_f32(ptr addrspace(1) %out, ptr addrspace(
; GFX1150-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1150-NEXT: v_fma_f32 v7, -v4, v6, v3
; GFX1150-NEXT: v_fmac_f32_e32 v6, v7, v5
-; GFX1150-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
+; GFX1150-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1150-NEXT: v_fma_f32 v3, -v4, v6, v3
; GFX1150-NEXT: s_denorm_mode 12
+; GFX1150-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX1150-NEXT: v_div_fmas_f32 v3, v3, v5, v6
; GFX1150-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1150-NEXT: v_div_fixup_f32 v3, v3, v2, v1
@@ -3573,9 +3611,10 @@ define amdgpu_kernel void @unsafe_frem_f32(ptr addrspace(1) %out, ptr addrspace(
; GFX1200-NEXT: s_wait_loadcnt 0x0
; GFX1200-NEXT: v_div_scale_f32 v4, null, v2, v2, v1
; GFX1200-NEXT: v_div_scale_f32 v3, vcc_lo, v1, v2, v1
-; GFX1200-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(TRANS32_DEP_1)
+; GFX1200-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX1200-NEXT: v_rcp_f32_e32 v5, v4
; GFX1200-NEXT: s_denorm_mode 15
+; GFX1200-NEXT: s_delay_alu instid0(TRANS32_DEP_1) | instid1(SALU_CYCLE_1)
; GFX1200-NEXT: v_fma_f32 v6, -v4, v5, 1.0
; GFX1200-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1200-NEXT: v_fmac_f32_e32 v5, v6, v5
@@ -3583,9 +3622,10 @@ define amdgpu_kernel void @unsafe_frem_f32(ptr addrspace(1) %out, ptr addrspace(
; GFX1200-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1200-NEXT: v_fma_f32 v7, -v4, v6, v3
; GFX1200-NEXT: v_fmac_f32_e32 v6, v7, v5
-; GFX1200-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
+; GFX1200-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1200-NEXT: v_fma_f32 v3, -v4, v6, v3
; GFX1200-NEXT: s_denorm_mode 12
+; GFX1200-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX1200-NEXT: v_div_fmas_f32 v3, v3, v5, v6
; GFX1200-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1200-NEXT: v_div_fixup_f32 v3, v3, v2, v1
@@ -4177,6 +4217,7 @@ define amdgpu_kernel void @frem_f64(ptr addrspace(1) %out, ptr addrspace(1) %in1
; GFX11-NEXT: s_waitcnt vmcnt(0)
; GFX11-NEXT: v_cmp_ngt_f64_e64 s2, |v[0:1]|, |v[2:3]|
; GFX11-NEXT: s_and_b32 vcc_lo, exec_lo, s2
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_vccz .LBB6_2
; GFX11-NEXT: ; %bb.1: ; %frem.else
; GFX11-NEXT: v_cmp_eq_f64_e64 vcc_lo, |v[0:1]|, |v[2:3]|
@@ -4194,6 +4235,7 @@ define amdgpu_kernel void @frem_f64(ptr addrspace(1) %out, ptr addrspace(1) %in1
; GFX11-NEXT: s_and_b32 s2, s2, exec_lo
; GFX11-NEXT: s_cselect_b32 s2, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s2, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB6_9
; GFX11-NEXT: ; %bb.4: ; %frem.compute
; GFX11-NEXT: v_frexp_mant_f64_e64 v[4:5], |v[0:1]|
@@ -4273,6 +4315,7 @@ define amdgpu_kernel void @frem_f64(ptr addrspace(1) %out, ptr addrspace(1) %in1
; GFX11-NEXT: v_cmp_lg_f64_e32 vcc_lo, 0, v[2:3]
; GFX11-NEXT: v_cmp_nle_f64_e64 s2, 0x7ff00000, |v[0:1]|
; GFX11-NEXT: s_and_b32 vcc_lo, s2, vcc_lo
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: v_dual_mov_b32 v6, 0 :: v_dual_cndmask_b32 v1, 0x7ff80000, v5
; GFX11-NEXT: v_cndmask_b32_e32 v0, 0, v4, vcc_lo
; GFX11-NEXT: global_store_b64 v6, v[0:1], s[0:1]
@@ -4291,6 +4334,7 @@ define amdgpu_kernel void @frem_f64(ptr addrspace(1) %out, ptr addrspace(1) %in1
; GFX1150-NEXT: s_waitcnt vmcnt(0)
; GFX1150-NEXT: v_cmp_ngt_f64_e64 s2, |v[0:1]|, |v[2:3]|
; GFX1150-NEXT: s_and_b32 vcc_lo, exec_lo, s2
+; GFX1150-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1150-NEXT: s_cbranch_vccz .LBB6_2
; GFX1150-NEXT: ; %bb.1: ; %frem.else
; GFX1150-NEXT: v_cmp_eq_f64_e64 vcc_lo, |v[0:1]|, |v[2:3]|
@@ -4308,6 +4352,7 @@ define amdgpu_kernel void @frem_f64(ptr addrspace(1) %out, ptr addrspace(1) %in1
; GFX1150-NEXT: s_and_b32 s2, s2, exec_lo
; GFX1150-NEXT: s_cselect_b32 s2, 1, 0
; GFX1150-NEXT: s_cmp_lg_u32 s2, 1
+; GFX1150-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1150-NEXT: s_cbranch_scc1 .LBB6_9
; GFX1150-NEXT: ; %bb.4: ; %frem.compute
; GFX1150-NEXT: v_frexp_mant_f64_e64 v[4:5], |v[0:1]|
@@ -4386,6 +4431,7 @@ define amdgpu_kernel void @frem_f64(ptr addrspace(1) %out, ptr addrspace(1) %in1
; GFX1150-NEXT: v_cmp_lg_f64_e32 vcc_lo, 0, v[2:3]
; GFX1150-NEXT: v_cmp_nle_f64_e64 s2, 0x7ff00000, |v[0:1]|
; GFX1150-NEXT: s_and_b32 vcc_lo, s2, vcc_lo
+; GFX1150-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1150-NEXT: v_dual_mov_b32 v6, 0 :: v_dual_cndmask_b32 v1, 0x7ff80000, v5
; GFX1150-NEXT: v_cndmask_b32_e32 v0, 0, v4, vcc_lo
; GFX1150-NEXT: global_store_b64 v6, v[0:1], s[0:1]
@@ -4404,6 +4450,7 @@ define amdgpu_kernel void @frem_f64(ptr addrspace(1) %out, ptr addrspace(1) %in1
; GFX1200-NEXT: s_wait_loadcnt 0x0
; GFX1200-NEXT: v_cmp_ngt_f64_e64 s2, |v[0:1]|, |v[2:3]|
; GFX1200-NEXT: s_and_b32 vcc_lo, exec_lo, s2
+; GFX1200-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1200-NEXT: s_cbranch_vccz .LBB6_2
; GFX1200-NEXT: ; %bb.1: ; %frem.else
; GFX1200-NEXT: v_cmp_eq_f64_e64 vcc_lo, |v[0:1]|, |v[2:3]|
@@ -4421,6 +4468,7 @@ define amdgpu_kernel void @frem_f64(ptr addrspace(1) %out, ptr addrspace(1) %in1
; GFX1200-NEXT: s_and_b32 s2, s2, exec_lo
; GFX1200-NEXT: s_cselect_b32 s2, 1, 0
; GFX1200-NEXT: s_cmp_lg_u32 s2, 1
+; GFX1200-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1200-NEXT: s_cbranch_scc1 .LBB6_9
; GFX1200-NEXT: ; %bb.4: ; %frem.compute
; GFX1200-NEXT: v_frexp_mant_f64_e64 v[4:5], |v[0:1]|
@@ -6045,6 +6093,7 @@ define amdgpu_kernel void @frem_v2f16(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX11-TRUE16-NEXT: s_and_b32 s2, s2, exec_lo
; GFX11-TRUE16-NEXT: s_cselect_b32 s2, 1, 0
; GFX11-TRUE16-NEXT: s_cmp_lg_u32 s2, 1
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_cbranch_scc1 .LBB9_8
; GFX11-TRUE16-NEXT: ; %bb.4: ; %frem.compute19
; GFX11-TRUE16-NEXT: v_frexp_exp_i32_f32_e32 v5, v4
@@ -6076,9 +6125,10 @@ define amdgpu_kernel void @frem_v2f16(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX11-TRUE16-NEXT: v_fmac_f32_e32 v8, v9, v7
; GFX11-TRUE16-NEXT: v_fma_f32 v4, -v6, v8, v4
; GFX11-TRUE16-NEXT: s_denorm_mode 12
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_div_fmas_f32 v4, v4, v7, v8
; GFX11-TRUE16-NEXT: v_cmp_gt_i32_e32 vcc_lo, 12, v5
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX11-TRUE16-NEXT: v_div_fixup_f32 v4, v4, v3, 1.0
; GFX11-TRUE16-NEXT: s_cbranch_vccnz .LBB9_7
; GFX11-TRUE16-NEXT: ; %bb.5: ; %frem.loop_body27.preheader
@@ -6127,6 +6177,7 @@ define amdgpu_kernel void @frem_v2f16(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX11-TRUE16-NEXT: s_and_b32 s2, s2, exec_lo
; GFX11-TRUE16-NEXT: s_cselect_b32 s2, 1, 0
; GFX11-TRUE16-NEXT: s_cmp_lg_u32 s2, 1
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_cbranch_scc1 .LBB9_16
; GFX11-TRUE16-NEXT: ; %bb.12: ; %frem.compute
; GFX11-TRUE16-NEXT: v_frexp_exp_i32_f32_e32 v8, v7
@@ -6158,9 +6209,10 @@ define amdgpu_kernel void @frem_v2f16(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX11-TRUE16-NEXT: v_fmac_f32_e32 v11, v12, v10
; GFX11-TRUE16-NEXT: v_fma_f32 v7, -v9, v11, v7
; GFX11-TRUE16-NEXT: s_denorm_mode 12
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_div_fmas_f32 v7, v7, v10, v11
; GFX11-TRUE16-NEXT: v_cmp_gt_i32_e32 vcc_lo, 12, v8
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX11-TRUE16-NEXT: v_div_fixup_f32 v7, v7, v6, 1.0
; GFX11-TRUE16-NEXT: s_cbranch_vccnz .LBB9_15
; GFX11-TRUE16-NEXT: ; %bb.13: ; %frem.loop_body.preheader
@@ -6231,6 +6283,7 @@ define amdgpu_kernel void @frem_v2f16(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX11-FAKE16-NEXT: s_and_b32 s2, s2, exec_lo
; GFX11-FAKE16-NEXT: s_cselect_b32 s2, 1, 0
; GFX11-FAKE16-NEXT: s_cmp_lg_u32 s2, 1
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_cbranch_scc1 .LBB9_9
; GFX11-FAKE16-NEXT: ; %bb.4: ; %frem.compute19
; GFX11-FAKE16-NEXT: v_frexp_mant_f32_e32 v2, v4
@@ -6260,9 +6313,10 @@ define amdgpu_kernel void @frem_v2f16(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_fma_f32 v10, -v7, v9, v5
; GFX11-FAKE16-NEXT: v_fmac_f32_e32 v9, v10, v8
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_fma_f32 v5, -v7, v9, v5
; GFX11-FAKE16-NEXT: s_denorm_mode 12
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_div_fmas_f32 v5, v5, v8, v9
; GFX11-FAKE16-NEXT: v_cmp_gt_i32_e32 vcc_lo, 12, v6
; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2)
@@ -6333,6 +6387,7 @@ define amdgpu_kernel void @frem_v2f16(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX11-FAKE16-NEXT: s_and_b32 s2, s2, exec_lo
; GFX11-FAKE16-NEXT: s_cselect_b32 s2, 1, 0
; GFX11-FAKE16-NEXT: s_cmp_lg_u32 s2, 1
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_cbranch_scc1 .LBB9_18
; GFX11-FAKE16-NEXT: ; %bb.13: ; %frem.compute
; GFX11-FAKE16-NEXT: v_frexp_mant_f32_e32 v5, v7
@@ -6362,9 +6417,10 @@ define amdgpu_kernel void @frem_v2f16(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_fma_f32 v13, -v10, v12, v8
; GFX11-FAKE16-NEXT: v_fmac_f32_e32 v12, v13, v11
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_fma_f32 v8, -v10, v12, v8
; GFX11-FAKE16-NEXT: s_denorm_mode 12
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_div_fmas_f32 v8, v8, v11, v12
; GFX11-FAKE16-NEXT: v_cmp_gt_i32_e32 vcc_lo, 12, v9
; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2)
@@ -6419,10 +6475,11 @@ define amdgpu_kernel void @frem_v2f16(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX11-FAKE16-NEXT: v_cmp_nle_f16_e64 s2, 0x7c00, |v3|
; GFX11-FAKE16-NEXT: v_cndmask_b32_e32 v0, 0x7e00, v2, vcc_lo
; GFX11-FAKE16-NEXT: v_cmp_lg_f16_e32 vcc_lo, 0, v4
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_2) | instid1(VALU_DEP_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_and_b32_e32 v0, 0xffff, v0
; GFX11-FAKE16-NEXT: s_and_b32 vcc_lo, s2, vcc_lo
; GFX11-FAKE16-NEXT: v_cndmask_b32_e32 v2, 0x7e00, v5, vcc_lo
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_lshl_or_b32 v0, v2, 16, v0
; GFX11-FAKE16-NEXT: global_store_b32 v1, v0, s[0:1]
; GFX11-FAKE16-NEXT: s_endpgm
@@ -6445,12 +6502,13 @@ define amdgpu_kernel void @frem_v2f16(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX1150-TRUE16-NEXT: s_and_b32 s5, s3, 0x7fff
; GFX1150-TRUE16-NEXT: s_cvt_f32_f16 s6, s2
; GFX1150-TRUE16-NEXT: s_cvt_f32_f16 s5, s5
-; GFX1150-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_3)
+; GFX1150-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_3) | instskip(NEXT) | instid1(SALU_CYCLE_3)
; GFX1150-TRUE16-NEXT: s_cmp_ngt_f32 s6, s5
; GFX1150-TRUE16-NEXT: s_cbranch_scc0 .LBB9_2
; GFX1150-TRUE16-NEXT: ; %bb.1: ; %frem.else20
; GFX1150-TRUE16-NEXT: s_cmp_eq_f32 s6, s5
; GFX1150-TRUE16-NEXT: v_and_b16 v0.l, 0x8000, s4
+; GFX1150-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_2)
; GFX1150-TRUE16-NEXT: s_cselect_b32 s7, -1, 0
; GFX1150-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX1150-TRUE16-NEXT: v_cndmask_b16 v0.l, s4, v0.l, s7
@@ -6464,6 +6522,7 @@ define amdgpu_kernel void @frem_v2f16(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX1150-TRUE16-NEXT: s_and_b32 s7, s7, exec_lo
; GFX1150-TRUE16-NEXT: s_cselect_b32 s7, 1, 0
; GFX1150-TRUE16-NEXT: s_cmp_lg_u32 s7, 1
+; GFX1150-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1150-TRUE16-NEXT: s_cbranch_scc1 .LBB9_8
; GFX1150-TRUE16-NEXT: ; %bb.4: ; %frem.compute19
; GFX1150-TRUE16-NEXT: v_frexp_mant_f32_e32 v1, s5
@@ -6480,10 +6539,11 @@ define amdgpu_kernel void @frem_v2f16(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX1150-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1150-TRUE16-NEXT: v_not_b32_e32 v3, v3
; GFX1150-TRUE16-NEXT: v_rcp_f32_e32 v5, v4
-; GFX1150-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_2) | instid1(TRANS32_DEP_1)
+; GFX1150-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1150-TRUE16-NEXT: v_add_nc_u32_e32 v3, v3, v2
; GFX1150-TRUE16-NEXT: v_div_scale_f32 v2, vcc_lo, 1.0, v1, 1.0
; GFX1150-TRUE16-NEXT: s_denorm_mode 15
+; GFX1150-TRUE16-NEXT: s_delay_alu instid0(TRANS32_DEP_1) | instid1(SALU_CYCLE_1)
; GFX1150-TRUE16-NEXT: v_fma_f32 v6, -v4, v5, 1.0
; GFX1150-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1150-TRUE16-NEXT: v_fmac_f32_e32 v5, v6, v5
@@ -6491,9 +6551,10 @@ define amdgpu_kernel void @frem_v2f16(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX1150-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1150-TRUE16-NEXT: v_fma_f32 v7, -v4, v6, v2
; GFX1150-TRUE16-NEXT: v_fmac_f32_e32 v6, v7, v5
-; GFX1150-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
+; GFX1150-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1150-TRUE16-NEXT: v_fma_f32 v2, -v4, v6, v2
; GFX1150-TRUE16-NEXT: s_denorm_mode 12
+; GFX1150-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX1150-TRUE16-NEXT: v_div_fmas_f32 v2, v2, v5, v6
; GFX1150-TRUE16-NEXT: v_cmp_gt_i32_e32 vcc_lo, 12, v3
; GFX1150-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2)
@@ -6529,12 +6590,13 @@ define amdgpu_kernel void @frem_v2f16(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX1150-TRUE16-NEXT: s_and_b32 s6, s5, 0x7fff
; GFX1150-TRUE16-NEXT: s_cvt_f32_f16 s7, s4
; GFX1150-TRUE16-NEXT: s_cvt_f32_f16 s6, s6
-; GFX1150-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_3)
+; GFX1150-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_3) | instskip(NEXT) | instid1(SALU_CYCLE_3)
; GFX1150-TRUE16-NEXT: s_cmp_ngt_f32 s7, s6
; GFX1150-TRUE16-NEXT: s_cbranch_scc0 .LBB9_10
; GFX1150-TRUE16-NEXT: ; %bb.9: ; %frem.else
; GFX1150-TRUE16-NEXT: s_cmp_eq_f32 s7, s6
; GFX1150-TRUE16-NEXT: v_and_b16 v0.h, 0x8000, s8
+; GFX1150-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_2)
; GFX1150-TRUE16-NEXT: s_cselect_b32 s9, -1, 0
; GFX1150-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX1150-TRUE16-NEXT: v_cndmask_b16 v1.l, s8, v0.h, s9
@@ -6548,6 +6610,7 @@ define amdgpu_kernel void @frem_v2f16(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX1150-TRUE16-NEXT: s_and_b32 s8, s8, exec_lo
; GFX1150-TRUE16-NEXT: s_cselect_b32 s8, 1, 0
; GFX1150-TRUE16-NEXT: s_cmp_lg_u32 s8, 1
+; GFX1150-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1150-TRUE16-NEXT: s_cbranch_scc1 .LBB9_16
; GFX1150-TRUE16-NEXT: ; %bb.12: ; %frem.compute
; GFX1150-TRUE16-NEXT: v_frexp_mant_f32_e32 v2, s6
@@ -6564,10 +6627,11 @@ define amdgpu_kernel void @frem_v2f16(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX1150-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1150-TRUE16-NEXT: v_not_b32_e32 v4, v4
; GFX1150-TRUE16-NEXT: v_rcp_f32_e32 v6, v5
-; GFX1150-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_2) | instid1(TRANS32_DEP_1)
+; GFX1150-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1150-TRUE16-NEXT: v_add_nc_u32_e32 v4, v4, v3
; GFX1150-TRUE16-NEXT: v_div_scale_f32 v3, vcc_lo, 1.0, v2, 1.0
; GFX1150-TRUE16-NEXT: s_denorm_mode 15
+; GFX1150-TRUE16-NEXT: s_delay_alu instid0(TRANS32_DEP_1) | instid1(SALU_CYCLE_1)
; GFX1150-TRUE16-NEXT: v_fma_f32 v7, -v5, v6, 1.0
; GFX1150-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1150-TRUE16-NEXT: v_fmac_f32_e32 v6, v7, v6
@@ -6575,9 +6639,10 @@ define amdgpu_kernel void @frem_v2f16(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX1150-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1150-TRUE16-NEXT: v_fma_f32 v8, -v5, v7, v3
; GFX1150-TRUE16-NEXT: v_fmac_f32_e32 v7, v8, v6
-; GFX1150-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
+; GFX1150-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1150-TRUE16-NEXT: v_fma_f32 v3, -v5, v7, v3
; GFX1150-TRUE16-NEXT: s_denorm_mode 12
+; GFX1150-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX1150-TRUE16-NEXT: v_div_fmas_f32 v3, v3, v6, v7
; GFX1150-TRUE16-NEXT: v_cmp_gt_i32_e32 vcc_lo, 12, v4
; GFX1150-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2)
@@ -6609,18 +6674,20 @@ define amdgpu_kernel void @frem_v2f16(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX1150-TRUE16-NEXT: .LBB9_16: ; %Flow54
; GFX1150-TRUE16-NEXT: s_cmp_lg_f16 s3, 0
; GFX1150-TRUE16-NEXT: v_mov_b32_e32 v2, 0
+; GFX1150-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_2) | instskip(SKIP_1) | instid1(SALU_CYCLE_3)
; GFX1150-TRUE16-NEXT: s_cselect_b32 s3, -1, 0
; GFX1150-TRUE16-NEXT: s_cmp_nge_f16 s2, 0x7c00
; GFX1150-TRUE16-NEXT: s_cselect_b32 s2, -1, 0
-; GFX1150-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1150-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_2)
; GFX1150-TRUE16-NEXT: s_and_b32 s2, s2, s3
; GFX1150-TRUE16-NEXT: s_cmp_lg_f16 s5, 0
; GFX1150-TRUE16-NEXT: v_cndmask_b16 v0.l, 0x7e00, v0.l, s2
; GFX1150-TRUE16-NEXT: s_cselect_b32 s2, -1, 0
; GFX1150-TRUE16-NEXT: s_cmp_nge_f16 s4, 0x7c00
+; GFX1150-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_3) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1150-TRUE16-NEXT: s_cselect_b32 s3, -1, 0
-; GFX1150-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1150-TRUE16-NEXT: s_and_b32 s2, s3, s2
+; GFX1150-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1150-TRUE16-NEXT: v_cndmask_b16 v0.h, 0x7e00, v1.l, s2
; GFX1150-TRUE16-NEXT: global_store_b32 v2, v0, s[0:1]
; GFX1150-TRUE16-NEXT: s_endpgm
@@ -6643,7 +6710,7 @@ define amdgpu_kernel void @frem_v2f16(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX1150-FAKE16-NEXT: s_and_b32 s5, s3, 0x7fff
; GFX1150-FAKE16-NEXT: s_cvt_f32_f16 s6, s2
; GFX1150-FAKE16-NEXT: s_cvt_f32_f16 s5, s5
-; GFX1150-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_3)
+; GFX1150-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_3) | instskip(NEXT) | instid1(SALU_CYCLE_3)
; GFX1150-FAKE16-NEXT: s_cmp_ngt_f32 s6, s5
; GFX1150-FAKE16-NEXT: s_cbranch_scc0 .LBB9_2
; GFX1150-FAKE16-NEXT: ; %bb.1: ; %frem.else20
@@ -6651,8 +6718,9 @@ define amdgpu_kernel void @frem_v2f16(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX1150-FAKE16-NEXT: s_cmp_eq_f32 s6, s5
; GFX1150-FAKE16-NEXT: v_mov_b32_e32 v0, s7
; GFX1150-FAKE16-NEXT: s_mov_b32 s7, 0
+; GFX1150-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1150-FAKE16-NEXT: s_cselect_b32 vcc_lo, -1, 0
-; GFX1150-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1150-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX1150-FAKE16-NEXT: v_cndmask_b32_e32 v0, s4, v0, vcc_lo
; GFX1150-FAKE16-NEXT: s_branch .LBB9_3
; GFX1150-FAKE16-NEXT: .LBB9_2:
@@ -6663,6 +6731,7 @@ define amdgpu_kernel void @frem_v2f16(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX1150-FAKE16-NEXT: s_and_b32 s7, s7, exec_lo
; GFX1150-FAKE16-NEXT: s_cselect_b32 s7, 1, 0
; GFX1150-FAKE16-NEXT: s_cmp_lg_u32 s7, 1
+; GFX1150-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1150-FAKE16-NEXT: s_cbranch_scc1 .LBB9_9
; GFX1150-FAKE16-NEXT: ; %bb.4: ; %frem.compute19
; GFX1150-FAKE16-NEXT: v_frexp_mant_f32_e32 v1, s5
@@ -6684,19 +6753,21 @@ define amdgpu_kernel void @frem_v2f16(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX1150-FAKE16-NEXT: v_add_nc_u32_e32 v4, v4, v3
; GFX1150-FAKE16-NEXT: v_div_scale_f32 v3, vcc_lo, 1.0, v1, 1.0
; GFX1150-FAKE16-NEXT: s_denorm_mode 15
-; GFX1150-FAKE16-NEXT: s_delay_alu instid0(TRANS32_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1150-FAKE16-NEXT: s_delay_alu instid0(TRANS32_DEP_1) | instid1(SALU_CYCLE_1)
; GFX1150-FAKE16-NEXT: v_fma_f32 v7, -v5, v6, 1.0
-; GFX1150-FAKE16-NEXT: v_fmac_f32_e32 v6, v7, v6
; GFX1150-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1150-FAKE16-NEXT: v_fmac_f32_e32 v6, v7, v6
; GFX1150-FAKE16-NEXT: v_mul_f32_e32 v7, v3, v6
-; GFX1150-FAKE16-NEXT: v_fma_f32 v8, -v5, v7, v3
; GFX1150-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1150-FAKE16-NEXT: v_fma_f32 v8, -v5, v7, v3
; GFX1150-FAKE16-NEXT: v_fmac_f32_e32 v7, v8, v6
+; GFX1150-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1150-FAKE16-NEXT: v_fma_f32 v3, -v5, v7, v3
; GFX1150-FAKE16-NEXT: s_denorm_mode 12
-; GFX1150-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
+; GFX1150-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX1150-FAKE16-NEXT: v_div_fmas_f32 v3, v3, v6, v7
; GFX1150-FAKE16-NEXT: v_cmp_gt_i32_e32 vcc_lo, 12, v4
+; GFX1150-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX1150-FAKE16-NEXT: v_div_fixup_f32 v3, v3, v1, 1.0
; GFX1150-FAKE16-NEXT: s_cbranch_vccnz .LBB9_8
; GFX1150-FAKE16-NEXT: ; %bb.5: ; %frem.loop_body27.preheader
@@ -6751,7 +6822,7 @@ define amdgpu_kernel void @frem_v2f16(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX1150-FAKE16-NEXT: s_and_b32 s7, s5, 0x7fff
; GFX1150-FAKE16-NEXT: s_cvt_f32_f16 s8, s4
; GFX1150-FAKE16-NEXT: s_cvt_f32_f16 s7, s7
-; GFX1150-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_3)
+; GFX1150-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_3) | instskip(NEXT) | instid1(SALU_CYCLE_3)
; GFX1150-FAKE16-NEXT: s_cmp_ngt_f32 s8, s7
; GFX1150-FAKE16-NEXT: s_cbranch_scc0 .LBB9_11
; GFX1150-FAKE16-NEXT: ; %bb.10: ; %frem.else
@@ -6759,8 +6830,9 @@ define amdgpu_kernel void @frem_v2f16(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX1150-FAKE16-NEXT: s_cmp_eq_f32 s8, s7
; GFX1150-FAKE16-NEXT: v_mov_b32_e32 v1, s9
; GFX1150-FAKE16-NEXT: s_mov_b32 s9, 0
+; GFX1150-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1150-FAKE16-NEXT: s_cselect_b32 vcc_lo, -1, 0
-; GFX1150-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1150-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX1150-FAKE16-NEXT: v_cndmask_b32_e32 v1, s6, v1, vcc_lo
; GFX1150-FAKE16-NEXT: s_branch .LBB9_12
; GFX1150-FAKE16-NEXT: .LBB9_11:
@@ -6771,6 +6843,7 @@ define amdgpu_kernel void @frem_v2f16(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX1150-FAKE16-NEXT: s_and_b32 s9, s9, exec_lo
; GFX1150-FAKE16-NEXT: s_cselect_b32 s9, 1, 0
; GFX1150-FAKE16-NEXT: s_cmp_lg_u32 s9, 1
+; GFX1150-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1150-FAKE16-NEXT: s_cbranch_scc1 .LBB9_18
; GFX1150-FAKE16-NEXT: ; %bb.13: ; %frem.compute
; GFX1150-FAKE16-NEXT: v_frexp_mant_f32_e32 v2, s7
@@ -6792,19 +6865,21 @@ define amdgpu_kernel void @frem_v2f16(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX1150-FAKE16-NEXT: v_add_nc_u32_e32 v5, v5, v4
; GFX1150-FAKE16-NEXT: v_div_scale_f32 v4, vcc_lo, 1.0, v2, 1.0
; GFX1150-FAKE16-NEXT: s_denorm_mode 15
-; GFX1150-FAKE16-NEXT: s_delay_alu instid0(TRANS32_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1150-FAKE16-NEXT: s_delay_alu instid0(TRANS32_DEP_1) | instid1(SALU_CYCLE_1)
; GFX1150-FAKE16-NEXT: v_fma_f32 v8, -v6, v7, 1.0
-; GFX1150-FAKE16-NEXT: v_fmac_f32_e32 v7, v8, v7
; GFX1150-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1150-FAKE16-NEXT: v_fmac_f32_e32 v7, v8, v7
; GFX1150-FAKE16-NEXT: v_mul_f32_e32 v8, v4, v7
-; GFX1150-FAKE16-NEXT: v_fma_f32 v9, -v6, v8, v4
; GFX1150-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1150-FAKE16-NEXT: v_fma_f32 v9, -v6, v8, v4
; GFX1150-FAKE16-NEXT: v_fmac_f32_e32 v8, v9, v7
+; GFX1150-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1150-FAKE16-NEXT: v_fma_f32 v4, -v6, v8, v4
; GFX1150-FAKE16-NEXT: s_denorm_mode 12
-; GFX1150-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
+; GFX1150-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX1150-FAKE16-NEXT: v_div_fmas_f32 v4, v4, v7, v8
; GFX1150-FAKE16-NEXT: v_cmp_gt_i32_e32 vcc_lo, 12, v5
+; GFX1150-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX1150-FAKE16-NEXT: v_div_fixup_f32 v4, v4, v2, 1.0
; GFX1150-FAKE16-NEXT: s_cbranch_vccnz .LBB9_17
; GFX1150-FAKE16-NEXT: ; %bb.14: ; %frem.loop_body.preheader
@@ -6855,18 +6930,20 @@ define amdgpu_kernel void @frem_v2f16(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX1150-FAKE16-NEXT: .LBB9_18: ; %Flow54
; GFX1150-FAKE16-NEXT: s_cmp_lg_f16 s3, 0
; GFX1150-FAKE16-NEXT: v_mov_b32_e32 v2, 0
+; GFX1150-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_2) | instskip(SKIP_1) | instid1(SALU_CYCLE_3)
; GFX1150-FAKE16-NEXT: s_cselect_b32 s3, -1, 0
; GFX1150-FAKE16-NEXT: s_cmp_nge_f16 s2, 0x7c00
; GFX1150-FAKE16-NEXT: s_cselect_b32 s2, -1, 0
-; GFX1150-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1150-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_2)
; GFX1150-FAKE16-NEXT: s_and_b32 vcc_lo, s2, s3
; GFX1150-FAKE16-NEXT: s_cmp_lg_f16 s5, 0
; GFX1150-FAKE16-NEXT: v_cndmask_b32_e32 v0, 0x7e00, v0, vcc_lo
; GFX1150-FAKE16-NEXT: s_cselect_b32 s2, -1, 0
; GFX1150-FAKE16-NEXT: s_cmp_nge_f16 s4, 0x7c00
+; GFX1150-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_3) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1150-FAKE16-NEXT: s_cselect_b32 s3, -1, 0
-; GFX1150-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(VALU_DEP_1)
; GFX1150-FAKE16-NEXT: s_and_b32 vcc_lo, s3, s2
+; GFX1150-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX1150-FAKE16-NEXT: v_cndmask_b32_e32 v1, 0x7e00, v1, vcc_lo
; GFX1150-FAKE16-NEXT: v_and_b32_e32 v0, 0xffff, v0
; GFX1150-FAKE16-NEXT: v_lshl_or_b32 v0, v1, 16, v0
@@ -6891,12 +6968,13 @@ define amdgpu_kernel void @frem_v2f16(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX1200-TRUE16-NEXT: s_and_b32 s5, s3, 0x7fff
; GFX1200-TRUE16-NEXT: s_cvt_f32_f16 s6, s2
; GFX1200-TRUE16-NEXT: s_cvt_f32_f16 s5, s5
-; GFX1200-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_3)
+; GFX1200-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_3) | instskip(NEXT) | instid1(SALU_CYCLE_3)
; GFX1200-TRUE16-NEXT: s_cmp_ngt_f32 s6, s5
; GFX1200-TRUE16-NEXT: s_cbranch_scc0 .LBB9_2
; GFX1200-TRUE16-NEXT: ; %bb.1: ; %frem.else20
; GFX1200-TRUE16-NEXT: s_cmp_eq_f32 s6, s5
; GFX1200-TRUE16-NEXT: v_and_b16 v0.l, 0x8000, s4
+; GFX1200-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_2)
; GFX1200-TRUE16-NEXT: s_cselect_b32 s7, -1, 0
; GFX1200-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX1200-TRUE16-NEXT: v_cndmask_b16 v0.l, s4, v0.l, s7
@@ -6911,6 +6989,7 @@ define amdgpu_kernel void @frem_v2f16(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX1200-TRUE16-NEXT: s_cselect_b32 s7, 1, 0
; GFX1200-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX1200-TRUE16-NEXT: s_cmp_lg_u32 s7, 1
+; GFX1200-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1200-TRUE16-NEXT: s_cbranch_scc1 .LBB9_8
; GFX1200-TRUE16-NEXT: ; %bb.4: ; %frem.compute19
; GFX1200-TRUE16-NEXT: v_frexp_mant_f32_e32 v1, s5
@@ -6927,10 +7006,11 @@ define amdgpu_kernel void @frem_v2f16(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX1200-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1200-TRUE16-NEXT: v_not_b32_e32 v3, v3
; GFX1200-TRUE16-NEXT: v_rcp_f32_e32 v5, v4
-; GFX1200-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_2) | instid1(TRANS32_DEP_1)
+; GFX1200-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1200-TRUE16-NEXT: v_add_nc_u32_e32 v3, v3, v2
; GFX1200-TRUE16-NEXT: v_div_scale_f32 v2, vcc_lo, 1.0, v1, 1.0
; GFX1200-TRUE16-NEXT: s_denorm_mode 15
+; GFX1200-TRUE16-NEXT: s_delay_alu instid0(TRANS32_DEP_1) | instid1(SALU_CYCLE_1)
; GFX1200-TRUE16-NEXT: v_fma_f32 v6, -v4, v5, 1.0
; GFX1200-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1200-TRUE16-NEXT: v_fmac_f32_e32 v5, v6, v5
@@ -6938,9 +7018,10 @@ define amdgpu_kernel void @frem_v2f16(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX1200-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1200-TRUE16-NEXT: v_fma_f32 v7, -v4, v6, v2
; GFX1200-TRUE16-NEXT: v_fmac_f32_e32 v6, v7, v5
-; GFX1200-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
+; GFX1200-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1200-TRUE16-NEXT: v_fma_f32 v2, -v4, v6, v2
; GFX1200-TRUE16-NEXT: s_denorm_mode 12
+; GFX1200-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX1200-TRUE16-NEXT: v_div_fmas_f32 v2, v2, v5, v6
; GFX1200-TRUE16-NEXT: v_cmp_gt_i32_e32 vcc_lo, 12, v3
; GFX1200-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2)
@@ -6982,15 +7063,15 @@ define amdgpu_kernel void @frem_v2f16(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX1200-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX1200-TRUE16-NEXT: s_cvt_f32_f16 s6, s6
; GFX1200-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
-; GFX1200-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_2)
+; GFX1200-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_2) | instskip(NEXT) | instid1(SALU_CYCLE_3)
; GFX1200-TRUE16-NEXT: s_cmp_ngt_f32 s7, s6
; GFX1200-TRUE16-NEXT: s_cbranch_scc0 .LBB9_10
; GFX1200-TRUE16-NEXT: ; %bb.9: ; %frem.else
; GFX1200-TRUE16-NEXT: s_cmp_eq_f32 s7, s6
; GFX1200-TRUE16-NEXT: v_and_b16 v0.h, 0x8000, s8
+; GFX1200-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_2) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX1200-TRUE16-NEXT: s_cselect_b32 s9, -1, 0
; GFX1200-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
-; GFX1200-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1200-TRUE16-NEXT: v_cndmask_b16 v1.l, s8, v0.h, s9
; GFX1200-TRUE16-NEXT: s_mov_b32 s8, 0
; GFX1200-TRUE16-NEXT: s_branch .LBB9_11
@@ -7003,6 +7084,7 @@ define amdgpu_kernel void @frem_v2f16(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX1200-TRUE16-NEXT: s_cselect_b32 s8, 1, 0
; GFX1200-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX1200-TRUE16-NEXT: s_cmp_lg_u32 s8, 1
+; GFX1200-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1200-TRUE16-NEXT: s_cbranch_scc1 .LBB9_16
; GFX1200-TRUE16-NEXT: ; %bb.12: ; %frem.compute
; GFX1200-TRUE16-NEXT: v_frexp_mant_f32_e32 v2, s6
@@ -7019,10 +7101,11 @@ define amdgpu_kernel void @frem_v2f16(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX1200-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1200-TRUE16-NEXT: v_not_b32_e32 v4, v4
; GFX1200-TRUE16-NEXT: v_rcp_f32_e32 v6, v5
-; GFX1200-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_2) | instid1(TRANS32_DEP_1)
+; GFX1200-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1200-TRUE16-NEXT: v_add_nc_u32_e32 v4, v4, v3
; GFX1200-TRUE16-NEXT: v_div_scale_f32 v3, vcc_lo, 1.0, v2, 1.0
; GFX1200-TRUE16-NEXT: s_denorm_mode 15
+; GFX1200-TRUE16-NEXT: s_delay_alu instid0(TRANS32_DEP_1) | instid1(SALU_CYCLE_1)
; GFX1200-TRUE16-NEXT: v_fma_f32 v7, -v5, v6, 1.0
; GFX1200-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1200-TRUE16-NEXT: v_fmac_f32_e32 v6, v7, v6
@@ -7068,15 +7151,17 @@ define amdgpu_kernel void @frem_v2f16(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX1200-TRUE16-NEXT: .LBB9_16: ; %Flow54
; GFX1200-TRUE16-NEXT: s_cmp_lg_f16 s3, 0
; GFX1200-TRUE16-NEXT: v_mov_b32_e32 v2, 0
+; GFX1200-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_2) | instskip(SKIP_1) | instid1(SALU_CYCLE_3)
; GFX1200-TRUE16-NEXT: s_cselect_b32 s3, -1, 0
; GFX1200-TRUE16-NEXT: s_cmp_nge_f16 s2, 0x7c00
; GFX1200-TRUE16-NEXT: s_cselect_b32 s2, -1, 0
-; GFX1200-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1200-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_2)
; GFX1200-TRUE16-NEXT: s_and_b32 s2, s2, s3
; GFX1200-TRUE16-NEXT: s_cmp_lg_f16 s5, 0
; GFX1200-TRUE16-NEXT: v_cndmask_b16 v0.l, 0x7e00, v0.l, s2
; GFX1200-TRUE16-NEXT: s_cselect_b32 s2, -1, 0
; GFX1200-TRUE16-NEXT: s_cmp_nge_f16 s4, 0x7c00
+; GFX1200-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_3)
; GFX1200-TRUE16-NEXT: s_cselect_b32 s3, -1, 0
; GFX1200-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX1200-TRUE16-NEXT: s_and_b32 s2, s3, s2
@@ -7103,7 +7188,7 @@ define amdgpu_kernel void @frem_v2f16(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX1200-FAKE16-NEXT: s_and_b32 s5, s3, 0x7fff
; GFX1200-FAKE16-NEXT: s_cvt_f32_f16 s6, s2
; GFX1200-FAKE16-NEXT: s_cvt_f32_f16 s5, s5
-; GFX1200-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_3)
+; GFX1200-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_3) | instskip(NEXT) | instid1(SALU_CYCLE_3)
; GFX1200-FAKE16-NEXT: s_cmp_ngt_f32 s6, s5
; GFX1200-FAKE16-NEXT: s_cbranch_scc0 .LBB9_2
; GFX1200-FAKE16-NEXT: ; %bb.1: ; %frem.else20
@@ -7111,8 +7196,9 @@ define amdgpu_kernel void @frem_v2f16(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX1200-FAKE16-NEXT: s_cmp_eq_f32 s6, s5
; GFX1200-FAKE16-NEXT: v_mov_b32_e32 v0, s7
; GFX1200-FAKE16-NEXT: s_mov_b32 s7, 0
+; GFX1200-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1200-FAKE16-NEXT: s_cselect_b32 vcc_lo, -1, 0
-; GFX1200-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1200-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX1200-FAKE16-NEXT: v_cndmask_b32_e32 v0, s4, v0, vcc_lo
; GFX1200-FAKE16-NEXT: s_branch .LBB9_3
; GFX1200-FAKE16-NEXT: .LBB9_2:
@@ -7124,6 +7210,7 @@ define amdgpu_kernel void @frem_v2f16(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX1200-FAKE16-NEXT: s_cselect_b32 s7, 1, 0
; GFX1200-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX1200-FAKE16-NEXT: s_cmp_lg_u32 s7, 1
+; GFX1200-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1200-FAKE16-NEXT: s_cbranch_scc1 .LBB9_9
; GFX1200-FAKE16-NEXT: ; %bb.4: ; %frem.compute19
; GFX1200-FAKE16-NEXT: v_frexp_mant_f32_e32 v1, s5
@@ -7145,20 +7232,21 @@ define amdgpu_kernel void @frem_v2f16(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX1200-FAKE16-NEXT: v_add_nc_u32_e32 v4, v4, v3
; GFX1200-FAKE16-NEXT: v_div_scale_f32 v3, vcc_lo, 1.0, v1, 1.0
; GFX1200-FAKE16-NEXT: s_denorm_mode 15
-; GFX1200-FAKE16-NEXT: s_delay_alu instid0(TRANS32_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1200-FAKE16-NEXT: s_delay_alu instid0(TRANS32_DEP_1) | instid1(SALU_CYCLE_1)
; GFX1200-FAKE16-NEXT: v_fma_f32 v7, -v5, v6, 1.0
-; GFX1200-FAKE16-NEXT: v_fmac_f32_e32 v6, v7, v6
; GFX1200-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1200-FAKE16-NEXT: v_fmac_f32_e32 v6, v7, v6
; GFX1200-FAKE16-NEXT: v_mul_f32_e32 v7, v3, v6
-; GFX1200-FAKE16-NEXT: v_fma_f32 v8, -v5, v7, v3
; GFX1200-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1200-FAKE16-NEXT: v_fma_f32 v8, -v5, v7, v3
; GFX1200-FAKE16-NEXT: v_fmac_f32_e32 v7, v8, v6
+; GFX1200-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_2) | instid1(VALU_DEP_1)
; GFX1200-FAKE16-NEXT: v_fma_f32 v3, -v5, v7, v3
; GFX1200-FAKE16-NEXT: s_denorm_mode 12
; GFX1200-FAKE16-NEXT: s_wait_alu depctr_va_vcc(0)
-; GFX1200-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX1200-FAKE16-NEXT: v_div_fmas_f32 v3, v3, v6, v7
; GFX1200-FAKE16-NEXT: v_cmp_gt_i32_e32 vcc_lo, 12, v4
+; GFX1200-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX1200-FAKE16-NEXT: v_div_fixup_f32 v3, v3, v1, 1.0
; GFX1200-FAKE16-NEXT: s_cbranch_vccnz .LBB9_8
; GFX1200-FAKE16-NEXT: ; %bb.5: ; %frem.loop_body27.preheader
@@ -7219,7 +7307,7 @@ define amdgpu_kernel void @frem_v2f16(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX1200-FAKE16-NEXT: s_cvt_f32_f16 s8, s4
; GFX1200-FAKE16-NEXT: s_cvt_f32_f16 s7, s7
; GFX1200-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
-; GFX1200-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_2)
+; GFX1200-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_2) | instskip(NEXT) | instid1(SALU_CYCLE_3)
; GFX1200-FAKE16-NEXT: s_cmp_ngt_f32 s8, s7
; GFX1200-FAKE16-NEXT: s_cbranch_scc0 .LBB9_11
; GFX1200-FAKE16-NEXT: ; %bb.10: ; %frem.else
@@ -7227,16 +7315,16 @@ define amdgpu_kernel void @frem_v2f16(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX1200-FAKE16-NEXT: s_cmp_eq_f32 s8, s7
; GFX1200-FAKE16-NEXT: v_mov_b32_e32 v1, s9
; GFX1200-FAKE16-NEXT: s_mov_b32 s9, 0
+; GFX1200-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX1200-FAKE16-NEXT: s_cselect_b32 vcc_lo, -1, 0
; GFX1200-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
-; GFX1200-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1200-FAKE16-NEXT: v_cndmask_b32_e32 v1, s6, v1, vcc_lo
; GFX1200-FAKE16-NEXT: s_branch .LBB9_12
; GFX1200-FAKE16-NEXT: .LBB9_11:
; GFX1200-FAKE16-NEXT: s_mov_b32 s9, -1
; GFX1200-FAKE16-NEXT: ; implicit-def: $vgpr1
; GFX1200-FAKE16-NEXT: .LBB9_12: ; %Flow53
-; GFX1200-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1200-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_3) | instid1(SALU_CYCLE_1)
; GFX1200-FAKE16-NEXT: s_and_b32 s9, s9, exec_lo
; GFX1200-FAKE16-NEXT: s_cselect_b32 s9, 1, 0
; GFX1200-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
@@ -7262,20 +7350,21 @@ define amdgpu_kernel void @frem_v2f16(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX1200-FAKE16-NEXT: v_add_nc_u32_e32 v5, v5, v4
; GFX1200-FAKE16-NEXT: v_div_scale_f32 v4, vcc_lo, 1.0, v2, 1.0
; GFX1200-FAKE16-NEXT: s_denorm_mode 15
-; GFX1200-FAKE16-NEXT: s_delay_alu instid0(TRANS32_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1200-FAKE16-NEXT: s_delay_alu instid0(TRANS32_DEP_1) | instid1(SALU_CYCLE_1)
; GFX1200-FAKE16-NEXT: v_fma_f32 v8, -v6, v7, 1.0
-; GFX1200-FAKE16-NEXT: v_fmac_f32_e32 v7, v8, v7
; GFX1200-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1200-FAKE16-NEXT: v_fmac_f32_e32 v7, v8, v7
; GFX1200-FAKE16-NEXT: v_mul_f32_e32 v8, v4, v7
-; GFX1200-FAKE16-NEXT: v_fma_f32 v9, -v6, v8, v4
; GFX1200-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1200-FAKE16-NEXT: v_fma_f32 v9, -v6, v8, v4
; GFX1200-FAKE16-NEXT: v_fmac_f32_e32 v8, v9, v7
+; GFX1200-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_2) | instid1(VALU_DEP_1)
; GFX1200-FAKE16-NEXT: v_fma_f32 v4, -v6, v8, v4
; GFX1200-FAKE16-NEXT: s_denorm_mode 12
; GFX1200-FAKE16-NEXT: s_wait_alu depctr_va_vcc(0)
-; GFX1200-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX1200-FAKE16-NEXT: v_div_fmas_f32 v4, v4, v7, v8
; GFX1200-FAKE16-NEXT: v_cmp_gt_i32_e32 vcc_lo, 12, v5
+; GFX1200-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX1200-FAKE16-NEXT: v_div_fixup_f32 v4, v4, v2, 1.0
; GFX1200-FAKE16-NEXT: s_cbranch_vccnz .LBB9_17
; GFX1200-FAKE16-NEXT: ; %bb.14: ; %frem.loop_body.preheader
@@ -7329,22 +7418,24 @@ define amdgpu_kernel void @frem_v2f16(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX1200-FAKE16-NEXT: .LBB9_18: ; %Flow54
; GFX1200-FAKE16-NEXT: s_cmp_lg_f16 s3, 0
; GFX1200-FAKE16-NEXT: v_mov_b32_e32 v2, 0
+; GFX1200-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_2) | instskip(SKIP_1) | instid1(SALU_CYCLE_3)
; GFX1200-FAKE16-NEXT: s_cselect_b32 s3, -1, 0
; GFX1200-FAKE16-NEXT: s_cmp_nge_f16 s2, 0x7c00
; GFX1200-FAKE16-NEXT: s_cselect_b32 s2, -1, 0
-; GFX1200-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1200-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_3) | instid1(SALU_CYCLE_1)
; GFX1200-FAKE16-NEXT: s_and_b32 vcc_lo, s2, s3
; GFX1200-FAKE16-NEXT: s_cmp_lg_f16 s5, 0
; GFX1200-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX1200-FAKE16-NEXT: v_cndmask_b32_e32 v0, 0x7e00, v0, vcc_lo
; GFX1200-FAKE16-NEXT: s_cselect_b32 s2, -1, 0
; GFX1200-FAKE16-NEXT: s_cmp_nge_f16 s4, 0x7c00
+; GFX1200-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_3) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1200-FAKE16-NEXT: s_cselect_b32 s3, -1, 0
-; GFX1200-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_3) | instid1(VALU_DEP_1)
; GFX1200-FAKE16-NEXT: s_and_b32 vcc_lo, s3, s2
; GFX1200-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX1200-FAKE16-NEXT: v_cndmask_b32_e32 v1, 0x7e00, v1, vcc_lo
; GFX1200-FAKE16-NEXT: v_and_b32_e32 v0, 0xffff, v0
+; GFX1200-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1200-FAKE16-NEXT: v_lshl_or_b32 v0, v1, 16, v0
; GFX1200-FAKE16-NEXT: global_store_b32 v2, v0, s[0:1]
; GFX1200-FAKE16-NEXT: s_endpgm
@@ -9217,6 +9308,7 @@ define amdgpu_kernel void @frem_v4f16(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX11-TRUE16-NEXT: s_and_b32 s2, s2, exec_lo
; GFX11-TRUE16-NEXT: s_cselect_b32 s2, 1, 0
; GFX11-TRUE16-NEXT: s_cmp_lg_u32 s2, 1
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_cbranch_scc1 .LBB10_8
; GFX11-TRUE16-NEXT: ; %bb.4: ; %frem.compute85
; GFX11-TRUE16-NEXT: v_frexp_exp_i32_f32_e32 v7, v6
@@ -9248,9 +9340,10 @@ define amdgpu_kernel void @frem_v4f16(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX11-TRUE16-NEXT: v_fmac_f32_e32 v10, v11, v9
; GFX11-TRUE16-NEXT: v_fma_f32 v6, -v8, v10, v6
; GFX11-TRUE16-NEXT: s_denorm_mode 12
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_div_fmas_f32 v6, v6, v9, v10
; GFX11-TRUE16-NEXT: v_cmp_gt_i32_e32 vcc_lo, 12, v7
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX11-TRUE16-NEXT: v_div_fixup_f32 v6, v6, v5, 1.0
; GFX11-TRUE16-NEXT: s_cbranch_vccnz .LBB10_7
; GFX11-TRUE16-NEXT: ; %bb.5: ; %frem.loop_body93.preheader
@@ -9299,6 +9392,7 @@ define amdgpu_kernel void @frem_v4f16(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX11-TRUE16-NEXT: s_and_b32 s2, s2, exec_lo
; GFX11-TRUE16-NEXT: s_cselect_b32 s2, 1, 0
; GFX11-TRUE16-NEXT: s_cmp_lg_u32 s2, 1
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_cbranch_scc1 .LBB10_16
; GFX11-TRUE16-NEXT: ; %bb.12: ; %frem.compute52
; GFX11-TRUE16-NEXT: v_frexp_exp_i32_f32_e32 v10, v9
@@ -9330,9 +9424,10 @@ define amdgpu_kernel void @frem_v4f16(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX11-TRUE16-NEXT: v_fmac_f32_e32 v13, v14, v12
; GFX11-TRUE16-NEXT: v_fma_f32 v9, -v11, v13, v9
; GFX11-TRUE16-NEXT: s_denorm_mode 12
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_div_fmas_f32 v9, v9, v12, v13
; GFX11-TRUE16-NEXT: v_cmp_gt_i32_e32 vcc_lo, 12, v10
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX11-TRUE16-NEXT: v_div_fixup_f32 v9, v9, v8, 1.0
; GFX11-TRUE16-NEXT: s_cbranch_vccnz .LBB10_15
; GFX11-TRUE16-NEXT: ; %bb.13: ; %frem.loop_body60.preheader
@@ -9378,6 +9473,7 @@ define amdgpu_kernel void @frem_v4f16(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX11-TRUE16-NEXT: s_and_b32 s2, s2, exec_lo
; GFX11-TRUE16-NEXT: s_cselect_b32 s2, 1, 0
; GFX11-TRUE16-NEXT: s_cmp_lg_u32 s2, 1
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_cbranch_scc1 .LBB10_24
; GFX11-TRUE16-NEXT: ; %bb.20: ; %frem.compute19
; GFX11-TRUE16-NEXT: v_frexp_exp_i32_f32_e32 v11, v10
@@ -9409,9 +9505,10 @@ define amdgpu_kernel void @frem_v4f16(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX11-TRUE16-NEXT: v_fmac_f32_e32 v14, v15, v13
; GFX11-TRUE16-NEXT: v_fma_f32 v10, -v12, v14, v10
; GFX11-TRUE16-NEXT: s_denorm_mode 12
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_div_fmas_f32 v10, v10, v13, v14
; GFX11-TRUE16-NEXT: v_cmp_gt_i32_e32 vcc_lo, 12, v11
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX11-TRUE16-NEXT: v_div_fixup_f32 v10, v10, v9, 1.0
; GFX11-TRUE16-NEXT: s_cbranch_vccnz .LBB10_23
; GFX11-TRUE16-NEXT: ; %bb.21: ; %frem.loop_body27.preheader
@@ -9460,6 +9557,7 @@ define amdgpu_kernel void @frem_v4f16(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX11-TRUE16-NEXT: s_and_b32 s2, s2, exec_lo
; GFX11-TRUE16-NEXT: s_cselect_b32 s2, 1, 0
; GFX11-TRUE16-NEXT: s_cmp_lg_u32 s2, 1
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_cbranch_scc1 .LBB10_32
; GFX11-TRUE16-NEXT: ; %bb.28: ; %frem.compute
; GFX11-TRUE16-NEXT: v_frexp_exp_i32_f32_e32 v14, v13
@@ -9491,9 +9589,10 @@ define amdgpu_kernel void @frem_v4f16(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX11-TRUE16-NEXT: v_fmac_f32_e32 v17, v18, v16
; GFX11-TRUE16-NEXT: v_fma_f32 v13, -v15, v17, v13
; GFX11-TRUE16-NEXT: s_denorm_mode 12
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_div_fmas_f32 v13, v13, v16, v17
; GFX11-TRUE16-NEXT: v_cmp_gt_i32_e32 vcc_lo, 12, v14
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX11-TRUE16-NEXT: v_div_fixup_f32 v13, v13, v12, 1.0
; GFX11-TRUE16-NEXT: s_cbranch_vccnz .LBB10_31
; GFX11-TRUE16-NEXT: ; %bb.29: ; %frem.loop_body.preheader
@@ -9572,6 +9671,7 @@ define amdgpu_kernel void @frem_v4f16(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX11-FAKE16-NEXT: s_and_b32 s2, s2, exec_lo
; GFX11-FAKE16-NEXT: s_cselect_b32 s2, 1, 0
; GFX11-FAKE16-NEXT: s_cmp_lg_u32 s2, 1
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_cbranch_scc1 .LBB10_9
; GFX11-FAKE16-NEXT: ; %bb.4: ; %frem.compute85
; GFX11-FAKE16-NEXT: v_frexp_mant_f32_e32 v4, v6
@@ -9601,9 +9701,10 @@ define amdgpu_kernel void @frem_v4f16(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_fma_f32 v12, -v9, v11, v7
; GFX11-FAKE16-NEXT: v_fmac_f32_e32 v11, v12, v10
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_fma_f32 v7, -v9, v11, v7
; GFX11-FAKE16-NEXT: s_denorm_mode 12
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_div_fmas_f32 v7, v7, v10, v11
; GFX11-FAKE16-NEXT: v_cmp_gt_i32_e32 vcc_lo, 12, v8
; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2)
@@ -9674,6 +9775,7 @@ define amdgpu_kernel void @frem_v4f16(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX11-FAKE16-NEXT: s_and_b32 s2, s2, exec_lo
; GFX11-FAKE16-NEXT: s_cselect_b32 s2, 1, 0
; GFX11-FAKE16-NEXT: s_cmp_lg_u32 s2, 1
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_cbranch_scc1 .LBB10_18
; GFX11-FAKE16-NEXT: ; %bb.13: ; %frem.compute52
; GFX11-FAKE16-NEXT: v_frexp_mant_f32_e32 v7, v9
@@ -9703,9 +9805,10 @@ define amdgpu_kernel void @frem_v4f16(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_fma_f32 v15, -v12, v14, v10
; GFX11-FAKE16-NEXT: v_fmac_f32_e32 v14, v15, v13
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_fma_f32 v10, -v12, v14, v10
; GFX11-FAKE16-NEXT: s_denorm_mode 12
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_div_fmas_f32 v10, v10, v13, v14
; GFX11-FAKE16-NEXT: v_cmp_gt_i32_e32 vcc_lo, 12, v11
; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2)
@@ -9773,6 +9876,7 @@ define amdgpu_kernel void @frem_v4f16(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX11-FAKE16-NEXT: s_and_b32 s2, s2, exec_lo
; GFX11-FAKE16-NEXT: s_cselect_b32 s2, 1, 0
; GFX11-FAKE16-NEXT: s_cmp_lg_u32 s2, 1
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_cbranch_scc1 .LBB10_27
; GFX11-FAKE16-NEXT: ; %bb.22: ; %frem.compute19
; GFX11-FAKE16-NEXT: v_frexp_mant_f32_e32 v8, v10
@@ -9802,9 +9906,10 @@ define amdgpu_kernel void @frem_v4f16(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_fma_f32 v16, -v13, v15, v11
; GFX11-FAKE16-NEXT: v_fmac_f32_e32 v15, v16, v14
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_fma_f32 v11, -v13, v15, v11
; GFX11-FAKE16-NEXT: s_denorm_mode 12
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_div_fmas_f32 v11, v11, v14, v15
; GFX11-FAKE16-NEXT: v_cmp_gt_i32_e32 vcc_lo, 12, v12
; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2)
@@ -9875,6 +9980,7 @@ define amdgpu_kernel void @frem_v4f16(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX11-FAKE16-NEXT: s_and_b32 s2, s2, exec_lo
; GFX11-FAKE16-NEXT: s_cselect_b32 s2, 1, 0
; GFX11-FAKE16-NEXT: s_cmp_lg_u32 s2, 1
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_cbranch_scc1 .LBB10_36
; GFX11-FAKE16-NEXT: ; %bb.31: ; %frem.compute
; GFX11-FAKE16-NEXT: v_frexp_mant_f32_e32 v11, v13
@@ -9904,9 +10010,10 @@ define amdgpu_kernel void @frem_v4f16(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_fma_f32 v19, -v16, v18, v14
; GFX11-FAKE16-NEXT: v_fmac_f32_e32 v18, v19, v17
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_fma_f32 v14, -v16, v18, v14
; GFX11-FAKE16-NEXT: s_denorm_mode 12
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_div_fmas_f32 v14, v14, v17, v18
; GFX11-FAKE16-NEXT: v_cmp_gt_i32_e32 vcc_lo, 12, v15
; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2)
@@ -9970,10 +10077,11 @@ define amdgpu_kernel void @frem_v4f16(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX11-FAKE16-NEXT: v_cndmask_b32_e32 v1, 0x7e00, v8, vcc_lo
; GFX11-FAKE16-NEXT: v_cmp_lg_f16_e32 vcc_lo, 0, v10
; GFX11-FAKE16-NEXT: v_lshl_or_b32 v0, v2, 16, v0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_2) | instid1(VALU_DEP_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_and_b32_e32 v1, 0xffff, v1
; GFX11-FAKE16-NEXT: s_and_b32 vcc_lo, s2, vcc_lo
; GFX11-FAKE16-NEXT: v_cndmask_b32_e32 v4, 0x7e00, v11, vcc_lo
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_lshl_or_b32 v1, v4, 16, v1
; GFX11-FAKE16-NEXT: global_store_b64 v3, v[0:1], s[0:1]
; GFX11-FAKE16-NEXT: s_endpgm
@@ -9998,12 +10106,13 @@ define amdgpu_kernel void @frem_v4f16(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX1150-TRUE16-NEXT: v_readfirstlane_b32 s2, v1
; GFX1150-TRUE16-NEXT: s_and_b32 s6, s4, 0x7fff
; GFX1150-TRUE16-NEXT: s_cvt_f32_f16 s6, s6
-; GFX1150-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_3)
+; GFX1150-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_3) | instskip(NEXT) | instid1(SALU_CYCLE_3)
; GFX1150-TRUE16-NEXT: s_cmp_ngt_f32 s8, s6
; GFX1150-TRUE16-NEXT: s_cbranch_scc0 .LBB10_2
; GFX1150-TRUE16-NEXT: ; %bb.1: ; %frem.else86
; GFX1150-TRUE16-NEXT: s_cmp_eq_f32 s8, s6
; GFX1150-TRUE16-NEXT: v_and_b16 v0.l, 0x8000, s5
+; GFX1150-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_2)
; GFX1150-TRUE16-NEXT: s_cselect_b32 s9, -1, 0
; GFX1150-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX1150-TRUE16-NEXT: v_cndmask_b16 v0.l, s5, v0.l, s9
@@ -10017,6 +10126,7 @@ define amdgpu_kernel void @frem_v4f16(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX1150-TRUE16-NEXT: s_and_b32 s9, s9, exec_lo
; GFX1150-TRUE16-NEXT: s_cselect_b32 s9, 1, 0
; GFX1150-TRUE16-NEXT: s_cmp_lg_u32 s9, 1
+; GFX1150-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1150-TRUE16-NEXT: s_cbranch_scc1 .LBB10_8
; GFX1150-TRUE16-NEXT: ; %bb.4: ; %frem.compute85
; GFX1150-TRUE16-NEXT: v_frexp_mant_f32_e32 v1, s6
@@ -10033,10 +10143,11 @@ define amdgpu_kernel void @frem_v4f16(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX1150-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1150-TRUE16-NEXT: v_not_b32_e32 v3, v3
; GFX1150-TRUE16-NEXT: v_rcp_f32_e32 v5, v4
-; GFX1150-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_2) | instid1(TRANS32_DEP_1)
+; GFX1150-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1150-TRUE16-NEXT: v_add_nc_u32_e32 v3, v3, v2
; GFX1150-TRUE16-NEXT: v_div_scale_f32 v2, vcc_lo, 1.0, v1, 1.0
; GFX1150-TRUE16-NEXT: s_denorm_mode 15
+; GFX1150-TRUE16-NEXT: s_delay_alu instid0(TRANS32_DEP_1) | instid1(SALU_CYCLE_1)
; GFX1150-TRUE16-NEXT: v_fma_f32 v6, -v4, v5, 1.0
; GFX1150-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1150-TRUE16-NEXT: v_fmac_f32_e32 v5, v6, v5
@@ -10044,9 +10155,10 @@ define amdgpu_kernel void @frem_v4f16(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX1150-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1150-TRUE16-NEXT: v_fma_f32 v7, -v4, v6, v2
; GFX1150-TRUE16-NEXT: v_fmac_f32_e32 v6, v7, v5
-; GFX1150-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
+; GFX1150-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1150-TRUE16-NEXT: v_fma_f32 v2, -v4, v6, v2
; GFX1150-TRUE16-NEXT: s_denorm_mode 12
+; GFX1150-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX1150-TRUE16-NEXT: v_div_fmas_f32 v2, v2, v5, v6
; GFX1150-TRUE16-NEXT: v_cmp_gt_i32_e32 vcc_lo, 12, v3
; GFX1150-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2)
@@ -10082,12 +10194,13 @@ define amdgpu_kernel void @frem_v4f16(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX1150-TRUE16-NEXT: s_and_b32 s8, s6, 0x7fff
; GFX1150-TRUE16-NEXT: s_cvt_f32_f16 s9, s5
; GFX1150-TRUE16-NEXT: s_cvt_f32_f16 s8, s8
-; GFX1150-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_3)
+; GFX1150-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_3) | instskip(NEXT) | instid1(SALU_CYCLE_3)
; GFX1150-TRUE16-NEXT: s_cmp_ngt_f32 s9, s8
; GFX1150-TRUE16-NEXT: s_cbranch_scc0 .LBB10_10
; GFX1150-TRUE16-NEXT: ; %bb.9: ; %frem.else53
; GFX1150-TRUE16-NEXT: s_cmp_eq_f32 s9, s8
; GFX1150-TRUE16-NEXT: v_and_b16 v0.h, 0x8000, s10
+; GFX1150-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_2)
; GFX1150-TRUE16-NEXT: s_cselect_b32 s11, -1, 0
; GFX1150-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX1150-TRUE16-NEXT: v_cndmask_b16 v1.l, s10, v0.h, s11
@@ -10101,6 +10214,7 @@ define amdgpu_kernel void @frem_v4f16(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX1150-TRUE16-NEXT: s_and_b32 s10, s10, exec_lo
; GFX1150-TRUE16-NEXT: s_cselect_b32 s10, 1, 0
; GFX1150-TRUE16-NEXT: s_cmp_lg_u32 s10, 1
+; GFX1150-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1150-TRUE16-NEXT: s_cbranch_scc1 .LBB10_16
; GFX1150-TRUE16-NEXT: ; %bb.12: ; %frem.compute52
; GFX1150-TRUE16-NEXT: v_frexp_mant_f32_e32 v2, s8
@@ -10117,10 +10231,11 @@ define amdgpu_kernel void @frem_v4f16(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX1150-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1150-TRUE16-NEXT: v_not_b32_e32 v4, v4
; GFX1150-TRUE16-NEXT: v_rcp_f32_e32 v6, v5
-; GFX1150-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_2) | instid1(TRANS32_DEP_1)
+; GFX1150-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1150-TRUE16-NEXT: v_add_nc_u32_e32 v4, v4, v3
; GFX1150-TRUE16-NEXT: v_div_scale_f32 v3, vcc_lo, 1.0, v2, 1.0
; GFX1150-TRUE16-NEXT: s_denorm_mode 15
+; GFX1150-TRUE16-NEXT: s_delay_alu instid0(TRANS32_DEP_1) | instid1(SALU_CYCLE_1)
; GFX1150-TRUE16-NEXT: v_fma_f32 v7, -v5, v6, 1.0
; GFX1150-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1150-TRUE16-NEXT: v_fmac_f32_e32 v6, v7, v6
@@ -10128,9 +10243,10 @@ define amdgpu_kernel void @frem_v4f16(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX1150-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1150-TRUE16-NEXT: v_fma_f32 v8, -v5, v7, v3
; GFX1150-TRUE16-NEXT: v_fmac_f32_e32 v7, v8, v6
-; GFX1150-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
+; GFX1150-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1150-TRUE16-NEXT: v_fma_f32 v3, -v5, v7, v3
; GFX1150-TRUE16-NEXT: s_denorm_mode 12
+; GFX1150-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX1150-TRUE16-NEXT: v_div_fmas_f32 v3, v3, v6, v7
; GFX1150-TRUE16-NEXT: v_cmp_gt_i32_e32 vcc_lo, 12, v4
; GFX1150-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2)
@@ -10164,12 +10280,13 @@ define amdgpu_kernel void @frem_v4f16(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX1150-TRUE16-NEXT: s_and_b32 s9, s2, 0x7fff
; GFX1150-TRUE16-NEXT: s_cvt_f32_f16 s10, s8
; GFX1150-TRUE16-NEXT: s_cvt_f32_f16 s9, s9
-; GFX1150-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_3)
+; GFX1150-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_3) | instskip(NEXT) | instid1(SALU_CYCLE_3)
; GFX1150-TRUE16-NEXT: s_cmp_ngt_f32 s10, s9
; GFX1150-TRUE16-NEXT: s_cbranch_scc0 .LBB10_18
; GFX1150-TRUE16-NEXT: ; %bb.17: ; %frem.else20
; GFX1150-TRUE16-NEXT: s_cmp_eq_f32 s10, s9
; GFX1150-TRUE16-NEXT: v_and_b16 v0.h, 0x8000, s7
+; GFX1150-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_2)
; GFX1150-TRUE16-NEXT: s_cselect_b32 s11, -1, 0
; GFX1150-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX1150-TRUE16-NEXT: v_cndmask_b16 v2.l, s7, v0.h, s11
@@ -10183,6 +10300,7 @@ define amdgpu_kernel void @frem_v4f16(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX1150-TRUE16-NEXT: s_and_b32 s11, s11, exec_lo
; GFX1150-TRUE16-NEXT: s_cselect_b32 s11, 1, 0
; GFX1150-TRUE16-NEXT: s_cmp_lg_u32 s11, 1
+; GFX1150-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1150-TRUE16-NEXT: s_cbranch_scc1 .LBB10_24
; GFX1150-TRUE16-NEXT: ; %bb.20: ; %frem.compute19
; GFX1150-TRUE16-NEXT: v_frexp_mant_f32_e32 v3, s9
@@ -10199,10 +10317,11 @@ define amdgpu_kernel void @frem_v4f16(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX1150-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1150-TRUE16-NEXT: v_not_b32_e32 v5, v5
; GFX1150-TRUE16-NEXT: v_rcp_f32_e32 v7, v6
-; GFX1150-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_2) | instid1(TRANS32_DEP_1)
+; GFX1150-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1150-TRUE16-NEXT: v_add_nc_u32_e32 v5, v5, v4
; GFX1150-TRUE16-NEXT: v_div_scale_f32 v4, vcc_lo, 1.0, v3, 1.0
; GFX1150-TRUE16-NEXT: s_denorm_mode 15
+; GFX1150-TRUE16-NEXT: s_delay_alu instid0(TRANS32_DEP_1) | instid1(SALU_CYCLE_1)
; GFX1150-TRUE16-NEXT: v_fma_f32 v8, -v6, v7, 1.0
; GFX1150-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1150-TRUE16-NEXT: v_fmac_f32_e32 v7, v8, v7
@@ -10210,9 +10329,10 @@ define amdgpu_kernel void @frem_v4f16(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX1150-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1150-TRUE16-NEXT: v_fma_f32 v9, -v6, v8, v4
; GFX1150-TRUE16-NEXT: v_fmac_f32_e32 v8, v9, v7
-; GFX1150-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
+; GFX1150-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1150-TRUE16-NEXT: v_fma_f32 v4, -v6, v8, v4
; GFX1150-TRUE16-NEXT: s_denorm_mode 12
+; GFX1150-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX1150-TRUE16-NEXT: v_div_fmas_f32 v4, v4, v7, v8
; GFX1150-TRUE16-NEXT: v_cmp_gt_i32_e32 vcc_lo, 12, v5
; GFX1150-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2)
@@ -10248,12 +10368,13 @@ define amdgpu_kernel void @frem_v4f16(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX1150-TRUE16-NEXT: s_and_b32 s10, s9, 0x7fff
; GFX1150-TRUE16-NEXT: s_cvt_f32_f16 s11, s7
; GFX1150-TRUE16-NEXT: s_cvt_f32_f16 s10, s10
-; GFX1150-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_3)
+; GFX1150-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_3) | instskip(NEXT) | instid1(SALU_CYCLE_3)
; GFX1150-TRUE16-NEXT: s_cmp_ngt_f32 s11, s10
; GFX1150-TRUE16-NEXT: s_cbranch_scc0 .LBB10_26
; GFX1150-TRUE16-NEXT: ; %bb.25: ; %frem.else
; GFX1150-TRUE16-NEXT: s_cmp_eq_f32 s11, s10
; GFX1150-TRUE16-NEXT: v_and_b16 v0.h, 0x8000, s12
+; GFX1150-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_2)
; GFX1150-TRUE16-NEXT: s_cselect_b32 s13, -1, 0
; GFX1150-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX1150-TRUE16-NEXT: v_cndmask_b16 v3.l, s12, v0.h, s13
@@ -10267,6 +10388,7 @@ define amdgpu_kernel void @frem_v4f16(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX1150-TRUE16-NEXT: s_and_b32 s12, s12, exec_lo
; GFX1150-TRUE16-NEXT: s_cselect_b32 s12, 1, 0
; GFX1150-TRUE16-NEXT: s_cmp_lg_u32 s12, 1
+; GFX1150-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1150-TRUE16-NEXT: s_cbranch_scc1 .LBB10_32
; GFX1150-TRUE16-NEXT: ; %bb.28: ; %frem.compute
; GFX1150-TRUE16-NEXT: v_frexp_mant_f32_e32 v4, s10
@@ -10283,10 +10405,11 @@ define amdgpu_kernel void @frem_v4f16(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX1150-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1150-TRUE16-NEXT: v_not_b32_e32 v6, v6
; GFX1150-TRUE16-NEXT: v_rcp_f32_e32 v8, v7
-; GFX1150-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_2) | instid1(TRANS32_DEP_1)
+; GFX1150-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1150-TRUE16-NEXT: v_add_nc_u32_e32 v6, v6, v5
; GFX1150-TRUE16-NEXT: v_div_scale_f32 v5, vcc_lo, 1.0, v4, 1.0
; GFX1150-TRUE16-NEXT: s_denorm_mode 15
+; GFX1150-TRUE16-NEXT: s_delay_alu instid0(TRANS32_DEP_1) | instid1(SALU_CYCLE_1)
; GFX1150-TRUE16-NEXT: v_fma_f32 v9, -v7, v8, 1.0
; GFX1150-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1150-TRUE16-NEXT: v_fmac_f32_e32 v8, v9, v8
@@ -10294,9 +10417,10 @@ define amdgpu_kernel void @frem_v4f16(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX1150-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1150-TRUE16-NEXT: v_fma_f32 v10, -v7, v9, v5
; GFX1150-TRUE16-NEXT: v_fmac_f32_e32 v9, v10, v8
-; GFX1150-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
+; GFX1150-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1150-TRUE16-NEXT: v_fma_f32 v5, -v7, v9, v5
; GFX1150-TRUE16-NEXT: s_denorm_mode 12
+; GFX1150-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX1150-TRUE16-NEXT: v_div_fmas_f32 v5, v5, v8, v9
; GFX1150-TRUE16-NEXT: v_cmp_gt_i32_e32 vcc_lo, 12, v6
; GFX1150-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2)
@@ -10327,33 +10451,36 @@ define amdgpu_kernel void @frem_v4f16(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX1150-TRUE16-NEXT: ; implicit-def: $vgpr3
; GFX1150-TRUE16-NEXT: .LBB10_32: ; %Flow124
; GFX1150-TRUE16-NEXT: s_cmp_lg_f16 s4, 0
+; GFX1150-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_3) | instskip(SKIP_1) | instid1(SALU_CYCLE_3)
; GFX1150-TRUE16-NEXT: s_cselect_b32 s4, -1, 0
; GFX1150-TRUE16-NEXT: s_cmp_nge_f16 s3, 0x7c00
; GFX1150-TRUE16-NEXT: s_cselect_b32 s3, -1, 0
-; GFX1150-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1150-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_2)
; GFX1150-TRUE16-NEXT: s_and_b32 s3, s3, s4
; GFX1150-TRUE16-NEXT: s_cmp_lg_f16 s6, 0
; GFX1150-TRUE16-NEXT: v_cndmask_b16 v0.l, 0x7e00, v0.l, s3
; GFX1150-TRUE16-NEXT: s_cselect_b32 s3, -1, 0
; GFX1150-TRUE16-NEXT: s_cmp_nge_f16 s5, 0x7c00
+; GFX1150-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_3) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1150-TRUE16-NEXT: s_cselect_b32 s4, -1, 0
-; GFX1150-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1150-TRUE16-NEXT: s_and_b32 s3, s4, s3
; GFX1150-TRUE16-NEXT: s_cmp_lg_f16 s2, 0
; GFX1150-TRUE16-NEXT: v_cndmask_b16 v0.h, 0x7e00, v1.l, s3
+; GFX1150-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_2) | instskip(SKIP_1) | instid1(SALU_CYCLE_3)
; GFX1150-TRUE16-NEXT: s_cselect_b32 s2, -1, 0
; GFX1150-TRUE16-NEXT: s_cmp_nge_f16 s8, 0x7c00
; GFX1150-TRUE16-NEXT: s_cselect_b32 s3, -1, 0
-; GFX1150-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1150-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_3) | instid1(SALU_CYCLE_1)
; GFX1150-TRUE16-NEXT: s_and_b32 s2, s3, s2
; GFX1150-TRUE16-NEXT: s_cmp_lg_f16 s9, 0
; GFX1150-TRUE16-NEXT: v_cndmask_b16 v1.l, 0x7e00, v2.l, s2
; GFX1150-TRUE16-NEXT: v_mov_b32_e32 v2, 0
; GFX1150-TRUE16-NEXT: s_cselect_b32 s2, -1, 0
; GFX1150-TRUE16-NEXT: s_cmp_nge_f16 s7, 0x7c00
+; GFX1150-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_3) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1150-TRUE16-NEXT: s_cselect_b32 s3, -1, 0
-; GFX1150-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1150-TRUE16-NEXT: s_and_b32 s2, s3, s2
+; GFX1150-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1150-TRUE16-NEXT: v_cndmask_b16 v1.h, 0x7e00, v3.l, s2
; GFX1150-TRUE16-NEXT: global_store_b64 v2, v[0:1], s[0:1]
; GFX1150-TRUE16-NEXT: s_endpgm
@@ -10378,7 +10505,7 @@ define amdgpu_kernel void @frem_v4f16(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX1150-FAKE16-NEXT: v_readfirstlane_b32 s2, v1
; GFX1150-FAKE16-NEXT: s_and_b32 s6, s4, 0x7fff
; GFX1150-FAKE16-NEXT: s_cvt_f32_f16 s6, s6
-; GFX1150-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_3)
+; GFX1150-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_3) | instskip(NEXT) | instid1(SALU_CYCLE_3)
; GFX1150-FAKE16-NEXT: s_cmp_ngt_f32 s8, s6
; GFX1150-FAKE16-NEXT: s_cbranch_scc0 .LBB10_2
; GFX1150-FAKE16-NEXT: ; %bb.1: ; %frem.else86
@@ -10386,8 +10513,9 @@ define amdgpu_kernel void @frem_v4f16(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX1150-FAKE16-NEXT: s_cmp_eq_f32 s8, s6
; GFX1150-FAKE16-NEXT: v_mov_b32_e32 v0, s9
; GFX1150-FAKE16-NEXT: s_mov_b32 s9, 0
+; GFX1150-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1150-FAKE16-NEXT: s_cselect_b32 vcc_lo, -1, 0
-; GFX1150-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1150-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX1150-FAKE16-NEXT: v_cndmask_b32_e32 v0, s5, v0, vcc_lo
; GFX1150-FAKE16-NEXT: s_branch .LBB10_3
; GFX1150-FAKE16-NEXT: .LBB10_2:
@@ -10398,6 +10526,7 @@ define amdgpu_kernel void @frem_v4f16(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX1150-FAKE16-NEXT: s_and_b32 s9, s9, exec_lo
; GFX1150-FAKE16-NEXT: s_cselect_b32 s9, 1, 0
; GFX1150-FAKE16-NEXT: s_cmp_lg_u32 s9, 1
+; GFX1150-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1150-FAKE16-NEXT: s_cbranch_scc1 .LBB10_9
; GFX1150-FAKE16-NEXT: ; %bb.4: ; %frem.compute85
; GFX1150-FAKE16-NEXT: v_frexp_mant_f32_e32 v1, s6
@@ -10419,19 +10548,21 @@ define amdgpu_kernel void @frem_v4f16(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX1150-FAKE16-NEXT: v_add_nc_u32_e32 v4, v4, v3
; GFX1150-FAKE16-NEXT: v_div_scale_f32 v3, vcc_lo, 1.0, v1, 1.0
; GFX1150-FAKE16-NEXT: s_denorm_mode 15
-; GFX1150-FAKE16-NEXT: s_delay_alu instid0(TRANS32_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1150-FAKE16-NEXT: s_delay_alu instid0(TRANS32_DEP_1) | instid1(SALU_CYCLE_1)
; GFX1150-FAKE16-NEXT: v_fma_f32 v7, -v5, v6, 1.0
-; GFX1150-FAKE16-NEXT: v_fmac_f32_e32 v6, v7, v6
; GFX1150-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1150-FAKE16-NEXT: v_fmac_f32_e32 v6, v7, v6
; GFX1150-FAKE16-NEXT: v_mul_f32_e32 v7, v3, v6
-; GFX1150-FAKE16-NEXT: v_fma_f32 v8, -v5, v7, v3
; GFX1150-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1150-FAKE16-NEXT: v_fma_f32 v8, -v5, v7, v3
; GFX1150-FAKE16-NEXT: v_fmac_f32_e32 v7, v8, v6
+; GFX1150-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1150-FAKE16-NEXT: v_fma_f32 v3, -v5, v7, v3
; GFX1150-FAKE16-NEXT: s_denorm_mode 12
-; GFX1150-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
+; GFX1150-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX1150-FAKE16-NEXT: v_div_fmas_f32 v3, v3, v6, v7
; GFX1150-FAKE16-NEXT: v_cmp_gt_i32_e32 vcc_lo, 12, v4
+; GFX1150-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX1150-FAKE16-NEXT: v_div_fixup_f32 v3, v3, v1, 1.0
; GFX1150-FAKE16-NEXT: s_cbranch_vccnz .LBB10_8
; GFX1150-FAKE16-NEXT: ; %bb.5: ; %frem.loop_body93.preheader
@@ -10486,7 +10617,7 @@ define amdgpu_kernel void @frem_v4f16(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX1150-FAKE16-NEXT: s_and_b32 s9, s6, 0x7fff
; GFX1150-FAKE16-NEXT: s_cvt_f32_f16 s10, s5
; GFX1150-FAKE16-NEXT: s_cvt_f32_f16 s9, s9
-; GFX1150-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_3)
+; GFX1150-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_3) | instskip(NEXT) | instid1(SALU_CYCLE_3)
; GFX1150-FAKE16-NEXT: s_cmp_ngt_f32 s10, s9
; GFX1150-FAKE16-NEXT: s_cbranch_scc0 .LBB10_11
; GFX1150-FAKE16-NEXT: ; %bb.10: ; %frem.else53
@@ -10494,8 +10625,9 @@ define amdgpu_kernel void @frem_v4f16(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX1150-FAKE16-NEXT: s_cmp_eq_f32 s10, s9
; GFX1150-FAKE16-NEXT: v_mov_b32_e32 v1, s11
; GFX1150-FAKE16-NEXT: s_mov_b32 s11, 0
+; GFX1150-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1150-FAKE16-NEXT: s_cselect_b32 vcc_lo, -1, 0
-; GFX1150-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1150-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX1150-FAKE16-NEXT: v_cndmask_b32_e32 v1, s8, v1, vcc_lo
; GFX1150-FAKE16-NEXT: s_branch .LBB10_12
; GFX1150-FAKE16-NEXT: .LBB10_11:
@@ -10506,6 +10638,7 @@ define amdgpu_kernel void @frem_v4f16(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX1150-FAKE16-NEXT: s_and_b32 s11, s11, exec_lo
; GFX1150-FAKE16-NEXT: s_cselect_b32 s11, 1, 0
; GFX1150-FAKE16-NEXT: s_cmp_lg_u32 s11, 1
+; GFX1150-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1150-FAKE16-NEXT: s_cbranch_scc1 .LBB10_18
; GFX1150-FAKE16-NEXT: ; %bb.13: ; %frem.compute52
; GFX1150-FAKE16-NEXT: v_frexp_mant_f32_e32 v2, s9
@@ -10527,19 +10660,21 @@ define amdgpu_kernel void @frem_v4f16(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX1150-FAKE16-NEXT: v_add_nc_u32_e32 v5, v5, v4
; GFX1150-FAKE16-NEXT: v_div_scale_f32 v4, vcc_lo, 1.0, v2, 1.0
; GFX1150-FAKE16-NEXT: s_denorm_mode 15
-; GFX1150-FAKE16-NEXT: s_delay_alu instid0(TRANS32_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1150-FAKE16-NEXT: s_delay_alu instid0(TRANS32_DEP_1) | instid1(SALU_CYCLE_1)
; GFX1150-FAKE16-NEXT: v_fma_f32 v8, -v6, v7, 1.0
-; GFX1150-FAKE16-NEXT: v_fmac_f32_e32 v7, v8, v7
; GFX1150-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1150-FAKE16-NEXT: v_fmac_f32_e32 v7, v8, v7
; GFX1150-FAKE16-NEXT: v_mul_f32_e32 v8, v4, v7
-; GFX1150-FAKE16-NEXT: v_fma_f32 v9, -v6, v8, v4
; GFX1150-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1150-FAKE16-NEXT: v_fma_f32 v9, -v6, v8, v4
; GFX1150-FAKE16-NEXT: v_fmac_f32_e32 v8, v9, v7
+; GFX1150-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1150-FAKE16-NEXT: v_fma_f32 v4, -v6, v8, v4
; GFX1150-FAKE16-NEXT: s_denorm_mode 12
-; GFX1150-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
+; GFX1150-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX1150-FAKE16-NEXT: v_div_fmas_f32 v4, v4, v7, v8
; GFX1150-FAKE16-NEXT: v_cmp_gt_i32_e32 vcc_lo, 12, v5
+; GFX1150-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX1150-FAKE16-NEXT: v_div_fixup_f32 v4, v4, v2, 1.0
; GFX1150-FAKE16-NEXT: s_cbranch_vccnz .LBB10_17
; GFX1150-FAKE16-NEXT: ; %bb.14: ; %frem.loop_body60.preheader
@@ -10592,7 +10727,7 @@ define amdgpu_kernel void @frem_v4f16(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX1150-FAKE16-NEXT: s_and_b32 s9, s2, 0x7fff
; GFX1150-FAKE16-NEXT: s_cvt_f32_f16 s10, s8
; GFX1150-FAKE16-NEXT: s_cvt_f32_f16 s9, s9
-; GFX1150-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_3)
+; GFX1150-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_3) | instskip(NEXT) | instid1(SALU_CYCLE_3)
; GFX1150-FAKE16-NEXT: s_cmp_ngt_f32 s10, s9
; GFX1150-FAKE16-NEXT: s_cbranch_scc0 .LBB10_20
; GFX1150-FAKE16-NEXT: ; %bb.19: ; %frem.else20
@@ -10600,8 +10735,9 @@ define amdgpu_kernel void @frem_v4f16(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX1150-FAKE16-NEXT: s_cmp_eq_f32 s10, s9
; GFX1150-FAKE16-NEXT: v_mov_b32_e32 v2, s11
; GFX1150-FAKE16-NEXT: s_mov_b32 s11, 0
+; GFX1150-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1150-FAKE16-NEXT: s_cselect_b32 vcc_lo, -1, 0
-; GFX1150-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1150-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX1150-FAKE16-NEXT: v_cndmask_b32_e32 v2, s7, v2, vcc_lo
; GFX1150-FAKE16-NEXT: s_branch .LBB10_21
; GFX1150-FAKE16-NEXT: .LBB10_20:
@@ -10612,6 +10748,7 @@ define amdgpu_kernel void @frem_v4f16(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX1150-FAKE16-NEXT: s_and_b32 s11, s11, exec_lo
; GFX1150-FAKE16-NEXT: s_cselect_b32 s11, 1, 0
; GFX1150-FAKE16-NEXT: s_cmp_lg_u32 s11, 1
+; GFX1150-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1150-FAKE16-NEXT: s_cbranch_scc1 .LBB10_27
; GFX1150-FAKE16-NEXT: ; %bb.22: ; %frem.compute19
; GFX1150-FAKE16-NEXT: v_frexp_mant_f32_e32 v3, s9
@@ -10633,19 +10770,21 @@ define amdgpu_kernel void @frem_v4f16(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX1150-FAKE16-NEXT: v_add_nc_u32_e32 v6, v6, v5
; GFX1150-FAKE16-NEXT: v_div_scale_f32 v5, vcc_lo, 1.0, v3, 1.0
; GFX1150-FAKE16-NEXT: s_denorm_mode 15
-; GFX1150-FAKE16-NEXT: s_delay_alu instid0(TRANS32_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1150-FAKE16-NEXT: s_delay_alu instid0(TRANS32_DEP_1) | instid1(SALU_CYCLE_1)
; GFX1150-FAKE16-NEXT: v_fma_f32 v9, -v7, v8, 1.0
-; GFX1150-FAKE16-NEXT: v_fmac_f32_e32 v8, v9, v8
; GFX1150-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1150-FAKE16-NEXT: v_fmac_f32_e32 v8, v9, v8
; GFX1150-FAKE16-NEXT: v_mul_f32_e32 v9, v5, v8
-; GFX1150-FAKE16-NEXT: v_fma_f32 v10, -v7, v9, v5
; GFX1150-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1150-FAKE16-NEXT: v_fma_f32 v10, -v7, v9, v5
; GFX1150-FAKE16-NEXT: v_fmac_f32_e32 v9, v10, v8
+; GFX1150-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1150-FAKE16-NEXT: v_fma_f32 v5, -v7, v9, v5
; GFX1150-FAKE16-NEXT: s_denorm_mode 12
-; GFX1150-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
+; GFX1150-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX1150-FAKE16-NEXT: v_div_fmas_f32 v5, v5, v8, v9
; GFX1150-FAKE16-NEXT: v_cmp_gt_i32_e32 vcc_lo, 12, v6
+; GFX1150-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX1150-FAKE16-NEXT: v_div_fixup_f32 v5, v5, v3, 1.0
; GFX1150-FAKE16-NEXT: s_cbranch_vccnz .LBB10_26
; GFX1150-FAKE16-NEXT: ; %bb.23: ; %frem.loop_body27.preheader
@@ -10700,7 +10839,7 @@ define amdgpu_kernel void @frem_v4f16(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX1150-FAKE16-NEXT: s_and_b32 s11, s9, 0x7fff
; GFX1150-FAKE16-NEXT: s_cvt_f32_f16 s12, s7
; GFX1150-FAKE16-NEXT: s_cvt_f32_f16 s11, s11
-; GFX1150-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_3)
+; GFX1150-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_3) | instskip(NEXT) | instid1(SALU_CYCLE_3)
; GFX1150-FAKE16-NEXT: s_cmp_ngt_f32 s12, s11
; GFX1150-FAKE16-NEXT: s_cbranch_scc0 .LBB10_29
; GFX1150-FAKE16-NEXT: ; %bb.28: ; %frem.else
@@ -10708,8 +10847,9 @@ define amdgpu_kernel void @frem_v4f16(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX1150-FAKE16-NEXT: s_cmp_eq_f32 s12, s11
; GFX1150-FAKE16-NEXT: v_mov_b32_e32 v3, s13
; GFX1150-FAKE16-NEXT: s_mov_b32 s13, 0
+; GFX1150-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1150-FAKE16-NEXT: s_cselect_b32 vcc_lo, -1, 0
-; GFX1150-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1150-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX1150-FAKE16-NEXT: v_cndmask_b32_e32 v3, s10, v3, vcc_lo
; GFX1150-FAKE16-NEXT: s_branch .LBB10_30
; GFX1150-FAKE16-NEXT: .LBB10_29:
@@ -10720,6 +10860,7 @@ define amdgpu_kernel void @frem_v4f16(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX1150-FAKE16-NEXT: s_and_b32 s13, s13, exec_lo
; GFX1150-FAKE16-NEXT: s_cselect_b32 s13, 1, 0
; GFX1150-FAKE16-NEXT: s_cmp_lg_u32 s13, 1
+; GFX1150-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1150-FAKE16-NEXT: s_cbranch_scc1 .LBB10_36
; GFX1150-FAKE16-NEXT: ; %bb.31: ; %frem.compute
; GFX1150-FAKE16-NEXT: v_frexp_mant_f32_e32 v4, s11
@@ -10741,19 +10882,21 @@ define amdgpu_kernel void @frem_v4f16(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX1150-FAKE16-NEXT: v_add_nc_u32_e32 v7, v7, v6
; GFX1150-FAKE16-NEXT: v_div_scale_f32 v6, vcc_lo, 1.0, v4, 1.0
; GFX1150-FAKE16-NEXT: s_denorm_mode 15
-; GFX1150-FAKE16-NEXT: s_delay_alu instid0(TRANS32_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1150-FAKE16-NEXT: s_delay_alu instid0(TRANS32_DEP_1) | instid1(SALU_CYCLE_1)
; GFX1150-FAKE16-NEXT: v_fma_f32 v10, -v8, v9, 1.0
-; GFX1150-FAKE16-NEXT: v_fmac_f32_e32 v9, v10, v9
; GFX1150-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1150-FAKE16-NEXT: v_fmac_f32_e32 v9, v10, v9
; GFX1150-FAKE16-NEXT: v_mul_f32_e32 v10, v6, v9
-; GFX1150-FAKE16-NEXT: v_fma_f32 v11, -v8, v10, v6
; GFX1150-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1150-FAKE16-NEXT: v_fma_f32 v11, -v8, v10, v6
; GFX1150-FAKE16-NEXT: v_fmac_f32_e32 v10, v11, v9
+; GFX1150-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1150-FAKE16-NEXT: v_fma_f32 v6, -v8, v10, v6
; GFX1150-FAKE16-NEXT: s_denorm_mode 12
-; GFX1150-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
+; GFX1150-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX1150-FAKE16-NEXT: v_div_fmas_f32 v6, v6, v9, v10
; GFX1150-FAKE16-NEXT: v_cmp_gt_i32_e32 vcc_lo, 12, v7
+; GFX1150-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX1150-FAKE16-NEXT: v_div_fixup_f32 v6, v6, v4, 1.0
; GFX1150-FAKE16-NEXT: s_cbranch_vccnz .LBB10_35
; GFX1150-FAKE16-NEXT: ; %bb.32: ; %frem.loop_body.preheader
@@ -10803,33 +10946,36 @@ define amdgpu_kernel void @frem_v4f16(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX1150-FAKE16-NEXT: v_bfi_b32 v3, 0x7fff, v3, s10
; GFX1150-FAKE16-NEXT: .LBB10_36: ; %Flow124
; GFX1150-FAKE16-NEXT: s_cmp_lg_f16 s4, 0
+; GFX1150-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_3) | instskip(SKIP_1) | instid1(SALU_CYCLE_3)
; GFX1150-FAKE16-NEXT: s_cselect_b32 s4, -1, 0
; GFX1150-FAKE16-NEXT: s_cmp_nge_f16 s3, 0x7c00
; GFX1150-FAKE16-NEXT: s_cselect_b32 s3, -1, 0
-; GFX1150-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1150-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_2)
; GFX1150-FAKE16-NEXT: s_and_b32 vcc_lo, s3, s4
; GFX1150-FAKE16-NEXT: s_cmp_lg_f16 s6, 0
; GFX1150-FAKE16-NEXT: v_cndmask_b32_e32 v0, 0x7e00, v0, vcc_lo
; GFX1150-FAKE16-NEXT: s_cselect_b32 s3, -1, 0
; GFX1150-FAKE16-NEXT: s_cmp_nge_f16 s5, 0x7c00
+; GFX1150-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_3) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1150-FAKE16-NEXT: s_cselect_b32 s4, -1, 0
-; GFX1150-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1150-FAKE16-NEXT: s_and_b32 vcc_lo, s4, s3
; GFX1150-FAKE16-NEXT: s_cmp_lg_f16 s2, 0
; GFX1150-FAKE16-NEXT: v_cndmask_b32_e32 v4, 0x7e00, v1, vcc_lo
+; GFX1150-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_2) | instskip(SKIP_1) | instid1(SALU_CYCLE_3)
; GFX1150-FAKE16-NEXT: s_cselect_b32 s2, -1, 0
; GFX1150-FAKE16-NEXT: s_cmp_nge_f16 s8, 0x7c00
; GFX1150-FAKE16-NEXT: s_cselect_b32 s3, -1, 0
-; GFX1150-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1150-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_3) | instid1(SALU_CYCLE_1)
; GFX1150-FAKE16-NEXT: s_and_b32 vcc_lo, s3, s2
; GFX1150-FAKE16-NEXT: s_cmp_lg_f16 s9, 0
; GFX1150-FAKE16-NEXT: v_dual_cndmask_b32 v1, 0x7e00, v2 :: v_dual_mov_b32 v2, 0
; GFX1150-FAKE16-NEXT: v_and_b32_e32 v0, 0xffff, v0
; GFX1150-FAKE16-NEXT: s_cselect_b32 s2, -1, 0
; GFX1150-FAKE16-NEXT: s_cmp_nge_f16 s7, 0x7c00
-; GFX1150-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
+; GFX1150-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(SALU_CYCLE_2)
; GFX1150-FAKE16-NEXT: v_and_b32_e32 v1, 0xffff, v1
; GFX1150-FAKE16-NEXT: s_cselect_b32 s3, -1, 0
+; GFX1150-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1150-FAKE16-NEXT: s_and_b32 vcc_lo, s3, s2
; GFX1150-FAKE16-NEXT: v_cndmask_b32_e32 v3, 0x7e00, v3, vcc_lo
; GFX1150-FAKE16-NEXT: v_lshl_or_b32 v0, v4, 16, v0
@@ -10858,12 +11004,13 @@ define amdgpu_kernel void @frem_v4f16(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX1200-TRUE16-NEXT: v_readfirstlane_b32 s2, v1
; GFX1200-TRUE16-NEXT: s_and_b32 s6, s4, 0x7fff
; GFX1200-TRUE16-NEXT: s_cvt_f32_f16 s6, s6
-; GFX1200-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_3)
+; GFX1200-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_3) | instskip(NEXT) | instid1(SALU_CYCLE_3)
; GFX1200-TRUE16-NEXT: s_cmp_ngt_f32 s8, s6
; GFX1200-TRUE16-NEXT: s_cbranch_scc0 .LBB10_2
; GFX1200-TRUE16-NEXT: ; %bb.1: ; %frem.else86
; GFX1200-TRUE16-NEXT: s_cmp_eq_f32 s8, s6
; GFX1200-TRUE16-NEXT: v_and_b16 v0.l, 0x8000, s5
+; GFX1200-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_2)
; GFX1200-TRUE16-NEXT: s_cselect_b32 s9, -1, 0
; GFX1200-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX1200-TRUE16-NEXT: v_cndmask_b16 v0.l, s5, v0.l, s9
@@ -10878,6 +11025,7 @@ define amdgpu_kernel void @frem_v4f16(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX1200-TRUE16-NEXT: s_cselect_b32 s9, 1, 0
; GFX1200-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX1200-TRUE16-NEXT: s_cmp_lg_u32 s9, 1
+; GFX1200-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1200-TRUE16-NEXT: s_cbranch_scc1 .LBB10_8
; GFX1200-TRUE16-NEXT: ; %bb.4: ; %frem.compute85
; GFX1200-TRUE16-NEXT: v_frexp_mant_f32_e32 v1, s6
@@ -10894,10 +11042,11 @@ define amdgpu_kernel void @frem_v4f16(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX1200-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1200-TRUE16-NEXT: v_not_b32_e32 v3, v3
; GFX1200-TRUE16-NEXT: v_rcp_f32_e32 v5, v4
-; GFX1200-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_2) | instid1(TRANS32_DEP_1)
+; GFX1200-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1200-TRUE16-NEXT: v_add_nc_u32_e32 v3, v3, v2
; GFX1200-TRUE16-NEXT: v_div_scale_f32 v2, vcc_lo, 1.0, v1, 1.0
; GFX1200-TRUE16-NEXT: s_denorm_mode 15
+; GFX1200-TRUE16-NEXT: s_delay_alu instid0(TRANS32_DEP_1) | instid1(SALU_CYCLE_1)
; GFX1200-TRUE16-NEXT: v_fma_f32 v6, -v4, v5, 1.0
; GFX1200-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1200-TRUE16-NEXT: v_fmac_f32_e32 v5, v6, v5
@@ -10905,9 +11054,10 @@ define amdgpu_kernel void @frem_v4f16(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX1200-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1200-TRUE16-NEXT: v_fma_f32 v7, -v4, v6, v2
; GFX1200-TRUE16-NEXT: v_fmac_f32_e32 v6, v7, v5
-; GFX1200-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
+; GFX1200-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1200-TRUE16-NEXT: v_fma_f32 v2, -v4, v6, v2
; GFX1200-TRUE16-NEXT: s_denorm_mode 12
+; GFX1200-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX1200-TRUE16-NEXT: v_div_fmas_f32 v2, v2, v5, v6
; GFX1200-TRUE16-NEXT: v_cmp_gt_i32_e32 vcc_lo, 12, v3
; GFX1200-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2)
@@ -10949,15 +11099,15 @@ define amdgpu_kernel void @frem_v4f16(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX1200-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX1200-TRUE16-NEXT: s_cvt_f32_f16 s8, s8
; GFX1200-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
-; GFX1200-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_2)
+; GFX1200-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_2) | instskip(NEXT) | instid1(SALU_CYCLE_3)
; GFX1200-TRUE16-NEXT: s_cmp_ngt_f32 s9, s8
; GFX1200-TRUE16-NEXT: s_cbranch_scc0 .LBB10_10
; GFX1200-TRUE16-NEXT: ; %bb.9: ; %frem.else53
; GFX1200-TRUE16-NEXT: s_cmp_eq_f32 s9, s8
; GFX1200-TRUE16-NEXT: v_and_b16 v0.h, 0x8000, s10
+; GFX1200-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_2) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX1200-TRUE16-NEXT: s_cselect_b32 s11, -1, 0
; GFX1200-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
-; GFX1200-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1200-TRUE16-NEXT: v_cndmask_b16 v1.l, s10, v0.h, s11
; GFX1200-TRUE16-NEXT: s_mov_b32 s10, 0
; GFX1200-TRUE16-NEXT: s_branch .LBB10_11
@@ -10970,6 +11120,7 @@ define amdgpu_kernel void @frem_v4f16(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX1200-TRUE16-NEXT: s_cselect_b32 s10, 1, 0
; GFX1200-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX1200-TRUE16-NEXT: s_cmp_lg_u32 s10, 1
+; GFX1200-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1200-TRUE16-NEXT: s_cbranch_scc1 .LBB10_16
; GFX1200-TRUE16-NEXT: ; %bb.12: ; %frem.compute52
; GFX1200-TRUE16-NEXT: v_frexp_mant_f32_e32 v2, s8
@@ -10986,10 +11137,11 @@ define amdgpu_kernel void @frem_v4f16(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX1200-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1200-TRUE16-NEXT: v_not_b32_e32 v4, v4
; GFX1200-TRUE16-NEXT: v_rcp_f32_e32 v6, v5
-; GFX1200-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_2) | instid1(TRANS32_DEP_1)
+; GFX1200-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1200-TRUE16-NEXT: v_add_nc_u32_e32 v4, v4, v3
; GFX1200-TRUE16-NEXT: v_div_scale_f32 v3, vcc_lo, 1.0, v2, 1.0
; GFX1200-TRUE16-NEXT: s_denorm_mode 15
+; GFX1200-TRUE16-NEXT: s_delay_alu instid0(TRANS32_DEP_1) | instid1(SALU_CYCLE_1)
; GFX1200-TRUE16-NEXT: v_fma_f32 v7, -v5, v6, 1.0
; GFX1200-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1200-TRUE16-NEXT: v_fmac_f32_e32 v6, v7, v6
@@ -11039,15 +11191,15 @@ define amdgpu_kernel void @frem_v4f16(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX1200-TRUE16-NEXT: s_cvt_f32_f16 s10, s8
; GFX1200-TRUE16-NEXT: s_cvt_f32_f16 s9, s9
; GFX1200-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
-; GFX1200-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_2)
+; GFX1200-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_2) | instskip(NEXT) | instid1(SALU_CYCLE_3)
; GFX1200-TRUE16-NEXT: s_cmp_ngt_f32 s10, s9
; GFX1200-TRUE16-NEXT: s_cbranch_scc0 .LBB10_18
; GFX1200-TRUE16-NEXT: ; %bb.17: ; %frem.else20
; GFX1200-TRUE16-NEXT: s_cmp_eq_f32 s10, s9
; GFX1200-TRUE16-NEXT: v_and_b16 v0.h, 0x8000, s7
+; GFX1200-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_2) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX1200-TRUE16-NEXT: s_cselect_b32 s11, -1, 0
; GFX1200-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
-; GFX1200-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1200-TRUE16-NEXT: v_cndmask_b16 v2.l, s7, v0.h, s11
; GFX1200-TRUE16-NEXT: s_mov_b32 s11, 0
; GFX1200-TRUE16-NEXT: s_branch .LBB10_19
@@ -11060,6 +11212,7 @@ define amdgpu_kernel void @frem_v4f16(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX1200-TRUE16-NEXT: s_cselect_b32 s11, 1, 0
; GFX1200-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX1200-TRUE16-NEXT: s_cmp_lg_u32 s11, 1
+; GFX1200-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1200-TRUE16-NEXT: s_cbranch_scc1 .LBB10_24
; GFX1200-TRUE16-NEXT: ; %bb.20: ; %frem.compute19
; GFX1200-TRUE16-NEXT: v_frexp_mant_f32_e32 v3, s9
@@ -11076,10 +11229,11 @@ define amdgpu_kernel void @frem_v4f16(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX1200-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1200-TRUE16-NEXT: v_not_b32_e32 v5, v5
; GFX1200-TRUE16-NEXT: v_rcp_f32_e32 v7, v6
-; GFX1200-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_2) | instid1(TRANS32_DEP_1)
+; GFX1200-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1200-TRUE16-NEXT: v_add_nc_u32_e32 v5, v5, v4
; GFX1200-TRUE16-NEXT: v_div_scale_f32 v4, vcc_lo, 1.0, v3, 1.0
; GFX1200-TRUE16-NEXT: s_denorm_mode 15
+; GFX1200-TRUE16-NEXT: s_delay_alu instid0(TRANS32_DEP_1) | instid1(SALU_CYCLE_1)
; GFX1200-TRUE16-NEXT: v_fma_f32 v8, -v6, v7, 1.0
; GFX1200-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1200-TRUE16-NEXT: v_fmac_f32_e32 v7, v8, v7
@@ -11132,15 +11286,15 @@ define amdgpu_kernel void @frem_v4f16(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX1200-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX1200-TRUE16-NEXT: s_cvt_f32_f16 s10, s10
; GFX1200-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
-; GFX1200-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_2)
+; GFX1200-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_2) | instskip(NEXT) | instid1(SALU_CYCLE_3)
; GFX1200-TRUE16-NEXT: s_cmp_ngt_f32 s11, s10
; GFX1200-TRUE16-NEXT: s_cbranch_scc0 .LBB10_26
; GFX1200-TRUE16-NEXT: ; %bb.25: ; %frem.else
; GFX1200-TRUE16-NEXT: s_cmp_eq_f32 s11, s10
; GFX1200-TRUE16-NEXT: v_and_b16 v0.h, 0x8000, s12
+; GFX1200-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_2) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX1200-TRUE16-NEXT: s_cselect_b32 s13, -1, 0
; GFX1200-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
-; GFX1200-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1200-TRUE16-NEXT: v_cndmask_b16 v3.l, s12, v0.h, s13
; GFX1200-TRUE16-NEXT: s_mov_b32 s12, 0
; GFX1200-TRUE16-NEXT: s_branch .LBB10_27
@@ -11153,6 +11307,7 @@ define amdgpu_kernel void @frem_v4f16(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX1200-TRUE16-NEXT: s_cselect_b32 s12, 1, 0
; GFX1200-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX1200-TRUE16-NEXT: s_cmp_lg_u32 s12, 1
+; GFX1200-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1200-TRUE16-NEXT: s_cbranch_scc1 .LBB10_32
; GFX1200-TRUE16-NEXT: ; %bb.28: ; %frem.compute
; GFX1200-TRUE16-NEXT: v_frexp_mant_f32_e32 v4, s10
@@ -11169,10 +11324,11 @@ define amdgpu_kernel void @frem_v4f16(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX1200-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1200-TRUE16-NEXT: v_not_b32_e32 v6, v6
; GFX1200-TRUE16-NEXT: v_rcp_f32_e32 v8, v7
-; GFX1200-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_2) | instid1(TRANS32_DEP_1)
+; GFX1200-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1200-TRUE16-NEXT: v_add_nc_u32_e32 v6, v6, v5
; GFX1200-TRUE16-NEXT: v_div_scale_f32 v5, vcc_lo, 1.0, v4, 1.0
; GFX1200-TRUE16-NEXT: s_denorm_mode 15
+; GFX1200-TRUE16-NEXT: s_delay_alu instid0(TRANS32_DEP_1) | instid1(SALU_CYCLE_1)
; GFX1200-TRUE16-NEXT: v_fma_f32 v9, -v7, v8, 1.0
; GFX1200-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1200-TRUE16-NEXT: v_fmac_f32_e32 v8, v9, v8
@@ -11217,6 +11373,7 @@ define amdgpu_kernel void @frem_v4f16(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX1200-TRUE16-NEXT: ; implicit-def: $vgpr3
; GFX1200-TRUE16-NEXT: .LBB10_32: ; %Flow124
; GFX1200-TRUE16-NEXT: s_cmp_lg_f16 s4, 0
+; GFX1200-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_3) | instskip(SKIP_1) | instid1(SALU_CYCLE_3)
; GFX1200-TRUE16-NEXT: s_cselect_b32 s4, -1, 0
; GFX1200-TRUE16-NEXT: s_cmp_nge_f16 s3, 0x7c00
; GFX1200-TRUE16-NEXT: s_cselect_b32 s3, -1, 0
@@ -11224,6 +11381,7 @@ define amdgpu_kernel void @frem_v4f16(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX1200-TRUE16-NEXT: s_and_b32 s3, s3, s4
; GFX1200-TRUE16-NEXT: s_cmp_lg_f16 s6, 0
; GFX1200-TRUE16-NEXT: v_cndmask_b16 v0.l, 0x7e00, v0.l, s3
+; GFX1200-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_2) | instskip(SKIP_1) | instid1(SALU_CYCLE_3)
; GFX1200-TRUE16-NEXT: s_cselect_b32 s3, -1, 0
; GFX1200-TRUE16-NEXT: s_cmp_nge_f16 s5, 0x7c00
; GFX1200-TRUE16-NEXT: s_cselect_b32 s4, -1, 0
@@ -11232,6 +11390,7 @@ define amdgpu_kernel void @frem_v4f16(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX1200-TRUE16-NEXT: s_cmp_lg_f16 s2, 0
; GFX1200-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX1200-TRUE16-NEXT: v_cndmask_b16 v0.h, 0x7e00, v1.l, s3
+; GFX1200-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_3)
; GFX1200-TRUE16-NEXT: s_cselect_b32 s2, -1, 0
; GFX1200-TRUE16-NEXT: s_cmp_nge_f16 s8, 0x7c00
; GFX1200-TRUE16-NEXT: s_cselect_b32 s3, -1, 0
@@ -11243,6 +11402,7 @@ define amdgpu_kernel void @frem_v4f16(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX1200-TRUE16-NEXT: v_mov_b32_e32 v2, 0
; GFX1200-TRUE16-NEXT: s_cselect_b32 s2, -1, 0
; GFX1200-TRUE16-NEXT: s_cmp_nge_f16 s7, 0x7c00
+; GFX1200-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_3)
; GFX1200-TRUE16-NEXT: s_cselect_b32 s3, -1, 0
; GFX1200-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX1200-TRUE16-NEXT: s_and_b32 s2, s3, s2
@@ -11271,7 +11431,7 @@ define amdgpu_kernel void @frem_v4f16(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX1200-FAKE16-NEXT: v_readfirstlane_b32 s2, v1
; GFX1200-FAKE16-NEXT: s_and_b32 s6, s4, 0x7fff
; GFX1200-FAKE16-NEXT: s_cvt_f32_f16 s6, s6
-; GFX1200-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_3)
+; GFX1200-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_3) | instskip(NEXT) | instid1(SALU_CYCLE_3)
; GFX1200-FAKE16-NEXT: s_cmp_ngt_f32 s8, s6
; GFX1200-FAKE16-NEXT: s_cbranch_scc0 .LBB10_2
; GFX1200-FAKE16-NEXT: ; %bb.1: ; %frem.else86
@@ -11279,8 +11439,9 @@ define amdgpu_kernel void @frem_v4f16(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX1200-FAKE16-NEXT: s_cmp_eq_f32 s8, s6
; GFX1200-FAKE16-NEXT: v_mov_b32_e32 v0, s9
; GFX1200-FAKE16-NEXT: s_mov_b32 s9, 0
+; GFX1200-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1200-FAKE16-NEXT: s_cselect_b32 vcc_lo, -1, 0
-; GFX1200-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1200-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX1200-FAKE16-NEXT: v_cndmask_b32_e32 v0, s5, v0, vcc_lo
; GFX1200-FAKE16-NEXT: s_branch .LBB10_3
; GFX1200-FAKE16-NEXT: .LBB10_2:
@@ -11292,6 +11453,7 @@ define amdgpu_kernel void @frem_v4f16(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX1200-FAKE16-NEXT: s_cselect_b32 s9, 1, 0
; GFX1200-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX1200-FAKE16-NEXT: s_cmp_lg_u32 s9, 1
+; GFX1200-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1200-FAKE16-NEXT: s_cbranch_scc1 .LBB10_9
; GFX1200-FAKE16-NEXT: ; %bb.4: ; %frem.compute85
; GFX1200-FAKE16-NEXT: v_frexp_mant_f32_e32 v1, s6
@@ -11313,20 +11475,21 @@ define amdgpu_kernel void @frem_v4f16(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX1200-FAKE16-NEXT: v_add_nc_u32_e32 v4, v4, v3
; GFX1200-FAKE16-NEXT: v_div_scale_f32 v3, vcc_lo, 1.0, v1, 1.0
; GFX1200-FAKE16-NEXT: s_denorm_mode 15
-; GFX1200-FAKE16-NEXT: s_delay_alu instid0(TRANS32_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1200-FAKE16-NEXT: s_delay_alu instid0(TRANS32_DEP_1) | instid1(SALU_CYCLE_1)
; GFX1200-FAKE16-NEXT: v_fma_f32 v7, -v5, v6, 1.0
-; GFX1200-FAKE16-NEXT: v_fmac_f32_e32 v6, v7, v6
; GFX1200-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1200-FAKE16-NEXT: v_fmac_f32_e32 v6, v7, v6
; GFX1200-FAKE16-NEXT: v_mul_f32_e32 v7, v3, v6
-; GFX1200-FAKE16-NEXT: v_fma_f32 v8, -v5, v7, v3
; GFX1200-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1200-FAKE16-NEXT: v_fma_f32 v8, -v5, v7, v3
; GFX1200-FAKE16-NEXT: v_fmac_f32_e32 v7, v8, v6
+; GFX1200-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_2) | instid1(VALU_DEP_1)
; GFX1200-FAKE16-NEXT: v_fma_f32 v3, -v5, v7, v3
; GFX1200-FAKE16-NEXT: s_denorm_mode 12
; GFX1200-FAKE16-NEXT: s_wait_alu depctr_va_vcc(0)
-; GFX1200-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX1200-FAKE16-NEXT: v_div_fmas_f32 v3, v3, v6, v7
; GFX1200-FAKE16-NEXT: v_cmp_gt_i32_e32 vcc_lo, 12, v4
+; GFX1200-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX1200-FAKE16-NEXT: v_div_fixup_f32 v3, v3, v1, 1.0
; GFX1200-FAKE16-NEXT: s_cbranch_vccnz .LBB10_8
; GFX1200-FAKE16-NEXT: ; %bb.5: ; %frem.loop_body93.preheader
@@ -11387,7 +11550,7 @@ define amdgpu_kernel void @frem_v4f16(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX1200-FAKE16-NEXT: s_cvt_f32_f16 s10, s5
; GFX1200-FAKE16-NEXT: s_cvt_f32_f16 s9, s9
; GFX1200-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
-; GFX1200-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_2)
+; GFX1200-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_2) | instskip(NEXT) | instid1(SALU_CYCLE_3)
; GFX1200-FAKE16-NEXT: s_cmp_ngt_f32 s10, s9
; GFX1200-FAKE16-NEXT: s_cbranch_scc0 .LBB10_11
; GFX1200-FAKE16-NEXT: ; %bb.10: ; %frem.else53
@@ -11395,16 +11558,16 @@ define amdgpu_kernel void @frem_v4f16(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX1200-FAKE16-NEXT: s_cmp_eq_f32 s10, s9
; GFX1200-FAKE16-NEXT: v_mov_b32_e32 v1, s11
; GFX1200-FAKE16-NEXT: s_mov_b32 s11, 0
+; GFX1200-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX1200-FAKE16-NEXT: s_cselect_b32 vcc_lo, -1, 0
; GFX1200-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
-; GFX1200-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1200-FAKE16-NEXT: v_cndmask_b32_e32 v1, s8, v1, vcc_lo
; GFX1200-FAKE16-NEXT: s_branch .LBB10_12
; GFX1200-FAKE16-NEXT: .LBB10_11:
; GFX1200-FAKE16-NEXT: s_mov_b32 s11, -1
; GFX1200-FAKE16-NEXT: ; implicit-def: $vgpr1
; GFX1200-FAKE16-NEXT: .LBB10_12: ; %Flow131
-; GFX1200-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1200-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_3) | instid1(SALU_CYCLE_1)
; GFX1200-FAKE16-NEXT: s_and_b32 s11, s11, exec_lo
; GFX1200-FAKE16-NEXT: s_cselect_b32 s11, 1, 0
; GFX1200-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
@@ -11430,20 +11593,21 @@ define amdgpu_kernel void @frem_v4f16(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX1200-FAKE16-NEXT: v_add_nc_u32_e32 v5, v5, v4
; GFX1200-FAKE16-NEXT: v_div_scale_f32 v4, vcc_lo, 1.0, v2, 1.0
; GFX1200-FAKE16-NEXT: s_denorm_mode 15
-; GFX1200-FAKE16-NEXT: s_delay_alu instid0(TRANS32_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1200-FAKE16-NEXT: s_delay_alu instid0(TRANS32_DEP_1) | instid1(SALU_CYCLE_1)
; GFX1200-FAKE16-NEXT: v_fma_f32 v8, -v6, v7, 1.0
-; GFX1200-FAKE16-NEXT: v_fmac_f32_e32 v7, v8, v7
; GFX1200-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1200-FAKE16-NEXT: v_fmac_f32_e32 v7, v8, v7
; GFX1200-FAKE16-NEXT: v_mul_f32_e32 v8, v4, v7
-; GFX1200-FAKE16-NEXT: v_fma_f32 v9, -v6, v8, v4
; GFX1200-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1200-FAKE16-NEXT: v_fma_f32 v9, -v6, v8, v4
; GFX1200-FAKE16-NEXT: v_fmac_f32_e32 v8, v9, v7
+; GFX1200-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_2) | instid1(VALU_DEP_1)
; GFX1200-FAKE16-NEXT: v_fma_f32 v4, -v6, v8, v4
; GFX1200-FAKE16-NEXT: s_denorm_mode 12
; GFX1200-FAKE16-NEXT: s_wait_alu depctr_va_vcc(0)
-; GFX1200-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX1200-FAKE16-NEXT: v_div_fmas_f32 v4, v4, v7, v8
; GFX1200-FAKE16-NEXT: v_cmp_gt_i32_e32 vcc_lo, 12, v5
+; GFX1200-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX1200-FAKE16-NEXT: v_div_fixup_f32 v4, v4, v2, 1.0
; GFX1200-FAKE16-NEXT: s_cbranch_vccnz .LBB10_17
; GFX1200-FAKE16-NEXT: ; %bb.14: ; %frem.loop_body60.preheader
@@ -11501,7 +11665,7 @@ define amdgpu_kernel void @frem_v4f16(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX1200-FAKE16-NEXT: s_cvt_f32_f16 s10, s8
; GFX1200-FAKE16-NEXT: s_cvt_f32_f16 s9, s9
; GFX1200-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
-; GFX1200-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_2)
+; GFX1200-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_2) | instskip(NEXT) | instid1(SALU_CYCLE_3)
; GFX1200-FAKE16-NEXT: s_cmp_ngt_f32 s10, s9
; GFX1200-FAKE16-NEXT: s_cbranch_scc0 .LBB10_20
; GFX1200-FAKE16-NEXT: ; %bb.19: ; %frem.else20
@@ -11524,6 +11688,7 @@ define amdgpu_kernel void @frem_v4f16(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX1200-FAKE16-NEXT: s_cselect_b32 s11, 1, 0
; GFX1200-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX1200-FAKE16-NEXT: s_cmp_lg_u32 s11, 1
+; GFX1200-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1200-FAKE16-NEXT: s_cbranch_scc1 .LBB10_27
; GFX1200-FAKE16-NEXT: ; %bb.22: ; %frem.compute19
; GFX1200-FAKE16-NEXT: v_frexp_mant_f32_e32 v3, s9
@@ -11545,20 +11710,21 @@ define amdgpu_kernel void @frem_v4f16(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX1200-FAKE16-NEXT: v_add_nc_u32_e32 v6, v6, v5
; GFX1200-FAKE16-NEXT: v_div_scale_f32 v5, vcc_lo, 1.0, v3, 1.0
; GFX1200-FAKE16-NEXT: s_denorm_mode 15
-; GFX1200-FAKE16-NEXT: s_delay_alu instid0(TRANS32_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1200-FAKE16-NEXT: s_delay_alu instid0(TRANS32_DEP_1) | instid1(SALU_CYCLE_1)
; GFX1200-FAKE16-NEXT: v_fma_f32 v9, -v7, v8, 1.0
-; GFX1200-FAKE16-NEXT: v_fmac_f32_e32 v8, v9, v8
; GFX1200-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1200-FAKE16-NEXT: v_fmac_f32_e32 v8, v9, v8
; GFX1200-FAKE16-NEXT: v_mul_f32_e32 v9, v5, v8
-; GFX1200-FAKE16-NEXT: v_fma_f32 v10, -v7, v9, v5
; GFX1200-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1200-FAKE16-NEXT: v_fma_f32 v10, -v7, v9, v5
; GFX1200-FAKE16-NEXT: v_fmac_f32_e32 v9, v10, v8
+; GFX1200-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_2) | instid1(VALU_DEP_1)
; GFX1200-FAKE16-NEXT: v_fma_f32 v5, -v7, v9, v5
; GFX1200-FAKE16-NEXT: s_denorm_mode 12
; GFX1200-FAKE16-NEXT: s_wait_alu depctr_va_vcc(0)
-; GFX1200-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX1200-FAKE16-NEXT: v_div_fmas_f32 v5, v5, v8, v9
; GFX1200-FAKE16-NEXT: v_cmp_gt_i32_e32 vcc_lo, 12, v6
+; GFX1200-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX1200-FAKE16-NEXT: v_div_fixup_f32 v5, v5, v3, 1.0
; GFX1200-FAKE16-NEXT: s_cbranch_vccnz .LBB10_26
; GFX1200-FAKE16-NEXT: ; %bb.23: ; %frem.loop_body27.preheader
@@ -11619,7 +11785,7 @@ define amdgpu_kernel void @frem_v4f16(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX1200-FAKE16-NEXT: s_cvt_f32_f16 s12, s7
; GFX1200-FAKE16-NEXT: s_cvt_f32_f16 s11, s11
; GFX1200-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
-; GFX1200-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_2)
+; GFX1200-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_2) | instskip(NEXT) | instid1(SALU_CYCLE_3)
; GFX1200-FAKE16-NEXT: s_cmp_ngt_f32 s12, s11
; GFX1200-FAKE16-NEXT: s_cbranch_scc0 .LBB10_29
; GFX1200-FAKE16-NEXT: ; %bb.28: ; %frem.else
@@ -11627,16 +11793,16 @@ define amdgpu_kernel void @frem_v4f16(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX1200-FAKE16-NEXT: s_cmp_eq_f32 s12, s11
; GFX1200-FAKE16-NEXT: v_mov_b32_e32 v3, s13
; GFX1200-FAKE16-NEXT: s_mov_b32 s13, 0
+; GFX1200-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX1200-FAKE16-NEXT: s_cselect_b32 vcc_lo, -1, 0
; GFX1200-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
-; GFX1200-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1200-FAKE16-NEXT: v_cndmask_b32_e32 v3, s10, v3, vcc_lo
; GFX1200-FAKE16-NEXT: s_branch .LBB10_30
; GFX1200-FAKE16-NEXT: .LBB10_29:
; GFX1200-FAKE16-NEXT: s_mov_b32 s13, -1
; GFX1200-FAKE16-NEXT: ; implicit-def: $vgpr3
; GFX1200-FAKE16-NEXT: .LBB10_30: ; %Flow123
-; GFX1200-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1200-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_3) | instid1(SALU_CYCLE_1)
; GFX1200-FAKE16-NEXT: s_and_b32 s13, s13, exec_lo
; GFX1200-FAKE16-NEXT: s_cselect_b32 s13, 1, 0
; GFX1200-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
@@ -11662,20 +11828,21 @@ define amdgpu_kernel void @frem_v4f16(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX1200-FAKE16-NEXT: v_add_nc_u32_e32 v7, v7, v6
; GFX1200-FAKE16-NEXT: v_div_scale_f32 v6, vcc_lo, 1.0, v4, 1.0
; GFX1200-FAKE16-NEXT: s_denorm_mode 15
-; GFX1200-FAKE16-NEXT: s_delay_alu instid0(TRANS32_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1200-FAKE16-NEXT: s_delay_alu instid0(TRANS32_DEP_1) | instid1(SALU_CYCLE_1)
; GFX1200-FAKE16-NEXT: v_fma_f32 v10, -v8, v9, 1.0
-; GFX1200-FAKE16-NEXT: v_fmac_f32_e32 v9, v10, v9
; GFX1200-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1200-FAKE16-NEXT: v_fmac_f32_e32 v9, v10, v9
; GFX1200-FAKE16-NEXT: v_mul_f32_e32 v10, v6, v9
-; GFX1200-FAKE16-NEXT: v_fma_f32 v11, -v8, v10, v6
; GFX1200-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1200-FAKE16-NEXT: v_fma_f32 v11, -v8, v10, v6
; GFX1200-FAKE16-NEXT: v_fmac_f32_e32 v10, v11, v9
+; GFX1200-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_2) | instid1(VALU_DEP_1)
; GFX1200-FAKE16-NEXT: v_fma_f32 v6, -v8, v10, v6
; GFX1200-FAKE16-NEXT: s_denorm_mode 12
; GFX1200-FAKE16-NEXT: s_wait_alu depctr_va_vcc(0)
-; GFX1200-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX1200-FAKE16-NEXT: v_div_fmas_f32 v6, v6, v9, v10
; GFX1200-FAKE16-NEXT: v_cmp_gt_i32_e32 vcc_lo, 12, v7
+; GFX1200-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX1200-FAKE16-NEXT: v_div_fixup_f32 v6, v6, v4, 1.0
; GFX1200-FAKE16-NEXT: s_cbranch_vccnz .LBB10_35
; GFX1200-FAKE16-NEXT: ; %bb.32: ; %frem.loop_body.preheader
@@ -11728,6 +11895,7 @@ define amdgpu_kernel void @frem_v4f16(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX1200-FAKE16-NEXT: v_bfi_b32 v3, 0x7fff, v3, s10
; GFX1200-FAKE16-NEXT: .LBB10_36: ; %Flow124
; GFX1200-FAKE16-NEXT: s_cmp_lg_f16 s4, 0
+; GFX1200-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_3) | instskip(SKIP_1) | instid1(SALU_CYCLE_3)
; GFX1200-FAKE16-NEXT: s_cselect_b32 s4, -1, 0
; GFX1200-FAKE16-NEXT: s_cmp_nge_f16 s3, 0x7c00
; GFX1200-FAKE16-NEXT: s_cselect_b32 s3, -1, 0
@@ -11736,6 +11904,7 @@ define amdgpu_kernel void @frem_v4f16(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX1200-FAKE16-NEXT: s_cmp_lg_f16 s6, 0
; GFX1200-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX1200-FAKE16-NEXT: v_cndmask_b32_e32 v0, 0x7e00, v0, vcc_lo
+; GFX1200-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_3)
; GFX1200-FAKE16-NEXT: s_cselect_b32 s3, -1, 0
; GFX1200-FAKE16-NEXT: s_cmp_nge_f16 s5, 0x7c00
; GFX1200-FAKE16-NEXT: s_cselect_b32 s4, -1, 0
@@ -11744,6 +11913,7 @@ define amdgpu_kernel void @frem_v4f16(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX1200-FAKE16-NEXT: s_cmp_lg_f16 s2, 0
; GFX1200-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX1200-FAKE16-NEXT: v_cndmask_b32_e32 v4, 0x7e00, v1, vcc_lo
+; GFX1200-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_3)
; GFX1200-FAKE16-NEXT: s_cselect_b32 s2, -1, 0
; GFX1200-FAKE16-NEXT: s_cmp_nge_f16 s8, 0x7c00
; GFX1200-FAKE16-NEXT: s_cselect_b32 s3, -1, 0
@@ -11755,14 +11925,14 @@ define amdgpu_kernel void @frem_v4f16(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX1200-FAKE16-NEXT: v_and_b32_e32 v0, 0xffff, v0
; GFX1200-FAKE16-NEXT: s_cselect_b32 s2, -1, 0
; GFX1200-FAKE16-NEXT: s_cmp_nge_f16 s7, 0x7c00
-; GFX1200-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
+; GFX1200-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(SALU_CYCLE_2)
; GFX1200-FAKE16-NEXT: v_and_b32_e32 v1, 0xffff, v1
; GFX1200-FAKE16-NEXT: s_cselect_b32 s3, -1, 0
+; GFX1200-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_3) | instid1(VALU_DEP_2)
; GFX1200-FAKE16-NEXT: s_and_b32 vcc_lo, s3, s2
; GFX1200-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX1200-FAKE16-NEXT: v_cndmask_b32_e32 v3, 0x7e00, v3, vcc_lo
; GFX1200-FAKE16-NEXT: v_lshl_or_b32 v0, v4, 16, v0
-; GFX1200-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX1200-FAKE16-NEXT: v_lshl_or_b32 v1, v3, 16, v1
; GFX1200-FAKE16-NEXT: global_store_b64 v2, v[0:1], s[0:1]
; GFX1200-FAKE16-NEXT: s_endpgm
@@ -12678,6 +12848,7 @@ define amdgpu_kernel void @frem_v2f32(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX11-NEXT: s_waitcnt vmcnt(0)
; GFX11-NEXT: v_cmp_ngt_f32_e64 s2, |v0|, |v2|
; GFX11-NEXT: s_and_b32 vcc_lo, exec_lo, s2
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_vccz .LBB11_2
; GFX11-NEXT: ; %bb.1: ; %frem.else16
; GFX11-NEXT: v_and_b32_e32 v4, 0x80000000, v0
@@ -12693,6 +12864,7 @@ define amdgpu_kernel void @frem_v2f32(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX11-NEXT: s_and_b32 s2, s2, exec_lo
; GFX11-NEXT: s_cselect_b32 s2, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s2, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB11_9
; GFX11-NEXT: ; %bb.4: ; %frem.compute15
; GFX11-NEXT: v_frexp_mant_f32_e64 v5, |v2|
@@ -12722,9 +12894,10 @@ define amdgpu_kernel void @frem_v2f32(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_fma_f32 v12, -v9, v11, v7
; GFX11-NEXT: v_fmac_f32_e32 v11, v12, v10
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_fma_f32 v7, -v9, v11, v7
; GFX11-NEXT: s_denorm_mode 12
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: v_div_fmas_f32 v7, v7, v10, v11
; GFX11-NEXT: v_cmp_gt_i32_e32 vcc_lo, 13, v8
; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2)
@@ -12773,6 +12946,7 @@ define amdgpu_kernel void @frem_v2f32(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX11-NEXT: .LBB11_9:
; GFX11-NEXT: v_cmp_ngt_f32_e64 s2, |v1|, |v3|
; GFX11-NEXT: s_and_b32 vcc_lo, exec_lo, s2
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_vccz .LBB11_11
; GFX11-NEXT: ; %bb.10: ; %frem.else
; GFX11-NEXT: v_and_b32_e32 v5, 0x80000000, v1
@@ -12788,6 +12962,7 @@ define amdgpu_kernel void @frem_v2f32(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX11-NEXT: s_and_b32 s2, s2, exec_lo
; GFX11-NEXT: s_cselect_b32 s2, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s2, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB11_18
; GFX11-NEXT: ; %bb.13: ; %frem.compute
; GFX11-NEXT: v_frexp_mant_f32_e64 v6, |v3|
@@ -12817,9 +12992,10 @@ define amdgpu_kernel void @frem_v2f32(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_fma_f32 v13, -v10, v12, v8
; GFX11-NEXT: v_fmac_f32_e32 v12, v13, v11
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_fma_f32 v8, -v10, v12, v8
; GFX11-NEXT: s_denorm_mode 12
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: v_div_fmas_f32 v8, v8, v11, v12
; GFX11-NEXT: v_cmp_gt_i32_e32 vcc_lo, 13, v9
; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2)
@@ -12874,6 +13050,7 @@ define amdgpu_kernel void @frem_v2f32(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX11-NEXT: v_cndmask_b32_e32 v0, 0x7fc00000, v4, vcc_lo
; GFX11-NEXT: v_cmp_lg_f32_e32 vcc_lo, 0, v3
; GFX11-NEXT: s_and_b32 vcc_lo, s2, vcc_lo
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: v_cndmask_b32_e32 v1, 0x7fc00000, v5, vcc_lo
; GFX11-NEXT: global_store_b64 v2, v[0:1], s[0:1]
; GFX11-NEXT: s_endpgm
@@ -12895,7 +13072,7 @@ define amdgpu_kernel void @frem_v2f32(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX1150-NEXT: v_readfirstlane_b32 s4, v1
; GFX1150-NEXT: v_readfirstlane_b32 s2, v2
; GFX1150-NEXT: s_and_b32 s7, s4, 0x7fffffff
-; GFX1150-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1150-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_3)
; GFX1150-NEXT: s_cmp_ngt_f32 s3, s7
; GFX1150-NEXT: s_cbranch_scc0 .LBB11_2
; GFX1150-NEXT: ; %bb.1: ; %frem.else16
@@ -12903,8 +13080,9 @@ define amdgpu_kernel void @frem_v2f32(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX1150-NEXT: s_cmp_eq_f32 s3, s7
; GFX1150-NEXT: v_mov_b32_e32 v0, s8
; GFX1150-NEXT: s_mov_b32 s7, 0
+; GFX1150-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1150-NEXT: s_cselect_b32 vcc_lo, -1, 0
-; GFX1150-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1150-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX1150-NEXT: v_cndmask_b32_e32 v0, s6, v0, vcc_lo
; GFX1150-NEXT: s_branch .LBB11_3
; GFX1150-NEXT: .LBB11_2:
@@ -12915,6 +13093,7 @@ define amdgpu_kernel void @frem_v2f32(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX1150-NEXT: s_and_b32 s7, s7, exec_lo
; GFX1150-NEXT: s_cselect_b32 s7, 1, 0
; GFX1150-NEXT: s_cmp_lg_u32 s7, 1
+; GFX1150-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1150-NEXT: s_cbranch_scc1 .LBB11_9
; GFX1150-NEXT: ; %bb.4: ; %frem.compute15
; GFX1150-NEXT: v_frexp_mant_f32_e64 v1, |s4|
@@ -12936,19 +13115,21 @@ define amdgpu_kernel void @frem_v2f32(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX1150-NEXT: v_add_nc_u32_e32 v4, v4, v3
; GFX1150-NEXT: v_div_scale_f32 v3, vcc_lo, 1.0, v1, 1.0
; GFX1150-NEXT: s_denorm_mode 15
-; GFX1150-NEXT: s_delay_alu instid0(TRANS32_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1150-NEXT: s_delay_alu instid0(TRANS32_DEP_1) | instid1(SALU_CYCLE_1)
; GFX1150-NEXT: v_fma_f32 v7, -v5, v6, 1.0
-; GFX1150-NEXT: v_fmac_f32_e32 v6, v7, v6
; GFX1150-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1150-NEXT: v_fmac_f32_e32 v6, v7, v6
; GFX1150-NEXT: v_mul_f32_e32 v7, v3, v6
-; GFX1150-NEXT: v_fma_f32 v8, -v5, v7, v3
; GFX1150-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1150-NEXT: v_fma_f32 v8, -v5, v7, v3
; GFX1150-NEXT: v_fmac_f32_e32 v7, v8, v6
+; GFX1150-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1150-NEXT: v_fma_f32 v3, -v5, v7, v3
; GFX1150-NEXT: s_denorm_mode 12
-; GFX1150-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
+; GFX1150-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX1150-NEXT: v_div_fmas_f32 v3, v3, v6, v7
; GFX1150-NEXT: v_cmp_gt_i32_e32 vcc_lo, 13, v4
+; GFX1150-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX1150-NEXT: v_div_fixup_f32 v3, v3, v1, 1.0
; GFX1150-NEXT: s_cbranch_vccnz .LBB11_8
; GFX1150-NEXT: ; %bb.5: ; %frem.loop_body23.preheader
@@ -12997,7 +13178,7 @@ define amdgpu_kernel void @frem_v2f32(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX1150-NEXT: .LBB11_9:
; GFX1150-NEXT: s_and_b32 s6, s5, 0x7fffffff
; GFX1150-NEXT: s_and_b32 s7, s2, 0x7fffffff
-; GFX1150-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1150-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_3)
; GFX1150-NEXT: s_cmp_ngt_f32 s6, s7
; GFX1150-NEXT: s_cbranch_scc0 .LBB11_11
; GFX1150-NEXT: ; %bb.10: ; %frem.else
@@ -13005,8 +13186,9 @@ define amdgpu_kernel void @frem_v2f32(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX1150-NEXT: s_cmp_eq_f32 s6, s7
; GFX1150-NEXT: v_mov_b32_e32 v1, s8
; GFX1150-NEXT: s_mov_b32 s7, 0
+; GFX1150-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1150-NEXT: s_cselect_b32 vcc_lo, -1, 0
-; GFX1150-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1150-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX1150-NEXT: v_cndmask_b32_e32 v1, s5, v1, vcc_lo
; GFX1150-NEXT: s_branch .LBB11_12
; GFX1150-NEXT: .LBB11_11:
@@ -13017,6 +13199,7 @@ define amdgpu_kernel void @frem_v2f32(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX1150-NEXT: s_and_b32 s7, s7, exec_lo
; GFX1150-NEXT: s_cselect_b32 s7, 1, 0
; GFX1150-NEXT: s_cmp_lg_u32 s7, 1
+; GFX1150-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1150-NEXT: s_cbranch_scc1 .LBB11_18
; GFX1150-NEXT: ; %bb.13: ; %frem.compute
; GFX1150-NEXT: v_frexp_mant_f32_e64 v2, |s2|
@@ -13038,19 +13221,21 @@ define amdgpu_kernel void @frem_v2f32(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX1150-NEXT: v_add_nc_u32_e32 v5, v5, v4
; GFX1150-NEXT: v_div_scale_f32 v4, vcc_lo, 1.0, v2, 1.0
; GFX1150-NEXT: s_denorm_mode 15
-; GFX1150-NEXT: s_delay_alu instid0(TRANS32_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1150-NEXT: s_delay_alu instid0(TRANS32_DEP_1) | instid1(SALU_CYCLE_1)
; GFX1150-NEXT: v_fma_f32 v8, -v6, v7, 1.0
-; GFX1150-NEXT: v_fmac_f32_e32 v7, v8, v7
; GFX1150-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1150-NEXT: v_fmac_f32_e32 v7, v8, v7
; GFX1150-NEXT: v_mul_f32_e32 v8, v4, v7
-; GFX1150-NEXT: v_fma_f32 v9, -v6, v8, v4
; GFX1150-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1150-NEXT: v_fma_f32 v9, -v6, v8, v4
; GFX1150-NEXT: v_fmac_f32_e32 v8, v9, v7
+; GFX1150-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1150-NEXT: v_fma_f32 v4, -v6, v8, v4
; GFX1150-NEXT: s_denorm_mode 12
-; GFX1150-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
+; GFX1150-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX1150-NEXT: v_div_fmas_f32 v4, v4, v7, v8
; GFX1150-NEXT: v_cmp_gt_i32_e32 vcc_lo, 13, v5
+; GFX1150-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX1150-NEXT: v_div_fixup_f32 v4, v4, v2, 1.0
; GFX1150-NEXT: s_cbranch_vccnz .LBB11_17
; GFX1150-NEXT: ; %bb.14: ; %frem.loop_body.preheader
@@ -13099,18 +13284,20 @@ define amdgpu_kernel void @frem_v2f32(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX1150-NEXT: .LBB11_18: ; %Flow50
; GFX1150-NEXT: s_cmp_lg_f32 s4, 0
; GFX1150-NEXT: v_mov_b32_e32 v2, 0
+; GFX1150-NEXT: s_delay_alu instid0(SALU_CYCLE_2) | instskip(SKIP_1) | instid1(SALU_CYCLE_3)
; GFX1150-NEXT: s_cselect_b32 s4, -1, 0
; GFX1150-NEXT: s_cmp_nge_f32 s3, 0x7f800000
; GFX1150-NEXT: s_cselect_b32 s3, -1, 0
-; GFX1150-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1150-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_2)
; GFX1150-NEXT: s_and_b32 vcc_lo, s3, s4
; GFX1150-NEXT: s_cmp_lg_f32 s2, 0
; GFX1150-NEXT: v_cndmask_b32_e32 v0, 0x7fc00000, v0, vcc_lo
; GFX1150-NEXT: s_cselect_b32 s2, -1, 0
; GFX1150-NEXT: s_cmp_nge_f32 s6, 0x7f800000
+; GFX1150-NEXT: s_delay_alu instid0(SALU_CYCLE_3) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1150-NEXT: s_cselect_b32 s3, -1, 0
-; GFX1150-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1150-NEXT: s_and_b32 vcc_lo, s3, s2
+; GFX1150-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1150-NEXT: v_cndmask_b32_e32 v1, 0x7fc00000, v1, vcc_lo
; GFX1150-NEXT: global_store_b64 v2, v[0:1], s[0:1]
; GFX1150-NEXT: s_endpgm
@@ -13132,12 +13319,13 @@ define amdgpu_kernel void @frem_v2f32(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX1200-NEXT: v_readfirstlane_b32 s4, v1
; GFX1200-NEXT: v_readfirstlane_b32 s2, v2
; GFX1200-NEXT: s_and_b32 s7, s4, 0x7fffffff
-; GFX1200-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1200-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_3)
; GFX1200-NEXT: s_cmp_ngt_f32 s3, s7
; GFX1200-NEXT: s_cbranch_scc0 .LBB11_2
; GFX1200-NEXT: ; %bb.1: ; %frem.else16
; GFX1200-NEXT: s_and_b32 s8, 0x80000000, s6
; GFX1200-NEXT: s_cmp_eq_f32 s3, s7
+; GFX1200-NEXT: s_delay_alu instid0(SALU_CYCLE_3)
; GFX1200-NEXT: s_cselect_b32 s7, s8, s6
; GFX1200-NEXT: s_mov_b32 s8, 0
; GFX1200-NEXT: s_branch .LBB11_3
@@ -13149,6 +13337,7 @@ define amdgpu_kernel void @frem_v2f32(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX1200-NEXT: s_and_b32 s8, s8, exec_lo
; GFX1200-NEXT: s_cselect_b32 s8, 1, 0
; GFX1200-NEXT: s_cmp_lg_u32 s8, 1
+; GFX1200-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1200-NEXT: s_cbranch_scc1 .LBB11_9
; GFX1200-NEXT: ; %bb.4: ; %frem.compute15
; GFX1200-NEXT: v_frexp_mant_f32_e64 v1, |s4|
@@ -13170,19 +13359,21 @@ define amdgpu_kernel void @frem_v2f32(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX1200-NEXT: v_add_nc_u32_e32 v4, v4, v3
; GFX1200-NEXT: v_div_scale_f32 v3, vcc_lo, 1.0, v1, 1.0
; GFX1200-NEXT: s_denorm_mode 15
-; GFX1200-NEXT: s_delay_alu instid0(TRANS32_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1200-NEXT: s_delay_alu instid0(TRANS32_DEP_1) | instid1(SALU_CYCLE_1)
; GFX1200-NEXT: v_fma_f32 v7, -v5, v6, 1.0
-; GFX1200-NEXT: v_fmac_f32_e32 v6, v7, v6
; GFX1200-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1200-NEXT: v_fmac_f32_e32 v6, v7, v6
; GFX1200-NEXT: v_mul_f32_e32 v7, v3, v6
-; GFX1200-NEXT: v_fma_f32 v8, -v5, v7, v3
; GFX1200-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1200-NEXT: v_fma_f32 v8, -v5, v7, v3
; GFX1200-NEXT: v_fmac_f32_e32 v7, v8, v6
+; GFX1200-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1200-NEXT: v_fma_f32 v3, -v5, v7, v3
; GFX1200-NEXT: s_denorm_mode 12
-; GFX1200-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
+; GFX1200-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX1200-NEXT: v_div_fmas_f32 v3, v3, v6, v7
; GFX1200-NEXT: v_cmp_gt_i32_e32 vcc_lo, 13, v4
+; GFX1200-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX1200-NEXT: v_div_fixup_f32 v3, v3, v1, 1.0
; GFX1200-NEXT: s_cbranch_vccnz .LBB11_8
; GFX1200-NEXT: ; %bb.5: ; %frem.loop_body23.preheader
@@ -13239,10 +13430,12 @@ define amdgpu_kernel void @frem_v2f32(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX1200-NEXT: s_and_b32 s7, s2, 0x7fffffff
; GFX1200-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX1200-NEXT: s_cmp_ngt_f32 s6, s7
+; GFX1200-NEXT: s_delay_alu instid0(SALU_CYCLE_3)
; GFX1200-NEXT: s_cbranch_scc0 .LBB11_12
; GFX1200-NEXT: ; %bb.11: ; %frem.else
; GFX1200-NEXT: s_and_b32 s8, 0x80000000, s5
; GFX1200-NEXT: s_cmp_eq_f32 s6, s7
+; GFX1200-NEXT: s_delay_alu instid0(SALU_CYCLE_3)
; GFX1200-NEXT: s_cselect_b32 s7, s8, s5
; GFX1200-NEXT: s_mov_b32 s8, 0
; GFX1200-NEXT: s_branch .LBB11_13
@@ -13254,6 +13447,7 @@ define amdgpu_kernel void @frem_v2f32(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX1200-NEXT: s_and_b32 s8, s8, exec_lo
; GFX1200-NEXT: s_cselect_b32 s8, 1, 0
; GFX1200-NEXT: s_cmp_lg_u32 s8, 1
+; GFX1200-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1200-NEXT: s_cbranch_scc1 .LBB11_19
; GFX1200-NEXT: ; %bb.14: ; %frem.compute
; GFX1200-NEXT: v_frexp_mant_f32_e64 v2, |s2|
@@ -13275,20 +13469,21 @@ define amdgpu_kernel void @frem_v2f32(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX1200-NEXT: v_add_nc_u32_e32 v5, v5, v4
; GFX1200-NEXT: v_div_scale_f32 v4, vcc_lo, 1.0, v2, 1.0
; GFX1200-NEXT: s_denorm_mode 15
-; GFX1200-NEXT: s_delay_alu instid0(TRANS32_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1200-NEXT: s_delay_alu instid0(TRANS32_DEP_1) | instid1(SALU_CYCLE_1)
; GFX1200-NEXT: v_fma_f32 v8, -v6, v7, 1.0
-; GFX1200-NEXT: v_fmac_f32_e32 v7, v8, v7
; GFX1200-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1200-NEXT: v_fmac_f32_e32 v7, v8, v7
; GFX1200-NEXT: v_mul_f32_e32 v8, v4, v7
-; GFX1200-NEXT: v_fma_f32 v9, -v6, v8, v4
; GFX1200-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1200-NEXT: v_fma_f32 v9, -v6, v8, v4
; GFX1200-NEXT: v_fmac_f32_e32 v8, v9, v7
+; GFX1200-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_2) | instid1(VALU_DEP_1)
; GFX1200-NEXT: v_fma_f32 v4, -v6, v8, v4
; GFX1200-NEXT: s_denorm_mode 12
; GFX1200-NEXT: s_wait_alu depctr_va_vcc(0)
-; GFX1200-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX1200-NEXT: v_div_fmas_f32 v4, v4, v7, v8
; GFX1200-NEXT: v_cmp_gt_i32_e32 vcc_lo, 13, v5
+; GFX1200-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX1200-NEXT: v_div_fixup_f32 v4, v4, v2, 1.0
; GFX1200-NEXT: s_cbranch_vccnz .LBB11_18
; GFX1200-NEXT: ; %bb.15: ; %frem.loop_body.preheader
@@ -13345,6 +13540,7 @@ define amdgpu_kernel void @frem_v2f32(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX1200-NEXT: .LBB11_20: ; %Flow50
; GFX1200-NEXT: s_cmp_lg_f32 s4, 0
; GFX1200-NEXT: v_mov_b32_e32 v2, 0
+; GFX1200-NEXT: s_delay_alu instid0(SALU_CYCLE_2) | instskip(SKIP_1) | instid1(SALU_CYCLE_3)
; GFX1200-NEXT: s_cselect_b32 s4, -1, 0
; GFX1200-NEXT: s_cmp_nge_f32 s3, 0x7f800000
; GFX1200-NEXT: s_cselect_b32 s3, -1, 0
@@ -13353,6 +13549,7 @@ define amdgpu_kernel void @frem_v2f32(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX1200-NEXT: s_cmp_lg_f32 s2, 0
; GFX1200-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX1200-NEXT: v_cndmask_b32_e32 v0, 0x7fc00000, v0, vcc_lo
+; GFX1200-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_3)
; GFX1200-NEXT: s_cselect_b32 s2, -1, 0
; GFX1200-NEXT: s_cmp_nge_f32 s6, 0x7f800000
; GFX1200-NEXT: s_cselect_b32 s3, -1, 0
@@ -15070,6 +15267,7 @@ define amdgpu_kernel void @frem_v4f32(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX11-NEXT: s_waitcnt vmcnt(0)
; GFX11-NEXT: v_cmp_ngt_f32_e64 s2, |v0|, |v4|
; GFX11-NEXT: s_and_b32 vcc_lo, exec_lo, s2
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_vccz .LBB12_2
; GFX11-NEXT: ; %bb.1: ; %frem.else78
; GFX11-NEXT: v_and_b32_e32 v8, 0x80000000, v0
@@ -15085,6 +15283,7 @@ define amdgpu_kernel void @frem_v4f32(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX11-NEXT: s_and_b32 s2, s2, exec_lo
; GFX11-NEXT: s_cselect_b32 s2, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s2, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB12_9
; GFX11-NEXT: ; %bb.4: ; %frem.compute77
; GFX11-NEXT: v_frexp_mant_f32_e64 v9, |v4|
@@ -15114,9 +15313,10 @@ define amdgpu_kernel void @frem_v4f32(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_fma_f32 v16, -v13, v15, v11
; GFX11-NEXT: v_fmac_f32_e32 v15, v16, v14
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_fma_f32 v11, -v13, v15, v11
; GFX11-NEXT: s_denorm_mode 12
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: v_div_fmas_f32 v11, v11, v14, v15
; GFX11-NEXT: v_cmp_gt_i32_e32 vcc_lo, 13, v12
; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2)
@@ -15165,6 +15365,7 @@ define amdgpu_kernel void @frem_v4f32(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX11-NEXT: .LBB12_9:
; GFX11-NEXT: v_cmp_ngt_f32_e64 s2, |v1|, |v5|
; GFX11-NEXT: s_and_b32 vcc_lo, exec_lo, s2
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_vccz .LBB12_11
; GFX11-NEXT: ; %bb.10: ; %frem.else47
; GFX11-NEXT: v_and_b32_e32 v9, 0x80000000, v1
@@ -15180,6 +15381,7 @@ define amdgpu_kernel void @frem_v4f32(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX11-NEXT: s_and_b32 s2, s2, exec_lo
; GFX11-NEXT: s_cselect_b32 s2, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s2, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB12_18
; GFX11-NEXT: ; %bb.13: ; %frem.compute46
; GFX11-NEXT: v_frexp_mant_f32_e64 v10, |v5|
@@ -15209,9 +15411,10 @@ define amdgpu_kernel void @frem_v4f32(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_fma_f32 v17, -v14, v16, v12
; GFX11-NEXT: v_fmac_f32_e32 v16, v17, v15
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_fma_f32 v12, -v14, v16, v12
; GFX11-NEXT: s_denorm_mode 12
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: v_div_fmas_f32 v12, v12, v15, v16
; GFX11-NEXT: v_cmp_gt_i32_e32 vcc_lo, 13, v13
; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2)
@@ -15260,6 +15463,7 @@ define amdgpu_kernel void @frem_v4f32(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX11-NEXT: .LBB12_18:
; GFX11-NEXT: v_cmp_ngt_f32_e64 s2, |v2|, |v6|
; GFX11-NEXT: s_and_b32 vcc_lo, exec_lo, s2
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_vccz .LBB12_20
; GFX11-NEXT: ; %bb.19: ; %frem.else16
; GFX11-NEXT: v_and_b32_e32 v10, 0x80000000, v2
@@ -15275,6 +15479,7 @@ define amdgpu_kernel void @frem_v4f32(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX11-NEXT: s_and_b32 s2, s2, exec_lo
; GFX11-NEXT: s_cselect_b32 s2, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s2, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB12_27
; GFX11-NEXT: ; %bb.22: ; %frem.compute15
; GFX11-NEXT: v_frexp_mant_f32_e64 v11, |v6|
@@ -15304,9 +15509,10 @@ define amdgpu_kernel void @frem_v4f32(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_fma_f32 v18, -v15, v17, v13
; GFX11-NEXT: v_fmac_f32_e32 v17, v18, v16
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_fma_f32 v13, -v15, v17, v13
; GFX11-NEXT: s_denorm_mode 12
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: v_div_fmas_f32 v13, v13, v16, v17
; GFX11-NEXT: v_cmp_gt_i32_e32 vcc_lo, 13, v14
; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2)
@@ -15355,6 +15561,7 @@ define amdgpu_kernel void @frem_v4f32(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX11-NEXT: .LBB12_27:
; GFX11-NEXT: v_cmp_ngt_f32_e64 s2, |v3|, |v7|
; GFX11-NEXT: s_and_b32 vcc_lo, exec_lo, s2
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_vccz .LBB12_29
; GFX11-NEXT: ; %bb.28: ; %frem.else
; GFX11-NEXT: v_and_b32_e32 v11, 0x80000000, v3
@@ -15370,6 +15577,7 @@ define amdgpu_kernel void @frem_v4f32(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX11-NEXT: s_and_b32 s2, s2, exec_lo
; GFX11-NEXT: s_cselect_b32 s2, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s2, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB12_36
; GFX11-NEXT: ; %bb.31: ; %frem.compute
; GFX11-NEXT: v_frexp_mant_f32_e64 v12, |v7|
@@ -15399,9 +15607,10 @@ define amdgpu_kernel void @frem_v4f32(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_fma_f32 v19, -v16, v18, v14
; GFX11-NEXT: v_fmac_f32_e32 v18, v19, v17
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_fma_f32 v14, -v16, v18, v14
; GFX11-NEXT: s_denorm_mode 12
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: v_div_fmas_f32 v14, v14, v17, v18
; GFX11-NEXT: v_cmp_gt_i32_e32 vcc_lo, 13, v15
; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2)
@@ -15464,6 +15673,7 @@ define amdgpu_kernel void @frem_v4f32(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX11-NEXT: v_cndmask_b32_e32 v2, 0x7fc00000, v10, vcc_lo
; GFX11-NEXT: v_cmp_lg_f32_e32 vcc_lo, 0, v7
; GFX11-NEXT: s_and_b32 vcc_lo, s2, vcc_lo
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: v_cndmask_b32_e32 v3, 0x7fc00000, v11, vcc_lo
; GFX11-NEXT: global_store_b128 v4, v[0:3], s[0:1]
; GFX11-NEXT: s_endpgm
@@ -15489,7 +15699,7 @@ define amdgpu_kernel void @frem_v4f32(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX1150-NEXT: v_readfirstlane_b32 s3, v3
; GFX1150-NEXT: v_readfirstlane_b32 s2, v4
; GFX1150-NEXT: s_and_b32 s11, s6, 0x7fffffff
-; GFX1150-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1150-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_3)
; GFX1150-NEXT: s_cmp_ngt_f32 s5, s11
; GFX1150-NEXT: s_cbranch_scc0 .LBB12_2
; GFX1150-NEXT: ; %bb.1: ; %frem.else78
@@ -15497,8 +15707,9 @@ define amdgpu_kernel void @frem_v4f32(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX1150-NEXT: s_cmp_eq_f32 s5, s11
; GFX1150-NEXT: v_mov_b32_e32 v0, s12
; GFX1150-NEXT: s_mov_b32 s11, 0
+; GFX1150-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1150-NEXT: s_cselect_b32 vcc_lo, -1, 0
-; GFX1150-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1150-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX1150-NEXT: v_cndmask_b32_e32 v0, s8, v0, vcc_lo
; GFX1150-NEXT: s_branch .LBB12_3
; GFX1150-NEXT: .LBB12_2:
@@ -15509,6 +15720,7 @@ define amdgpu_kernel void @frem_v4f32(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX1150-NEXT: s_and_b32 s11, s11, exec_lo
; GFX1150-NEXT: s_cselect_b32 s11, 1, 0
; GFX1150-NEXT: s_cmp_lg_u32 s11, 1
+; GFX1150-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1150-NEXT: s_cbranch_scc1 .LBB12_9
; GFX1150-NEXT: ; %bb.4: ; %frem.compute77
; GFX1150-NEXT: v_frexp_mant_f32_e64 v1, |s6|
@@ -15530,19 +15742,21 @@ define amdgpu_kernel void @frem_v4f32(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX1150-NEXT: v_add_nc_u32_e32 v4, v4, v3
; GFX1150-NEXT: v_div_scale_f32 v3, vcc_lo, 1.0, v1, 1.0
; GFX1150-NEXT: s_denorm_mode 15
-; GFX1150-NEXT: s_delay_alu instid0(TRANS32_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1150-NEXT: s_delay_alu instid0(TRANS32_DEP_1) | instid1(SALU_CYCLE_1)
; GFX1150-NEXT: v_fma_f32 v7, -v5, v6, 1.0
-; GFX1150-NEXT: v_fmac_f32_e32 v6, v7, v6
; GFX1150-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1150-NEXT: v_fmac_f32_e32 v6, v7, v6
; GFX1150-NEXT: v_mul_f32_e32 v7, v3, v6
-; GFX1150-NEXT: v_fma_f32 v8, -v5, v7, v3
; GFX1150-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1150-NEXT: v_fma_f32 v8, -v5, v7, v3
; GFX1150-NEXT: v_fmac_f32_e32 v7, v8, v6
+; GFX1150-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1150-NEXT: v_fma_f32 v3, -v5, v7, v3
; GFX1150-NEXT: s_denorm_mode 12
-; GFX1150-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
+; GFX1150-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX1150-NEXT: v_div_fmas_f32 v3, v3, v6, v7
; GFX1150-NEXT: v_cmp_gt_i32_e32 vcc_lo, 13, v4
+; GFX1150-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX1150-NEXT: v_div_fixup_f32 v3, v3, v1, 1.0
; GFX1150-NEXT: s_cbranch_vccnz .LBB12_8
; GFX1150-NEXT: ; %bb.5: ; %frem.loop_body85.preheader
@@ -15591,7 +15805,7 @@ define amdgpu_kernel void @frem_v4f32(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX1150-NEXT: .LBB12_9:
; GFX1150-NEXT: s_and_b32 s8, s10, 0x7fffffff
; GFX1150-NEXT: s_and_b32 s11, s4, 0x7fffffff
-; GFX1150-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1150-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_3)
; GFX1150-NEXT: s_cmp_ngt_f32 s8, s11
; GFX1150-NEXT: s_cbranch_scc0 .LBB12_11
; GFX1150-NEXT: ; %bb.10: ; %frem.else47
@@ -15599,8 +15813,9 @@ define amdgpu_kernel void @frem_v4f32(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX1150-NEXT: s_cmp_eq_f32 s8, s11
; GFX1150-NEXT: v_mov_b32_e32 v1, s12
; GFX1150-NEXT: s_mov_b32 s11, 0
+; GFX1150-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1150-NEXT: s_cselect_b32 vcc_lo, -1, 0
-; GFX1150-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1150-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX1150-NEXT: v_cndmask_b32_e32 v1, s10, v1, vcc_lo
; GFX1150-NEXT: s_branch .LBB12_12
; GFX1150-NEXT: .LBB12_11:
@@ -15611,6 +15826,7 @@ define amdgpu_kernel void @frem_v4f32(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX1150-NEXT: s_and_b32 s11, s11, exec_lo
; GFX1150-NEXT: s_cselect_b32 s11, 1, 0
; GFX1150-NEXT: s_cmp_lg_u32 s11, 1
+; GFX1150-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1150-NEXT: s_cbranch_scc1 .LBB12_18
; GFX1150-NEXT: ; %bb.13: ; %frem.compute46
; GFX1150-NEXT: v_frexp_mant_f32_e64 v2, |s4|
@@ -15632,19 +15848,21 @@ define amdgpu_kernel void @frem_v4f32(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX1150-NEXT: v_add_nc_u32_e32 v5, v5, v4
; GFX1150-NEXT: v_div_scale_f32 v4, vcc_lo, 1.0, v2, 1.0
; GFX1150-NEXT: s_denorm_mode 15
-; GFX1150-NEXT: s_delay_alu instid0(TRANS32_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1150-NEXT: s_delay_alu instid0(TRANS32_DEP_1) | instid1(SALU_CYCLE_1)
; GFX1150-NEXT: v_fma_f32 v8, -v6, v7, 1.0
-; GFX1150-NEXT: v_fmac_f32_e32 v7, v8, v7
; GFX1150-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1150-NEXT: v_fmac_f32_e32 v7, v8, v7
; GFX1150-NEXT: v_mul_f32_e32 v8, v4, v7
-; GFX1150-NEXT: v_fma_f32 v9, -v6, v8, v4
; GFX1150-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1150-NEXT: v_fma_f32 v9, -v6, v8, v4
; GFX1150-NEXT: v_fmac_f32_e32 v8, v9, v7
+; GFX1150-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1150-NEXT: v_fma_f32 v4, -v6, v8, v4
; GFX1150-NEXT: s_denorm_mode 12
-; GFX1150-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
+; GFX1150-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX1150-NEXT: v_div_fmas_f32 v4, v4, v7, v8
; GFX1150-NEXT: v_cmp_gt_i32_e32 vcc_lo, 13, v5
+; GFX1150-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX1150-NEXT: v_div_fixup_f32 v4, v4, v2, 1.0
; GFX1150-NEXT: s_cbranch_vccnz .LBB12_17
; GFX1150-NEXT: ; %bb.14: ; %frem.loop_body54.preheader
@@ -15693,7 +15911,7 @@ define amdgpu_kernel void @frem_v4f32(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX1150-NEXT: .LBB12_18:
; GFX1150-NEXT: s_and_b32 s10, s9, 0x7fffffff
; GFX1150-NEXT: s_and_b32 s11, s3, 0x7fffffff
-; GFX1150-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1150-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_3)
; GFX1150-NEXT: s_cmp_ngt_f32 s10, s11
; GFX1150-NEXT: s_cbranch_scc0 .LBB12_20
; GFX1150-NEXT: ; %bb.19: ; %frem.else16
@@ -15701,8 +15919,9 @@ define amdgpu_kernel void @frem_v4f32(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX1150-NEXT: s_cmp_eq_f32 s10, s11
; GFX1150-NEXT: v_mov_b32_e32 v2, s12
; GFX1150-NEXT: s_mov_b32 s11, 0
+; GFX1150-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1150-NEXT: s_cselect_b32 vcc_lo, -1, 0
-; GFX1150-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1150-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX1150-NEXT: v_cndmask_b32_e32 v2, s9, v2, vcc_lo
; GFX1150-NEXT: s_branch .LBB12_21
; GFX1150-NEXT: .LBB12_20:
@@ -15713,6 +15932,7 @@ define amdgpu_kernel void @frem_v4f32(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX1150-NEXT: s_and_b32 s11, s11, exec_lo
; GFX1150-NEXT: s_cselect_b32 s11, 1, 0
; GFX1150-NEXT: s_cmp_lg_u32 s11, 1
+; GFX1150-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1150-NEXT: s_cbranch_scc1 .LBB12_27
; GFX1150-NEXT: ; %bb.22: ; %frem.compute15
; GFX1150-NEXT: v_frexp_mant_f32_e64 v3, |s3|
@@ -15734,19 +15954,21 @@ define amdgpu_kernel void @frem_v4f32(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX1150-NEXT: v_add_nc_u32_e32 v6, v6, v5
; GFX1150-NEXT: v_div_scale_f32 v5, vcc_lo, 1.0, v3, 1.0
; GFX1150-NEXT: s_denorm_mode 15
-; GFX1150-NEXT: s_delay_alu instid0(TRANS32_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1150-NEXT: s_delay_alu instid0(TRANS32_DEP_1) | instid1(SALU_CYCLE_1)
; GFX1150-NEXT: v_fma_f32 v9, -v7, v8, 1.0
-; GFX1150-NEXT: v_fmac_f32_e32 v8, v9, v8
; GFX1150-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1150-NEXT: v_fmac_f32_e32 v8, v9, v8
; GFX1150-NEXT: v_mul_f32_e32 v9, v5, v8
-; GFX1150-NEXT: v_fma_f32 v10, -v7, v9, v5
; GFX1150-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1150-NEXT: v_fma_f32 v10, -v7, v9, v5
; GFX1150-NEXT: v_fmac_f32_e32 v9, v10, v8
+; GFX1150-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1150-NEXT: v_fma_f32 v5, -v7, v9, v5
; GFX1150-NEXT: s_denorm_mode 12
-; GFX1150-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
+; GFX1150-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX1150-NEXT: v_div_fmas_f32 v5, v5, v8, v9
; GFX1150-NEXT: v_cmp_gt_i32_e32 vcc_lo, 13, v6
+; GFX1150-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX1150-NEXT: v_div_fixup_f32 v5, v5, v3, 1.0
; GFX1150-NEXT: s_cbranch_vccnz .LBB12_26
; GFX1150-NEXT: ; %bb.23: ; %frem.loop_body23.preheader
@@ -15795,7 +16017,7 @@ define amdgpu_kernel void @frem_v4f32(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX1150-NEXT: .LBB12_27:
; GFX1150-NEXT: s_and_b32 s9, s7, 0x7fffffff
; GFX1150-NEXT: s_and_b32 s11, s2, 0x7fffffff
-; GFX1150-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1150-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_3)
; GFX1150-NEXT: s_cmp_ngt_f32 s9, s11
; GFX1150-NEXT: s_cbranch_scc0 .LBB12_29
; GFX1150-NEXT: ; %bb.28: ; %frem.else
@@ -15803,8 +16025,9 @@ define amdgpu_kernel void @frem_v4f32(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX1150-NEXT: s_cmp_eq_f32 s9, s11
; GFX1150-NEXT: v_mov_b32_e32 v3, s12
; GFX1150-NEXT: s_mov_b32 s11, 0
+; GFX1150-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1150-NEXT: s_cselect_b32 vcc_lo, -1, 0
-; GFX1150-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1150-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX1150-NEXT: v_cndmask_b32_e32 v3, s7, v3, vcc_lo
; GFX1150-NEXT: s_branch .LBB12_30
; GFX1150-NEXT: .LBB12_29:
@@ -15815,6 +16038,7 @@ define amdgpu_kernel void @frem_v4f32(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX1150-NEXT: s_and_b32 s11, s11, exec_lo
; GFX1150-NEXT: s_cselect_b32 s11, 1, 0
; GFX1150-NEXT: s_cmp_lg_u32 s11, 1
+; GFX1150-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1150-NEXT: s_cbranch_scc1 .LBB12_36
; GFX1150-NEXT: ; %bb.31: ; %frem.compute
; GFX1150-NEXT: v_frexp_mant_f32_e64 v4, |s2|
@@ -15836,19 +16060,21 @@ define amdgpu_kernel void @frem_v4f32(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX1150-NEXT: v_add_nc_u32_e32 v7, v7, v6
; GFX1150-NEXT: v_div_scale_f32 v6, vcc_lo, 1.0, v4, 1.0
; GFX1150-NEXT: s_denorm_mode 15
-; GFX1150-NEXT: s_delay_alu instid0(TRANS32_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1150-NEXT: s_delay_alu instid0(TRANS32_DEP_1) | instid1(SALU_CYCLE_1)
; GFX1150-NEXT: v_fma_f32 v10, -v8, v9, 1.0
-; GFX1150-NEXT: v_fmac_f32_e32 v9, v10, v9
; GFX1150-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1150-NEXT: v_fmac_f32_e32 v9, v10, v9
; GFX1150-NEXT: v_mul_f32_e32 v10, v6, v9
-; GFX1150-NEXT: v_fma_f32 v11, -v8, v10, v6
; GFX1150-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1150-NEXT: v_fma_f32 v11, -v8, v10, v6
; GFX1150-NEXT: v_fmac_f32_e32 v10, v11, v9
+; GFX1150-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1150-NEXT: v_fma_f32 v6, -v8, v10, v6
; GFX1150-NEXT: s_denorm_mode 12
-; GFX1150-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
+; GFX1150-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX1150-NEXT: v_div_fmas_f32 v6, v6, v9, v10
; GFX1150-NEXT: v_cmp_gt_i32_e32 vcc_lo, 13, v7
+; GFX1150-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX1150-NEXT: v_div_fixup_f32 v6, v6, v4, 1.0
; GFX1150-NEXT: s_cbranch_vccnz .LBB12_35
; GFX1150-NEXT: ; %bb.32: ; %frem.loop_body.preheader
@@ -15897,32 +16123,35 @@ define amdgpu_kernel void @frem_v4f32(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX1150-NEXT: .LBB12_36: ; %Flow116
; GFX1150-NEXT: s_cmp_lg_f32 s6, 0
; GFX1150-NEXT: v_mov_b32_e32 v4, 0
+; GFX1150-NEXT: s_delay_alu instid0(SALU_CYCLE_2) | instskip(SKIP_1) | instid1(SALU_CYCLE_3)
; GFX1150-NEXT: s_cselect_b32 s6, -1, 0
; GFX1150-NEXT: s_cmp_nge_f32 s5, 0x7f800000
; GFX1150-NEXT: s_cselect_b32 s5, -1, 0
-; GFX1150-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1150-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_2)
; GFX1150-NEXT: s_and_b32 vcc_lo, s5, s6
; GFX1150-NEXT: s_cmp_lg_f32 s4, 0
; GFX1150-NEXT: v_cndmask_b32_e32 v0, 0x7fc00000, v0, vcc_lo
; GFX1150-NEXT: s_cselect_b32 s4, -1, 0
; GFX1150-NEXT: s_cmp_nge_f32 s8, 0x7f800000
+; GFX1150-NEXT: s_delay_alu instid0(SALU_CYCLE_3) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1150-NEXT: s_cselect_b32 s5, -1, 0
-; GFX1150-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1150-NEXT: s_and_b32 vcc_lo, s5, s4
; GFX1150-NEXT: s_cmp_lg_f32 s3, 0
; GFX1150-NEXT: v_cndmask_b32_e32 v1, 0x7fc00000, v1, vcc_lo
+; GFX1150-NEXT: s_delay_alu instid0(SALU_CYCLE_2) | instskip(SKIP_1) | instid1(SALU_CYCLE_3)
; GFX1150-NEXT: s_cselect_b32 s3, -1, 0
; GFX1150-NEXT: s_cmp_nge_f32 s10, 0x7f800000
; GFX1150-NEXT: s_cselect_b32 s4, -1, 0
-; GFX1150-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1150-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_2)
; GFX1150-NEXT: s_and_b32 vcc_lo, s4, s3
; GFX1150-NEXT: s_cmp_lg_f32 s2, 0
; GFX1150-NEXT: v_cndmask_b32_e32 v2, 0x7fc00000, v2, vcc_lo
; GFX1150-NEXT: s_cselect_b32 s2, -1, 0
; GFX1150-NEXT: s_cmp_nge_f32 s9, 0x7f800000
+; GFX1150-NEXT: s_delay_alu instid0(SALU_CYCLE_3) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1150-NEXT: s_cselect_b32 s3, -1, 0
-; GFX1150-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1150-NEXT: s_and_b32 vcc_lo, s3, s2
+; GFX1150-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1150-NEXT: v_cndmask_b32_e32 v3, 0x7fc00000, v3, vcc_lo
; GFX1150-NEXT: global_store_b128 v4, v[0:3], s[0:1]
; GFX1150-NEXT: s_endpgm
@@ -15948,12 +16177,13 @@ define amdgpu_kernel void @frem_v4f32(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX1200-NEXT: v_readfirstlane_b32 s3, v3
; GFX1200-NEXT: v_readfirstlane_b32 s2, v4
; GFX1200-NEXT: s_and_b32 s11, s6, 0x7fffffff
-; GFX1200-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1200-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_3)
; GFX1200-NEXT: s_cmp_ngt_f32 s5, s11
; GFX1200-NEXT: s_cbranch_scc0 .LBB12_2
; GFX1200-NEXT: ; %bb.1: ; %frem.else78
; GFX1200-NEXT: s_and_b32 s12, 0x80000000, s8
; GFX1200-NEXT: s_cmp_eq_f32 s5, s11
+; GFX1200-NEXT: s_delay_alu instid0(SALU_CYCLE_3)
; GFX1200-NEXT: s_cselect_b32 s11, s12, s8
; GFX1200-NEXT: s_mov_b32 s12, 0
; GFX1200-NEXT: s_branch .LBB12_3
@@ -15965,6 +16195,7 @@ define amdgpu_kernel void @frem_v4f32(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX1200-NEXT: s_and_b32 s12, s12, exec_lo
; GFX1200-NEXT: s_cselect_b32 s12, 1, 0
; GFX1200-NEXT: s_cmp_lg_u32 s12, 1
+; GFX1200-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1200-NEXT: s_cbranch_scc1 .LBB12_9
; GFX1200-NEXT: ; %bb.4: ; %frem.compute77
; GFX1200-NEXT: v_frexp_mant_f32_e64 v1, |s6|
@@ -15986,19 +16217,21 @@ define amdgpu_kernel void @frem_v4f32(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX1200-NEXT: v_add_nc_u32_e32 v4, v4, v3
; GFX1200-NEXT: v_div_scale_f32 v3, vcc_lo, 1.0, v1, 1.0
; GFX1200-NEXT: s_denorm_mode 15
-; GFX1200-NEXT: s_delay_alu instid0(TRANS32_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1200-NEXT: s_delay_alu instid0(TRANS32_DEP_1) | instid1(SALU_CYCLE_1)
; GFX1200-NEXT: v_fma_f32 v7, -v5, v6, 1.0
-; GFX1200-NEXT: v_fmac_f32_e32 v6, v7, v6
; GFX1200-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1200-NEXT: v_fmac_f32_e32 v6, v7, v6
; GFX1200-NEXT: v_mul_f32_e32 v7, v3, v6
-; GFX1200-NEXT: v_fma_f32 v8, -v5, v7, v3
; GFX1200-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1200-NEXT: v_fma_f32 v8, -v5, v7, v3
; GFX1200-NEXT: v_fmac_f32_e32 v7, v8, v6
+; GFX1200-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1200-NEXT: v_fma_f32 v3, -v5, v7, v3
; GFX1200-NEXT: s_denorm_mode 12
-; GFX1200-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
+; GFX1200-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX1200-NEXT: v_div_fmas_f32 v3, v3, v6, v7
; GFX1200-NEXT: v_cmp_gt_i32_e32 vcc_lo, 13, v4
+; GFX1200-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX1200-NEXT: v_div_fixup_f32 v3, v3, v1, 1.0
; GFX1200-NEXT: s_cbranch_vccnz .LBB12_8
; GFX1200-NEXT: ; %bb.5: ; %frem.loop_body85.preheader
@@ -16054,10 +16287,12 @@ define amdgpu_kernel void @frem_v4f32(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX1200-NEXT: s_and_b32 s11, s4, 0x7fffffff
; GFX1200-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX1200-NEXT: s_cmp_ngt_f32 s8, s11
+; GFX1200-NEXT: s_delay_alu instid0(SALU_CYCLE_3)
; GFX1200-NEXT: s_cbranch_scc0 .LBB12_12
; GFX1200-NEXT: ; %bb.11: ; %frem.else47
; GFX1200-NEXT: s_and_b32 s12, 0x80000000, s10
; GFX1200-NEXT: s_cmp_eq_f32 s8, s11
+; GFX1200-NEXT: s_delay_alu instid0(SALU_CYCLE_3)
; GFX1200-NEXT: s_cselect_b32 s11, s12, s10
; GFX1200-NEXT: s_mov_b32 s12, 0
; GFX1200-NEXT: s_branch .LBB12_13
@@ -16069,6 +16304,7 @@ define amdgpu_kernel void @frem_v4f32(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX1200-NEXT: s_and_b32 s12, s12, exec_lo
; GFX1200-NEXT: s_cselect_b32 s12, 1, 0
; GFX1200-NEXT: s_cmp_lg_u32 s12, 1
+; GFX1200-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1200-NEXT: s_cbranch_scc1 .LBB12_19
; GFX1200-NEXT: ; %bb.14: ; %frem.compute46
; GFX1200-NEXT: v_frexp_mant_f32_e64 v2, |s4|
@@ -16090,20 +16326,21 @@ define amdgpu_kernel void @frem_v4f32(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX1200-NEXT: v_add_nc_u32_e32 v5, v5, v4
; GFX1200-NEXT: v_div_scale_f32 v4, vcc_lo, 1.0, v2, 1.0
; GFX1200-NEXT: s_denorm_mode 15
-; GFX1200-NEXT: s_delay_alu instid0(TRANS32_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1200-NEXT: s_delay_alu instid0(TRANS32_DEP_1) | instid1(SALU_CYCLE_1)
; GFX1200-NEXT: v_fma_f32 v8, -v6, v7, 1.0
-; GFX1200-NEXT: v_fmac_f32_e32 v7, v8, v7
; GFX1200-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1200-NEXT: v_fmac_f32_e32 v7, v8, v7
; GFX1200-NEXT: v_mul_f32_e32 v8, v4, v7
-; GFX1200-NEXT: v_fma_f32 v9, -v6, v8, v4
; GFX1200-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1200-NEXT: v_fma_f32 v9, -v6, v8, v4
; GFX1200-NEXT: v_fmac_f32_e32 v8, v9, v7
+; GFX1200-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_2) | instid1(VALU_DEP_1)
; GFX1200-NEXT: v_fma_f32 v4, -v6, v8, v4
; GFX1200-NEXT: s_denorm_mode 12
; GFX1200-NEXT: s_wait_alu depctr_va_vcc(0)
-; GFX1200-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX1200-NEXT: v_div_fmas_f32 v4, v4, v7, v8
; GFX1200-NEXT: v_cmp_gt_i32_e32 vcc_lo, 13, v5
+; GFX1200-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX1200-NEXT: v_div_fixup_f32 v4, v4, v2, 1.0
; GFX1200-NEXT: s_cbranch_vccnz .LBB12_18
; GFX1200-NEXT: ; %bb.15: ; %frem.loop_body54.preheader
@@ -16162,10 +16399,12 @@ define amdgpu_kernel void @frem_v4f32(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX1200-NEXT: s_and_b32 s11, s3, 0x7fffffff
; GFX1200-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX1200-NEXT: s_cmp_ngt_f32 s10, s11
+; GFX1200-NEXT: s_delay_alu instid0(SALU_CYCLE_3)
; GFX1200-NEXT: s_cbranch_scc0 .LBB12_22
; GFX1200-NEXT: ; %bb.21: ; %frem.else16
; GFX1200-NEXT: s_and_b32 s12, 0x80000000, s9
; GFX1200-NEXT: s_cmp_eq_f32 s10, s11
+; GFX1200-NEXT: s_delay_alu instid0(SALU_CYCLE_3)
; GFX1200-NEXT: s_cselect_b32 s11, s12, s9
; GFX1200-NEXT: s_mov_b32 s12, 0
; GFX1200-NEXT: s_branch .LBB12_23
@@ -16177,6 +16416,7 @@ define amdgpu_kernel void @frem_v4f32(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX1200-NEXT: s_and_b32 s12, s12, exec_lo
; GFX1200-NEXT: s_cselect_b32 s12, 1, 0
; GFX1200-NEXT: s_cmp_lg_u32 s12, 1
+; GFX1200-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1200-NEXT: s_cbranch_scc1 .LBB12_29
; GFX1200-NEXT: ; %bb.24: ; %frem.compute15
; GFX1200-NEXT: v_frexp_mant_f32_e64 v3, |s3|
@@ -16198,20 +16438,21 @@ define amdgpu_kernel void @frem_v4f32(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX1200-NEXT: v_add_nc_u32_e32 v6, v6, v5
; GFX1200-NEXT: v_div_scale_f32 v5, vcc_lo, 1.0, v3, 1.0
; GFX1200-NEXT: s_denorm_mode 15
-; GFX1200-NEXT: s_delay_alu instid0(TRANS32_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1200-NEXT: s_delay_alu instid0(TRANS32_DEP_1) | instid1(SALU_CYCLE_1)
; GFX1200-NEXT: v_fma_f32 v9, -v7, v8, 1.0
-; GFX1200-NEXT: v_fmac_f32_e32 v8, v9, v8
; GFX1200-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1200-NEXT: v_fmac_f32_e32 v8, v9, v8
; GFX1200-NEXT: v_mul_f32_e32 v9, v5, v8
-; GFX1200-NEXT: v_fma_f32 v10, -v7, v9, v5
; GFX1200-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1200-NEXT: v_fma_f32 v10, -v7, v9, v5
; GFX1200-NEXT: v_fmac_f32_e32 v9, v10, v8
+; GFX1200-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_2) | instid1(VALU_DEP_1)
; GFX1200-NEXT: v_fma_f32 v5, -v7, v9, v5
; GFX1200-NEXT: s_denorm_mode 12
; GFX1200-NEXT: s_wait_alu depctr_va_vcc(0)
-; GFX1200-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX1200-NEXT: v_div_fmas_f32 v5, v5, v8, v9
; GFX1200-NEXT: v_cmp_gt_i32_e32 vcc_lo, 13, v6
+; GFX1200-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX1200-NEXT: v_div_fixup_f32 v5, v5, v3, 1.0
; GFX1200-NEXT: s_cbranch_vccnz .LBB12_28
; GFX1200-NEXT: ; %bb.25: ; %frem.loop_body23.preheader
@@ -16270,10 +16511,12 @@ define amdgpu_kernel void @frem_v4f32(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX1200-NEXT: s_and_b32 s11, s2, 0x7fffffff
; GFX1200-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX1200-NEXT: s_cmp_ngt_f32 s9, s11
+; GFX1200-NEXT: s_delay_alu instid0(SALU_CYCLE_3)
; GFX1200-NEXT: s_cbranch_scc0 .LBB12_32
; GFX1200-NEXT: ; %bb.31: ; %frem.else
; GFX1200-NEXT: s_and_b32 s12, 0x80000000, s7
; GFX1200-NEXT: s_cmp_eq_f32 s9, s11
+; GFX1200-NEXT: s_delay_alu instid0(SALU_CYCLE_3)
; GFX1200-NEXT: s_cselect_b32 s11, s12, s7
; GFX1200-NEXT: s_mov_b32 s12, 0
; GFX1200-NEXT: s_branch .LBB12_33
@@ -16285,6 +16528,7 @@ define amdgpu_kernel void @frem_v4f32(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX1200-NEXT: s_and_b32 s12, s12, exec_lo
; GFX1200-NEXT: s_cselect_b32 s12, 1, 0
; GFX1200-NEXT: s_cmp_lg_u32 s12, 1
+; GFX1200-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1200-NEXT: s_cbranch_scc1 .LBB12_39
; GFX1200-NEXT: ; %bb.34: ; %frem.compute
; GFX1200-NEXT: v_frexp_mant_f32_e64 v4, |s2|
@@ -16306,20 +16550,21 @@ define amdgpu_kernel void @frem_v4f32(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX1200-NEXT: v_add_nc_u32_e32 v7, v7, v6
; GFX1200-NEXT: v_div_scale_f32 v6, vcc_lo, 1.0, v4, 1.0
; GFX1200-NEXT: s_denorm_mode 15
-; GFX1200-NEXT: s_delay_alu instid0(TRANS32_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1200-NEXT: s_delay_alu instid0(TRANS32_DEP_1) | instid1(SALU_CYCLE_1)
; GFX1200-NEXT: v_fma_f32 v10, -v8, v9, 1.0
-; GFX1200-NEXT: v_fmac_f32_e32 v9, v10, v9
; GFX1200-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1200-NEXT: v_fmac_f32_e32 v9, v10, v9
; GFX1200-NEXT: v_mul_f32_e32 v10, v6, v9
-; GFX1200-NEXT: v_fma_f32 v11, -v8, v10, v6
; GFX1200-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1200-NEXT: v_fma_f32 v11, -v8, v10, v6
; GFX1200-NEXT: v_fmac_f32_e32 v10, v11, v9
+; GFX1200-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_2) | instid1(VALU_DEP_1)
; GFX1200-NEXT: v_fma_f32 v6, -v8, v10, v6
; GFX1200-NEXT: s_denorm_mode 12
; GFX1200-NEXT: s_wait_alu depctr_va_vcc(0)
-; GFX1200-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX1200-NEXT: v_div_fmas_f32 v6, v6, v9, v10
; GFX1200-NEXT: v_cmp_gt_i32_e32 vcc_lo, 13, v7
+; GFX1200-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX1200-NEXT: v_div_fixup_f32 v6, v6, v4, 1.0
; GFX1200-NEXT: s_cbranch_vccnz .LBB12_38
; GFX1200-NEXT: ; %bb.35: ; %frem.loop_body.preheader
@@ -16376,6 +16621,7 @@ define amdgpu_kernel void @frem_v4f32(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX1200-NEXT: .LBB12_40: ; %Flow116
; GFX1200-NEXT: s_cmp_lg_f32 s6, 0
; GFX1200-NEXT: v_mov_b32_e32 v4, 0
+; GFX1200-NEXT: s_delay_alu instid0(SALU_CYCLE_2) | instskip(SKIP_1) | instid1(SALU_CYCLE_3)
; GFX1200-NEXT: s_cselect_b32 s6, -1, 0
; GFX1200-NEXT: s_cmp_nge_f32 s5, 0x7f800000
; GFX1200-NEXT: s_cselect_b32 s5, -1, 0
@@ -16384,6 +16630,7 @@ define amdgpu_kernel void @frem_v4f32(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX1200-NEXT: s_cmp_lg_f32 s4, 0
; GFX1200-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX1200-NEXT: v_cndmask_b32_e32 v0, 0x7fc00000, v0, vcc_lo
+; GFX1200-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_3)
; GFX1200-NEXT: s_cselect_b32 s4, -1, 0
; GFX1200-NEXT: s_cmp_nge_f32 s8, 0x7f800000
; GFX1200-NEXT: s_cselect_b32 s5, -1, 0
@@ -16392,6 +16639,7 @@ define amdgpu_kernel void @frem_v4f32(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX1200-NEXT: s_cmp_lg_f32 s3, 0
; GFX1200-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX1200-NEXT: v_cndmask_b32_e32 v1, 0x7fc00000, v1, vcc_lo
+; GFX1200-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_3)
; GFX1200-NEXT: s_cselect_b32 s3, -1, 0
; GFX1200-NEXT: s_cmp_nge_f32 s10, 0x7f800000
; GFX1200-NEXT: s_cselect_b32 s4, -1, 0
@@ -16400,6 +16648,7 @@ define amdgpu_kernel void @frem_v4f32(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX1200-NEXT: s_cmp_lg_f32 s2, 0
; GFX1200-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX1200-NEXT: v_cndmask_b32_e32 v2, 0x7fc00000, v2, vcc_lo
+; GFX1200-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_3)
; GFX1200-NEXT: s_cselect_b32 s2, -1, 0
; GFX1200-NEXT: s_cmp_nge_f32 s9, 0x7f800000
; GFX1200-NEXT: s_cselect_b32 s3, -1, 0
@@ -17460,6 +17709,7 @@ define amdgpu_kernel void @frem_v2f64(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX11-NEXT: s_waitcnt vmcnt(0)
; GFX11-NEXT: v_cmp_ngt_f64_e64 s2, |v[0:1]|, |v[4:5]|
; GFX11-NEXT: s_and_b32 vcc_lo, exec_lo, s2
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_vccz .LBB13_2
; GFX11-NEXT: ; %bb.1: ; %frem.else16
; GFX11-NEXT: v_cmp_eq_f64_e64 vcc_lo, |v[0:1]|, |v[4:5]|
@@ -17477,6 +17727,7 @@ define amdgpu_kernel void @frem_v2f64(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX11-NEXT: s_and_b32 s2, s2, exec_lo
; GFX11-NEXT: s_cselect_b32 s2, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s2, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB13_9
; GFX11-NEXT: ; %bb.4: ; %frem.compute15
; GFX11-NEXT: v_frexp_mant_f64_e64 v[8:9], |v[0:1]|
@@ -17555,6 +17806,7 @@ define amdgpu_kernel void @frem_v2f64(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX11-NEXT: .LBB13_9:
; GFX11-NEXT: v_cmp_ngt_f64_e64 s2, |v[2:3]|, |v[6:7]|
; GFX11-NEXT: s_and_b32 vcc_lo, exec_lo, s2
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_vccz .LBB13_11
; GFX11-NEXT: ; %bb.10: ; %frem.else
; GFX11-NEXT: v_cmp_eq_f64_e64 vcc_lo, |v[2:3]|, |v[6:7]|
@@ -17572,6 +17824,7 @@ define amdgpu_kernel void @frem_v2f64(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX11-NEXT: s_and_b32 s2, s2, exec_lo
; GFX11-NEXT: s_cselect_b32 s2, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s2, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB13_18
; GFX11-NEXT: ; %bb.13: ; %frem.compute
; GFX11-NEXT: v_frexp_mant_f64_e64 v[10:11], |v[2:3]|
@@ -17655,6 +17908,7 @@ define amdgpu_kernel void @frem_v2f64(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX11-NEXT: v_dual_cndmask_b32 v1, 0x7ff80000, v9 :: v_dual_cndmask_b32 v0, 0, v8
; GFX11-NEXT: v_cmp_lg_f64_e32 vcc_lo, 0, v[6:7]
; GFX11-NEXT: s_and_b32 vcc_lo, s2, vcc_lo
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: v_dual_mov_b32 v4, 0 :: v_dual_cndmask_b32 v3, 0x7ff80000, v11
; GFX11-NEXT: v_cndmask_b32_e32 v2, 0, v10, vcc_lo
; GFX11-NEXT: global_store_b128 v4, v[0:3], s[0:1]
@@ -17673,6 +17927,7 @@ define amdgpu_kernel void @frem_v2f64(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX1150-NEXT: s_waitcnt vmcnt(0)
; GFX1150-NEXT: v_cmp_ngt_f64_e64 s2, |v[0:1]|, |v[4:5]|
; GFX1150-NEXT: s_and_b32 vcc_lo, exec_lo, s2
+; GFX1150-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1150-NEXT: s_cbranch_vccz .LBB13_2
; GFX1150-NEXT: ; %bb.1: ; %frem.else16
; GFX1150-NEXT: v_cmp_eq_f64_e64 vcc_lo, |v[0:1]|, |v[4:5]|
@@ -17690,6 +17945,7 @@ define amdgpu_kernel void @frem_v2f64(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX1150-NEXT: s_and_b32 s2, s2, exec_lo
; GFX1150-NEXT: s_cselect_b32 s2, 1, 0
; GFX1150-NEXT: s_cmp_lg_u32 s2, 1
+; GFX1150-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1150-NEXT: s_cbranch_scc1 .LBB13_9
; GFX1150-NEXT: ; %bb.4: ; %frem.compute15
; GFX1150-NEXT: v_frexp_mant_f64_e64 v[8:9], |v[0:1]|
@@ -17767,6 +18023,7 @@ define amdgpu_kernel void @frem_v2f64(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX1150-NEXT: .LBB13_9:
; GFX1150-NEXT: v_cmp_ngt_f64_e64 s2, |v[2:3]|, |v[6:7]|
; GFX1150-NEXT: s_and_b32 vcc_lo, exec_lo, s2
+; GFX1150-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1150-NEXT: s_cbranch_vccz .LBB13_11
; GFX1150-NEXT: ; %bb.10: ; %frem.else
; GFX1150-NEXT: v_cmp_eq_f64_e64 vcc_lo, |v[2:3]|, |v[6:7]|
@@ -17784,6 +18041,7 @@ define amdgpu_kernel void @frem_v2f64(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX1150-NEXT: s_and_b32 s2, s2, exec_lo
; GFX1150-NEXT: s_cselect_b32 s2, 1, 0
; GFX1150-NEXT: s_cmp_lg_u32 s2, 1
+; GFX1150-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1150-NEXT: s_cbranch_scc1 .LBB13_18
; GFX1150-NEXT: ; %bb.13: ; %frem.compute
; GFX1150-NEXT: v_frexp_mant_f64_e64 v[10:11], |v[2:3]|
@@ -17866,6 +18124,7 @@ define amdgpu_kernel void @frem_v2f64(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX1150-NEXT: v_dual_cndmask_b32 v1, 0x7ff80000, v9 :: v_dual_cndmask_b32 v0, 0, v8
; GFX1150-NEXT: v_cmp_lg_f64_e32 vcc_lo, 0, v[6:7]
; GFX1150-NEXT: s_and_b32 vcc_lo, s2, vcc_lo
+; GFX1150-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1150-NEXT: v_dual_mov_b32 v4, 0 :: v_dual_cndmask_b32 v3, 0x7ff80000, v11
; GFX1150-NEXT: v_cndmask_b32_e32 v2, 0, v10, vcc_lo
; GFX1150-NEXT: global_store_b128 v4, v[0:3], s[0:1]
@@ -17884,6 +18143,7 @@ define amdgpu_kernel void @frem_v2f64(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX1200-NEXT: s_wait_loadcnt 0x0
; GFX1200-NEXT: v_cmp_ngt_f64_e64 s2, |v[0:1]|, |v[4:5]|
; GFX1200-NEXT: s_and_b32 vcc_lo, exec_lo, s2
+; GFX1200-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1200-NEXT: s_cbranch_vccz .LBB13_2
; GFX1200-NEXT: ; %bb.1: ; %frem.else16
; GFX1200-NEXT: v_cmp_eq_f64_e64 vcc_lo, |v[0:1]|, |v[4:5]|
@@ -17901,6 +18161,7 @@ define amdgpu_kernel void @frem_v2f64(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX1200-NEXT: s_and_b32 s2, s2, exec_lo
; GFX1200-NEXT: s_cselect_b32 s2, 1, 0
; GFX1200-NEXT: s_cmp_lg_u32 s2, 1
+; GFX1200-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1200-NEXT: s_cbranch_scc1 .LBB13_9
; GFX1200-NEXT: ; %bb.4: ; %frem.compute15
; GFX1200-NEXT: v_frexp_mant_f64_e64 v[8:9], |v[0:1]|
@@ -18000,6 +18261,7 @@ define amdgpu_kernel void @frem_v2f64(ptr addrspace(1) %out, ptr addrspace(1) %i
; GFX1200-NEXT: s_cselect_b32 s2, 1, 0
; GFX1200-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX1200-NEXT: s_cmp_lg_u32 s2, 1
+; GFX1200-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1200-NEXT: s_cbranch_scc1 .LBB13_18
; GFX1200-NEXT: ; %bb.13: ; %frem.compute
; GFX1200-NEXT: v_frexp_mant_f64_e64 v[10:11], |v[2:3]|
@@ -19031,6 +19293,7 @@ define amdgpu_kernel void @frem_v2f64_const_one_denum(ptr addrspace(1) %out, ptr
; GFX11-NEXT: s_waitcnt vmcnt(0)
; GFX11-NEXT: v_cmp_ngt_f64_e64 s2, |v[0:1]|, 1.0
; GFX11-NEXT: s_and_b32 vcc_lo, exec_lo, s2
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_vccz .LBB15_2
; GFX11-NEXT: ; %bb.1: ; %frem.else16
; GFX11-NEXT: v_cmp_eq_f64_e64 vcc_lo, |v[0:1]|, 1.0
@@ -19048,6 +19311,7 @@ define amdgpu_kernel void @frem_v2f64_const_one_denum(ptr addrspace(1) %out, ptr
; GFX11-NEXT: s_and_b32 s2, s2, exec_lo
; GFX11-NEXT: s_cselect_b32 s2, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s2, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB15_9
; GFX11-NEXT: ; %bb.4: ; %frem.compute15
; GFX11-NEXT: v_frexp_mant_f64_e64 v[4:5], |v[0:1]|
@@ -19096,6 +19360,7 @@ define amdgpu_kernel void @frem_v2f64_const_one_denum(ptr addrspace(1) %out, ptr
; GFX11-NEXT: .LBB15_9:
; GFX11-NEXT: v_cmp_ngt_f64_e64 s2, |v[2:3]|, 1.0
; GFX11-NEXT: s_and_b32 vcc_lo, exec_lo, s2
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_vccz .LBB15_11
; GFX11-NEXT: ; %bb.10: ; %frem.else
; GFX11-NEXT: v_cmp_eq_f64_e64 vcc_lo, |v[2:3]|, 1.0
@@ -19113,6 +19378,7 @@ define amdgpu_kernel void @frem_v2f64_const_one_denum(ptr addrspace(1) %out, ptr
; GFX11-NEXT: s_and_b32 s2, s2, exec_lo
; GFX11-NEXT: s_cselect_b32 s2, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s2, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB15_18
; GFX11-NEXT: ; %bb.13: ; %frem.compute
; GFX11-NEXT: v_frexp_mant_f64_e64 v[6:7], |v[2:3]|
@@ -19177,6 +19443,7 @@ define amdgpu_kernel void @frem_v2f64_const_one_denum(ptr addrspace(1) %out, ptr
; GFX1150-NEXT: s_waitcnt vmcnt(0)
; GFX1150-NEXT: v_cmp_ngt_f64_e64 s2, |v[0:1]|, 1.0
; GFX1150-NEXT: s_and_b32 vcc_lo, exec_lo, s2
+; GFX1150-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1150-NEXT: s_cbranch_vccz .LBB15_2
; GFX1150-NEXT: ; %bb.1: ; %frem.else16
; GFX1150-NEXT: v_cmp_eq_f64_e64 vcc_lo, |v[0:1]|, 1.0
@@ -19194,6 +19461,7 @@ define amdgpu_kernel void @frem_v2f64_const_one_denum(ptr addrspace(1) %out, ptr
; GFX1150-NEXT: s_and_b32 s2, s2, exec_lo
; GFX1150-NEXT: s_cselect_b32 s2, 1, 0
; GFX1150-NEXT: s_cmp_lg_u32 s2, 1
+; GFX1150-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1150-NEXT: s_cbranch_scc1 .LBB15_9
; GFX1150-NEXT: ; %bb.4: ; %frem.compute15
; GFX1150-NEXT: v_frexp_mant_f64_e64 v[4:5], |v[0:1]|
@@ -19242,6 +19510,7 @@ define amdgpu_kernel void @frem_v2f64_const_one_denum(ptr addrspace(1) %out, ptr
; GFX1150-NEXT: .LBB15_9:
; GFX1150-NEXT: v_cmp_ngt_f64_e64 s2, |v[2:3]|, 1.0
; GFX1150-NEXT: s_and_b32 vcc_lo, exec_lo, s2
+; GFX1150-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1150-NEXT: s_cbranch_vccz .LBB15_11
; GFX1150-NEXT: ; %bb.10: ; %frem.else
; GFX1150-NEXT: v_cmp_eq_f64_e64 vcc_lo, |v[2:3]|, 1.0
@@ -19259,6 +19528,7 @@ define amdgpu_kernel void @frem_v2f64_const_one_denum(ptr addrspace(1) %out, ptr
; GFX1150-NEXT: s_and_b32 s2, s2, exec_lo
; GFX1150-NEXT: s_cselect_b32 s2, 1, 0
; GFX1150-NEXT: s_cmp_lg_u32 s2, 1
+; GFX1150-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1150-NEXT: s_cbranch_scc1 .LBB15_18
; GFX1150-NEXT: ; %bb.13: ; %frem.compute
; GFX1150-NEXT: v_frexp_mant_f64_e64 v[6:7], |v[2:3]|
@@ -19323,6 +19593,7 @@ define amdgpu_kernel void @frem_v2f64_const_one_denum(ptr addrspace(1) %out, ptr
; GFX1200-NEXT: s_wait_loadcnt 0x0
; GFX1200-NEXT: v_cmp_ngt_f64_e64 s2, |v[0:1]|, 1.0
; GFX1200-NEXT: s_and_b32 vcc_lo, exec_lo, s2
+; GFX1200-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1200-NEXT: s_cbranch_vccz .LBB15_2
; GFX1200-NEXT: ; %bb.1: ; %frem.else16
; GFX1200-NEXT: v_cmp_eq_f64_e64 vcc_lo, |v[0:1]|, 1.0
@@ -19340,6 +19611,7 @@ define amdgpu_kernel void @frem_v2f64_const_one_denum(ptr addrspace(1) %out, ptr
; GFX1200-NEXT: s_and_b32 s2, s2, exec_lo
; GFX1200-NEXT: s_cselect_b32 s2, 1, 0
; GFX1200-NEXT: s_cmp_lg_u32 s2, 1
+; GFX1200-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1200-NEXT: s_cbranch_scc1 .LBB15_9
; GFX1200-NEXT: ; %bb.4: ; %frem.compute15
; GFX1200-NEXT: v_frexp_mant_f64_e64 v[4:5], |v[0:1]|
@@ -19410,6 +19682,7 @@ define amdgpu_kernel void @frem_v2f64_const_one_denum(ptr addrspace(1) %out, ptr
; GFX1200-NEXT: s_cselect_b32 s2, 1, 0
; GFX1200-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX1200-NEXT: s_cmp_lg_u32 s2, 1
+; GFX1200-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1200-NEXT: s_cbranch_scc1 .LBB15_18
; GFX1200-NEXT: ; %bb.13: ; %frem.compute
; GFX1200-NEXT: v_frexp_mant_f64_e64 v[6:7], |v[2:3]|
diff --git a/llvm/test/CodeGen/AMDGPU/frexp-inf-nan-combine.ll b/llvm/test/CodeGen/AMDGPU/frexp-inf-nan-combine.ll
index 30775bd463ac61..0e199f1597b66f 100644
--- a/llvm/test/CodeGen/AMDGPU/frexp-inf-nan-combine.ll
+++ b/llvm/test/CodeGen/AMDGPU/frexp-inf-nan-combine.ll
@@ -195,7 +195,7 @@ define i32 @frexp_not_inf_clamp_exp_f32(float %x) {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: v_frexp_exp_i32_f32_e32 v1, v0
; GFX11-NEXT: v_cmp_class_f32_e64 vcc_lo, v0, 0x1f8
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_cndmask_b32_e32 v0, 0, v1, vcc_lo
; GFX11-NEXT: s_setpc_b64 s[30:31]
;
@@ -253,7 +253,7 @@ define i32 @frexp_inf_or_nan_clamp_exp_f32(float %x) {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: v_frexp_exp_i32_f32_e32 v1, v0
; GFX11-NEXT: v_cmp_class_f32_e64 vcc_lo, v0, 0x1f8
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_cndmask_b32_e32 v0, 0, v1, vcc_lo
; GFX11-NEXT: s_setpc_b64 s[30:31]
;
@@ -613,7 +613,7 @@ define <2 x i16> @frexp_nan_clamp_exp_v2f16(<2 x half> %x) {
; GFX11-NEXT: v_frexp_exp_i16_f16_e32 v1.l, v2.l
; GFX11-NEXT: v_cmp_o_f16_e64 s0, v2.l, v2.l
; GFX11-NEXT: v_cndmask_b16 v0.l, 0, v0.h, vcc_lo
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX11-NEXT: v_cndmask_b16 v0.h, 0, v1.l, s0
; GFX11-NEXT: s_setpc_b64 s[30:31]
;
@@ -700,7 +700,7 @@ define <2 x i16> @frexp_nan_clamp_exp_v2bf16(<2 x bfloat> %x) {
; GFX11-NEXT: v_frexp_exp_i32_f32_e32 v3, v0
; GFX11-NEXT: v_cmp_o_f32_e32 vcc_lo, v0, v0
; GFX11-NEXT: v_cmp_o_f32_e64 s0, v1, v1
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX11-NEXT: v_cndmask_b16 v0.h, 0, v3.l, vcc_lo
; GFX11-NEXT: v_cndmask_b16 v0.l, 0, v2.l, s0
; GFX11-NEXT: s_setpc_b64 s[30:31]
diff --git a/llvm/test/CodeGen/AMDGPU/function-esm2-prologue-epilogue.ll b/llvm/test/CodeGen/AMDGPU/function-esm2-prologue-epilogue.ll
index 87b60b8d8df1cf..33eaec74318349 100644
--- a/llvm/test/CodeGen/AMDGPU/function-esm2-prologue-epilogue.ll
+++ b/llvm/test/CodeGen/AMDGPU/function-esm2-prologue-epilogue.ll
@@ -15,7 +15,7 @@ define float @missing_truncate_promote_bswap(i32 %arg) {
; GFX12-NEXT: s_wait_kmcnt 0x0
; GFX12-NEXT: v_perm_b32 v0, 0, v0, 0xc0c0001
; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1)
-; GFX12-NEXT: v_cvt_f32_f16_e32 v0, v0
+; GFX12-NEXT: v_cvt_f32_f16_e32 v0, v0.l
; GFX12-NEXT: s_setpc_b64 s[30:31]
;
; GFX12-ESM-LABEL: missing_truncate_promote_bswap:
@@ -28,7 +28,7 @@ define float @missing_truncate_promote_bswap(i32 %arg) {
; GFX12-ESM-NEXT: s_wait_kmcnt 0x0
; GFX12-ESM-NEXT: v_perm_b32 v0, 0, v0, 0xc0c0001
; GFX12-ESM-NEXT: s_delay_alu instid0(VALU_DEP_1)
-; GFX12-ESM-NEXT: v_cvt_f32_f16_e32 v0, v0
+; GFX12-ESM-NEXT: v_cvt_f32_f16_e32 v0, v0.l
; GFX12-ESM-NEXT: s_wait_alu depctr_va_vdst(0)
; GFX12-ESM-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_WAVE_SCHED_MODE, 0, 2), 0
; GFX12-ESM-NEXT: s_setpc_b64 s[30:31]
diff --git a/llvm/test/CodeGen/AMDGPU/gfx12_scalar_subword_loads.ll b/llvm/test/CodeGen/AMDGPU/gfx12_scalar_subword_loads.ll
index 56ae6abcefe779..5fbab1f3ef2491 100644
--- a/llvm/test/CodeGen/AMDGPU/gfx12_scalar_subword_loads.ll
+++ b/llvm/test/CodeGen/AMDGPU/gfx12_scalar_subword_loads.ll
@@ -272,7 +272,7 @@ define amdgpu_ps void @test_s_load_i16_divergent(ptr addrspace(4) inreg %in, i32
; DAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; DAG-NEXT: v_lshlrev_b64_e32 v[3:4], 1, v[3:4]
; DAG-NEXT: v_add_co_u32 v3, vcc_lo, s0, v3
-; DAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; DAG-NEXT: s_delay_alu instid0(VALU_DEP_2)
; DAG-NEXT: v_add_co_ci_u32_e64 v4, null, s1, v4, vcc_lo
; DAG-NEXT: global_load_i16 v0, v[3:4], off offset:32
; DAG-NEXT: s_wait_loadcnt 0x0
@@ -287,7 +287,7 @@ define amdgpu_ps void @test_s_load_i16_divergent(ptr addrspace(4) inreg %in, i32
; GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GISEL-NEXT: v_lshlrev_b64_e32 v[0:1], 1, v[0:1]
; GISEL-NEXT: v_add_co_u32 v0, vcc_lo, v5, v0
-; GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, v6, v1, vcc_lo
; GISEL-NEXT: global_load_i16 v0, v[0:1], off offset:32
; GISEL-NEXT: s_wait_loadcnt 0x0
@@ -388,7 +388,7 @@ define amdgpu_ps void @test_s_load_u16_divergent(ptr addrspace(4) inreg %in, i32
; DAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; DAG-NEXT: v_lshlrev_b64_e32 v[3:4], 1, v[3:4]
; DAG-NEXT: v_add_co_u32 v3, vcc_lo, s0, v3
-; DAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; DAG-NEXT: s_delay_alu instid0(VALU_DEP_2)
; DAG-NEXT: v_add_co_ci_u32_e64 v4, null, s1, v4, vcc_lo
; DAG-NEXT: global_load_u16 v0, v[3:4], off offset:32
; DAG-NEXT: s_wait_loadcnt 0x0
@@ -403,7 +403,7 @@ define amdgpu_ps void @test_s_load_u16_divergent(ptr addrspace(4) inreg %in, i32
; GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GISEL-NEXT: v_lshlrev_b64_e32 v[0:1], 1, v[0:1]
; GISEL-NEXT: v_add_co_u32 v0, vcc_lo, v5, v0
-; GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, v6, v1, vcc_lo
; GISEL-NEXT: global_load_u16 v0, v[0:1], off offset:32
; GISEL-NEXT: s_wait_loadcnt 0x0
diff --git a/llvm/test/CodeGen/AMDGPU/global-atomicrmw-fadd.ll b/llvm/test/CodeGen/AMDGPU/global-atomicrmw-fadd.ll
index c1089c70368be9..3898a7d308bd01 100644
--- a/llvm/test/CodeGen/AMDGPU/global-atomicrmw-fadd.ll
+++ b/llvm/test/CodeGen/AMDGPU/global-atomicrmw-fadd.ll
@@ -1673,11 +1673,12 @@ define float @global_agent_atomic_fadd_ret_f32_maybe_remote(ptr addrspace(1) %pt
; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB8_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: v_mov_b32_e32 v0, v3
; GFX11-NEXT: s_setpc_b64 s[30:31]
;
@@ -1889,11 +1890,12 @@ define float @global_agent_atomic_fadd_ret_f32_maybe_remote__amdgpu_ignore_denor
; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB9_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: v_mov_b32_e32 v0, v3
; GFX11-NEXT: s_setpc_b64 s[30:31]
;
@@ -2104,7 +2106,7 @@ define void @global_agent_atomic_fadd_noret_f32_maybe_remote__amdgpu_ignore_deno
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-NEXT: v_mov_b32_e32 v4, v3
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB10_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -2700,11 +2702,12 @@ define float @global_agent_atomic_fadd_ret_f32_amdgpu_ignore_denormal_mode(ptr a
; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB13_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: v_mov_b32_e32 v0, v3
; GFX11-NEXT: s_setpc_b64 s[30:31]
;
@@ -2915,7 +2918,7 @@ define void @global_agent_atomic_fadd_noret_f32_maybe_remote(ptr addrspace(1) %p
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-NEXT: v_mov_b32_e32 v4, v3
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB14_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -3488,7 +3491,7 @@ define void @global_agent_atomic_fadd_noret_f32_amdgpu_ignore_denormal_mode(ptr
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-NEXT: v_mov_b32_e32 v4, v3
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB17_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -3698,11 +3701,12 @@ define float @global_agent_atomic_fadd_ret_f32__amdgpu_no_remote_memory(ptr addr
; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB18_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: v_mov_b32_e32 v0, v3
; GFX11-NEXT: s_setpc_b64 s[30:31]
;
@@ -3911,7 +3915,7 @@ define void @global_agent_atomic_fadd_noret_f32__amdgpu_no_remote_memory(ptr add
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-NEXT: v_mov_b32_e32 v4, v3
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB19_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -4118,11 +4122,12 @@ define float @global_agent_atomic_fadd_ret_f32__amdgpu_no_remote_memory__amdgpu_
; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB20_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: v_mov_b32_e32 v0, v3
; GFX11-NEXT: s_setpc_b64 s[30:31]
;
@@ -4331,7 +4336,7 @@ define void @global_agent_atomic_fadd_noret_f32__amdgpu_no_remote_memory__amdgpu
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-NEXT: v_mov_b32_e32 v4, v3
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB21_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -6728,11 +6733,12 @@ define float @global_agent_atomic_fadd_ret_f32__ftz__amdgpu_no_remote_memory(ptr
; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB34_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: v_mov_b32_e32 v0, v3
; GFX11-NEXT: s_setpc_b64 s[30:31]
;
@@ -6941,7 +6947,7 @@ define void @global_agent_atomic_fadd_noret_f32__ftz__amdgpu_no_remote_memory(pt
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-NEXT: v_mov_b32_e32 v4, v3
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB35_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -7485,9 +7491,11 @@ define double @global_agent_atomic_fadd_ret_f64__amdgpu_no_fine_grained_memory(p
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB38_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: v_dual_mov_b32 v0, v4 :: v_dual_mov_b32 v1, v5
; GFX12-NEXT: s_setpc_b64 s[30:31]
;
@@ -7519,11 +7527,12 @@ define double @global_agent_atomic_fadd_ret_f64__amdgpu_no_fine_grained_memory(p
; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: v_cmp_eq_u64_e32 vcc_lo, v[4:5], v[6:7]
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB38_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: v_dual_mov_b32 v0, v4 :: v_dual_mov_b32 v1, v5
; GFX11-NEXT: s_setpc_b64 s[30:31]
;
@@ -7720,9 +7729,11 @@ define double @global_agent_atomic_fadd_ret_f64__offset12b_pos__amdgpu_no_fine_g
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB39_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: v_dual_mov_b32 v0, v4 :: v_dual_mov_b32 v1, v5
; GFX12-NEXT: s_setpc_b64 s[30:31]
;
@@ -7754,11 +7765,12 @@ define double @global_agent_atomic_fadd_ret_f64__offset12b_pos__amdgpu_no_fine_g
; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: v_cmp_eq_u64_e32 vcc_lo, v[4:5], v[6:7]
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB39_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: v_dual_mov_b32 v0, v4 :: v_dual_mov_b32 v1, v5
; GFX11-NEXT: s_setpc_b64 s[30:31]
;
@@ -7956,9 +7968,11 @@ define double @global_agent_atomic_fadd_ret_f64__offset12b_neg__amdgpu_no_fine_g
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB40_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: v_dual_mov_b32 v0, v4 :: v_dual_mov_b32 v1, v5
; GFX12-NEXT: s_setpc_b64 s[30:31]
;
@@ -7990,11 +8004,12 @@ define double @global_agent_atomic_fadd_ret_f64__offset12b_neg__amdgpu_no_fine_g
; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: v_cmp_eq_u64_e32 vcc_lo, v[4:5], v[6:7]
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB40_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: v_dual_mov_b32 v0, v4 :: v_dual_mov_b32 v1, v5
; GFX11-NEXT: s_setpc_b64 s[30:31]
;
@@ -8195,6 +8210,7 @@ define void @global_agent_atomic_fadd_noret_f64__amdgpu_no_fine_grained_memory(p
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB41_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -8227,7 +8243,7 @@ define void @global_agent_atomic_fadd_noret_f64__amdgpu_no_fine_grained_memory(p
; GFX11-NEXT: v_cmp_eq_u64_e32 vcc_lo, v[4:5], v[6:7]
; GFX11-NEXT: v_dual_mov_b32 v7, v5 :: v_dual_mov_b32 v6, v4
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB41_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -8412,6 +8428,7 @@ define void @global_agent_atomic_fadd_noret_f64__offset12b_pos__amdgpu_no_fine_g
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB42_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -8444,7 +8461,7 @@ define void @global_agent_atomic_fadd_noret_f64__offset12b_pos__amdgpu_no_fine_g
; GFX11-NEXT: v_cmp_eq_u64_e32 vcc_lo, v[4:5], v[6:7]
; GFX11-NEXT: v_dual_mov_b32 v7, v5 :: v_dual_mov_b32 v6, v4
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB42_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -8632,6 +8649,7 @@ define void @global_agent_atomic_fadd_noret_f64__offset12b_neg__amdgpu_no_fine_g
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB43_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -8664,7 +8682,7 @@ define void @global_agent_atomic_fadd_noret_f64__offset12b_neg__amdgpu_no_fine_g
; GFX11-NEXT: v_cmp_eq_u64_e32 vcc_lo, v[4:5], v[6:7]
; GFX11-NEXT: v_dual_mov_b32 v7, v5 :: v_dual_mov_b32 v6, v4
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB43_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -8865,11 +8883,12 @@ define half @global_agent_atomic_fadd_ret_f16__amdgpu_no_fine_grained_memory(ptr
; GFX1250-TRUE16-NEXT: s_wait_loadcnt 0x0
; GFX1250-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v5, v7
; GFX1250-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX1250-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1250-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1250-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX1250-TRUE16-NEXT: s_cbranch_execnz .LBB44_1
; GFX1250-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX1250-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX1250-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1250-TRUE16-NEXT: v_lshrrev_b32_e32 v0, v3, v5
; GFX1250-TRUE16-NEXT: s_set_pc_i64 s[30:31]
;
@@ -8909,11 +8928,12 @@ define half @global_agent_atomic_fadd_ret_f16__amdgpu_no_fine_grained_memory(ptr
; GFX1250-FAKE16-NEXT: s_wait_loadcnt 0x0
; GFX1250-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v5, v7
; GFX1250-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX1250-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1250-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1250-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX1250-FAKE16-NEXT: s_cbranch_execnz .LBB44_1
; GFX1250-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX1250-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX1250-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1250-FAKE16-NEXT: v_lshrrev_b32_e32 v0, v3, v5
; GFX1250-FAKE16-NEXT: s_set_pc_i64 s[30:31]
;
@@ -8955,9 +8975,11 @@ define half @global_agent_atomic_fadd_ret_f16__amdgpu_no_fine_grained_memory(ptr
; GFX12-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_cbranch_execnz .LBB44_1
; GFX12-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: v_lshrrev_b32_e32 v0, v3, v5
; GFX12-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -8999,9 +9021,11 @@ define half @global_agent_atomic_fadd_ret_f16__amdgpu_no_fine_grained_memory(ptr
; GFX12-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: s_cbranch_execnz .LBB44_1
; GFX12-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: v_lshrrev_b32_e32 v0, v3, v5
; GFX12-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -9071,11 +9095,12 @@ define half @global_agent_atomic_fadd_ret_f16__amdgpu_no_fine_grained_memory(ptr
; GFX11-TRUE16-NEXT: buffer_gl0_inv
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v5, v6
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB44_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v0, v3, v5
; GFX11-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -9111,11 +9136,12 @@ define half @global_agent_atomic_fadd_ret_f16__amdgpu_no_fine_grained_memory(ptr
; GFX11-FAKE16-NEXT: buffer_gl0_inv
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v5, v6
; GFX11-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB44_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v0, v3, v5
; GFX11-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -9373,11 +9399,12 @@ define half @global_agent_atomic_fadd_ret_f16__offset12b_pos__amdgpu_no_fine_gra
; GFX1250-TRUE16-NEXT: s_wait_loadcnt 0x0
; GFX1250-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v5, v7
; GFX1250-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX1250-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1250-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1250-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX1250-TRUE16-NEXT: s_cbranch_execnz .LBB45_1
; GFX1250-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX1250-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX1250-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1250-TRUE16-NEXT: v_lshrrev_b32_e32 v0, v3, v5
; GFX1250-TRUE16-NEXT: s_set_pc_i64 s[30:31]
;
@@ -9417,11 +9444,12 @@ define half @global_agent_atomic_fadd_ret_f16__offset12b_pos__amdgpu_no_fine_gra
; GFX1250-FAKE16-NEXT: s_wait_loadcnt 0x0
; GFX1250-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v5, v7
; GFX1250-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX1250-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1250-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1250-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX1250-FAKE16-NEXT: s_cbranch_execnz .LBB45_1
; GFX1250-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX1250-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX1250-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1250-FAKE16-NEXT: v_lshrrev_b32_e32 v0, v3, v5
; GFX1250-FAKE16-NEXT: s_set_pc_i64 s[30:31]
;
@@ -9464,9 +9492,11 @@ define half @global_agent_atomic_fadd_ret_f16__offset12b_pos__amdgpu_no_fine_gra
; GFX12-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_cbranch_execnz .LBB45_1
; GFX12-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: v_lshrrev_b32_e32 v0, v3, v5
; GFX12-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -9509,9 +9539,11 @@ define half @global_agent_atomic_fadd_ret_f16__offset12b_pos__amdgpu_no_fine_gra
; GFX12-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: s_cbranch_execnz .LBB45_1
; GFX12-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: v_lshrrev_b32_e32 v0, v3, v5
; GFX12-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -9555,7 +9587,6 @@ define half @global_agent_atomic_fadd_ret_f16__offset12b_pos__amdgpu_no_fine_gra
; GFX11-TRUE16: ; %bb.0:
; GFX11-TRUE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-TRUE16-NEXT: v_add_co_u32 v3, vcc_lo, 0x7fe, v0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-TRUE16-NEXT: s_mov_b32 s0, 0
; GFX11-TRUE16-NEXT: v_and_b32_e32 v0, -4, v3
@@ -9584,11 +9615,12 @@ define half @global_agent_atomic_fadd_ret_f16__offset12b_pos__amdgpu_no_fine_gra
; GFX11-TRUE16-NEXT: buffer_gl0_inv
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v5, v6
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB45_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v0, v3, v5
; GFX11-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -9596,7 +9628,6 @@ define half @global_agent_atomic_fadd_ret_f16__offset12b_pos__amdgpu_no_fine_gra
; GFX11-FAKE16: ; %bb.0:
; GFX11-FAKE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-FAKE16-NEXT: v_add_co_u32 v3, vcc_lo, 0x7fe, v0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-FAKE16-NEXT: s_mov_b32 s0, 0
; GFX11-FAKE16-NEXT: v_and_b32_e32 v0, -4, v3
@@ -9625,11 +9656,12 @@ define half @global_agent_atomic_fadd_ret_f16__offset12b_pos__amdgpu_no_fine_gra
; GFX11-FAKE16-NEXT: buffer_gl0_inv
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v5, v6
; GFX11-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB45_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v0, v3, v5
; GFX11-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -9895,11 +9927,12 @@ define half @global_agent_atomic_fadd_ret_f16__offset12b_neg__amdgpu_no_fine_gra
; GFX1250-TRUE16-NEXT: s_wait_loadcnt 0x0
; GFX1250-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v5, v7
; GFX1250-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX1250-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1250-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1250-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX1250-TRUE16-NEXT: s_cbranch_execnz .LBB46_1
; GFX1250-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX1250-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX1250-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1250-TRUE16-NEXT: v_lshrrev_b32_e32 v0, v3, v5
; GFX1250-TRUE16-NEXT: s_set_pc_i64 s[30:31]
;
@@ -9940,11 +9973,12 @@ define half @global_agent_atomic_fadd_ret_f16__offset12b_neg__amdgpu_no_fine_gra
; GFX1250-FAKE16-NEXT: s_wait_loadcnt 0x0
; GFX1250-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v5, v7
; GFX1250-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX1250-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1250-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1250-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX1250-FAKE16-NEXT: s_cbranch_execnz .LBB46_1
; GFX1250-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX1250-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX1250-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1250-FAKE16-NEXT: v_lshrrev_b32_e32 v0, v3, v5
; GFX1250-FAKE16-NEXT: s_set_pc_i64 s[30:31]
;
@@ -9987,9 +10021,11 @@ define half @global_agent_atomic_fadd_ret_f16__offset12b_neg__amdgpu_no_fine_gra
; GFX12-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_cbranch_execnz .LBB46_1
; GFX12-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: v_lshrrev_b32_e32 v0, v3, v5
; GFX12-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -10032,9 +10068,11 @@ define half @global_agent_atomic_fadd_ret_f16__offset12b_neg__amdgpu_no_fine_gra
; GFX12-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: s_cbranch_execnz .LBB46_1
; GFX12-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: v_lshrrev_b32_e32 v0, v3, v5
; GFX12-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -10079,7 +10117,6 @@ define half @global_agent_atomic_fadd_ret_f16__offset12b_neg__amdgpu_no_fine_gra
; GFX11-TRUE16: ; %bb.0:
; GFX11-TRUE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-TRUE16-NEXT: v_add_co_u32 v3, vcc_lo, 0xfffff800, v0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v1, null, -1, v1, vcc_lo
; GFX11-TRUE16-NEXT: s_mov_b32 s0, 0
; GFX11-TRUE16-NEXT: v_and_b32_e32 v0, -4, v3
@@ -10108,11 +10145,12 @@ define half @global_agent_atomic_fadd_ret_f16__offset12b_neg__amdgpu_no_fine_gra
; GFX11-TRUE16-NEXT: buffer_gl0_inv
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v5, v6
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB46_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v0, v3, v5
; GFX11-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -10120,7 +10158,6 @@ define half @global_agent_atomic_fadd_ret_f16__offset12b_neg__amdgpu_no_fine_gra
; GFX11-FAKE16: ; %bb.0:
; GFX11-FAKE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-FAKE16-NEXT: v_add_co_u32 v3, vcc_lo, 0xfffff800, v0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v1, null, -1, v1, vcc_lo
; GFX11-FAKE16-NEXT: s_mov_b32 s0, 0
; GFX11-FAKE16-NEXT: v_and_b32_e32 v0, -4, v3
@@ -10149,11 +10186,12 @@ define half @global_agent_atomic_fadd_ret_f16__offset12b_neg__amdgpu_no_fine_gra
; GFX11-FAKE16-NEXT: buffer_gl0_inv
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v5, v6
; GFX11-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB46_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v0, v3, v5
; GFX11-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -10417,7 +10455,7 @@ define void @global_agent_atomic_fadd_noret_f16__amdgpu_no_fine_grained_memory(p
; GFX1250-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v4, v5
; GFX1250-TRUE16-NEXT: v_mov_b32_e32 v5, v4
; GFX1250-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX1250-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1250-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1250-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX1250-TRUE16-NEXT: s_cbranch_execnz .LBB47_1
; GFX1250-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -10459,7 +10497,7 @@ define void @global_agent_atomic_fadd_noret_f16__amdgpu_no_fine_grained_memory(p
; GFX1250-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v4, v5
; GFX1250-FAKE16-NEXT: v_mov_b32_e32 v5, v4
; GFX1250-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX1250-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1250-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1250-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX1250-FAKE16-NEXT: s_cbranch_execnz .LBB47_1
; GFX1250-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -10503,6 +10541,7 @@ define void @global_agent_atomic_fadd_noret_f16__amdgpu_no_fine_grained_memory(p
; GFX12-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_cbranch_execnz .LBB47_1
; GFX12-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -10545,6 +10584,7 @@ define void @global_agent_atomic_fadd_noret_f16__amdgpu_no_fine_grained_memory(p
; GFX12-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: s_cbranch_execnz .LBB47_1
; GFX12-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -10614,7 +10654,7 @@ define void @global_agent_atomic_fadd_noret_f16__amdgpu_no_fine_grained_memory(p
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-TRUE16-NEXT: v_mov_b32_e32 v4, v3
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB47_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -10652,7 +10692,7 @@ define void @global_agent_atomic_fadd_noret_f16__amdgpu_no_fine_grained_memory(p
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v4, v3
; GFX11-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB47_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -10906,7 +10946,7 @@ define void @global_agent_atomic_fadd_noret_f16__offset12b_pos__amdgpu_no_fine_g
; GFX1250-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v4, v5
; GFX1250-TRUE16-NEXT: v_mov_b32_e32 v5, v4
; GFX1250-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX1250-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1250-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1250-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX1250-TRUE16-NEXT: s_cbranch_execnz .LBB48_1
; GFX1250-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -10948,7 +10988,7 @@ define void @global_agent_atomic_fadd_noret_f16__offset12b_pos__amdgpu_no_fine_g
; GFX1250-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v4, v5
; GFX1250-FAKE16-NEXT: v_mov_b32_e32 v5, v4
; GFX1250-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX1250-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1250-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1250-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX1250-FAKE16-NEXT: s_cbranch_execnz .LBB48_1
; GFX1250-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -10993,6 +11033,7 @@ define void @global_agent_atomic_fadd_noret_f16__offset12b_pos__amdgpu_no_fine_g
; GFX12-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_cbranch_execnz .LBB48_1
; GFX12-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -11036,6 +11077,7 @@ define void @global_agent_atomic_fadd_noret_f16__offset12b_pos__amdgpu_no_fine_g
; GFX12-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: s_cbranch_execnz .LBB48_1
; GFX12-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -11080,7 +11122,6 @@ define void @global_agent_atomic_fadd_noret_f16__offset12b_pos__amdgpu_no_fine_g
; GFX11-TRUE16: ; %bb.0:
; GFX11-TRUE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-TRUE16-NEXT: v_add_co_u32 v3, vcc_lo, 0x7fe, v0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-TRUE16-NEXT: s_mov_b32 s0, 0
; GFX11-TRUE16-NEXT: v_and_b32_e32 v0, -4, v3
@@ -11108,7 +11149,7 @@ define void @global_agent_atomic_fadd_noret_f16__offset12b_pos__amdgpu_no_fine_g
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-TRUE16-NEXT: v_mov_b32_e32 v4, v3
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB48_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -11119,7 +11160,6 @@ define void @global_agent_atomic_fadd_noret_f16__offset12b_pos__amdgpu_no_fine_g
; GFX11-FAKE16: ; %bb.0:
; GFX11-FAKE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-FAKE16-NEXT: v_add_co_u32 v3, vcc_lo, 0x7fe, v0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-FAKE16-NEXT: s_mov_b32 s0, 0
; GFX11-FAKE16-NEXT: v_and_b32_e32 v0, -4, v3
@@ -11147,7 +11187,7 @@ define void @global_agent_atomic_fadd_noret_f16__offset12b_pos__amdgpu_no_fine_g
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v4, v3
; GFX11-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB48_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -11409,7 +11449,7 @@ define void @global_agent_atomic_fadd_noret_f16__offset12b_neg__amdgpu_no_fine_g
; GFX1250-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v4, v5
; GFX1250-TRUE16-NEXT: v_mov_b32_e32 v5, v4
; GFX1250-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX1250-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1250-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1250-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX1250-TRUE16-NEXT: s_cbranch_execnz .LBB49_1
; GFX1250-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -11452,7 +11492,7 @@ define void @global_agent_atomic_fadd_noret_f16__offset12b_neg__amdgpu_no_fine_g
; GFX1250-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v4, v5
; GFX1250-FAKE16-NEXT: v_mov_b32_e32 v5, v4
; GFX1250-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX1250-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1250-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1250-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX1250-FAKE16-NEXT: s_cbranch_execnz .LBB49_1
; GFX1250-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -11497,6 +11537,7 @@ define void @global_agent_atomic_fadd_noret_f16__offset12b_neg__amdgpu_no_fine_g
; GFX12-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_cbranch_execnz .LBB49_1
; GFX12-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -11540,6 +11581,7 @@ define void @global_agent_atomic_fadd_noret_f16__offset12b_neg__amdgpu_no_fine_g
; GFX12-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: s_cbranch_execnz .LBB49_1
; GFX12-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -11585,7 +11627,6 @@ define void @global_agent_atomic_fadd_noret_f16__offset12b_neg__amdgpu_no_fine_g
; GFX11-TRUE16: ; %bb.0:
; GFX11-TRUE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-TRUE16-NEXT: v_add_co_u32 v3, vcc_lo, 0xfffff800, v0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v1, null, -1, v1, vcc_lo
; GFX11-TRUE16-NEXT: s_mov_b32 s0, 0
; GFX11-TRUE16-NEXT: v_and_b32_e32 v0, -4, v3
@@ -11613,7 +11654,7 @@ define void @global_agent_atomic_fadd_noret_f16__offset12b_neg__amdgpu_no_fine_g
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-TRUE16-NEXT: v_mov_b32_e32 v4, v3
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB49_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -11624,7 +11665,6 @@ define void @global_agent_atomic_fadd_noret_f16__offset12b_neg__amdgpu_no_fine_g
; GFX11-FAKE16: ; %bb.0:
; GFX11-FAKE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-FAKE16-NEXT: v_add_co_u32 v3, vcc_lo, 0xfffff800, v0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v1, null, -1, v1, vcc_lo
; GFX11-FAKE16-NEXT: s_mov_b32 s0, 0
; GFX11-FAKE16-NEXT: v_and_b32_e32 v0, -4, v3
@@ -11652,7 +11692,7 @@ define void @global_agent_atomic_fadd_noret_f16__offset12b_neg__amdgpu_no_fine_g
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v4, v3
; GFX11-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB49_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -11903,11 +11943,12 @@ define half @global_agent_atomic_fadd_ret_f16__offset12b_pos__align4__amdgpu_no_
; GFX1250-TRUE16-NEXT: s_wait_loadcnt 0x0
; GFX1250-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v5
; GFX1250-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX1250-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1250-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1250-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX1250-TRUE16-NEXT: s_cbranch_execnz .LBB50_1
; GFX1250-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX1250-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX1250-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1250-TRUE16-NEXT: v_mov_b16_e32 v0.l, v3.l
; GFX1250-TRUE16-NEXT: s_set_pc_i64 s[30:31]
;
@@ -11936,11 +11977,12 @@ define half @global_agent_atomic_fadd_ret_f16__offset12b_pos__align4__amdgpu_no_
; GFX1250-FAKE16-NEXT: s_wait_loadcnt 0x0
; GFX1250-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v5
; GFX1250-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX1250-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1250-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1250-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX1250-FAKE16-NEXT: s_cbranch_execnz .LBB50_1
; GFX1250-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX1250-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX1250-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1250-FAKE16-NEXT: v_mov_b32_e32 v0, v3
; GFX1250-FAKE16-NEXT: s_set_pc_i64 s[30:31]
;
@@ -11971,9 +12013,11 @@ define half @global_agent_atomic_fadd_ret_f16__offset12b_pos__align4__amdgpu_no_
; GFX12-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_cbranch_execnz .LBB50_1
; GFX12-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: v_mov_b16_e32 v0.l, v3.l
; GFX12-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -12004,9 +12048,11 @@ define half @global_agent_atomic_fadd_ret_f16__offset12b_pos__align4__amdgpu_no_
; GFX12-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: s_cbranch_execnz .LBB50_1
; GFX12-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: v_mov_b32_e32 v0, v3
; GFX12-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -12057,11 +12103,12 @@ define half @global_agent_atomic_fadd_ret_f16__offset12b_pos__align4__amdgpu_no_
; GFX11-TRUE16-NEXT: buffer_gl0_inv
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB50_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v0.l, v3.l
; GFX11-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -12086,11 +12133,12 @@ define half @global_agent_atomic_fadd_ret_f16__offset12b_pos__align4__amdgpu_no_
; GFX11-FAKE16-NEXT: buffer_gl0_inv
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB50_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v0, v3
; GFX11-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -12290,7 +12338,7 @@ define void @global_agent_atomic_fadd_noret_f16__offset12b__align4_pos__amdgpu_n
; GFX1250-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v5
; GFX1250-TRUE16-NEXT: v_mov_b32_e32 v5, v3
; GFX1250-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX1250-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1250-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1250-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX1250-TRUE16-NEXT: s_cbranch_execnz .LBB51_1
; GFX1250-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -12321,7 +12369,7 @@ define void @global_agent_atomic_fadd_noret_f16__offset12b__align4_pos__amdgpu_n
; GFX1250-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v5
; GFX1250-FAKE16-NEXT: v_mov_b32_e32 v5, v3
; GFX1250-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX1250-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1250-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1250-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX1250-FAKE16-NEXT: s_cbranch_execnz .LBB51_1
; GFX1250-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -12354,6 +12402,7 @@ define void @global_agent_atomic_fadd_noret_f16__offset12b__align4_pos__amdgpu_n
; GFX12-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_cbranch_execnz .LBB51_1
; GFX12-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -12385,6 +12434,7 @@ define void @global_agent_atomic_fadd_noret_f16__offset12b__align4_pos__amdgpu_n
; GFX12-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: s_cbranch_execnz .LBB51_1
; GFX12-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -12435,7 +12485,7 @@ define void @global_agent_atomic_fadd_noret_f16__offset12b__align4_pos__amdgpu_n
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-TRUE16-NEXT: v_mov_b32_e32 v4, v3
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB51_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -12462,7 +12512,7 @@ define void @global_agent_atomic_fadd_noret_f16__offset12b__align4_pos__amdgpu_n
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v4, v3
; GFX11-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB51_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -12672,11 +12722,12 @@ define half @global_system_atomic_fadd_ret_f16__offset12b_pos__amdgpu_no_fine_gr
; GFX1250-TRUE16-NEXT: s_wait_loadcnt 0x0
; GFX1250-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v5, v7
; GFX1250-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX1250-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1250-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1250-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX1250-TRUE16-NEXT: s_cbranch_execnz .LBB52_1
; GFX1250-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX1250-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX1250-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1250-TRUE16-NEXT: v_lshrrev_b32_e32 v0, v3, v5
; GFX1250-TRUE16-NEXT: s_set_pc_i64 s[30:31]
;
@@ -12716,11 +12767,12 @@ define half @global_system_atomic_fadd_ret_f16__offset12b_pos__amdgpu_no_fine_gr
; GFX1250-FAKE16-NEXT: s_wait_loadcnt 0x0
; GFX1250-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v5, v7
; GFX1250-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX1250-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1250-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1250-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX1250-FAKE16-NEXT: s_cbranch_execnz .LBB52_1
; GFX1250-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX1250-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX1250-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1250-FAKE16-NEXT: v_lshrrev_b32_e32 v0, v3, v5
; GFX1250-FAKE16-NEXT: s_set_pc_i64 s[30:31]
;
@@ -12764,9 +12816,11 @@ define half @global_system_atomic_fadd_ret_f16__offset12b_pos__amdgpu_no_fine_gr
; GFX12-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_cbranch_execnz .LBB52_1
; GFX12-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: v_lshrrev_b32_e32 v0, v3, v5
; GFX12-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -12810,9 +12864,11 @@ define half @global_system_atomic_fadd_ret_f16__offset12b_pos__amdgpu_no_fine_gr
; GFX12-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: s_cbranch_execnz .LBB52_1
; GFX12-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: v_lshrrev_b32_e32 v0, v3, v5
; GFX12-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -12856,7 +12912,6 @@ define half @global_system_atomic_fadd_ret_f16__offset12b_pos__amdgpu_no_fine_gr
; GFX11-TRUE16: ; %bb.0:
; GFX11-TRUE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-TRUE16-NEXT: v_add_co_u32 v3, vcc_lo, 0x7fe, v0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-TRUE16-NEXT: s_mov_b32 s0, 0
; GFX11-TRUE16-NEXT: v_and_b32_e32 v0, -4, v3
@@ -12885,11 +12940,12 @@ define half @global_system_atomic_fadd_ret_f16__offset12b_pos__amdgpu_no_fine_gr
; GFX11-TRUE16-NEXT: buffer_gl0_inv
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v5, v6
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB52_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v0, v3, v5
; GFX11-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -12897,7 +12953,6 @@ define half @global_system_atomic_fadd_ret_f16__offset12b_pos__amdgpu_no_fine_gr
; GFX11-FAKE16: ; %bb.0:
; GFX11-FAKE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-FAKE16-NEXT: v_add_co_u32 v3, vcc_lo, 0x7fe, v0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-FAKE16-NEXT: s_mov_b32 s0, 0
; GFX11-FAKE16-NEXT: v_and_b32_e32 v0, -4, v3
@@ -12926,11 +12981,12 @@ define half @global_system_atomic_fadd_ret_f16__offset12b_pos__amdgpu_no_fine_gr
; GFX11-FAKE16-NEXT: buffer_gl0_inv
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v5, v6
; GFX11-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB52_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v0, v3, v5
; GFX11-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -13197,7 +13253,7 @@ define void @global_system_atomic_fadd_noret_f16__offset12b_pos__amdgpu_no_fine_
; GFX1250-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v4, v5
; GFX1250-TRUE16-NEXT: v_mov_b32_e32 v5, v4
; GFX1250-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX1250-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1250-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1250-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX1250-TRUE16-NEXT: s_cbranch_execnz .LBB53_1
; GFX1250-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -13239,7 +13295,7 @@ define void @global_system_atomic_fadd_noret_f16__offset12b_pos__amdgpu_no_fine_
; GFX1250-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v4, v5
; GFX1250-FAKE16-NEXT: v_mov_b32_e32 v5, v4
; GFX1250-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX1250-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1250-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1250-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX1250-FAKE16-NEXT: s_cbranch_execnz .LBB53_1
; GFX1250-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -13285,6 +13341,7 @@ define void @global_system_atomic_fadd_noret_f16__offset12b_pos__amdgpu_no_fine_
; GFX12-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_cbranch_execnz .LBB53_1
; GFX12-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -13329,6 +13386,7 @@ define void @global_system_atomic_fadd_noret_f16__offset12b_pos__amdgpu_no_fine_
; GFX12-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: s_cbranch_execnz .LBB53_1
; GFX12-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -13373,7 +13431,6 @@ define void @global_system_atomic_fadd_noret_f16__offset12b_pos__amdgpu_no_fine_
; GFX11-TRUE16: ; %bb.0:
; GFX11-TRUE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-TRUE16-NEXT: v_add_co_u32 v3, vcc_lo, 0x7fe, v0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-TRUE16-NEXT: s_mov_b32 s0, 0
; GFX11-TRUE16-NEXT: v_and_b32_e32 v0, -4, v3
@@ -13401,7 +13458,7 @@ define void @global_system_atomic_fadd_noret_f16__offset12b_pos__amdgpu_no_fine_
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-TRUE16-NEXT: v_mov_b32_e32 v4, v3
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB53_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -13412,7 +13469,6 @@ define void @global_system_atomic_fadd_noret_f16__offset12b_pos__amdgpu_no_fine_
; GFX11-FAKE16: ; %bb.0:
; GFX11-FAKE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-FAKE16-NEXT: v_add_co_u32 v3, vcc_lo, 0x7fe, v0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-FAKE16-NEXT: s_mov_b32 s0, 0
; GFX11-FAKE16-NEXT: v_and_b32_e32 v0, -4, v3
@@ -13440,7 +13496,7 @@ define void @global_system_atomic_fadd_noret_f16__offset12b_pos__amdgpu_no_fine_
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v4, v3
; GFX11-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB53_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -13709,11 +13765,12 @@ define bfloat @global_agent_atomic_fadd_ret_bf16__amdgpu_no_fine_grained_memory(
; GFX1250-NEXT: s_wait_loadcnt 0x0
; GFX1250-NEXT: v_cmp_eq_u32_e32 vcc_lo, v5, v7
; GFX1250-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX1250-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1250-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1250-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX1250-NEXT: s_cbranch_execnz .LBB54_1
; GFX1250-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX1250-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX1250-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1250-NEXT: v_lshrrev_b32_e32 v0, v3, v5
; GFX1250-NEXT: s_set_pc_i64 s[30:31]
;
@@ -13764,9 +13821,11 @@ define bfloat @global_agent_atomic_fadd_ret_bf16__amdgpu_no_fine_grained_memory(
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB54_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: v_lshrrev_b32_e32 v0, v3, v5
; GFX12-NEXT: s_setpc_b64 s[30:31]
;
@@ -13854,11 +13913,12 @@ define bfloat @global_agent_atomic_fadd_ret_bf16__amdgpu_no_fine_grained_memory(
; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v5, v6
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB54_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: v_lshrrev_b32_e32 v0, v3, v5
; GFX11-NEXT: s_setpc_b64 s[30:31]
;
@@ -14146,11 +14206,12 @@ define bfloat @global_agent_atomic_fadd_ret_bf16__offset12b_pos__amdgpu_no_fine_
; GFX1250-NEXT: s_wait_loadcnt 0x0
; GFX1250-NEXT: v_cmp_eq_u32_e32 vcc_lo, v5, v7
; GFX1250-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX1250-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1250-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1250-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX1250-NEXT: s_cbranch_execnz .LBB55_1
; GFX1250-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX1250-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX1250-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1250-NEXT: v_lshrrev_b32_e32 v0, v3, v5
; GFX1250-NEXT: s_set_pc_i64 s[30:31]
;
@@ -14204,9 +14265,11 @@ define bfloat @global_agent_atomic_fadd_ret_bf16__offset12b_pos__amdgpu_no_fine_
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB55_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: v_lshrrev_b32_e32 v0, v3, v5
; GFX12-NEXT: s_setpc_b64 s[30:31]
;
@@ -14259,16 +14322,16 @@ define bfloat @global_agent_atomic_fadd_ret_bf16__offset12b_pos__amdgpu_no_fine_
; GFX11: ; %bb.0:
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: v_add_co_u32 v3, vcc_lo, 0x7fe, v0
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_3)
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-NEXT: v_lshlrev_b32_e32 v2, 16, v2
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_4) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_and_b32_e32 v0, -4, v3
; GFX11-NEXT: v_and_b32_e32 v3, 3, v3
; GFX11-NEXT: s_mov_b32 s0, 0
; GFX11-NEXT: global_load_b32 v5, v[0:1], off
; GFX11-NEXT: v_lshlrev_b32_e32 v3, 3, v3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_lshlrev_b32_e64 v4, v3, 0xffff
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_not_b32_e32 v4, v4
; GFX11-NEXT: .p2align 6
; GFX11-NEXT: .LBB55_1: ; %atomicrmw.start
@@ -14298,11 +14361,12 @@ define bfloat @global_agent_atomic_fadd_ret_bf16__offset12b_pos__amdgpu_no_fine_
; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v5, v6
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB55_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: v_lshrrev_b32_e32 v0, v3, v5
; GFX11-NEXT: s_setpc_b64 s[30:31]
;
@@ -14599,11 +14663,12 @@ define bfloat @global_agent_atomic_fadd_ret_bf16__offset12b_neg__amdgpu_no_fine_
; GFX1250-NEXT: s_wait_loadcnt 0x0
; GFX1250-NEXT: v_cmp_eq_u32_e32 vcc_lo, v5, v7
; GFX1250-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX1250-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1250-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1250-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX1250-NEXT: s_cbranch_execnz .LBB56_1
; GFX1250-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX1250-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX1250-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1250-NEXT: v_lshrrev_b32_e32 v0, v3, v5
; GFX1250-NEXT: s_set_pc_i64 s[30:31]
;
@@ -14657,9 +14722,11 @@ define bfloat @global_agent_atomic_fadd_ret_bf16__offset12b_neg__amdgpu_no_fine_
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB56_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: v_lshrrev_b32_e32 v0, v3, v5
; GFX12-NEXT: s_setpc_b64 s[30:31]
;
@@ -14713,16 +14780,16 @@ define bfloat @global_agent_atomic_fadd_ret_bf16__offset12b_neg__amdgpu_no_fine_
; GFX11: ; %bb.0:
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: v_add_co_u32 v3, vcc_lo, 0xfffff800, v0
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_3)
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, -1, v1, vcc_lo
; GFX11-NEXT: v_lshlrev_b32_e32 v2, 16, v2
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_4) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_and_b32_e32 v0, -4, v3
; GFX11-NEXT: v_and_b32_e32 v3, 3, v3
; GFX11-NEXT: s_mov_b32 s0, 0
; GFX11-NEXT: global_load_b32 v5, v[0:1], off
; GFX11-NEXT: v_lshlrev_b32_e32 v3, 3, v3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_lshlrev_b32_e64 v4, v3, 0xffff
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_not_b32_e32 v4, v4
; GFX11-NEXT: .p2align 6
; GFX11-NEXT: .LBB56_1: ; %atomicrmw.start
@@ -14752,11 +14819,12 @@ define bfloat @global_agent_atomic_fadd_ret_bf16__offset12b_neg__amdgpu_no_fine_
; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v5, v6
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB56_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: v_lshrrev_b32_e32 v0, v3, v5
; GFX11-NEXT: s_setpc_b64 s[30:31]
;
@@ -15051,7 +15119,7 @@ define void @global_agent_atomic_fadd_noret_bf16__amdgpu_no_fine_grained_memory(
; GFX1250-NEXT: v_cmp_eq_u32_e32 vcc_lo, v4, v5
; GFX1250-NEXT: v_mov_b32_e32 v5, v4
; GFX1250-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX1250-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1250-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1250-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX1250-NEXT: s_cbranch_execnz .LBB57_1
; GFX1250-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -15104,6 +15172,7 @@ define void @global_agent_atomic_fadd_noret_bf16__amdgpu_no_fine_grained_memory(
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB57_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -15191,7 +15260,7 @@ define void @global_agent_atomic_fadd_noret_bf16__amdgpu_no_fine_grained_memory(
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-NEXT: v_mov_b32_e32 v4, v3
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB57_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -15477,7 +15546,7 @@ define void @global_agent_atomic_fadd_noret_bf16__offset12b_pos__amdgpu_no_fine_
; GFX1250-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v3
; GFX1250-TRUE16-NEXT: v_mov_b32_e32 v3, v2
; GFX1250-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX1250-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1250-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1250-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX1250-TRUE16-NEXT: s_cbranch_execnz .LBB58_1
; GFX1250-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -15519,7 +15588,7 @@ define void @global_agent_atomic_fadd_noret_bf16__offset12b_pos__amdgpu_no_fine_
; GFX1250-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v4, v5
; GFX1250-FAKE16-NEXT: v_mov_b32_e32 v5, v4
; GFX1250-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX1250-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1250-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1250-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX1250-FAKE16-NEXT: s_cbranch_execnz .LBB58_1
; GFX1250-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -15575,6 +15644,7 @@ define void @global_agent_atomic_fadd_noret_bf16__offset12b_pos__amdgpu_no_fine_
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB58_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -15628,16 +15698,16 @@ define void @global_agent_atomic_fadd_noret_bf16__offset12b_pos__amdgpu_no_fine_
; GFX11: ; %bb.0:
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: v_add_co_u32 v4, vcc_lo, 0x7fe, v0
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_3)
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-NEXT: v_lshlrev_b32_e32 v6, 16, v2
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_4) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_and_b32_e32 v0, -4, v4
; GFX11-NEXT: v_and_b32_e32 v4, 3, v4
; GFX11-NEXT: s_mov_b32 s0, 0
; GFX11-NEXT: global_load_b32 v3, v[0:1], off
; GFX11-NEXT: v_lshlrev_b32_e32 v4, 3, v4
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_lshlrev_b32_e64 v5, v4, 0xffff
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_not_b32_e32 v5, v5
; GFX11-NEXT: .p2align 6
; GFX11-NEXT: .LBB58_1: ; %atomicrmw.start
@@ -15666,7 +15736,7 @@ define void @global_agent_atomic_fadd_noret_bf16__offset12b_pos__amdgpu_no_fine_
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v3
; GFX11-NEXT: v_mov_b32_e32 v3, v2
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB58_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -15961,7 +16031,7 @@ define void @global_agent_atomic_fadd_noret_bf16__offset12b_neg__amdgpu_no_fine_
; GFX1250-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v3
; GFX1250-TRUE16-NEXT: v_mov_b32_e32 v3, v2
; GFX1250-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX1250-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1250-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1250-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX1250-TRUE16-NEXT: s_cbranch_execnz .LBB59_1
; GFX1250-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -16004,7 +16074,7 @@ define void @global_agent_atomic_fadd_noret_bf16__offset12b_neg__amdgpu_no_fine_
; GFX1250-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v4, v5
; GFX1250-FAKE16-NEXT: v_mov_b32_e32 v5, v4
; GFX1250-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX1250-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1250-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1250-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX1250-FAKE16-NEXT: s_cbranch_execnz .LBB59_1
; GFX1250-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -16060,6 +16130,7 @@ define void @global_agent_atomic_fadd_noret_bf16__offset12b_neg__amdgpu_no_fine_
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB59_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -16114,16 +16185,16 @@ define void @global_agent_atomic_fadd_noret_bf16__offset12b_neg__amdgpu_no_fine_
; GFX11: ; %bb.0:
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: v_add_co_u32 v4, vcc_lo, 0xfffff800, v0
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_3)
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, -1, v1, vcc_lo
; GFX11-NEXT: v_lshlrev_b32_e32 v6, 16, v2
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_4) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_and_b32_e32 v0, -4, v4
; GFX11-NEXT: v_and_b32_e32 v4, 3, v4
; GFX11-NEXT: s_mov_b32 s0, 0
; GFX11-NEXT: global_load_b32 v3, v[0:1], off
; GFX11-NEXT: v_lshlrev_b32_e32 v4, 3, v4
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_lshlrev_b32_e64 v5, v4, 0xffff
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_not_b32_e32 v5, v5
; GFX11-NEXT: .p2align 6
; GFX11-NEXT: .LBB59_1: ; %atomicrmw.start
@@ -16152,7 +16223,7 @@ define void @global_agent_atomic_fadd_noret_bf16__offset12b_neg__amdgpu_no_fine_
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v3
; GFX11-NEXT: v_mov_b32_e32 v3, v2
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB59_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -16433,11 +16504,12 @@ define bfloat @global_agent_atomic_fadd_ret_bf16__offset12b_pos__align4__amdgpu_
; GFX1250-TRUE16-NEXT: s_wait_loadcnt 0x0
; GFX1250-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v5
; GFX1250-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX1250-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1250-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1250-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX1250-TRUE16-NEXT: s_cbranch_execnz .LBB60_1
; GFX1250-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX1250-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX1250-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1250-TRUE16-NEXT: v_mov_b16_e32 v0.l, v3.l
; GFX1250-TRUE16-NEXT: s_set_pc_i64 s[30:31]
;
@@ -16464,11 +16536,12 @@ define bfloat @global_agent_atomic_fadd_ret_bf16__offset12b_pos__align4__amdgpu_
; GFX1250-FAKE16-NEXT: s_wait_loadcnt 0x0
; GFX1250-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v5
; GFX1250-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX1250-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1250-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1250-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX1250-FAKE16-NEXT: s_cbranch_execnz .LBB60_1
; GFX1250-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX1250-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX1250-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1250-FAKE16-NEXT: v_mov_b32_e32 v0, v3
; GFX1250-FAKE16-NEXT: s_set_pc_i64 s[30:31]
;
@@ -16509,9 +16582,11 @@ define bfloat @global_agent_atomic_fadd_ret_bf16__offset12b_pos__align4__amdgpu_
; GFX12-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_cbranch_execnz .LBB60_1
; GFX12-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: v_mov_b16_e32 v0.l, v3.l
; GFX12-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -16552,9 +16627,11 @@ define bfloat @global_agent_atomic_fadd_ret_bf16__offset12b_pos__align4__amdgpu_
; GFX12-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: s_cbranch_execnz .LBB60_1
; GFX12-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: v_mov_b32_e32 v0, v3
; GFX12-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -16625,11 +16702,12 @@ define bfloat @global_agent_atomic_fadd_ret_bf16__offset12b_pos__align4__amdgpu_
; GFX11-TRUE16-NEXT: buffer_gl0_inv
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB60_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v0.l, v3.l
; GFX11-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -16664,11 +16742,12 @@ define bfloat @global_agent_atomic_fadd_ret_bf16__offset12b_pos__align4__amdgpu_
; GFX11-FAKE16-NEXT: buffer_gl0_inv
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB60_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v0, v3
; GFX11-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -16906,7 +16985,7 @@ define void @global_agent_atomic_fadd_noret_bf16__offset12b__align4_pos__amdgpu_
; GFX1250-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v3
; GFX1250-TRUE16-NEXT: v_mov_b32_e32 v3, v2
; GFX1250-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX1250-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1250-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1250-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX1250-TRUE16-NEXT: s_cbranch_execnz .LBB61_1
; GFX1250-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -16936,7 +17015,7 @@ define void @global_agent_atomic_fadd_noret_bf16__offset12b__align4_pos__amdgpu_
; GFX1250-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v5
; GFX1250-FAKE16-NEXT: v_mov_b32_e32 v5, v3
; GFX1250-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX1250-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1250-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1250-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX1250-FAKE16-NEXT: s_cbranch_execnz .LBB61_1
; GFX1250-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -16979,6 +17058,7 @@ define void @global_agent_atomic_fadd_noret_bf16__offset12b__align4_pos__amdgpu_
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB61_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -17049,7 +17129,7 @@ define void @global_agent_atomic_fadd_noret_bf16__offset12b__align4_pos__amdgpu_
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v3
; GFX11-NEXT: v_mov_b32_e32 v3, v2
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB61_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -17297,11 +17377,12 @@ define bfloat @global_system_atomic_fadd_ret_bf16__offset12b_pos__amdgpu_no_fine
; GFX1250-NEXT: s_wait_loadcnt 0x0
; GFX1250-NEXT: v_cmp_eq_u32_e32 vcc_lo, v5, v7
; GFX1250-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX1250-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1250-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1250-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX1250-NEXT: s_cbranch_execnz .LBB62_1
; GFX1250-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX1250-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX1250-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1250-NEXT: v_lshrrev_b32_e32 v0, v3, v5
; GFX1250-NEXT: s_set_pc_i64 s[30:31]
;
@@ -17356,9 +17437,11 @@ define bfloat @global_system_atomic_fadd_ret_bf16__offset12b_pos__amdgpu_no_fine
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB62_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: v_lshrrev_b32_e32 v0, v3, v5
; GFX12-NEXT: s_setpc_b64 s[30:31]
;
@@ -17411,16 +17494,16 @@ define bfloat @global_system_atomic_fadd_ret_bf16__offset12b_pos__amdgpu_no_fine
; GFX11: ; %bb.0:
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: v_add_co_u32 v3, vcc_lo, 0x7fe, v0
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_3)
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-NEXT: v_lshlrev_b32_e32 v2, 16, v2
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_4) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_and_b32_e32 v0, -4, v3
; GFX11-NEXT: v_and_b32_e32 v3, 3, v3
; GFX11-NEXT: s_mov_b32 s0, 0
; GFX11-NEXT: global_load_b32 v5, v[0:1], off
; GFX11-NEXT: v_lshlrev_b32_e32 v3, 3, v3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_lshlrev_b32_e64 v4, v3, 0xffff
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_not_b32_e32 v4, v4
; GFX11-NEXT: .p2align 6
; GFX11-NEXT: .LBB62_1: ; %atomicrmw.start
@@ -17450,11 +17533,12 @@ define bfloat @global_system_atomic_fadd_ret_bf16__offset12b_pos__amdgpu_no_fine
; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v5, v6
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB62_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: v_lshrrev_b32_e32 v0, v3, v5
; GFX11-NEXT: s_setpc_b64 s[30:31]
;
@@ -17754,7 +17838,7 @@ define void @global_system_atomic_fadd_noret_bf16__offset12b_pos__amdgpu_no_fine
; GFX1250-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v3
; GFX1250-TRUE16-NEXT: v_mov_b32_e32 v3, v2
; GFX1250-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX1250-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1250-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1250-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX1250-TRUE16-NEXT: s_cbranch_execnz .LBB63_1
; GFX1250-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -17796,7 +17880,7 @@ define void @global_system_atomic_fadd_noret_bf16__offset12b_pos__amdgpu_no_fine
; GFX1250-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v4, v5
; GFX1250-FAKE16-NEXT: v_mov_b32_e32 v5, v4
; GFX1250-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX1250-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1250-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1250-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX1250-FAKE16-NEXT: s_cbranch_execnz .LBB63_1
; GFX1250-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -17853,6 +17937,7 @@ define void @global_system_atomic_fadd_noret_bf16__offset12b_pos__amdgpu_no_fine
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB63_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -17906,16 +17991,16 @@ define void @global_system_atomic_fadd_noret_bf16__offset12b_pos__amdgpu_no_fine
; GFX11: ; %bb.0:
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: v_add_co_u32 v4, vcc_lo, 0x7fe, v0
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_3)
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-NEXT: v_lshlrev_b32_e32 v6, 16, v2
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_4) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_and_b32_e32 v0, -4, v4
; GFX11-NEXT: v_and_b32_e32 v4, 3, v4
; GFX11-NEXT: s_mov_b32 s0, 0
; GFX11-NEXT: global_load_b32 v3, v[0:1], off
; GFX11-NEXT: v_lshlrev_b32_e32 v4, 3, v4
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_lshlrev_b32_e64 v5, v4, 0xffff
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_not_b32_e32 v5, v5
; GFX11-NEXT: .p2align 6
; GFX11-NEXT: .LBB63_1: ; %atomicrmw.start
@@ -17944,7 +18029,7 @@ define void @global_system_atomic_fadd_noret_bf16__offset12b_pos__amdgpu_no_fine
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v3
; GFX11-NEXT: v_mov_b32_e32 v3, v2
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB63_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -18262,11 +18347,12 @@ define <2 x half> @global_agent_atomic_fadd_ret_v2f16__amdgpu_no_fine_grained_me
; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB64_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: v_mov_b32_e32 v0, v3
; GFX11-NEXT: s_setpc_b64 s[30:31]
;
@@ -18495,11 +18581,12 @@ define <2 x half> @global_agent_atomic_fadd_ret_v2f16__offset12b_pos__amdgpu_no_
; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB65_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: v_mov_b32_e32 v0, v3
; GFX11-NEXT: s_setpc_b64 s[30:31]
;
@@ -18730,11 +18817,12 @@ define <2 x half> @global_agent_atomic_fadd_ret_v2f16__offset12b_neg__amdgpu_no_
; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB66_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: v_mov_b32_e32 v0, v3
; GFX11-NEXT: s_setpc_b64 s[30:31]
;
@@ -18972,7 +19060,7 @@ define void @global_agent_atomic_fadd_noret_v2f16__amdgpu_no_fine_grained_memory
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-NEXT: v_mov_b32_e32 v4, v3
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB67_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -19181,7 +19269,7 @@ define void @global_agent_atomic_fadd_noret_v2f16__offset12b_pos__amdgpu_no_fine
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-NEXT: v_mov_b32_e32 v4, v3
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB68_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -19393,7 +19481,7 @@ define void @global_agent_atomic_fadd_noret_v2f16__offset12b_neg__amdgpu_no_fine
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-NEXT: v_mov_b32_e32 v4, v3
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB69_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -19615,11 +19703,12 @@ define <2 x half> @global_system_atomic_fadd_ret_v2f16__offset12b_pos__amdgpu_no
; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB70_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: v_mov_b32_e32 v0, v3
; GFX11-NEXT: s_setpc_b64 s[30:31]
;
@@ -19853,7 +19942,7 @@ define void @global_system_atomic_fadd_noret_v2f16__offset12b_pos__amdgpu_no_fin
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-NEXT: v_mov_b32_e32 v4, v3
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB71_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -20069,11 +20158,12 @@ define <2 x half> @global_agent_atomic_fadd_ret_v2f16__amdgpu_no_remote_memory(p
; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB72_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: v_mov_b32_e32 v0, v3
; GFX11-NEXT: s_setpc_b64 s[30:31]
;
@@ -20315,7 +20405,7 @@ define void @global_agent_atomic_fadd_noret_v2f16__amdgpu_no_remote_memory(ptr a
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-NEXT: v_mov_b32_e32 v4, v3
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB73_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -20551,11 +20641,12 @@ define <2 x half> @global_agent_atomic_fadd_ret_v2f16__amdgpu_no_fine_grained_me
; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB74_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: v_mov_b32_e32 v0, v3
; GFX11-NEXT: s_setpc_b64 s[30:31]
;
@@ -20783,7 +20874,7 @@ define void @global_agent_atomic_fadd_noret_v2f16__amdgpu_no_fine_grained_memory
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-NEXT: v_mov_b32_e32 v4, v3
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB75_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -20993,11 +21084,12 @@ define <2 x half> @global_agent_atomic_fadd_ret_v2f16__maybe_remote(ptr addrspac
; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB76_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: v_mov_b32_e32 v0, v3
; GFX11-NEXT: s_setpc_b64 s[30:31]
;
@@ -21239,7 +21331,7 @@ define void @global_agent_atomic_fadd_noret_v2f16__maybe_remote(ptr addrspace(1)
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-NEXT: v_mov_b32_e32 v4, v3
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB77_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -21504,12 +21596,13 @@ define <2 x bfloat> @global_agent_atomic_fadd_ret_v2bf16__amdgpu_no_fine_grained
; GFX11-TRUE16-NEXT: buffer_gl0_inv
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v6
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB78_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-TRUE16-NEXT: s_set_inst_prefetch_distance 0x2
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_mov_b32_e32 v0, v3
; GFX11-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -21553,12 +21646,13 @@ define <2 x bfloat> @global_agent_atomic_fadd_ret_v2bf16__amdgpu_no_fine_grained
; GFX11-FAKE16-NEXT: buffer_gl0_inv
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v6
; GFX11-FAKE16-NEXT: s_or_b32 s1, vcc_lo, s1
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB78_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-FAKE16-NEXT: s_set_inst_prefetch_distance 0x2
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s1
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v0, v3
; GFX11-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -21906,12 +22000,13 @@ define <2 x bfloat> @global_agent_atomic_fadd_ret_v2bf16__offset12b_pos__amdgpu_
; GFX11-TRUE16-NEXT: buffer_gl0_inv
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v6
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB79_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-TRUE16-NEXT: s_set_inst_prefetch_distance 0x2
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_mov_b32_e32 v0, v3
; GFX11-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -21955,12 +22050,13 @@ define <2 x bfloat> @global_agent_atomic_fadd_ret_v2bf16__offset12b_pos__amdgpu_
; GFX11-FAKE16-NEXT: buffer_gl0_inv
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v6
; GFX11-FAKE16-NEXT: s_or_b32 s1, vcc_lo, s1
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB79_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-FAKE16-NEXT: s_set_inst_prefetch_distance 0x2
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s1
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v0, v3
; GFX11-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -22310,12 +22406,13 @@ define <2 x bfloat> @global_agent_atomic_fadd_ret_v2bf16__offset12b_neg__amdgpu_
; GFX11-TRUE16-NEXT: buffer_gl0_inv
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v6
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB80_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-TRUE16-NEXT: s_set_inst_prefetch_distance 0x2
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_mov_b32_e32 v0, v3
; GFX11-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -22359,12 +22456,13 @@ define <2 x bfloat> @global_agent_atomic_fadd_ret_v2bf16__offset12b_neg__amdgpu_
; GFX11-FAKE16-NEXT: buffer_gl0_inv
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v6
; GFX11-FAKE16-NEXT: s_or_b32 s1, vcc_lo, s1
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB80_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-FAKE16-NEXT: s_set_inst_prefetch_distance 0x2
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s1
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v0, v3
; GFX11-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -22721,7 +22819,7 @@ define void @global_agent_atomic_fadd_noret_v2bf16__amdgpu_no_fine_grained_memor
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v3
; GFX11-TRUE16-NEXT: v_mov_b32_e32 v3, v2
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB81_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -22755,7 +22853,7 @@ define void @global_agent_atomic_fadd_noret_v2bf16__amdgpu_no_fine_grained_memor
; GFX11-FAKE16-NEXT: v_add3_u32 v7, v7, v2, 0x7fff
; GFX11-FAKE16-NEXT: v_add3_u32 v8, v8, v6, 0x7fff
; GFX11-FAKE16-NEXT: v_cmp_u_f32_e64 s0, v2, v2
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_4)
; GFX11-FAKE16-NEXT: v_cndmask_b32_e32 v6, v8, v10, vcc_lo
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v2, v7, v9, s0
; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
@@ -22768,7 +22866,7 @@ define void @global_agent_atomic_fadd_noret_v2bf16__amdgpu_no_fine_grained_memor
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v3
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v3, v2
; GFX11-FAKE16-NEXT: s_or_b32 s1, vcc_lo, s1
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB81_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -23107,7 +23205,7 @@ define void @global_agent_atomic_fadd_noret_v2bf16__offset12b_pos__amdgpu_no_fin
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v3
; GFX11-TRUE16-NEXT: v_mov_b32_e32 v3, v2
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB82_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -23141,7 +23239,7 @@ define void @global_agent_atomic_fadd_noret_v2bf16__offset12b_pos__amdgpu_no_fin
; GFX11-FAKE16-NEXT: v_add3_u32 v7, v7, v2, 0x7fff
; GFX11-FAKE16-NEXT: v_add3_u32 v8, v8, v6, 0x7fff
; GFX11-FAKE16-NEXT: v_cmp_u_f32_e64 s0, v2, v2
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_4)
; GFX11-FAKE16-NEXT: v_cndmask_b32_e32 v6, v8, v10, vcc_lo
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v2, v7, v9, s0
; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
@@ -23154,7 +23252,7 @@ define void @global_agent_atomic_fadd_noret_v2bf16__offset12b_pos__amdgpu_no_fin
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v3
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v3, v2
; GFX11-FAKE16-NEXT: s_or_b32 s1, vcc_lo, s1
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB82_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -23496,7 +23594,7 @@ define void @global_agent_atomic_fadd_noret_v2bf16__offset12b_neg__amdgpu_no_fin
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v3
; GFX11-TRUE16-NEXT: v_mov_b32_e32 v3, v2
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB83_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -23530,7 +23628,7 @@ define void @global_agent_atomic_fadd_noret_v2bf16__offset12b_neg__amdgpu_no_fin
; GFX11-FAKE16-NEXT: v_add3_u32 v7, v7, v2, 0x7fff
; GFX11-FAKE16-NEXT: v_add3_u32 v8, v8, v6, 0x7fff
; GFX11-FAKE16-NEXT: v_cmp_u_f32_e64 s0, v2, v2
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_4)
; GFX11-FAKE16-NEXT: v_cndmask_b32_e32 v6, v8, v10, vcc_lo
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v2, v7, v9, s0
; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
@@ -23543,7 +23641,7 @@ define void @global_agent_atomic_fadd_noret_v2bf16__offset12b_neg__amdgpu_no_fin
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v3
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v3, v2
; GFX11-FAKE16-NEXT: s_or_b32 s1, vcc_lo, s1
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB83_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -23895,12 +23993,13 @@ define <2 x bfloat> @global_system_atomic_fadd_ret_v2bf16__offset12b_pos__amdgpu
; GFX11-TRUE16-NEXT: buffer_gl0_inv
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v6
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB84_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-TRUE16-NEXT: s_set_inst_prefetch_distance 0x2
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_mov_b32_e32 v0, v3
; GFX11-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -23944,12 +24043,13 @@ define <2 x bfloat> @global_system_atomic_fadd_ret_v2bf16__offset12b_pos__amdgpu
; GFX11-FAKE16-NEXT: buffer_gl0_inv
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v6
; GFX11-FAKE16-NEXT: s_or_b32 s1, vcc_lo, s1
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB84_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-FAKE16-NEXT: s_set_inst_prefetch_distance 0x2
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s1
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v0, v3
; GFX11-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -24302,7 +24402,7 @@ define void @global_system_atomic_fadd_noret_v2bf16__offset12b_pos__amdgpu_no_fi
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v3
; GFX11-TRUE16-NEXT: v_mov_b32_e32 v3, v2
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB85_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -24336,7 +24436,7 @@ define void @global_system_atomic_fadd_noret_v2bf16__offset12b_pos__amdgpu_no_fi
; GFX11-FAKE16-NEXT: v_add3_u32 v7, v7, v2, 0x7fff
; GFX11-FAKE16-NEXT: v_add3_u32 v8, v8, v6, 0x7fff
; GFX11-FAKE16-NEXT: v_cmp_u_f32_e64 s0, v2, v2
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_4)
; GFX11-FAKE16-NEXT: v_cndmask_b32_e32 v6, v8, v10, vcc_lo
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v2, v7, v9, s0
; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
@@ -24349,7 +24449,7 @@ define void @global_system_atomic_fadd_noret_v2bf16__offset12b_pos__amdgpu_no_fi
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v3
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v3, v2
; GFX11-FAKE16-NEXT: s_or_b32 s1, vcc_lo, s1
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB85_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -24695,12 +24795,13 @@ define <2 x bfloat> @global_agent_atomic_fadd_ret_v2bf16__amdgpu_no_remote_memor
; GFX11-TRUE16-NEXT: buffer_gl0_inv
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v6
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB86_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-TRUE16-NEXT: s_set_inst_prefetch_distance 0x2
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_mov_b32_e32 v0, v3
; GFX11-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -24744,12 +24845,13 @@ define <2 x bfloat> @global_agent_atomic_fadd_ret_v2bf16__amdgpu_no_remote_memor
; GFX11-FAKE16-NEXT: buffer_gl0_inv
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v6
; GFX11-FAKE16-NEXT: s_or_b32 s1, vcc_lo, s1
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB86_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-FAKE16-NEXT: s_set_inst_prefetch_distance 0x2
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s1
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v0, v3
; GFX11-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -25096,7 +25198,7 @@ define void @global_agent_atomic_fadd_noret_v2bf16__amdgpu_no_remote_memory(ptr
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v3
; GFX11-TRUE16-NEXT: v_mov_b32_e32 v3, v2
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB87_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -25130,7 +25232,7 @@ define void @global_agent_atomic_fadd_noret_v2bf16__amdgpu_no_remote_memory(ptr
; GFX11-FAKE16-NEXT: v_add3_u32 v7, v7, v2, 0x7fff
; GFX11-FAKE16-NEXT: v_add3_u32 v8, v8, v6, 0x7fff
; GFX11-FAKE16-NEXT: v_cmp_u_f32_e64 s0, v2, v2
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_4)
; GFX11-FAKE16-NEXT: v_cndmask_b32_e32 v6, v8, v10, vcc_lo
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v2, v7, v9, s0
; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
@@ -25143,7 +25245,7 @@ define void @global_agent_atomic_fadd_noret_v2bf16__amdgpu_no_remote_memory(ptr
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v3
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v3, v2
; GFX11-FAKE16-NEXT: s_or_b32 s1, vcc_lo, s1
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB87_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -25483,12 +25585,13 @@ define <2 x bfloat> @global_agent_atomic_fadd_ret_v2bf16__amdgpu_no_fine_grained
; GFX11-TRUE16-NEXT: buffer_gl0_inv
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v6
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB88_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-TRUE16-NEXT: s_set_inst_prefetch_distance 0x2
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_mov_b32_e32 v0, v3
; GFX11-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -25532,12 +25635,13 @@ define <2 x bfloat> @global_agent_atomic_fadd_ret_v2bf16__amdgpu_no_fine_grained
; GFX11-FAKE16-NEXT: buffer_gl0_inv
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v6
; GFX11-FAKE16-NEXT: s_or_b32 s1, vcc_lo, s1
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB88_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-FAKE16-NEXT: s_set_inst_prefetch_distance 0x2
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s1
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v0, v3
; GFX11-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -25884,7 +25988,7 @@ define void @global_agent_atomic_fadd_noret_v2bf16__amdgpu_no_fine_grained_memor
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v3
; GFX11-TRUE16-NEXT: v_mov_b32_e32 v3, v2
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB89_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -25918,7 +26022,7 @@ define void @global_agent_atomic_fadd_noret_v2bf16__amdgpu_no_fine_grained_memor
; GFX11-FAKE16-NEXT: v_add3_u32 v7, v7, v2, 0x7fff
; GFX11-FAKE16-NEXT: v_add3_u32 v8, v8, v6, 0x7fff
; GFX11-FAKE16-NEXT: v_cmp_u_f32_e64 s0, v2, v2
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_4)
; GFX11-FAKE16-NEXT: v_cndmask_b32_e32 v6, v8, v10, vcc_lo
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v2, v7, v9, s0
; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
@@ -25931,7 +26035,7 @@ define void @global_agent_atomic_fadd_noret_v2bf16__amdgpu_no_fine_grained_memor
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v3
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v3, v2
; GFX11-FAKE16-NEXT: s_or_b32 s1, vcc_lo, s1
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB89_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -26271,12 +26375,13 @@ define <2 x bfloat> @global_agent_atomic_fadd_ret_v2bf16__maybe_remote(ptr addrs
; GFX11-TRUE16-NEXT: buffer_gl0_inv
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v6
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB90_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-TRUE16-NEXT: s_set_inst_prefetch_distance 0x2
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_mov_b32_e32 v0, v3
; GFX11-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -26320,12 +26425,13 @@ define <2 x bfloat> @global_agent_atomic_fadd_ret_v2bf16__maybe_remote(ptr addrs
; GFX11-FAKE16-NEXT: buffer_gl0_inv
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v6
; GFX11-FAKE16-NEXT: s_or_b32 s1, vcc_lo, s1
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB90_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-FAKE16-NEXT: s_set_inst_prefetch_distance 0x2
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s1
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v0, v3
; GFX11-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -26672,7 +26778,7 @@ define void @global_agent_atomic_fadd_noret_v2bf16__maybe_remote(ptr addrspace(1
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v3
; GFX11-TRUE16-NEXT: v_mov_b32_e32 v3, v2
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB91_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -26706,7 +26812,7 @@ define void @global_agent_atomic_fadd_noret_v2bf16__maybe_remote(ptr addrspace(1
; GFX11-FAKE16-NEXT: v_add3_u32 v7, v7, v2, 0x7fff
; GFX11-FAKE16-NEXT: v_add3_u32 v8, v8, v6, 0x7fff
; GFX11-FAKE16-NEXT: v_cmp_u_f32_e64 s0, v2, v2
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_4)
; GFX11-FAKE16-NEXT: v_cndmask_b32_e32 v6, v8, v10, vcc_lo
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v2, v7, v9, s0
; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
@@ -26719,7 +26825,7 @@ define void @global_agent_atomic_fadd_noret_v2bf16__maybe_remote(ptr addrspace(1
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v3
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v3, v2
; GFX11-FAKE16-NEXT: s_or_b32 s1, vcc_lo, s1
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB91_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
diff --git a/llvm/test/CodeGen/AMDGPU/global-atomicrmw-fmax.ll b/llvm/test/CodeGen/AMDGPU/global-atomicrmw-fmax.ll
index 2b5feeb6867313..4c84597a93b21f 100644
--- a/llvm/test/CodeGen/AMDGPU/global-atomicrmw-fmax.ll
+++ b/llvm/test/CodeGen/AMDGPU/global-atomicrmw-fmax.ll
@@ -1370,11 +1370,12 @@ define float @global_agent_atomic_fmax_ret_f32__amdgpu_no_remote_memory(ptr addr
; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB8_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: v_mov_b32_e32 v0, v3
; GFX11-NEXT: s_setpc_b64 s[30:31]
;
@@ -3033,9 +3034,11 @@ define double @global_agent_atomic_fmax_ret_f64__amdgpu_no_fine_grained_memory(p
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB18_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: v_dual_mov_b32 v0, v4 :: v_dual_mov_b32 v1, v5
; GFX12-NEXT: s_setpc_b64 s[30:31]
;
@@ -3069,11 +3072,12 @@ define double @global_agent_atomic_fmax_ret_f64__amdgpu_no_fine_grained_memory(p
; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: v_cmp_eq_u64_e32 vcc_lo, v[4:5], v[6:7]
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB18_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: v_dual_mov_b32 v0, v4 :: v_dual_mov_b32 v1, v5
; GFX11-NEXT: s_setpc_b64 s[30:31]
;
@@ -3206,9 +3210,11 @@ define double @global_agent_atomic_fmax_ret_f64__offset12b_pos__amdgpu_no_fine_g
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB19_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: v_dual_mov_b32 v0, v4 :: v_dual_mov_b32 v1, v5
; GFX12-NEXT: s_setpc_b64 s[30:31]
;
@@ -3242,11 +3248,12 @@ define double @global_agent_atomic_fmax_ret_f64__offset12b_pos__amdgpu_no_fine_g
; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: v_cmp_eq_u64_e32 vcc_lo, v[4:5], v[6:7]
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB19_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: v_dual_mov_b32 v0, v4 :: v_dual_mov_b32 v1, v5
; GFX11-NEXT: s_setpc_b64 s[30:31]
;
@@ -3380,9 +3387,11 @@ define double @global_agent_atomic_fmax_ret_f64__offset12b_neg__amdgpu_no_fine_g
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB20_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: v_dual_mov_b32 v0, v4 :: v_dual_mov_b32 v1, v5
; GFX12-NEXT: s_setpc_b64 s[30:31]
;
@@ -3416,11 +3425,12 @@ define double @global_agent_atomic_fmax_ret_f64__offset12b_neg__amdgpu_no_fine_g
; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: v_cmp_eq_u64_e32 vcc_lo, v[4:5], v[6:7]
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB20_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: v_dual_mov_b32 v0, v4 :: v_dual_mov_b32 v1, v5
; GFX11-NEXT: s_setpc_b64 s[30:31]
;
@@ -3554,6 +3564,7 @@ define void @global_agent_atomic_fmax_noret_f64__amdgpu_no_fine_grained_memory(p
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB21_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -3589,7 +3600,7 @@ define void @global_agent_atomic_fmax_noret_f64__amdgpu_no_fine_grained_memory(p
; GFX11-NEXT: v_cmp_eq_u64_e32 vcc_lo, v[2:3], v[4:5]
; GFX11-NEXT: v_dual_mov_b32 v5, v3 :: v_dual_mov_b32 v4, v2
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB21_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -3717,6 +3728,7 @@ define void @global_agent_atomic_fmax_noret_f64__offset12b_pos__amdgpu_no_fine_g
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB22_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -3752,7 +3764,7 @@ define void @global_agent_atomic_fmax_noret_f64__offset12b_pos__amdgpu_no_fine_g
; GFX11-NEXT: v_cmp_eq_u64_e32 vcc_lo, v[2:3], v[4:5]
; GFX11-NEXT: v_dual_mov_b32 v5, v3 :: v_dual_mov_b32 v4, v2
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB22_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -3883,6 +3895,7 @@ define void @global_agent_atomic_fmax_noret_f64__offset12b_neg__amdgpu_no_fine_g
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB23_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -3918,7 +3931,7 @@ define void @global_agent_atomic_fmax_noret_f64__offset12b_neg__amdgpu_no_fine_g
; GFX11-NEXT: v_cmp_eq_u64_e32 vcc_lo, v[2:3], v[4:5]
; GFX11-NEXT: v_dual_mov_b32 v5, v3 :: v_dual_mov_b32 v4, v2
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB23_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -4049,9 +4062,11 @@ define double @global_agent_atomic_fmax_ret_f64__amdgpu_no_remote_memory(ptr add
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB24_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: v_dual_mov_b32 v0, v4 :: v_dual_mov_b32 v1, v5
; GFX12-NEXT: s_setpc_b64 s[30:31]
;
@@ -4085,11 +4100,12 @@ define double @global_agent_atomic_fmax_ret_f64__amdgpu_no_remote_memory(ptr add
; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: v_cmp_eq_u64_e32 vcc_lo, v[4:5], v[6:7]
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB24_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: v_dual_mov_b32 v0, v4 :: v_dual_mov_b32 v1, v5
; GFX11-NEXT: s_setpc_b64 s[30:31]
;
@@ -4297,9 +4313,11 @@ define double @global_agent_atomic_fmax_ret_f64__amdgpu_no_fine_grained_memory__
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB25_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: v_dual_mov_b32 v0, v4 :: v_dual_mov_b32 v1, v5
; GFX12-NEXT: s_setpc_b64 s[30:31]
;
@@ -4333,11 +4351,12 @@ define double @global_agent_atomic_fmax_ret_f64__amdgpu_no_fine_grained_memory__
; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: v_cmp_eq_u64_e32 vcc_lo, v[4:5], v[6:7]
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB25_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: v_dual_mov_b32 v0, v4 :: v_dual_mov_b32 v1, v5
; GFX11-NEXT: s_setpc_b64 s[30:31]
;
@@ -4488,9 +4507,11 @@ define half @global_agent_atomic_fmax_ret_f16__amdgpu_no_fine_grained_memory(ptr
; GFX12-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_cbranch_execnz .LBB26_1
; GFX12-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: v_lshrrev_b32_e32 v0, v3, v5
; GFX12-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -4534,9 +4555,11 @@ define half @global_agent_atomic_fmax_ret_f16__amdgpu_no_fine_grained_memory(ptr
; GFX12-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: s_cbranch_execnz .LBB26_1
; GFX12-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: v_lshrrev_b32_e32 v0, v3, v5
; GFX12-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -4610,11 +4633,12 @@ define half @global_agent_atomic_fmax_ret_f16__amdgpu_no_fine_grained_memory(ptr
; GFX11-TRUE16-NEXT: buffer_gl0_inv
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v5, v6
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB26_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v0, v3, v5
; GFX11-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -4652,11 +4676,12 @@ define half @global_agent_atomic_fmax_ret_f16__amdgpu_no_fine_grained_memory(ptr
; GFX11-FAKE16-NEXT: buffer_gl0_inv
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v5, v6
; GFX11-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB26_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v0, v3, v5
; GFX11-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -4928,9 +4953,11 @@ define half @global_agent_atomic_fmax_ret_f16__offset12b_pos__amdgpu_no_fine_gra
; GFX12-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_cbranch_execnz .LBB27_1
; GFX12-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: v_lshrrev_b32_e32 v0, v1, v5
; GFX12-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -4976,9 +5003,11 @@ define half @global_agent_atomic_fmax_ret_f16__offset12b_pos__amdgpu_no_fine_gra
; GFX12-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: s_cbranch_execnz .LBB27_1
; GFX12-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: v_lshrrev_b32_e32 v0, v3, v5
; GFX12-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -5024,7 +5053,6 @@ define half @global_agent_atomic_fmax_ret_f16__offset12b_pos__amdgpu_no_fine_gra
; GFX11-TRUE16: ; %bb.0:
; GFX11-TRUE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-TRUE16-NEXT: v_add_co_u32 v0, vcc_lo, 0x7fe, v0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v4, null, 0, v1, vcc_lo
; GFX11-TRUE16-NEXT: s_mov_b32 s0, 0
; GFX11-TRUE16-NEXT: v_and_b32_e32 v3, -4, v0
@@ -5057,11 +5085,12 @@ define half @global_agent_atomic_fmax_ret_f16__offset12b_pos__amdgpu_no_fine_gra
; GFX11-TRUE16-NEXT: buffer_gl0_inv
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v5, v6
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB27_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v0, v1, v5
; GFX11-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -5069,16 +5098,16 @@ define half @global_agent_atomic_fmax_ret_f16__offset12b_pos__amdgpu_no_fine_gra
; GFX11-FAKE16: ; %bb.0:
; GFX11-FAKE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-FAKE16-NEXT: v_add_co_u32 v3, vcc_lo, 0x7fe, v0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-FAKE16-NEXT: v_max_f16_e32 v2, v2, v2
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_4) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_and_b32_e32 v0, -4, v3
; GFX11-FAKE16-NEXT: v_and_b32_e32 v3, 3, v3
; GFX11-FAKE16-NEXT: s_mov_b32 s0, 0
; GFX11-FAKE16-NEXT: global_load_b32 v5, v[0:1], off
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v3, 3, v3
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e64 v4, v3, 0xffff
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_not_b32_e32 v4, v4
; GFX11-FAKE16-NEXT: .LBB27_1: ; %atomicrmw.start
; GFX11-FAKE16-NEXT: ; =>This Inner Loop Header: Depth=1
@@ -5100,11 +5129,12 @@ define half @global_agent_atomic_fmax_ret_f16__offset12b_pos__amdgpu_no_fine_gra
; GFX11-FAKE16-NEXT: buffer_gl0_inv
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v5, v6
; GFX11-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB27_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v0, v3, v5
; GFX11-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -5384,9 +5414,11 @@ define half @global_agent_atomic_fmax_ret_f16__offset12b_neg__amdgpu_no_fine_gra
; GFX12-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_cbranch_execnz .LBB28_1
; GFX12-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: v_lshrrev_b32_e32 v0, v1, v5
; GFX12-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -5432,9 +5464,11 @@ define half @global_agent_atomic_fmax_ret_f16__offset12b_neg__amdgpu_no_fine_gra
; GFX12-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: s_cbranch_execnz .LBB28_1
; GFX12-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: v_lshrrev_b32_e32 v0, v3, v5
; GFX12-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -5481,7 +5515,6 @@ define half @global_agent_atomic_fmax_ret_f16__offset12b_neg__amdgpu_no_fine_gra
; GFX11-TRUE16: ; %bb.0:
; GFX11-TRUE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-TRUE16-NEXT: v_add_co_u32 v0, vcc_lo, 0xfffff800, v0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v4, null, -1, v1, vcc_lo
; GFX11-TRUE16-NEXT: s_mov_b32 s0, 0
; GFX11-TRUE16-NEXT: v_and_b32_e32 v3, -4, v0
@@ -5514,11 +5547,12 @@ define half @global_agent_atomic_fmax_ret_f16__offset12b_neg__amdgpu_no_fine_gra
; GFX11-TRUE16-NEXT: buffer_gl0_inv
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v5, v6
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB28_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v0, v1, v5
; GFX11-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -5526,16 +5560,16 @@ define half @global_agent_atomic_fmax_ret_f16__offset12b_neg__amdgpu_no_fine_gra
; GFX11-FAKE16: ; %bb.0:
; GFX11-FAKE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-FAKE16-NEXT: v_add_co_u32 v3, vcc_lo, 0xfffff800, v0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v1, null, -1, v1, vcc_lo
; GFX11-FAKE16-NEXT: v_max_f16_e32 v2, v2, v2
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_4) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_and_b32_e32 v0, -4, v3
; GFX11-FAKE16-NEXT: v_and_b32_e32 v3, 3, v3
; GFX11-FAKE16-NEXT: s_mov_b32 s0, 0
; GFX11-FAKE16-NEXT: global_load_b32 v5, v[0:1], off
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v3, 3, v3
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e64 v4, v3, 0xffff
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_not_b32_e32 v4, v4
; GFX11-FAKE16-NEXT: .LBB28_1: ; %atomicrmw.start
; GFX11-FAKE16-NEXT: ; =>This Inner Loop Header: Depth=1
@@ -5557,11 +5591,12 @@ define half @global_agent_atomic_fmax_ret_f16__offset12b_neg__amdgpu_no_fine_gra
; GFX11-FAKE16-NEXT: buffer_gl0_inv
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v5, v6
; GFX11-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB28_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v0, v3, v5
; GFX11-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -5838,6 +5873,7 @@ define void @global_agent_atomic_fmax_noret_f16__amdgpu_no_fine_grained_memory(p
; GFX12-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_cbranch_execnz .LBB29_1
; GFX12-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -5883,6 +5919,7 @@ define void @global_agent_atomic_fmax_noret_f16__amdgpu_no_fine_grained_memory(p
; GFX12-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: s_cbranch_execnz .LBB29_1
; GFX12-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -5957,7 +5994,7 @@ define void @global_agent_atomic_fmax_noret_f16__amdgpu_no_fine_grained_memory(p
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-TRUE16-NEXT: v_mov_b32_e32 v4, v3
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB29_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -5998,7 +6035,7 @@ define void @global_agent_atomic_fmax_noret_f16__amdgpu_no_fine_grained_memory(p
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v4, v3
; GFX11-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB29_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -6268,6 +6305,7 @@ define void @global_agent_atomic_fmax_noret_f16__offset12b_pos__amdgpu_no_fine_g
; GFX12-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_cbranch_execnz .LBB30_1
; GFX12-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -6315,6 +6353,7 @@ define void @global_agent_atomic_fmax_noret_f16__offset12b_pos__amdgpu_no_fine_g
; GFX12-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: s_cbranch_execnz .LBB30_1
; GFX12-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -6361,7 +6400,6 @@ define void @global_agent_atomic_fmax_noret_f16__offset12b_pos__amdgpu_no_fine_g
; GFX11-TRUE16: ; %bb.0:
; GFX11-TRUE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-TRUE16-NEXT: v_add_co_u32 v0, vcc_lo, 0x7fe, v0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v4, null, 0, v1, vcc_lo
; GFX11-TRUE16-NEXT: s_mov_b32 s0, 0
; GFX11-TRUE16-NEXT: v_and_b32_e32 v3, -4, v0
@@ -6394,7 +6432,7 @@ define void @global_agent_atomic_fmax_noret_f16__offset12b_pos__amdgpu_no_fine_g
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v5, v6
; GFX11-TRUE16-NEXT: v_mov_b32_e32 v6, v5
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB30_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -6405,16 +6443,16 @@ define void @global_agent_atomic_fmax_noret_f16__offset12b_pos__amdgpu_no_fine_g
; GFX11-FAKE16: ; %bb.0:
; GFX11-FAKE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-FAKE16-NEXT: v_add_co_u32 v4, vcc_lo, 0x7fe, v0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-FAKE16-NEXT: v_max_f16_e32 v6, v2, v2
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_4) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_and_b32_e32 v0, -4, v4
; GFX11-FAKE16-NEXT: v_and_b32_e32 v4, 3, v4
; GFX11-FAKE16-NEXT: s_mov_b32 s0, 0
; GFX11-FAKE16-NEXT: global_load_b32 v3, v[0:1], off
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v4, 3, v4
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e64 v5, v4, 0xffff
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_not_b32_e32 v5, v5
; GFX11-FAKE16-NEXT: .LBB30_1: ; %atomicrmw.start
; GFX11-FAKE16-NEXT: ; =>This Inner Loop Header: Depth=1
@@ -6436,7 +6474,7 @@ define void @global_agent_atomic_fmax_noret_f16__offset12b_pos__amdgpu_no_fine_g
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v3
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v3, v2
; GFX11-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB30_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -6713,6 +6751,7 @@ define void @global_agent_atomic_fmax_noret_f16__offset12b_neg__amdgpu_no_fine_g
; GFX12-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_cbranch_execnz .LBB31_1
; GFX12-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -6760,6 +6799,7 @@ define void @global_agent_atomic_fmax_noret_f16__offset12b_neg__amdgpu_no_fine_g
; GFX12-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: s_cbranch_execnz .LBB31_1
; GFX12-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -6807,7 +6847,6 @@ define void @global_agent_atomic_fmax_noret_f16__offset12b_neg__amdgpu_no_fine_g
; GFX11-TRUE16: ; %bb.0:
; GFX11-TRUE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-TRUE16-NEXT: v_add_co_u32 v0, vcc_lo, 0xfffff800, v0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v4, null, -1, v1, vcc_lo
; GFX11-TRUE16-NEXT: s_mov_b32 s0, 0
; GFX11-TRUE16-NEXT: v_and_b32_e32 v3, -4, v0
@@ -6840,7 +6879,7 @@ define void @global_agent_atomic_fmax_noret_f16__offset12b_neg__amdgpu_no_fine_g
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v5, v6
; GFX11-TRUE16-NEXT: v_mov_b32_e32 v6, v5
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB31_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -6851,16 +6890,16 @@ define void @global_agent_atomic_fmax_noret_f16__offset12b_neg__amdgpu_no_fine_g
; GFX11-FAKE16: ; %bb.0:
; GFX11-FAKE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-FAKE16-NEXT: v_add_co_u32 v4, vcc_lo, 0xfffff800, v0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v1, null, -1, v1, vcc_lo
; GFX11-FAKE16-NEXT: v_max_f16_e32 v6, v2, v2
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_4) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_and_b32_e32 v0, -4, v4
; GFX11-FAKE16-NEXT: v_and_b32_e32 v4, 3, v4
; GFX11-FAKE16-NEXT: s_mov_b32 s0, 0
; GFX11-FAKE16-NEXT: global_load_b32 v3, v[0:1], off
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v4, 3, v4
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e64 v5, v4, 0xffff
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_not_b32_e32 v5, v5
; GFX11-FAKE16-NEXT: .LBB31_1: ; %atomicrmw.start
; GFX11-FAKE16-NEXT: ; =>This Inner Loop Header: Depth=1
@@ -6882,7 +6921,7 @@ define void @global_agent_atomic_fmax_noret_f16__offset12b_neg__amdgpu_no_fine_g
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v3
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v3, v2
; GFX11-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB31_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -7145,9 +7184,11 @@ define half @global_agent_atomic_fmax_ret_f16__offset12b_pos__align4__amdgpu_no_
; GFX12-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_cbranch_execnz .LBB32_1
; GFX12-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: v_mov_b16_e32 v0.l, v3.l
; GFX12-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -7180,9 +7221,11 @@ define half @global_agent_atomic_fmax_ret_f16__offset12b_pos__align4__amdgpu_no_
; GFX12-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: s_cbranch_execnz .LBB32_1
; GFX12-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: v_mov_b32_e32 v0, v3
; GFX12-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -7237,11 +7280,12 @@ define half @global_agent_atomic_fmax_ret_f16__offset12b_pos__align4__amdgpu_no_
; GFX11-TRUE16-NEXT: buffer_gl0_inv
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB32_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v0.l, v3.l
; GFX11-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -7268,11 +7312,12 @@ define half @global_agent_atomic_fmax_ret_f16__offset12b_pos__align4__amdgpu_no_
; GFX11-FAKE16-NEXT: buffer_gl0_inv
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB32_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v0, v3
; GFX11-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -7485,6 +7530,7 @@ define void @global_agent_atomic_fmax_noret_f16__offset12b__align4_pos__amdgpu_n
; GFX12-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_cbranch_execnz .LBB33_1
; GFX12-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -7519,6 +7565,7 @@ define void @global_agent_atomic_fmax_noret_f16__offset12b__align4_pos__amdgpu_n
; GFX12-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: s_cbranch_execnz .LBB33_1
; GFX12-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -7574,7 +7621,7 @@ define void @global_agent_atomic_fmax_noret_f16__offset12b__align4_pos__amdgpu_n
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-TRUE16-NEXT: v_mov_b32_e32 v4, v3
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB33_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -7604,7 +7651,7 @@ define void @global_agent_atomic_fmax_noret_f16__offset12b__align4_pos__amdgpu_n
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v3
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v3, v2
; GFX11-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB33_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -7830,9 +7877,11 @@ define half @global_system_atomic_fmax_ret_f16__offset12b_pos__amdgpu_no_fine_gr
; GFX12-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_cbranch_execnz .LBB34_1
; GFX12-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: v_lshrrev_b32_e32 v0, v1, v5
; GFX12-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -7879,9 +7928,11 @@ define half @global_system_atomic_fmax_ret_f16__offset12b_pos__amdgpu_no_fine_gr
; GFX12-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: s_cbranch_execnz .LBB34_1
; GFX12-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: v_lshrrev_b32_e32 v0, v3, v5
; GFX12-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -7927,7 +7978,6 @@ define half @global_system_atomic_fmax_ret_f16__offset12b_pos__amdgpu_no_fine_gr
; GFX11-TRUE16: ; %bb.0:
; GFX11-TRUE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-TRUE16-NEXT: v_add_co_u32 v0, vcc_lo, 0x7fe, v0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v4, null, 0, v1, vcc_lo
; GFX11-TRUE16-NEXT: s_mov_b32 s0, 0
; GFX11-TRUE16-NEXT: v_and_b32_e32 v3, -4, v0
@@ -7960,11 +8010,12 @@ define half @global_system_atomic_fmax_ret_f16__offset12b_pos__amdgpu_no_fine_gr
; GFX11-TRUE16-NEXT: buffer_gl0_inv
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v5, v6
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB34_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v0, v1, v5
; GFX11-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -7972,16 +8023,16 @@ define half @global_system_atomic_fmax_ret_f16__offset12b_pos__amdgpu_no_fine_gr
; GFX11-FAKE16: ; %bb.0:
; GFX11-FAKE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-FAKE16-NEXT: v_add_co_u32 v3, vcc_lo, 0x7fe, v0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-FAKE16-NEXT: v_max_f16_e32 v2, v2, v2
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_4) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_and_b32_e32 v0, -4, v3
; GFX11-FAKE16-NEXT: v_and_b32_e32 v3, 3, v3
; GFX11-FAKE16-NEXT: s_mov_b32 s0, 0
; GFX11-FAKE16-NEXT: global_load_b32 v5, v[0:1], off
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v3, 3, v3
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e64 v4, v3, 0xffff
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_not_b32_e32 v4, v4
; GFX11-FAKE16-NEXT: .LBB34_1: ; %atomicrmw.start
; GFX11-FAKE16-NEXT: ; =>This Inner Loop Header: Depth=1
@@ -8003,11 +8054,12 @@ define half @global_system_atomic_fmax_ret_f16__offset12b_pos__amdgpu_no_fine_gr
; GFX11-FAKE16-NEXT: buffer_gl0_inv
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v5, v6
; GFX11-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB34_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v0, v3, v5
; GFX11-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -8291,6 +8343,7 @@ define void @global_system_atomic_fmax_noret_f16__offset12b_pos__amdgpu_no_fine_
; GFX12-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_cbranch_execnz .LBB35_1
; GFX12-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -8339,6 +8392,7 @@ define void @global_system_atomic_fmax_noret_f16__offset12b_pos__amdgpu_no_fine_
; GFX12-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: s_cbranch_execnz .LBB35_1
; GFX12-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -8385,7 +8439,6 @@ define void @global_system_atomic_fmax_noret_f16__offset12b_pos__amdgpu_no_fine_
; GFX11-TRUE16: ; %bb.0:
; GFX11-TRUE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-TRUE16-NEXT: v_add_co_u32 v0, vcc_lo, 0x7fe, v0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v4, null, 0, v1, vcc_lo
; GFX11-TRUE16-NEXT: s_mov_b32 s0, 0
; GFX11-TRUE16-NEXT: v_and_b32_e32 v3, -4, v0
@@ -8418,7 +8471,7 @@ define void @global_system_atomic_fmax_noret_f16__offset12b_pos__amdgpu_no_fine_
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v5, v6
; GFX11-TRUE16-NEXT: v_mov_b32_e32 v6, v5
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB35_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -8429,16 +8482,16 @@ define void @global_system_atomic_fmax_noret_f16__offset12b_pos__amdgpu_no_fine_
; GFX11-FAKE16: ; %bb.0:
; GFX11-FAKE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-FAKE16-NEXT: v_add_co_u32 v4, vcc_lo, 0x7fe, v0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-FAKE16-NEXT: v_max_f16_e32 v6, v2, v2
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_4) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_and_b32_e32 v0, -4, v4
; GFX11-FAKE16-NEXT: v_and_b32_e32 v4, 3, v4
; GFX11-FAKE16-NEXT: s_mov_b32 s0, 0
; GFX11-FAKE16-NEXT: global_load_b32 v3, v[0:1], off
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v4, 3, v4
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e64 v5, v4, 0xffff
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_not_b32_e32 v5, v5
; GFX11-FAKE16-NEXT: .LBB35_1: ; %atomicrmw.start
; GFX11-FAKE16-NEXT: ; =>This Inner Loop Header: Depth=1
@@ -8460,7 +8513,7 @@ define void @global_system_atomic_fmax_noret_f16__offset12b_pos__amdgpu_no_fine_
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v3
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v3, v2
; GFX11-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB35_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -8748,9 +8801,11 @@ define bfloat @global_agent_atomic_fmax_ret_bf16__amdgpu_no_fine_grained_memory(
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB36_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: v_lshrrev_b32_e32 v0, v3, v5
; GFX12-NEXT: s_setpc_b64 s[30:31]
;
@@ -8838,11 +8893,12 @@ define bfloat @global_agent_atomic_fmax_ret_bf16__amdgpu_no_fine_grained_memory(
; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v5, v6
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB36_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: v_lshrrev_b32_e32 v0, v3, v5
; GFX11-NEXT: s_setpc_b64 s[30:31]
;
@@ -9146,9 +9202,11 @@ define bfloat @global_agent_atomic_fmax_ret_bf16__offset12b_pos__amdgpu_no_fine_
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB37_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: v_lshrrev_b32_e32 v0, v3, v5
; GFX12-NEXT: s_setpc_b64 s[30:31]
;
@@ -9201,16 +9259,16 @@ define bfloat @global_agent_atomic_fmax_ret_bf16__offset12b_pos__amdgpu_no_fine_
; GFX11: ; %bb.0:
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: v_add_co_u32 v3, vcc_lo, 0x7fe, v0
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_3)
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-NEXT: v_lshlrev_b32_e32 v2, 16, v2
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_4) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_and_b32_e32 v0, -4, v3
; GFX11-NEXT: v_and_b32_e32 v3, 3, v3
; GFX11-NEXT: s_mov_b32 s0, 0
; GFX11-NEXT: global_load_b32 v5, v[0:1], off
; GFX11-NEXT: v_lshlrev_b32_e32 v3, 3, v3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_lshlrev_b32_e64 v4, v3, 0xffff
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_not_b32_e32 v4, v4
; GFX11-NEXT: .p2align 6
; GFX11-NEXT: .LBB37_1: ; %atomicrmw.start
@@ -9240,11 +9298,12 @@ define bfloat @global_agent_atomic_fmax_ret_bf16__offset12b_pos__amdgpu_no_fine_
; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v5, v6
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB37_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: v_lshrrev_b32_e32 v0, v3, v5
; GFX11-NEXT: s_setpc_b64 s[30:31]
;
@@ -9556,9 +9615,11 @@ define bfloat @global_agent_atomic_fmax_ret_bf16__offset12b_neg__amdgpu_no_fine_
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB38_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: v_lshrrev_b32_e32 v0, v3, v5
; GFX12-NEXT: s_setpc_b64 s[30:31]
;
@@ -9612,16 +9673,16 @@ define bfloat @global_agent_atomic_fmax_ret_bf16__offset12b_neg__amdgpu_no_fine_
; GFX11: ; %bb.0:
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: v_add_co_u32 v3, vcc_lo, 0xfffff800, v0
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_3)
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, -1, v1, vcc_lo
; GFX11-NEXT: v_lshlrev_b32_e32 v2, 16, v2
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_4) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_and_b32_e32 v0, -4, v3
; GFX11-NEXT: v_and_b32_e32 v3, 3, v3
; GFX11-NEXT: s_mov_b32 s0, 0
; GFX11-NEXT: global_load_b32 v5, v[0:1], off
; GFX11-NEXT: v_lshlrev_b32_e32 v3, 3, v3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_lshlrev_b32_e64 v4, v3, 0xffff
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_not_b32_e32 v4, v4
; GFX11-NEXT: .p2align 6
; GFX11-NEXT: .LBB38_1: ; %atomicrmw.start
@@ -9651,11 +9712,12 @@ define bfloat @global_agent_atomic_fmax_ret_bf16__offset12b_neg__amdgpu_no_fine_
; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v5, v6
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB38_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: v_lshrrev_b32_e32 v0, v3, v5
; GFX11-NEXT: s_setpc_b64 s[30:31]
;
@@ -9963,6 +10025,7 @@ define void @global_agent_atomic_fmax_noret_bf16__amdgpu_no_fine_grained_memory(
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB39_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -10050,7 +10113,7 @@ define void @global_agent_atomic_fmax_noret_bf16__amdgpu_no_fine_grained_memory(
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-NEXT: v_mov_b32_e32 v4, v3
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB39_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -10351,6 +10414,7 @@ define void @global_agent_atomic_fmax_noret_bf16__offset12b_pos__amdgpu_no_fine_
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB40_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -10404,16 +10468,16 @@ define void @global_agent_atomic_fmax_noret_bf16__offset12b_pos__amdgpu_no_fine_
; GFX11: ; %bb.0:
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: v_add_co_u32 v4, vcc_lo, 0x7fe, v0
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_3)
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-NEXT: v_lshlrev_b32_e32 v6, 16, v2
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_4) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_and_b32_e32 v0, -4, v4
; GFX11-NEXT: v_and_b32_e32 v4, 3, v4
; GFX11-NEXT: s_mov_b32 s0, 0
; GFX11-NEXT: global_load_b32 v3, v[0:1], off
; GFX11-NEXT: v_lshlrev_b32_e32 v4, 3, v4
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_lshlrev_b32_e64 v5, v4, 0xffff
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_not_b32_e32 v5, v5
; GFX11-NEXT: .p2align 6
; GFX11-NEXT: .LBB40_1: ; %atomicrmw.start
@@ -10442,7 +10506,7 @@ define void @global_agent_atomic_fmax_noret_bf16__offset12b_pos__amdgpu_no_fine_
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v3
; GFX11-NEXT: v_mov_b32_e32 v3, v2
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB40_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -10750,6 +10814,7 @@ define void @global_agent_atomic_fmax_noret_bf16__offset12b_neg__amdgpu_no_fine_
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB41_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -10804,16 +10869,16 @@ define void @global_agent_atomic_fmax_noret_bf16__offset12b_neg__amdgpu_no_fine_
; GFX11: ; %bb.0:
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: v_add_co_u32 v4, vcc_lo, 0xfffff800, v0
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_3)
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, -1, v1, vcc_lo
; GFX11-NEXT: v_lshlrev_b32_e32 v6, 16, v2
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_4) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_and_b32_e32 v0, -4, v4
; GFX11-NEXT: v_and_b32_e32 v4, 3, v4
; GFX11-NEXT: s_mov_b32 s0, 0
; GFX11-NEXT: global_load_b32 v3, v[0:1], off
; GFX11-NEXT: v_lshlrev_b32_e32 v4, 3, v4
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_lshlrev_b32_e64 v5, v4, 0xffff
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_not_b32_e32 v5, v5
; GFX11-NEXT: .p2align 6
; GFX11-NEXT: .LBB41_1: ; %atomicrmw.start
@@ -10842,7 +10907,7 @@ define void @global_agent_atomic_fmax_noret_bf16__offset12b_neg__amdgpu_no_fine_
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v3
; GFX11-NEXT: v_mov_b32_e32 v3, v2
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB41_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -11138,9 +11203,11 @@ define bfloat @global_agent_atomic_fmax_ret_bf16__offset12b_pos__align4__amdgpu_
; GFX12-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_cbranch_execnz .LBB42_1
; GFX12-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: v_mov_b16_e32 v0.l, v3.l
; GFX12-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -11181,9 +11248,11 @@ define bfloat @global_agent_atomic_fmax_ret_bf16__offset12b_pos__align4__amdgpu_
; GFX12-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: s_cbranch_execnz .LBB42_1
; GFX12-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: v_mov_b32_e32 v0, v3
; GFX12-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -11254,11 +11323,12 @@ define bfloat @global_agent_atomic_fmax_ret_bf16__offset12b_pos__align4__amdgpu_
; GFX11-TRUE16-NEXT: buffer_gl0_inv
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB42_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v0.l, v3.l
; GFX11-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -11293,11 +11363,12 @@ define bfloat @global_agent_atomic_fmax_ret_bf16__offset12b_pos__align4__amdgpu_
; GFX11-FAKE16-NEXT: buffer_gl0_inv
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB42_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v0, v3
; GFX11-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -11548,6 +11619,7 @@ define void @global_agent_atomic_fmax_noret_bf16__offset12b__align4_pos__amdgpu_
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB43_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -11618,7 +11690,7 @@ define void @global_agent_atomic_fmax_noret_bf16__offset12b__align4_pos__amdgpu_
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v3
; GFX11-NEXT: v_mov_b32_e32 v3, v2
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB43_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -11882,9 +11954,11 @@ define bfloat @global_system_atomic_fmax_ret_bf16__offset12b_pos__amdgpu_no_fine
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB44_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: v_lshrrev_b32_e32 v0, v3, v5
; GFX12-NEXT: s_setpc_b64 s[30:31]
;
@@ -11937,16 +12011,16 @@ define bfloat @global_system_atomic_fmax_ret_bf16__offset12b_pos__amdgpu_no_fine
; GFX11: ; %bb.0:
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: v_add_co_u32 v3, vcc_lo, 0x7fe, v0
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_3)
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-NEXT: v_lshlrev_b32_e32 v2, 16, v2
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_4) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_and_b32_e32 v0, -4, v3
; GFX11-NEXT: v_and_b32_e32 v3, 3, v3
; GFX11-NEXT: s_mov_b32 s0, 0
; GFX11-NEXT: global_load_b32 v5, v[0:1], off
; GFX11-NEXT: v_lshlrev_b32_e32 v3, 3, v3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_lshlrev_b32_e64 v4, v3, 0xffff
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_not_b32_e32 v4, v4
; GFX11-NEXT: .p2align 6
; GFX11-NEXT: .LBB44_1: ; %atomicrmw.start
@@ -11976,11 +12050,12 @@ define bfloat @global_system_atomic_fmax_ret_bf16__offset12b_pos__amdgpu_no_fine
; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v5, v6
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB44_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: v_lshrrev_b32_e32 v0, v3, v5
; GFX11-NEXT: s_setpc_b64 s[30:31]
;
@@ -12295,6 +12370,7 @@ define void @global_system_atomic_fmax_noret_bf16__offset12b_pos__amdgpu_no_fine
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB45_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -12348,16 +12424,16 @@ define void @global_system_atomic_fmax_noret_bf16__offset12b_pos__amdgpu_no_fine
; GFX11: ; %bb.0:
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: v_add_co_u32 v4, vcc_lo, 0x7fe, v0
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_3)
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-NEXT: v_lshlrev_b32_e32 v6, 16, v2
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_4) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_and_b32_e32 v0, -4, v4
; GFX11-NEXT: v_and_b32_e32 v4, 3, v4
; GFX11-NEXT: s_mov_b32 s0, 0
; GFX11-NEXT: global_load_b32 v3, v[0:1], off
; GFX11-NEXT: v_lshlrev_b32_e32 v4, 3, v4
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_lshlrev_b32_e64 v5, v4, 0xffff
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_not_b32_e32 v5, v5
; GFX11-NEXT: .p2align 6
; GFX11-NEXT: .LBB45_1: ; %atomicrmw.start
@@ -12386,7 +12462,7 @@ define void @global_system_atomic_fmax_noret_bf16__offset12b_pos__amdgpu_no_fine
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v3
; GFX11-NEXT: v_mov_b32_e32 v3, v2
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB45_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -12678,9 +12754,11 @@ define <2 x half> @global_agent_atomic_fmax_ret_v2f16__amdgpu_no_fine_grained_me
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB46_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: v_mov_b32_e32 v0, v3
; GFX12-NEXT: s_setpc_b64 s[30:31]
;
@@ -12731,11 +12809,12 @@ define <2 x half> @global_agent_atomic_fmax_ret_v2f16__amdgpu_no_fine_grained_me
; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB46_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: v_mov_b32_e32 v0, v3
; GFX11-NEXT: s_setpc_b64 s[30:31]
;
@@ -12960,9 +13039,11 @@ define <2 x half> @global_agent_atomic_fmax_ret_v2f16__offset12b_pos__amdgpu_no_
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB47_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: v_mov_b32_e32 v0, v3
; GFX12-NEXT: s_setpc_b64 s[30:31]
;
@@ -13013,11 +13094,12 @@ define <2 x half> @global_agent_atomic_fmax_ret_v2f16__offset12b_pos__amdgpu_no_
; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB47_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: v_mov_b32_e32 v0, v3
; GFX11-NEXT: s_setpc_b64 s[30:31]
;
@@ -13244,9 +13326,11 @@ define <2 x half> @global_agent_atomic_fmax_ret_v2f16__offset12b_neg__amdgpu_no_
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB48_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: v_mov_b32_e32 v0, v3
; GFX12-NEXT: s_setpc_b64 s[30:31]
;
@@ -13297,11 +13381,12 @@ define <2 x half> @global_agent_atomic_fmax_ret_v2f16__offset12b_neg__amdgpu_no_
; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB48_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: v_mov_b32_e32 v0, v3
; GFX11-NEXT: s_setpc_b64 s[30:31]
;
@@ -13536,6 +13621,7 @@ define void @global_agent_atomic_fmax_noret_v2f16__amdgpu_no_fine_grained_memory
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB49_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -13587,7 +13673,7 @@ define void @global_agent_atomic_fmax_noret_v2f16__amdgpu_no_fine_grained_memory
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v3
; GFX11-NEXT: v_mov_b32_e32 v3, v2
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB49_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -13805,6 +13891,7 @@ define void @global_agent_atomic_fmax_noret_v2f16__offset12b_pos__amdgpu_no_fine
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB50_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -13856,7 +13943,7 @@ define void @global_agent_atomic_fmax_noret_v2f16__offset12b_pos__amdgpu_no_fine
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v3
; GFX11-NEXT: v_mov_b32_e32 v3, v2
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB50_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -14077,6 +14164,7 @@ define void @global_agent_atomic_fmax_noret_v2f16__offset12b_neg__amdgpu_no_fine
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB51_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -14128,7 +14216,7 @@ define void @global_agent_atomic_fmax_noret_v2f16__offset12b_neg__amdgpu_no_fine
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v3
; GFX11-NEXT: v_mov_b32_e32 v3, v2
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB51_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -14358,9 +14446,11 @@ define <2 x half> @global_system_atomic_fmax_ret_v2f16__offset12b_pos__amdgpu_no
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB52_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: v_mov_b32_e32 v0, v3
; GFX12-NEXT: s_setpc_b64 s[30:31]
;
@@ -14411,11 +14501,12 @@ define <2 x half> @global_system_atomic_fmax_ret_v2f16__offset12b_pos__amdgpu_no
; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB52_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: v_mov_b32_e32 v0, v3
; GFX11-NEXT: s_setpc_b64 s[30:31]
;
@@ -14646,6 +14737,7 @@ define void @global_system_atomic_fmax_noret_v2f16__offset12b_pos__amdgpu_no_fin
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB53_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -14697,7 +14789,7 @@ define void @global_system_atomic_fmax_noret_v2f16__offset12b_pos__amdgpu_no_fin
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v3
; GFX11-NEXT: v_mov_b32_e32 v3, v2
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB53_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -14948,9 +15040,11 @@ define <2 x bfloat> @global_agent_atomic_fmax_ret_v2bf16__amdgpu_no_fine_grained
; GFX12-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_cbranch_execnz .LBB54_1
; GFX12-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: v_mov_b32_e32 v0, v3
; GFX12-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -14999,9 +15093,11 @@ define <2 x bfloat> @global_agent_atomic_fmax_ret_v2bf16__amdgpu_no_fine_grained
; GFX12-FAKE16-NEXT: s_or_b32 s1, vcc_lo, s1
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: s_cbranch_execnz .LBB54_1
; GFX12-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s1
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: v_mov_b32_e32 v0, v3
; GFX12-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -15091,12 +15187,13 @@ define <2 x bfloat> @global_agent_atomic_fmax_ret_v2bf16__amdgpu_no_fine_grained
; GFX11-TRUE16-NEXT: buffer_gl0_inv
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v6
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB54_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-TRUE16-NEXT: s_set_inst_prefetch_distance 0x2
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_mov_b32_e32 v0, v3
; GFX11-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -15140,12 +15237,13 @@ define <2 x bfloat> @global_agent_atomic_fmax_ret_v2bf16__amdgpu_no_fine_grained
; GFX11-FAKE16-NEXT: buffer_gl0_inv
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v6
; GFX11-FAKE16-NEXT: s_or_b32 s1, vcc_lo, s1
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB54_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-FAKE16-NEXT: s_set_inst_prefetch_distance 0x2
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s1
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v0, v3
; GFX11-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -15463,9 +15561,11 @@ define <2 x bfloat> @global_agent_atomic_fmax_ret_v2bf16__offset12b_pos__amdgpu_
; GFX12-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_cbranch_execnz .LBB55_1
; GFX12-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: v_mov_b32_e32 v0, v3
; GFX12-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -15514,9 +15614,11 @@ define <2 x bfloat> @global_agent_atomic_fmax_ret_v2bf16__offset12b_pos__amdgpu_
; GFX12-FAKE16-NEXT: s_or_b32 s1, vcc_lo, s1
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: s_cbranch_execnz .LBB55_1
; GFX12-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s1
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: v_mov_b32_e32 v0, v3
; GFX12-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -15606,12 +15708,13 @@ define <2 x bfloat> @global_agent_atomic_fmax_ret_v2bf16__offset12b_pos__amdgpu_
; GFX11-TRUE16-NEXT: buffer_gl0_inv
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v6
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB55_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-TRUE16-NEXT: s_set_inst_prefetch_distance 0x2
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_mov_b32_e32 v0, v3
; GFX11-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -15655,12 +15758,13 @@ define <2 x bfloat> @global_agent_atomic_fmax_ret_v2bf16__offset12b_pos__amdgpu_
; GFX11-FAKE16-NEXT: buffer_gl0_inv
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v6
; GFX11-FAKE16-NEXT: s_or_b32 s1, vcc_lo, s1
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB55_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-FAKE16-NEXT: s_set_inst_prefetch_distance 0x2
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s1
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v0, v3
; GFX11-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -15980,9 +16084,11 @@ define <2 x bfloat> @global_agent_atomic_fmax_ret_v2bf16__offset12b_neg__amdgpu_
; GFX12-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_cbranch_execnz .LBB56_1
; GFX12-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: v_mov_b32_e32 v0, v3
; GFX12-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -16031,9 +16137,11 @@ define <2 x bfloat> @global_agent_atomic_fmax_ret_v2bf16__offset12b_neg__amdgpu_
; GFX12-FAKE16-NEXT: s_or_b32 s1, vcc_lo, s1
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: s_cbranch_execnz .LBB56_1
; GFX12-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s1
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: v_mov_b32_e32 v0, v3
; GFX12-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -16123,12 +16231,13 @@ define <2 x bfloat> @global_agent_atomic_fmax_ret_v2bf16__offset12b_neg__amdgpu_
; GFX11-TRUE16-NEXT: buffer_gl0_inv
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v6
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB56_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-TRUE16-NEXT: s_set_inst_prefetch_distance 0x2
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_mov_b32_e32 v0, v3
; GFX11-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -16172,12 +16281,13 @@ define <2 x bfloat> @global_agent_atomic_fmax_ret_v2bf16__offset12b_neg__amdgpu_
; GFX11-FAKE16-NEXT: buffer_gl0_inv
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v6
; GFX11-FAKE16-NEXT: s_or_b32 s1, vcc_lo, s1
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB56_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-FAKE16-NEXT: s_set_inst_prefetch_distance 0x2
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s1
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v0, v3
; GFX11-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -16503,6 +16613,7 @@ define void @global_agent_atomic_fmax_noret_v2bf16__amdgpu_no_fine_grained_memor
; GFX12-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_cbranch_execnz .LBB57_1
; GFX12-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -16537,11 +16648,10 @@ define void @global_agent_atomic_fmax_noret_v2bf16__amdgpu_no_fine_grained_memor
; GFX12-FAKE16-NEXT: v_add3_u32 v8, v8, v6, 0x7fff
; GFX12-FAKE16-NEXT: v_cmp_u_f32_e64 s0, v2, v2
; GFX12-FAKE16-NEXT: s_wait_alu depctr_va_vcc(0)
-; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_2)
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_2) | instid1(VALU_DEP_1)
; GFX12-FAKE16-NEXT: v_cndmask_b32_e32 v6, v8, v10, vcc_lo
; GFX12-FAKE16-NEXT: s_wait_alu depctr_va_sdst(0)
; GFX12-FAKE16-NEXT: v_cndmask_b32_e64 v2, v7, v9, s0
-; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-FAKE16-NEXT: v_perm_b32 v2, v6, v2, 0x7060302
; GFX12-FAKE16-NEXT: s_wait_storecnt 0x0
; GFX12-FAKE16-NEXT: global_atomic_cmpswap_b32 v2, v[0:1], v[2:3], off th:TH_ATOMIC_RETURN scope:SCOPE_DEV
@@ -16553,6 +16663,7 @@ define void @global_agent_atomic_fmax_noret_v2bf16__amdgpu_no_fine_grained_memor
; GFX12-FAKE16-NEXT: s_or_b32 s1, vcc_lo, s1
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: s_cbranch_execnz .LBB57_1
; GFX12-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s1
@@ -16642,7 +16753,7 @@ define void @global_agent_atomic_fmax_noret_v2bf16__amdgpu_no_fine_grained_memor
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v3
; GFX11-TRUE16-NEXT: v_mov_b32_e32 v3, v2
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB57_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -16676,7 +16787,7 @@ define void @global_agent_atomic_fmax_noret_v2bf16__amdgpu_no_fine_grained_memor
; GFX11-FAKE16-NEXT: v_add3_u32 v7, v7, v2, 0x7fff
; GFX11-FAKE16-NEXT: v_add3_u32 v8, v8, v6, 0x7fff
; GFX11-FAKE16-NEXT: v_cmp_u_f32_e64 s0, v2, v2
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_4)
; GFX11-FAKE16-NEXT: v_cndmask_b32_e32 v6, v8, v10, vcc_lo
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v2, v7, v9, s0
; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
@@ -16689,7 +16800,7 @@ define void @global_agent_atomic_fmax_noret_v2bf16__amdgpu_no_fine_grained_memor
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v3
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v3, v2
; GFX11-FAKE16-NEXT: s_or_b32 s1, vcc_lo, s1
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB57_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -16997,6 +17108,7 @@ define void @global_agent_atomic_fmax_noret_v2bf16__offset12b_pos__amdgpu_no_fin
; GFX12-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_cbranch_execnz .LBB58_1
; GFX12-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -17031,11 +17143,10 @@ define void @global_agent_atomic_fmax_noret_v2bf16__offset12b_pos__amdgpu_no_fin
; GFX12-FAKE16-NEXT: v_add3_u32 v8, v8, v6, 0x7fff
; GFX12-FAKE16-NEXT: v_cmp_u_f32_e64 s0, v2, v2
; GFX12-FAKE16-NEXT: s_wait_alu depctr_va_vcc(0)
-; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_2)
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_2) | instid1(VALU_DEP_1)
; GFX12-FAKE16-NEXT: v_cndmask_b32_e32 v6, v8, v10, vcc_lo
; GFX12-FAKE16-NEXT: s_wait_alu depctr_va_sdst(0)
; GFX12-FAKE16-NEXT: v_cndmask_b32_e64 v2, v7, v9, s0
-; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-FAKE16-NEXT: v_perm_b32 v2, v6, v2, 0x7060302
; GFX12-FAKE16-NEXT: s_wait_storecnt 0x0
; GFX12-FAKE16-NEXT: global_atomic_cmpswap_b32 v2, v[0:1], v[2:3], off offset:2044 th:TH_ATOMIC_RETURN scope:SCOPE_DEV
@@ -17047,6 +17158,7 @@ define void @global_agent_atomic_fmax_noret_v2bf16__offset12b_pos__amdgpu_no_fin
; GFX12-FAKE16-NEXT: s_or_b32 s1, vcc_lo, s1
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: s_cbranch_execnz .LBB58_1
; GFX12-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s1
@@ -17136,7 +17248,7 @@ define void @global_agent_atomic_fmax_noret_v2bf16__offset12b_pos__amdgpu_no_fin
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v3
; GFX11-TRUE16-NEXT: v_mov_b32_e32 v3, v2
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB58_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -17170,7 +17282,7 @@ define void @global_agent_atomic_fmax_noret_v2bf16__offset12b_pos__amdgpu_no_fin
; GFX11-FAKE16-NEXT: v_add3_u32 v7, v7, v2, 0x7fff
; GFX11-FAKE16-NEXT: v_add3_u32 v8, v8, v6, 0x7fff
; GFX11-FAKE16-NEXT: v_cmp_u_f32_e64 s0, v2, v2
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_4)
; GFX11-FAKE16-NEXT: v_cndmask_b32_e32 v6, v8, v10, vcc_lo
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v2, v7, v9, s0
; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
@@ -17183,7 +17295,7 @@ define void @global_agent_atomic_fmax_noret_v2bf16__offset12b_pos__amdgpu_no_fin
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v3
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v3, v2
; GFX11-FAKE16-NEXT: s_or_b32 s1, vcc_lo, s1
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB58_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -17494,6 +17606,7 @@ define void @global_agent_atomic_fmax_noret_v2bf16__offset12b_neg__amdgpu_no_fin
; GFX12-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_cbranch_execnz .LBB59_1
; GFX12-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -17528,11 +17641,10 @@ define void @global_agent_atomic_fmax_noret_v2bf16__offset12b_neg__amdgpu_no_fin
; GFX12-FAKE16-NEXT: v_add3_u32 v8, v8, v6, 0x7fff
; GFX12-FAKE16-NEXT: v_cmp_u_f32_e64 s0, v2, v2
; GFX12-FAKE16-NEXT: s_wait_alu depctr_va_vcc(0)
-; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_2)
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_2) | instid1(VALU_DEP_1)
; GFX12-FAKE16-NEXT: v_cndmask_b32_e32 v6, v8, v10, vcc_lo
; GFX12-FAKE16-NEXT: s_wait_alu depctr_va_sdst(0)
; GFX12-FAKE16-NEXT: v_cndmask_b32_e64 v2, v7, v9, s0
-; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-FAKE16-NEXT: v_perm_b32 v2, v6, v2, 0x7060302
; GFX12-FAKE16-NEXT: s_wait_storecnt 0x0
; GFX12-FAKE16-NEXT: global_atomic_cmpswap_b32 v2, v[0:1], v[2:3], off offset:-2048 th:TH_ATOMIC_RETURN scope:SCOPE_DEV
@@ -17544,6 +17656,7 @@ define void @global_agent_atomic_fmax_noret_v2bf16__offset12b_neg__amdgpu_no_fin
; GFX12-FAKE16-NEXT: s_or_b32 s1, vcc_lo, s1
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: s_cbranch_execnz .LBB59_1
; GFX12-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s1
@@ -17633,7 +17746,7 @@ define void @global_agent_atomic_fmax_noret_v2bf16__offset12b_neg__amdgpu_no_fin
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v3
; GFX11-TRUE16-NEXT: v_mov_b32_e32 v3, v2
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB59_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -17667,7 +17780,7 @@ define void @global_agent_atomic_fmax_noret_v2bf16__offset12b_neg__amdgpu_no_fin
; GFX11-FAKE16-NEXT: v_add3_u32 v7, v7, v2, 0x7fff
; GFX11-FAKE16-NEXT: v_add3_u32 v8, v8, v6, 0x7fff
; GFX11-FAKE16-NEXT: v_cmp_u_f32_e64 s0, v2, v2
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_4)
; GFX11-FAKE16-NEXT: v_cndmask_b32_e32 v6, v8, v10, vcc_lo
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v2, v7, v9, s0
; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
@@ -17680,7 +17793,7 @@ define void @global_agent_atomic_fmax_noret_v2bf16__offset12b_neg__amdgpu_no_fin
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v3
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v3, v2
; GFX11-FAKE16-NEXT: s_or_b32 s1, vcc_lo, s1
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB59_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -18002,9 +18115,11 @@ define <2 x bfloat> @global_system_atomic_fmax_ret_v2bf16__offset12b_pos__amdgpu
; GFX12-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_cbranch_execnz .LBB60_1
; GFX12-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: v_mov_b32_e32 v0, v3
; GFX12-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -18054,9 +18169,11 @@ define <2 x bfloat> @global_system_atomic_fmax_ret_v2bf16__offset12b_pos__amdgpu
; GFX12-FAKE16-NEXT: s_or_b32 s1, vcc_lo, s1
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: s_cbranch_execnz .LBB60_1
; GFX12-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s1
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: v_mov_b32_e32 v0, v3
; GFX12-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -18146,12 +18263,13 @@ define <2 x bfloat> @global_system_atomic_fmax_ret_v2bf16__offset12b_pos__amdgpu
; GFX11-TRUE16-NEXT: buffer_gl0_inv
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v6
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB60_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-TRUE16-NEXT: s_set_inst_prefetch_distance 0x2
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_mov_b32_e32 v0, v3
; GFX11-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -18195,12 +18313,13 @@ define <2 x bfloat> @global_system_atomic_fmax_ret_v2bf16__offset12b_pos__amdgpu
; GFX11-FAKE16-NEXT: buffer_gl0_inv
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v6
; GFX11-FAKE16-NEXT: s_or_b32 s1, vcc_lo, s1
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB60_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-FAKE16-NEXT: s_set_inst_prefetch_distance 0x2
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s1
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v0, v3
; GFX11-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -18522,6 +18641,7 @@ define void @global_system_atomic_fmax_noret_v2bf16__offset12b_pos__amdgpu_no_fi
; GFX12-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_cbranch_execnz .LBB61_1
; GFX12-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -18556,11 +18676,10 @@ define void @global_system_atomic_fmax_noret_v2bf16__offset12b_pos__amdgpu_no_fi
; GFX12-FAKE16-NEXT: v_add3_u32 v8, v8, v6, 0x7fff
; GFX12-FAKE16-NEXT: v_cmp_u_f32_e64 s0, v2, v2
; GFX12-FAKE16-NEXT: s_wait_alu depctr_va_vcc(0)
-; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_2)
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_2) | instid1(VALU_DEP_1)
; GFX12-FAKE16-NEXT: v_cndmask_b32_e32 v6, v8, v10, vcc_lo
; GFX12-FAKE16-NEXT: s_wait_alu depctr_va_sdst(0)
; GFX12-FAKE16-NEXT: v_cndmask_b32_e64 v2, v7, v9, s0
-; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-FAKE16-NEXT: v_perm_b32 v2, v6, v2, 0x7060302
; GFX12-FAKE16-NEXT: global_wb scope:SCOPE_SYS
; GFX12-FAKE16-NEXT: s_wait_storecnt 0x0
@@ -18573,6 +18692,7 @@ define void @global_system_atomic_fmax_noret_v2bf16__offset12b_pos__amdgpu_no_fi
; GFX12-FAKE16-NEXT: s_or_b32 s1, vcc_lo, s1
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: s_cbranch_execnz .LBB61_1
; GFX12-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s1
@@ -18662,7 +18782,7 @@ define void @global_system_atomic_fmax_noret_v2bf16__offset12b_pos__amdgpu_no_fi
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v3
; GFX11-TRUE16-NEXT: v_mov_b32_e32 v3, v2
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB61_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -18696,7 +18816,7 @@ define void @global_system_atomic_fmax_noret_v2bf16__offset12b_pos__amdgpu_no_fi
; GFX11-FAKE16-NEXT: v_add3_u32 v7, v7, v2, 0x7fff
; GFX11-FAKE16-NEXT: v_add3_u32 v8, v8, v6, 0x7fff
; GFX11-FAKE16-NEXT: v_cmp_u_f32_e64 s0, v2, v2
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_4)
; GFX11-FAKE16-NEXT: v_cndmask_b32_e32 v6, v8, v10, vcc_lo
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v2, v7, v9, s0
; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
@@ -18709,7 +18829,7 @@ define void @global_system_atomic_fmax_noret_v2bf16__offset12b_pos__amdgpu_no_fi
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v3
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v3, v2
; GFX11-FAKE16-NEXT: s_or_b32 s1, vcc_lo, s1
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB61_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
diff --git a/llvm/test/CodeGen/AMDGPU/global-atomicrmw-fmin.ll b/llvm/test/CodeGen/AMDGPU/global-atomicrmw-fmin.ll
index 1510e0efeeee23..ebc2144befc876 100644
--- a/llvm/test/CodeGen/AMDGPU/global-atomicrmw-fmin.ll
+++ b/llvm/test/CodeGen/AMDGPU/global-atomicrmw-fmin.ll
@@ -1370,11 +1370,12 @@ define float @global_agent_atomic_fmin_ret_f32__amdgpu_no_remote_memory(ptr addr
; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB8_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: v_mov_b32_e32 v0, v3
; GFX11-NEXT: s_setpc_b64 s[30:31]
;
@@ -3033,9 +3034,11 @@ define double @global_agent_atomic_fmin_ret_f64__amdgpu_no_fine_grained_memory(p
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB18_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: v_dual_mov_b32 v0, v4 :: v_dual_mov_b32 v1, v5
; GFX12-NEXT: s_setpc_b64 s[30:31]
;
@@ -3069,11 +3072,12 @@ define double @global_agent_atomic_fmin_ret_f64__amdgpu_no_fine_grained_memory(p
; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: v_cmp_eq_u64_e32 vcc_lo, v[4:5], v[6:7]
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB18_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: v_dual_mov_b32 v0, v4 :: v_dual_mov_b32 v1, v5
; GFX11-NEXT: s_setpc_b64 s[30:31]
;
@@ -3206,9 +3210,11 @@ define double @global_agent_atomic_fmin_ret_f64__offset12b_pos__amdgpu_no_fine_g
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB19_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: v_dual_mov_b32 v0, v4 :: v_dual_mov_b32 v1, v5
; GFX12-NEXT: s_setpc_b64 s[30:31]
;
@@ -3242,11 +3248,12 @@ define double @global_agent_atomic_fmin_ret_f64__offset12b_pos__amdgpu_no_fine_g
; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: v_cmp_eq_u64_e32 vcc_lo, v[4:5], v[6:7]
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB19_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: v_dual_mov_b32 v0, v4 :: v_dual_mov_b32 v1, v5
; GFX11-NEXT: s_setpc_b64 s[30:31]
;
@@ -3380,9 +3387,11 @@ define double @global_agent_atomic_fmin_ret_f64__offset12b_neg__amdgpu_no_fine_g
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB20_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: v_dual_mov_b32 v0, v4 :: v_dual_mov_b32 v1, v5
; GFX12-NEXT: s_setpc_b64 s[30:31]
;
@@ -3416,11 +3425,12 @@ define double @global_agent_atomic_fmin_ret_f64__offset12b_neg__amdgpu_no_fine_g
; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: v_cmp_eq_u64_e32 vcc_lo, v[4:5], v[6:7]
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB20_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: v_dual_mov_b32 v0, v4 :: v_dual_mov_b32 v1, v5
; GFX11-NEXT: s_setpc_b64 s[30:31]
;
@@ -3554,6 +3564,7 @@ define void @global_agent_atomic_fmin_noret_f64__amdgpu_no_fine_grained_memory(p
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB21_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -3589,7 +3600,7 @@ define void @global_agent_atomic_fmin_noret_f64__amdgpu_no_fine_grained_memory(p
; GFX11-NEXT: v_cmp_eq_u64_e32 vcc_lo, v[2:3], v[4:5]
; GFX11-NEXT: v_dual_mov_b32 v5, v3 :: v_dual_mov_b32 v4, v2
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB21_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -3717,6 +3728,7 @@ define void @global_agent_atomic_fmin_noret_f64__offset12b_pos__amdgpu_no_fine_g
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB22_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -3752,7 +3764,7 @@ define void @global_agent_atomic_fmin_noret_f64__offset12b_pos__amdgpu_no_fine_g
; GFX11-NEXT: v_cmp_eq_u64_e32 vcc_lo, v[2:3], v[4:5]
; GFX11-NEXT: v_dual_mov_b32 v5, v3 :: v_dual_mov_b32 v4, v2
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB22_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -3883,6 +3895,7 @@ define void @global_agent_atomic_fmin_noret_f64__offset12b_neg__amdgpu_no_fine_g
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB23_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -3918,7 +3931,7 @@ define void @global_agent_atomic_fmin_noret_f64__offset12b_neg__amdgpu_no_fine_g
; GFX11-NEXT: v_cmp_eq_u64_e32 vcc_lo, v[2:3], v[4:5]
; GFX11-NEXT: v_dual_mov_b32 v5, v3 :: v_dual_mov_b32 v4, v2
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB23_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -4049,9 +4062,11 @@ define double @global_agent_atomic_fmin_ret_f64__amdgpu_no_remote_memory(ptr add
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB24_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: v_dual_mov_b32 v0, v4 :: v_dual_mov_b32 v1, v5
; GFX12-NEXT: s_setpc_b64 s[30:31]
;
@@ -4085,11 +4100,12 @@ define double @global_agent_atomic_fmin_ret_f64__amdgpu_no_remote_memory(ptr add
; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: v_cmp_eq_u64_e32 vcc_lo, v[4:5], v[6:7]
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB24_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: v_dual_mov_b32 v0, v4 :: v_dual_mov_b32 v1, v5
; GFX11-NEXT: s_setpc_b64 s[30:31]
;
@@ -4297,9 +4313,11 @@ define double @global_agent_atomic_fmin_ret_f64__amdgpu_no_fine_grained_memory__
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB25_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: v_dual_mov_b32 v0, v4 :: v_dual_mov_b32 v1, v5
; GFX12-NEXT: s_setpc_b64 s[30:31]
;
@@ -4333,11 +4351,12 @@ define double @global_agent_atomic_fmin_ret_f64__amdgpu_no_fine_grained_memory__
; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: v_cmp_eq_u64_e32 vcc_lo, v[4:5], v[6:7]
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB25_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: v_dual_mov_b32 v0, v4 :: v_dual_mov_b32 v1, v5
; GFX11-NEXT: s_setpc_b64 s[30:31]
;
@@ -4488,9 +4507,11 @@ define half @global_agent_atomic_fmin_ret_f16__amdgpu_no_fine_grained_memory(ptr
; GFX12-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_cbranch_execnz .LBB26_1
; GFX12-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: v_lshrrev_b32_e32 v0, v3, v5
; GFX12-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -4534,9 +4555,11 @@ define half @global_agent_atomic_fmin_ret_f16__amdgpu_no_fine_grained_memory(ptr
; GFX12-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: s_cbranch_execnz .LBB26_1
; GFX12-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: v_lshrrev_b32_e32 v0, v3, v5
; GFX12-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -4610,11 +4633,12 @@ define half @global_agent_atomic_fmin_ret_f16__amdgpu_no_fine_grained_memory(ptr
; GFX11-TRUE16-NEXT: buffer_gl0_inv
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v5, v6
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB26_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v0, v3, v5
; GFX11-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -4652,11 +4676,12 @@ define half @global_agent_atomic_fmin_ret_f16__amdgpu_no_fine_grained_memory(ptr
; GFX11-FAKE16-NEXT: buffer_gl0_inv
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v5, v6
; GFX11-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB26_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v0, v3, v5
; GFX11-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -4928,9 +4953,11 @@ define half @global_agent_atomic_fmin_ret_f16__offset12b_pos__amdgpu_no_fine_gra
; GFX12-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_cbranch_execnz .LBB27_1
; GFX12-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: v_lshrrev_b32_e32 v0, v1, v5
; GFX12-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -4976,9 +5003,11 @@ define half @global_agent_atomic_fmin_ret_f16__offset12b_pos__amdgpu_no_fine_gra
; GFX12-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: s_cbranch_execnz .LBB27_1
; GFX12-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: v_lshrrev_b32_e32 v0, v3, v5
; GFX12-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -5024,7 +5053,6 @@ define half @global_agent_atomic_fmin_ret_f16__offset12b_pos__amdgpu_no_fine_gra
; GFX11-TRUE16: ; %bb.0:
; GFX11-TRUE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-TRUE16-NEXT: v_add_co_u32 v0, vcc_lo, 0x7fe, v0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v4, null, 0, v1, vcc_lo
; GFX11-TRUE16-NEXT: s_mov_b32 s0, 0
; GFX11-TRUE16-NEXT: v_and_b32_e32 v3, -4, v0
@@ -5057,11 +5085,12 @@ define half @global_agent_atomic_fmin_ret_f16__offset12b_pos__amdgpu_no_fine_gra
; GFX11-TRUE16-NEXT: buffer_gl0_inv
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v5, v6
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB27_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v0, v1, v5
; GFX11-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -5069,16 +5098,16 @@ define half @global_agent_atomic_fmin_ret_f16__offset12b_pos__amdgpu_no_fine_gra
; GFX11-FAKE16: ; %bb.0:
; GFX11-FAKE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-FAKE16-NEXT: v_add_co_u32 v3, vcc_lo, 0x7fe, v0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-FAKE16-NEXT: v_max_f16_e32 v2, v2, v2
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_4) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_and_b32_e32 v0, -4, v3
; GFX11-FAKE16-NEXT: v_and_b32_e32 v3, 3, v3
; GFX11-FAKE16-NEXT: s_mov_b32 s0, 0
; GFX11-FAKE16-NEXT: global_load_b32 v5, v[0:1], off
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v3, 3, v3
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e64 v4, v3, 0xffff
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_not_b32_e32 v4, v4
; GFX11-FAKE16-NEXT: .LBB27_1: ; %atomicrmw.start
; GFX11-FAKE16-NEXT: ; =>This Inner Loop Header: Depth=1
@@ -5100,11 +5129,12 @@ define half @global_agent_atomic_fmin_ret_f16__offset12b_pos__amdgpu_no_fine_gra
; GFX11-FAKE16-NEXT: buffer_gl0_inv
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v5, v6
; GFX11-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB27_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v0, v3, v5
; GFX11-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -5384,9 +5414,11 @@ define half @global_agent_atomic_fmin_ret_f16__offset12b_neg__amdgpu_no_fine_gra
; GFX12-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_cbranch_execnz .LBB28_1
; GFX12-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: v_lshrrev_b32_e32 v0, v1, v5
; GFX12-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -5432,9 +5464,11 @@ define half @global_agent_atomic_fmin_ret_f16__offset12b_neg__amdgpu_no_fine_gra
; GFX12-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: s_cbranch_execnz .LBB28_1
; GFX12-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: v_lshrrev_b32_e32 v0, v3, v5
; GFX12-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -5481,7 +5515,6 @@ define half @global_agent_atomic_fmin_ret_f16__offset12b_neg__amdgpu_no_fine_gra
; GFX11-TRUE16: ; %bb.0:
; GFX11-TRUE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-TRUE16-NEXT: v_add_co_u32 v0, vcc_lo, 0xfffff800, v0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v4, null, -1, v1, vcc_lo
; GFX11-TRUE16-NEXT: s_mov_b32 s0, 0
; GFX11-TRUE16-NEXT: v_and_b32_e32 v3, -4, v0
@@ -5514,11 +5547,12 @@ define half @global_agent_atomic_fmin_ret_f16__offset12b_neg__amdgpu_no_fine_gra
; GFX11-TRUE16-NEXT: buffer_gl0_inv
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v5, v6
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB28_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v0, v1, v5
; GFX11-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -5526,16 +5560,16 @@ define half @global_agent_atomic_fmin_ret_f16__offset12b_neg__amdgpu_no_fine_gra
; GFX11-FAKE16: ; %bb.0:
; GFX11-FAKE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-FAKE16-NEXT: v_add_co_u32 v3, vcc_lo, 0xfffff800, v0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v1, null, -1, v1, vcc_lo
; GFX11-FAKE16-NEXT: v_max_f16_e32 v2, v2, v2
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_4) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_and_b32_e32 v0, -4, v3
; GFX11-FAKE16-NEXT: v_and_b32_e32 v3, 3, v3
; GFX11-FAKE16-NEXT: s_mov_b32 s0, 0
; GFX11-FAKE16-NEXT: global_load_b32 v5, v[0:1], off
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v3, 3, v3
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e64 v4, v3, 0xffff
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_not_b32_e32 v4, v4
; GFX11-FAKE16-NEXT: .LBB28_1: ; %atomicrmw.start
; GFX11-FAKE16-NEXT: ; =>This Inner Loop Header: Depth=1
@@ -5557,11 +5591,12 @@ define half @global_agent_atomic_fmin_ret_f16__offset12b_neg__amdgpu_no_fine_gra
; GFX11-FAKE16-NEXT: buffer_gl0_inv
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v5, v6
; GFX11-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB28_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v0, v3, v5
; GFX11-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -5838,6 +5873,7 @@ define void @global_agent_atomic_fmin_noret_f16__amdgpu_no_fine_grained_memory(p
; GFX12-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_cbranch_execnz .LBB29_1
; GFX12-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -5883,6 +5919,7 @@ define void @global_agent_atomic_fmin_noret_f16__amdgpu_no_fine_grained_memory(p
; GFX12-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: s_cbranch_execnz .LBB29_1
; GFX12-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -5957,7 +5994,7 @@ define void @global_agent_atomic_fmin_noret_f16__amdgpu_no_fine_grained_memory(p
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-TRUE16-NEXT: v_mov_b32_e32 v4, v3
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB29_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -5998,7 +6035,7 @@ define void @global_agent_atomic_fmin_noret_f16__amdgpu_no_fine_grained_memory(p
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v4, v3
; GFX11-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB29_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -6268,6 +6305,7 @@ define void @global_agent_atomic_fmin_noret_f16__offset12b_pos__amdgpu_no_fine_g
; GFX12-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_cbranch_execnz .LBB30_1
; GFX12-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -6315,6 +6353,7 @@ define void @global_agent_atomic_fmin_noret_f16__offset12b_pos__amdgpu_no_fine_g
; GFX12-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: s_cbranch_execnz .LBB30_1
; GFX12-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -6361,7 +6400,6 @@ define void @global_agent_atomic_fmin_noret_f16__offset12b_pos__amdgpu_no_fine_g
; GFX11-TRUE16: ; %bb.0:
; GFX11-TRUE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-TRUE16-NEXT: v_add_co_u32 v0, vcc_lo, 0x7fe, v0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v4, null, 0, v1, vcc_lo
; GFX11-TRUE16-NEXT: s_mov_b32 s0, 0
; GFX11-TRUE16-NEXT: v_and_b32_e32 v3, -4, v0
@@ -6394,7 +6432,7 @@ define void @global_agent_atomic_fmin_noret_f16__offset12b_pos__amdgpu_no_fine_g
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v5, v6
; GFX11-TRUE16-NEXT: v_mov_b32_e32 v6, v5
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB30_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -6405,16 +6443,16 @@ define void @global_agent_atomic_fmin_noret_f16__offset12b_pos__amdgpu_no_fine_g
; GFX11-FAKE16: ; %bb.0:
; GFX11-FAKE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-FAKE16-NEXT: v_add_co_u32 v4, vcc_lo, 0x7fe, v0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-FAKE16-NEXT: v_max_f16_e32 v6, v2, v2
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_4) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_and_b32_e32 v0, -4, v4
; GFX11-FAKE16-NEXT: v_and_b32_e32 v4, 3, v4
; GFX11-FAKE16-NEXT: s_mov_b32 s0, 0
; GFX11-FAKE16-NEXT: global_load_b32 v3, v[0:1], off
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v4, 3, v4
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e64 v5, v4, 0xffff
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_not_b32_e32 v5, v5
; GFX11-FAKE16-NEXT: .LBB30_1: ; %atomicrmw.start
; GFX11-FAKE16-NEXT: ; =>This Inner Loop Header: Depth=1
@@ -6436,7 +6474,7 @@ define void @global_agent_atomic_fmin_noret_f16__offset12b_pos__amdgpu_no_fine_g
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v3
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v3, v2
; GFX11-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB30_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -6713,6 +6751,7 @@ define void @global_agent_atomic_fmin_noret_f16__offset12b_neg__amdgpu_no_fine_g
; GFX12-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_cbranch_execnz .LBB31_1
; GFX12-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -6760,6 +6799,7 @@ define void @global_agent_atomic_fmin_noret_f16__offset12b_neg__amdgpu_no_fine_g
; GFX12-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: s_cbranch_execnz .LBB31_1
; GFX12-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -6807,7 +6847,6 @@ define void @global_agent_atomic_fmin_noret_f16__offset12b_neg__amdgpu_no_fine_g
; GFX11-TRUE16: ; %bb.0:
; GFX11-TRUE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-TRUE16-NEXT: v_add_co_u32 v0, vcc_lo, 0xfffff800, v0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v4, null, -1, v1, vcc_lo
; GFX11-TRUE16-NEXT: s_mov_b32 s0, 0
; GFX11-TRUE16-NEXT: v_and_b32_e32 v3, -4, v0
@@ -6840,7 +6879,7 @@ define void @global_agent_atomic_fmin_noret_f16__offset12b_neg__amdgpu_no_fine_g
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v5, v6
; GFX11-TRUE16-NEXT: v_mov_b32_e32 v6, v5
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB31_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -6851,16 +6890,16 @@ define void @global_agent_atomic_fmin_noret_f16__offset12b_neg__amdgpu_no_fine_g
; GFX11-FAKE16: ; %bb.0:
; GFX11-FAKE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-FAKE16-NEXT: v_add_co_u32 v4, vcc_lo, 0xfffff800, v0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v1, null, -1, v1, vcc_lo
; GFX11-FAKE16-NEXT: v_max_f16_e32 v6, v2, v2
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_4) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_and_b32_e32 v0, -4, v4
; GFX11-FAKE16-NEXT: v_and_b32_e32 v4, 3, v4
; GFX11-FAKE16-NEXT: s_mov_b32 s0, 0
; GFX11-FAKE16-NEXT: global_load_b32 v3, v[0:1], off
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v4, 3, v4
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e64 v5, v4, 0xffff
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_not_b32_e32 v5, v5
; GFX11-FAKE16-NEXT: .LBB31_1: ; %atomicrmw.start
; GFX11-FAKE16-NEXT: ; =>This Inner Loop Header: Depth=1
@@ -6882,7 +6921,7 @@ define void @global_agent_atomic_fmin_noret_f16__offset12b_neg__amdgpu_no_fine_g
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v3
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v3, v2
; GFX11-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB31_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -7145,9 +7184,11 @@ define half @global_agent_atomic_fmin_ret_f16__offset12b_pos__align4__amdgpu_no_
; GFX12-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_cbranch_execnz .LBB32_1
; GFX12-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: v_mov_b16_e32 v0.l, v3.l
; GFX12-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -7180,9 +7221,11 @@ define half @global_agent_atomic_fmin_ret_f16__offset12b_pos__align4__amdgpu_no_
; GFX12-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: s_cbranch_execnz .LBB32_1
; GFX12-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: v_mov_b32_e32 v0, v3
; GFX12-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -7237,11 +7280,12 @@ define half @global_agent_atomic_fmin_ret_f16__offset12b_pos__align4__amdgpu_no_
; GFX11-TRUE16-NEXT: buffer_gl0_inv
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB32_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v0.l, v3.l
; GFX11-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -7268,11 +7312,12 @@ define half @global_agent_atomic_fmin_ret_f16__offset12b_pos__align4__amdgpu_no_
; GFX11-FAKE16-NEXT: buffer_gl0_inv
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB32_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v0, v3
; GFX11-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -7485,6 +7530,7 @@ define void @global_agent_atomic_fmin_noret_f16__offset12b__align4_pos__amdgpu_n
; GFX12-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_cbranch_execnz .LBB33_1
; GFX12-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -7519,6 +7565,7 @@ define void @global_agent_atomic_fmin_noret_f16__offset12b__align4_pos__amdgpu_n
; GFX12-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: s_cbranch_execnz .LBB33_1
; GFX12-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -7574,7 +7621,7 @@ define void @global_agent_atomic_fmin_noret_f16__offset12b__align4_pos__amdgpu_n
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-TRUE16-NEXT: v_mov_b32_e32 v4, v3
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB33_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -7604,7 +7651,7 @@ define void @global_agent_atomic_fmin_noret_f16__offset12b__align4_pos__amdgpu_n
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v3
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v3, v2
; GFX11-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB33_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -7830,9 +7877,11 @@ define half @global_system_atomic_fmin_ret_f16__offset12b_pos__amdgpu_no_fine_gr
; GFX12-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_cbranch_execnz .LBB34_1
; GFX12-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: v_lshrrev_b32_e32 v0, v1, v5
; GFX12-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -7879,9 +7928,11 @@ define half @global_system_atomic_fmin_ret_f16__offset12b_pos__amdgpu_no_fine_gr
; GFX12-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: s_cbranch_execnz .LBB34_1
; GFX12-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: v_lshrrev_b32_e32 v0, v3, v5
; GFX12-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -7927,7 +7978,6 @@ define half @global_system_atomic_fmin_ret_f16__offset12b_pos__amdgpu_no_fine_gr
; GFX11-TRUE16: ; %bb.0:
; GFX11-TRUE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-TRUE16-NEXT: v_add_co_u32 v0, vcc_lo, 0x7fe, v0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v4, null, 0, v1, vcc_lo
; GFX11-TRUE16-NEXT: s_mov_b32 s0, 0
; GFX11-TRUE16-NEXT: v_and_b32_e32 v3, -4, v0
@@ -7960,11 +8010,12 @@ define half @global_system_atomic_fmin_ret_f16__offset12b_pos__amdgpu_no_fine_gr
; GFX11-TRUE16-NEXT: buffer_gl0_inv
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v5, v6
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB34_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v0, v1, v5
; GFX11-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -7972,16 +8023,16 @@ define half @global_system_atomic_fmin_ret_f16__offset12b_pos__amdgpu_no_fine_gr
; GFX11-FAKE16: ; %bb.0:
; GFX11-FAKE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-FAKE16-NEXT: v_add_co_u32 v3, vcc_lo, 0x7fe, v0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-FAKE16-NEXT: v_max_f16_e32 v2, v2, v2
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_4) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_and_b32_e32 v0, -4, v3
; GFX11-FAKE16-NEXT: v_and_b32_e32 v3, 3, v3
; GFX11-FAKE16-NEXT: s_mov_b32 s0, 0
; GFX11-FAKE16-NEXT: global_load_b32 v5, v[0:1], off
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v3, 3, v3
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e64 v4, v3, 0xffff
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_not_b32_e32 v4, v4
; GFX11-FAKE16-NEXT: .LBB34_1: ; %atomicrmw.start
; GFX11-FAKE16-NEXT: ; =>This Inner Loop Header: Depth=1
@@ -8003,11 +8054,12 @@ define half @global_system_atomic_fmin_ret_f16__offset12b_pos__amdgpu_no_fine_gr
; GFX11-FAKE16-NEXT: buffer_gl0_inv
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v5, v6
; GFX11-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB34_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v0, v3, v5
; GFX11-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -8291,6 +8343,7 @@ define void @global_system_atomic_fmin_noret_f16__offset12b_pos__amdgpu_no_fine_
; GFX12-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_cbranch_execnz .LBB35_1
; GFX12-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -8339,6 +8392,7 @@ define void @global_system_atomic_fmin_noret_f16__offset12b_pos__amdgpu_no_fine_
; GFX12-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: s_cbranch_execnz .LBB35_1
; GFX12-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -8385,7 +8439,6 @@ define void @global_system_atomic_fmin_noret_f16__offset12b_pos__amdgpu_no_fine_
; GFX11-TRUE16: ; %bb.0:
; GFX11-TRUE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-TRUE16-NEXT: v_add_co_u32 v0, vcc_lo, 0x7fe, v0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v4, null, 0, v1, vcc_lo
; GFX11-TRUE16-NEXT: s_mov_b32 s0, 0
; GFX11-TRUE16-NEXT: v_and_b32_e32 v3, -4, v0
@@ -8418,7 +8471,7 @@ define void @global_system_atomic_fmin_noret_f16__offset12b_pos__amdgpu_no_fine_
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v5, v6
; GFX11-TRUE16-NEXT: v_mov_b32_e32 v6, v5
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB35_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -8429,16 +8482,16 @@ define void @global_system_atomic_fmin_noret_f16__offset12b_pos__amdgpu_no_fine_
; GFX11-FAKE16: ; %bb.0:
; GFX11-FAKE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-FAKE16-NEXT: v_add_co_u32 v4, vcc_lo, 0x7fe, v0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-FAKE16-NEXT: v_max_f16_e32 v6, v2, v2
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_4) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_and_b32_e32 v0, -4, v4
; GFX11-FAKE16-NEXT: v_and_b32_e32 v4, 3, v4
; GFX11-FAKE16-NEXT: s_mov_b32 s0, 0
; GFX11-FAKE16-NEXT: global_load_b32 v3, v[0:1], off
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v4, 3, v4
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e64 v5, v4, 0xffff
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_not_b32_e32 v5, v5
; GFX11-FAKE16-NEXT: .LBB35_1: ; %atomicrmw.start
; GFX11-FAKE16-NEXT: ; =>This Inner Loop Header: Depth=1
@@ -8460,7 +8513,7 @@ define void @global_system_atomic_fmin_noret_f16__offset12b_pos__amdgpu_no_fine_
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v3
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v3, v2
; GFX11-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB35_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -8748,9 +8801,11 @@ define bfloat @global_agent_atomic_fmin_ret_bf16__amdgpu_no_fine_grained_memory(
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB36_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: v_lshrrev_b32_e32 v0, v3, v5
; GFX12-NEXT: s_setpc_b64 s[30:31]
;
@@ -8838,11 +8893,12 @@ define bfloat @global_agent_atomic_fmin_ret_bf16__amdgpu_no_fine_grained_memory(
; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v5, v6
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB36_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: v_lshrrev_b32_e32 v0, v3, v5
; GFX11-NEXT: s_setpc_b64 s[30:31]
;
@@ -9146,9 +9202,11 @@ define bfloat @global_agent_atomic_fmin_ret_bf16__offset12b_pos__amdgpu_no_fine_
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB37_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: v_lshrrev_b32_e32 v0, v3, v5
; GFX12-NEXT: s_setpc_b64 s[30:31]
;
@@ -9201,16 +9259,16 @@ define bfloat @global_agent_atomic_fmin_ret_bf16__offset12b_pos__amdgpu_no_fine_
; GFX11: ; %bb.0:
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: v_add_co_u32 v3, vcc_lo, 0x7fe, v0
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_3)
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-NEXT: v_lshlrev_b32_e32 v2, 16, v2
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_4) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_and_b32_e32 v0, -4, v3
; GFX11-NEXT: v_and_b32_e32 v3, 3, v3
; GFX11-NEXT: s_mov_b32 s0, 0
; GFX11-NEXT: global_load_b32 v5, v[0:1], off
; GFX11-NEXT: v_lshlrev_b32_e32 v3, 3, v3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_lshlrev_b32_e64 v4, v3, 0xffff
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_not_b32_e32 v4, v4
; GFX11-NEXT: .p2align 6
; GFX11-NEXT: .LBB37_1: ; %atomicrmw.start
@@ -9240,11 +9298,12 @@ define bfloat @global_agent_atomic_fmin_ret_bf16__offset12b_pos__amdgpu_no_fine_
; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v5, v6
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB37_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: v_lshrrev_b32_e32 v0, v3, v5
; GFX11-NEXT: s_setpc_b64 s[30:31]
;
@@ -9556,9 +9615,11 @@ define bfloat @global_agent_atomic_fmin_ret_bf16__offset12b_neg__amdgpu_no_fine_
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB38_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: v_lshrrev_b32_e32 v0, v3, v5
; GFX12-NEXT: s_setpc_b64 s[30:31]
;
@@ -9612,16 +9673,16 @@ define bfloat @global_agent_atomic_fmin_ret_bf16__offset12b_neg__amdgpu_no_fine_
; GFX11: ; %bb.0:
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: v_add_co_u32 v3, vcc_lo, 0xfffff800, v0
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_3)
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, -1, v1, vcc_lo
; GFX11-NEXT: v_lshlrev_b32_e32 v2, 16, v2
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_4) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_and_b32_e32 v0, -4, v3
; GFX11-NEXT: v_and_b32_e32 v3, 3, v3
; GFX11-NEXT: s_mov_b32 s0, 0
; GFX11-NEXT: global_load_b32 v5, v[0:1], off
; GFX11-NEXT: v_lshlrev_b32_e32 v3, 3, v3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_lshlrev_b32_e64 v4, v3, 0xffff
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_not_b32_e32 v4, v4
; GFX11-NEXT: .p2align 6
; GFX11-NEXT: .LBB38_1: ; %atomicrmw.start
@@ -9651,11 +9712,12 @@ define bfloat @global_agent_atomic_fmin_ret_bf16__offset12b_neg__amdgpu_no_fine_
; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v5, v6
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB38_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: v_lshrrev_b32_e32 v0, v3, v5
; GFX11-NEXT: s_setpc_b64 s[30:31]
;
@@ -9963,6 +10025,7 @@ define void @global_agent_atomic_fmin_noret_bf16__amdgpu_no_fine_grained_memory(
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB39_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -10050,7 +10113,7 @@ define void @global_agent_atomic_fmin_noret_bf16__amdgpu_no_fine_grained_memory(
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-NEXT: v_mov_b32_e32 v4, v3
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB39_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -10351,6 +10414,7 @@ define void @global_agent_atomic_fmin_noret_bf16__offset12b_pos__amdgpu_no_fine_
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB40_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -10404,16 +10468,16 @@ define void @global_agent_atomic_fmin_noret_bf16__offset12b_pos__amdgpu_no_fine_
; GFX11: ; %bb.0:
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: v_add_co_u32 v4, vcc_lo, 0x7fe, v0
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_3)
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-NEXT: v_lshlrev_b32_e32 v6, 16, v2
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_4) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_and_b32_e32 v0, -4, v4
; GFX11-NEXT: v_and_b32_e32 v4, 3, v4
; GFX11-NEXT: s_mov_b32 s0, 0
; GFX11-NEXT: global_load_b32 v3, v[0:1], off
; GFX11-NEXT: v_lshlrev_b32_e32 v4, 3, v4
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_lshlrev_b32_e64 v5, v4, 0xffff
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_not_b32_e32 v5, v5
; GFX11-NEXT: .p2align 6
; GFX11-NEXT: .LBB40_1: ; %atomicrmw.start
@@ -10442,7 +10506,7 @@ define void @global_agent_atomic_fmin_noret_bf16__offset12b_pos__amdgpu_no_fine_
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v3
; GFX11-NEXT: v_mov_b32_e32 v3, v2
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB40_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -10750,6 +10814,7 @@ define void @global_agent_atomic_fmin_noret_bf16__offset12b_neg__amdgpu_no_fine_
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB41_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -10804,16 +10869,16 @@ define void @global_agent_atomic_fmin_noret_bf16__offset12b_neg__amdgpu_no_fine_
; GFX11: ; %bb.0:
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: v_add_co_u32 v4, vcc_lo, 0xfffff800, v0
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_3)
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, -1, v1, vcc_lo
; GFX11-NEXT: v_lshlrev_b32_e32 v6, 16, v2
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_4) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_and_b32_e32 v0, -4, v4
; GFX11-NEXT: v_and_b32_e32 v4, 3, v4
; GFX11-NEXT: s_mov_b32 s0, 0
; GFX11-NEXT: global_load_b32 v3, v[0:1], off
; GFX11-NEXT: v_lshlrev_b32_e32 v4, 3, v4
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_lshlrev_b32_e64 v5, v4, 0xffff
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_not_b32_e32 v5, v5
; GFX11-NEXT: .p2align 6
; GFX11-NEXT: .LBB41_1: ; %atomicrmw.start
@@ -10842,7 +10907,7 @@ define void @global_agent_atomic_fmin_noret_bf16__offset12b_neg__amdgpu_no_fine_
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v3
; GFX11-NEXT: v_mov_b32_e32 v3, v2
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB41_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -11138,9 +11203,11 @@ define bfloat @global_agent_atomic_fmin_ret_bf16__offset12b_pos__align4__amdgpu_
; GFX12-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_cbranch_execnz .LBB42_1
; GFX12-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: v_mov_b16_e32 v0.l, v3.l
; GFX12-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -11181,9 +11248,11 @@ define bfloat @global_agent_atomic_fmin_ret_bf16__offset12b_pos__align4__amdgpu_
; GFX12-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: s_cbranch_execnz .LBB42_1
; GFX12-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: v_mov_b32_e32 v0, v3
; GFX12-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -11254,11 +11323,12 @@ define bfloat @global_agent_atomic_fmin_ret_bf16__offset12b_pos__align4__amdgpu_
; GFX11-TRUE16-NEXT: buffer_gl0_inv
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB42_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v0.l, v3.l
; GFX11-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -11293,11 +11363,12 @@ define bfloat @global_agent_atomic_fmin_ret_bf16__offset12b_pos__align4__amdgpu_
; GFX11-FAKE16-NEXT: buffer_gl0_inv
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB42_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v0, v3
; GFX11-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -11548,6 +11619,7 @@ define void @global_agent_atomic_fmin_noret_bf16__offset12b__align4_pos__amdgpu_
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB43_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -11618,7 +11690,7 @@ define void @global_agent_atomic_fmin_noret_bf16__offset12b__align4_pos__amdgpu_
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v3
; GFX11-NEXT: v_mov_b32_e32 v3, v2
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB43_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -11882,9 +11954,11 @@ define bfloat @global_system_atomic_fmin_ret_bf16__offset12b_pos__amdgpu_no_fine
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB44_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: v_lshrrev_b32_e32 v0, v3, v5
; GFX12-NEXT: s_setpc_b64 s[30:31]
;
@@ -11937,16 +12011,16 @@ define bfloat @global_system_atomic_fmin_ret_bf16__offset12b_pos__amdgpu_no_fine
; GFX11: ; %bb.0:
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: v_add_co_u32 v3, vcc_lo, 0x7fe, v0
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_3)
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-NEXT: v_lshlrev_b32_e32 v2, 16, v2
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_4) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_and_b32_e32 v0, -4, v3
; GFX11-NEXT: v_and_b32_e32 v3, 3, v3
; GFX11-NEXT: s_mov_b32 s0, 0
; GFX11-NEXT: global_load_b32 v5, v[0:1], off
; GFX11-NEXT: v_lshlrev_b32_e32 v3, 3, v3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_lshlrev_b32_e64 v4, v3, 0xffff
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_not_b32_e32 v4, v4
; GFX11-NEXT: .p2align 6
; GFX11-NEXT: .LBB44_1: ; %atomicrmw.start
@@ -11976,11 +12050,12 @@ define bfloat @global_system_atomic_fmin_ret_bf16__offset12b_pos__amdgpu_no_fine
; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v5, v6
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB44_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: v_lshrrev_b32_e32 v0, v3, v5
; GFX11-NEXT: s_setpc_b64 s[30:31]
;
@@ -12295,6 +12370,7 @@ define void @global_system_atomic_fmin_noret_bf16__offset12b_pos__amdgpu_no_fine
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB45_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -12348,16 +12424,16 @@ define void @global_system_atomic_fmin_noret_bf16__offset12b_pos__amdgpu_no_fine
; GFX11: ; %bb.0:
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: v_add_co_u32 v4, vcc_lo, 0x7fe, v0
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_3)
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-NEXT: v_lshlrev_b32_e32 v6, 16, v2
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_4) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_and_b32_e32 v0, -4, v4
; GFX11-NEXT: v_and_b32_e32 v4, 3, v4
; GFX11-NEXT: s_mov_b32 s0, 0
; GFX11-NEXT: global_load_b32 v3, v[0:1], off
; GFX11-NEXT: v_lshlrev_b32_e32 v4, 3, v4
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_lshlrev_b32_e64 v5, v4, 0xffff
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_not_b32_e32 v5, v5
; GFX11-NEXT: .p2align 6
; GFX11-NEXT: .LBB45_1: ; %atomicrmw.start
@@ -12386,7 +12462,7 @@ define void @global_system_atomic_fmin_noret_bf16__offset12b_pos__amdgpu_no_fine
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v3
; GFX11-NEXT: v_mov_b32_e32 v3, v2
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB45_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -12678,9 +12754,11 @@ define <2 x half> @global_agent_atomic_fmin_ret_v2f16__amdgpu_no_fine_grained_me
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB46_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: v_mov_b32_e32 v0, v3
; GFX12-NEXT: s_setpc_b64 s[30:31]
;
@@ -12731,11 +12809,12 @@ define <2 x half> @global_agent_atomic_fmin_ret_v2f16__amdgpu_no_fine_grained_me
; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB46_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: v_mov_b32_e32 v0, v3
; GFX11-NEXT: s_setpc_b64 s[30:31]
;
@@ -12960,9 +13039,11 @@ define <2 x half> @global_agent_atomic_fmin_ret_v2f16__offset12b_pos__amdgpu_no_
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB47_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: v_mov_b32_e32 v0, v3
; GFX12-NEXT: s_setpc_b64 s[30:31]
;
@@ -13013,11 +13094,12 @@ define <2 x half> @global_agent_atomic_fmin_ret_v2f16__offset12b_pos__amdgpu_no_
; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB47_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: v_mov_b32_e32 v0, v3
; GFX11-NEXT: s_setpc_b64 s[30:31]
;
@@ -13244,9 +13326,11 @@ define <2 x half> @global_agent_atomic_fmin_ret_v2f16__offset12b_neg__amdgpu_no_
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB48_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: v_mov_b32_e32 v0, v3
; GFX12-NEXT: s_setpc_b64 s[30:31]
;
@@ -13297,11 +13381,12 @@ define <2 x half> @global_agent_atomic_fmin_ret_v2f16__offset12b_neg__amdgpu_no_
; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB48_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: v_mov_b32_e32 v0, v3
; GFX11-NEXT: s_setpc_b64 s[30:31]
;
@@ -13536,6 +13621,7 @@ define void @global_agent_atomic_fmin_noret_v2f16__amdgpu_no_fine_grained_memory
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB49_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -13587,7 +13673,7 @@ define void @global_agent_atomic_fmin_noret_v2f16__amdgpu_no_fine_grained_memory
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v3
; GFX11-NEXT: v_mov_b32_e32 v3, v2
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB49_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -13805,6 +13891,7 @@ define void @global_agent_atomic_fmin_noret_v2f16__offset12b_pos__amdgpu_no_fine
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB50_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -13856,7 +13943,7 @@ define void @global_agent_atomic_fmin_noret_v2f16__offset12b_pos__amdgpu_no_fine
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v3
; GFX11-NEXT: v_mov_b32_e32 v3, v2
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB50_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -14077,6 +14164,7 @@ define void @global_agent_atomic_fmin_noret_v2f16__offset12b_neg__amdgpu_no_fine
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB51_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -14128,7 +14216,7 @@ define void @global_agent_atomic_fmin_noret_v2f16__offset12b_neg__amdgpu_no_fine
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v3
; GFX11-NEXT: v_mov_b32_e32 v3, v2
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB51_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -14358,9 +14446,11 @@ define <2 x half> @global_system_atomic_fmin_ret_v2f16__offset12b_pos__amdgpu_no
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB52_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: v_mov_b32_e32 v0, v3
; GFX12-NEXT: s_setpc_b64 s[30:31]
;
@@ -14411,11 +14501,12 @@ define <2 x half> @global_system_atomic_fmin_ret_v2f16__offset12b_pos__amdgpu_no
; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB52_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: v_mov_b32_e32 v0, v3
; GFX11-NEXT: s_setpc_b64 s[30:31]
;
@@ -14646,6 +14737,7 @@ define void @global_system_atomic_fmin_noret_v2f16__offset12b_pos__amdgpu_no_fin
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB53_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -14697,7 +14789,7 @@ define void @global_system_atomic_fmin_noret_v2f16__offset12b_pos__amdgpu_no_fin
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v3
; GFX11-NEXT: v_mov_b32_e32 v3, v2
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB53_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -14948,9 +15040,11 @@ define <2 x bfloat> @global_agent_atomic_fmin_ret_v2bf16__amdgpu_no_fine_grained
; GFX12-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_cbranch_execnz .LBB54_1
; GFX12-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: v_mov_b32_e32 v0, v3
; GFX12-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -14999,9 +15093,11 @@ define <2 x bfloat> @global_agent_atomic_fmin_ret_v2bf16__amdgpu_no_fine_grained
; GFX12-FAKE16-NEXT: s_or_b32 s1, vcc_lo, s1
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: s_cbranch_execnz .LBB54_1
; GFX12-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s1
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: v_mov_b32_e32 v0, v3
; GFX12-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -15091,12 +15187,13 @@ define <2 x bfloat> @global_agent_atomic_fmin_ret_v2bf16__amdgpu_no_fine_grained
; GFX11-TRUE16-NEXT: buffer_gl0_inv
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v6
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB54_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-TRUE16-NEXT: s_set_inst_prefetch_distance 0x2
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_mov_b32_e32 v0, v3
; GFX11-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -15140,12 +15237,13 @@ define <2 x bfloat> @global_agent_atomic_fmin_ret_v2bf16__amdgpu_no_fine_grained
; GFX11-FAKE16-NEXT: buffer_gl0_inv
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v6
; GFX11-FAKE16-NEXT: s_or_b32 s1, vcc_lo, s1
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB54_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-FAKE16-NEXT: s_set_inst_prefetch_distance 0x2
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s1
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v0, v3
; GFX11-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -15463,9 +15561,11 @@ define <2 x bfloat> @global_agent_atomic_fmin_ret_v2bf16__offset12b_pos__amdgpu_
; GFX12-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_cbranch_execnz .LBB55_1
; GFX12-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: v_mov_b32_e32 v0, v3
; GFX12-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -15514,9 +15614,11 @@ define <2 x bfloat> @global_agent_atomic_fmin_ret_v2bf16__offset12b_pos__amdgpu_
; GFX12-FAKE16-NEXT: s_or_b32 s1, vcc_lo, s1
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: s_cbranch_execnz .LBB55_1
; GFX12-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s1
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: v_mov_b32_e32 v0, v3
; GFX12-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -15606,12 +15708,13 @@ define <2 x bfloat> @global_agent_atomic_fmin_ret_v2bf16__offset12b_pos__amdgpu_
; GFX11-TRUE16-NEXT: buffer_gl0_inv
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v6
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB55_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-TRUE16-NEXT: s_set_inst_prefetch_distance 0x2
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_mov_b32_e32 v0, v3
; GFX11-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -15655,12 +15758,13 @@ define <2 x bfloat> @global_agent_atomic_fmin_ret_v2bf16__offset12b_pos__amdgpu_
; GFX11-FAKE16-NEXT: buffer_gl0_inv
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v6
; GFX11-FAKE16-NEXT: s_or_b32 s1, vcc_lo, s1
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB55_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-FAKE16-NEXT: s_set_inst_prefetch_distance 0x2
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s1
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v0, v3
; GFX11-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -15980,9 +16084,11 @@ define <2 x bfloat> @global_agent_atomic_fmin_ret_v2bf16__offset12b_neg__amdgpu_
; GFX12-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_cbranch_execnz .LBB56_1
; GFX12-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: v_mov_b32_e32 v0, v3
; GFX12-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -16031,9 +16137,11 @@ define <2 x bfloat> @global_agent_atomic_fmin_ret_v2bf16__offset12b_neg__amdgpu_
; GFX12-FAKE16-NEXT: s_or_b32 s1, vcc_lo, s1
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: s_cbranch_execnz .LBB56_1
; GFX12-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s1
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: v_mov_b32_e32 v0, v3
; GFX12-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -16123,12 +16231,13 @@ define <2 x bfloat> @global_agent_atomic_fmin_ret_v2bf16__offset12b_neg__amdgpu_
; GFX11-TRUE16-NEXT: buffer_gl0_inv
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v6
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB56_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-TRUE16-NEXT: s_set_inst_prefetch_distance 0x2
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_mov_b32_e32 v0, v3
; GFX11-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -16172,12 +16281,13 @@ define <2 x bfloat> @global_agent_atomic_fmin_ret_v2bf16__offset12b_neg__amdgpu_
; GFX11-FAKE16-NEXT: buffer_gl0_inv
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v6
; GFX11-FAKE16-NEXT: s_or_b32 s1, vcc_lo, s1
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB56_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-FAKE16-NEXT: s_set_inst_prefetch_distance 0x2
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s1
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v0, v3
; GFX11-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -16503,6 +16613,7 @@ define void @global_agent_atomic_fmin_noret_v2bf16__amdgpu_no_fine_grained_memor
; GFX12-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_cbranch_execnz .LBB57_1
; GFX12-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -16537,11 +16648,10 @@ define void @global_agent_atomic_fmin_noret_v2bf16__amdgpu_no_fine_grained_memor
; GFX12-FAKE16-NEXT: v_add3_u32 v8, v8, v6, 0x7fff
; GFX12-FAKE16-NEXT: v_cmp_u_f32_e64 s0, v2, v2
; GFX12-FAKE16-NEXT: s_wait_alu depctr_va_vcc(0)
-; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_2)
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_2) | instid1(VALU_DEP_1)
; GFX12-FAKE16-NEXT: v_cndmask_b32_e32 v6, v8, v10, vcc_lo
; GFX12-FAKE16-NEXT: s_wait_alu depctr_va_sdst(0)
; GFX12-FAKE16-NEXT: v_cndmask_b32_e64 v2, v7, v9, s0
-; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-FAKE16-NEXT: v_perm_b32 v2, v6, v2, 0x7060302
; GFX12-FAKE16-NEXT: s_wait_storecnt 0x0
; GFX12-FAKE16-NEXT: global_atomic_cmpswap_b32 v2, v[0:1], v[2:3], off th:TH_ATOMIC_RETURN scope:SCOPE_DEV
@@ -16553,6 +16663,7 @@ define void @global_agent_atomic_fmin_noret_v2bf16__amdgpu_no_fine_grained_memor
; GFX12-FAKE16-NEXT: s_or_b32 s1, vcc_lo, s1
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: s_cbranch_execnz .LBB57_1
; GFX12-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s1
@@ -16642,7 +16753,7 @@ define void @global_agent_atomic_fmin_noret_v2bf16__amdgpu_no_fine_grained_memor
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v3
; GFX11-TRUE16-NEXT: v_mov_b32_e32 v3, v2
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB57_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -16676,7 +16787,7 @@ define void @global_agent_atomic_fmin_noret_v2bf16__amdgpu_no_fine_grained_memor
; GFX11-FAKE16-NEXT: v_add3_u32 v7, v7, v2, 0x7fff
; GFX11-FAKE16-NEXT: v_add3_u32 v8, v8, v6, 0x7fff
; GFX11-FAKE16-NEXT: v_cmp_u_f32_e64 s0, v2, v2
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_4)
; GFX11-FAKE16-NEXT: v_cndmask_b32_e32 v6, v8, v10, vcc_lo
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v2, v7, v9, s0
; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
@@ -16689,7 +16800,7 @@ define void @global_agent_atomic_fmin_noret_v2bf16__amdgpu_no_fine_grained_memor
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v3
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v3, v2
; GFX11-FAKE16-NEXT: s_or_b32 s1, vcc_lo, s1
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB57_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -16997,6 +17108,7 @@ define void @global_agent_atomic_fmin_noret_v2bf16__offset12b_pos__amdgpu_no_fin
; GFX12-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_cbranch_execnz .LBB58_1
; GFX12-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -17031,11 +17143,10 @@ define void @global_agent_atomic_fmin_noret_v2bf16__offset12b_pos__amdgpu_no_fin
; GFX12-FAKE16-NEXT: v_add3_u32 v8, v8, v6, 0x7fff
; GFX12-FAKE16-NEXT: v_cmp_u_f32_e64 s0, v2, v2
; GFX12-FAKE16-NEXT: s_wait_alu depctr_va_vcc(0)
-; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_2)
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_2) | instid1(VALU_DEP_1)
; GFX12-FAKE16-NEXT: v_cndmask_b32_e32 v6, v8, v10, vcc_lo
; GFX12-FAKE16-NEXT: s_wait_alu depctr_va_sdst(0)
; GFX12-FAKE16-NEXT: v_cndmask_b32_e64 v2, v7, v9, s0
-; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-FAKE16-NEXT: v_perm_b32 v2, v6, v2, 0x7060302
; GFX12-FAKE16-NEXT: s_wait_storecnt 0x0
; GFX12-FAKE16-NEXT: global_atomic_cmpswap_b32 v2, v[0:1], v[2:3], off offset:2044 th:TH_ATOMIC_RETURN scope:SCOPE_DEV
@@ -17047,6 +17158,7 @@ define void @global_agent_atomic_fmin_noret_v2bf16__offset12b_pos__amdgpu_no_fin
; GFX12-FAKE16-NEXT: s_or_b32 s1, vcc_lo, s1
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: s_cbranch_execnz .LBB58_1
; GFX12-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s1
@@ -17136,7 +17248,7 @@ define void @global_agent_atomic_fmin_noret_v2bf16__offset12b_pos__amdgpu_no_fin
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v3
; GFX11-TRUE16-NEXT: v_mov_b32_e32 v3, v2
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB58_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -17170,7 +17282,7 @@ define void @global_agent_atomic_fmin_noret_v2bf16__offset12b_pos__amdgpu_no_fin
; GFX11-FAKE16-NEXT: v_add3_u32 v7, v7, v2, 0x7fff
; GFX11-FAKE16-NEXT: v_add3_u32 v8, v8, v6, 0x7fff
; GFX11-FAKE16-NEXT: v_cmp_u_f32_e64 s0, v2, v2
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_4)
; GFX11-FAKE16-NEXT: v_cndmask_b32_e32 v6, v8, v10, vcc_lo
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v2, v7, v9, s0
; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
@@ -17183,7 +17295,7 @@ define void @global_agent_atomic_fmin_noret_v2bf16__offset12b_pos__amdgpu_no_fin
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v3
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v3, v2
; GFX11-FAKE16-NEXT: s_or_b32 s1, vcc_lo, s1
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB58_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -17494,6 +17606,7 @@ define void @global_agent_atomic_fmin_noret_v2bf16__offset12b_neg__amdgpu_no_fin
; GFX12-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_cbranch_execnz .LBB59_1
; GFX12-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -17528,11 +17641,10 @@ define void @global_agent_atomic_fmin_noret_v2bf16__offset12b_neg__amdgpu_no_fin
; GFX12-FAKE16-NEXT: v_add3_u32 v8, v8, v6, 0x7fff
; GFX12-FAKE16-NEXT: v_cmp_u_f32_e64 s0, v2, v2
; GFX12-FAKE16-NEXT: s_wait_alu depctr_va_vcc(0)
-; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_2)
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_2) | instid1(VALU_DEP_1)
; GFX12-FAKE16-NEXT: v_cndmask_b32_e32 v6, v8, v10, vcc_lo
; GFX12-FAKE16-NEXT: s_wait_alu depctr_va_sdst(0)
; GFX12-FAKE16-NEXT: v_cndmask_b32_e64 v2, v7, v9, s0
-; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-FAKE16-NEXT: v_perm_b32 v2, v6, v2, 0x7060302
; GFX12-FAKE16-NEXT: s_wait_storecnt 0x0
; GFX12-FAKE16-NEXT: global_atomic_cmpswap_b32 v2, v[0:1], v[2:3], off offset:-2048 th:TH_ATOMIC_RETURN scope:SCOPE_DEV
@@ -17544,6 +17656,7 @@ define void @global_agent_atomic_fmin_noret_v2bf16__offset12b_neg__amdgpu_no_fin
; GFX12-FAKE16-NEXT: s_or_b32 s1, vcc_lo, s1
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: s_cbranch_execnz .LBB59_1
; GFX12-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s1
@@ -17633,7 +17746,7 @@ define void @global_agent_atomic_fmin_noret_v2bf16__offset12b_neg__amdgpu_no_fin
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v3
; GFX11-TRUE16-NEXT: v_mov_b32_e32 v3, v2
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB59_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -17667,7 +17780,7 @@ define void @global_agent_atomic_fmin_noret_v2bf16__offset12b_neg__amdgpu_no_fin
; GFX11-FAKE16-NEXT: v_add3_u32 v7, v7, v2, 0x7fff
; GFX11-FAKE16-NEXT: v_add3_u32 v8, v8, v6, 0x7fff
; GFX11-FAKE16-NEXT: v_cmp_u_f32_e64 s0, v2, v2
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_4)
; GFX11-FAKE16-NEXT: v_cndmask_b32_e32 v6, v8, v10, vcc_lo
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v2, v7, v9, s0
; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
@@ -17680,7 +17793,7 @@ define void @global_agent_atomic_fmin_noret_v2bf16__offset12b_neg__amdgpu_no_fin
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v3
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v3, v2
; GFX11-FAKE16-NEXT: s_or_b32 s1, vcc_lo, s1
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB59_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -18002,9 +18115,11 @@ define <2 x bfloat> @global_system_atomic_fmin_ret_v2bf16__offset12b_pos__amdgpu
; GFX12-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_cbranch_execnz .LBB60_1
; GFX12-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: v_mov_b32_e32 v0, v3
; GFX12-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -18054,9 +18169,11 @@ define <2 x bfloat> @global_system_atomic_fmin_ret_v2bf16__offset12b_pos__amdgpu
; GFX12-FAKE16-NEXT: s_or_b32 s1, vcc_lo, s1
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: s_cbranch_execnz .LBB60_1
; GFX12-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s1
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: v_mov_b32_e32 v0, v3
; GFX12-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -18146,12 +18263,13 @@ define <2 x bfloat> @global_system_atomic_fmin_ret_v2bf16__offset12b_pos__amdgpu
; GFX11-TRUE16-NEXT: buffer_gl0_inv
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v6
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB60_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-TRUE16-NEXT: s_set_inst_prefetch_distance 0x2
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_mov_b32_e32 v0, v3
; GFX11-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -18195,12 +18313,13 @@ define <2 x bfloat> @global_system_atomic_fmin_ret_v2bf16__offset12b_pos__amdgpu
; GFX11-FAKE16-NEXT: buffer_gl0_inv
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v6
; GFX11-FAKE16-NEXT: s_or_b32 s1, vcc_lo, s1
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB60_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-FAKE16-NEXT: s_set_inst_prefetch_distance 0x2
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s1
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v0, v3
; GFX11-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -18522,6 +18641,7 @@ define void @global_system_atomic_fmin_noret_v2bf16__offset12b_pos__amdgpu_no_fi
; GFX12-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_cbranch_execnz .LBB61_1
; GFX12-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -18556,11 +18676,10 @@ define void @global_system_atomic_fmin_noret_v2bf16__offset12b_pos__amdgpu_no_fi
; GFX12-FAKE16-NEXT: v_add3_u32 v8, v8, v6, 0x7fff
; GFX12-FAKE16-NEXT: v_cmp_u_f32_e64 s0, v2, v2
; GFX12-FAKE16-NEXT: s_wait_alu depctr_va_vcc(0)
-; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_2)
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_2) | instid1(VALU_DEP_1)
; GFX12-FAKE16-NEXT: v_cndmask_b32_e32 v6, v8, v10, vcc_lo
; GFX12-FAKE16-NEXT: s_wait_alu depctr_va_sdst(0)
; GFX12-FAKE16-NEXT: v_cndmask_b32_e64 v2, v7, v9, s0
-; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-FAKE16-NEXT: v_perm_b32 v2, v6, v2, 0x7060302
; GFX12-FAKE16-NEXT: global_wb scope:SCOPE_SYS
; GFX12-FAKE16-NEXT: s_wait_storecnt 0x0
@@ -18573,6 +18692,7 @@ define void @global_system_atomic_fmin_noret_v2bf16__offset12b_pos__amdgpu_no_fi
; GFX12-FAKE16-NEXT: s_or_b32 s1, vcc_lo, s1
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: s_cbranch_execnz .LBB61_1
; GFX12-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s1
@@ -18662,7 +18782,7 @@ define void @global_system_atomic_fmin_noret_v2bf16__offset12b_pos__amdgpu_no_fi
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v3
; GFX11-TRUE16-NEXT: v_mov_b32_e32 v3, v2
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB61_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -18696,7 +18816,7 @@ define void @global_system_atomic_fmin_noret_v2bf16__offset12b_pos__amdgpu_no_fi
; GFX11-FAKE16-NEXT: v_add3_u32 v7, v7, v2, 0x7fff
; GFX11-FAKE16-NEXT: v_add3_u32 v8, v8, v6, 0x7fff
; GFX11-FAKE16-NEXT: v_cmp_u_f32_e64 s0, v2, v2
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_4)
; GFX11-FAKE16-NEXT: v_cndmask_b32_e32 v6, v8, v10, vcc_lo
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v2, v7, v9, s0
; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
@@ -18709,7 +18829,7 @@ define void @global_system_atomic_fmin_noret_v2bf16__offset12b_pos__amdgpu_no_fi
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v3
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v3, v2
; GFX11-FAKE16-NEXT: s_or_b32 s1, vcc_lo, s1
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB61_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
diff --git a/llvm/test/CodeGen/AMDGPU/global-atomicrmw-fsub.ll b/llvm/test/CodeGen/AMDGPU/global-atomicrmw-fsub.ll
index ef314c8f68b1fe..a13e8d56415532 100644
--- a/llvm/test/CodeGen/AMDGPU/global-atomicrmw-fsub.ll
+++ b/llvm/test/CodeGen/AMDGPU/global-atomicrmw-fsub.ll
@@ -40,9 +40,11 @@ define float @global_agent_atomic_fsub_ret_f32(ptr addrspace(1) %ptr, float %val
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB0_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: v_mov_b32_e32 v0, v3
; GFX12-NEXT: s_setpc_b64 s[30:31]
;
@@ -88,11 +90,12 @@ define float @global_agent_atomic_fsub_ret_f32(ptr addrspace(1) %ptr, float %val
; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB0_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: v_mov_b32_e32 v0, v3
; GFX11-NEXT: s_setpc_b64 s[30:31]
;
@@ -272,9 +275,11 @@ define float @global_agent_atomic_fsub_ret_f32__offset12b_pos(ptr addrspace(1) %
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB1_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: v_mov_b32_e32 v0, v3
; GFX12-NEXT: s_setpc_b64 s[30:31]
;
@@ -320,11 +325,12 @@ define float @global_agent_atomic_fsub_ret_f32__offset12b_pos(ptr addrspace(1) %
; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB1_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: v_mov_b32_e32 v0, v3
; GFX11-NEXT: s_setpc_b64 s[30:31]
;
@@ -506,9 +512,11 @@ define float @global_agent_atomic_fsub_ret_f32__offset12b_neg(ptr addrspace(1) %
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB2_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: v_mov_b32_e32 v0, v3
; GFX12-NEXT: s_setpc_b64 s[30:31]
;
@@ -554,11 +562,12 @@ define float @global_agent_atomic_fsub_ret_f32__offset12b_neg(ptr addrspace(1) %
; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB2_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: v_mov_b32_e32 v0, v3
; GFX11-NEXT: s_setpc_b64 s[30:31]
;
@@ -749,6 +758,7 @@ define void @global_agent_atomic_fsub_noret_f32(ptr addrspace(1) %ptr, float %va
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB3_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -794,7 +804,7 @@ define void @global_agent_atomic_fsub_noret_f32(ptr addrspace(1) %ptr, float %va
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-NEXT: v_mov_b32_e32 v4, v3
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB3_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -970,6 +980,7 @@ define void @global_agent_atomic_fsub_noret_f32__offset12b_pos(ptr addrspace(1)
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB4_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -1015,7 +1026,7 @@ define void @global_agent_atomic_fsub_noret_f32__offset12b_pos(ptr addrspace(1)
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-NEXT: v_mov_b32_e32 v4, v3
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB4_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -1194,6 +1205,7 @@ define void @global_agent_atomic_fsub_noret_f32__offset12b_neg(ptr addrspace(1)
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB5_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -1239,7 +1251,7 @@ define void @global_agent_atomic_fsub_noret_f32__offset12b_neg(ptr addrspace(1)
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-NEXT: v_mov_b32_e32 v4, v3
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB5_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -1428,9 +1440,11 @@ define float @global_system_atomic_fsub_ret_f32__offset12b_pos(ptr addrspace(1)
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB6_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: v_mov_b32_e32 v0, v3
; GFX12-NEXT: s_setpc_b64 s[30:31]
;
@@ -1476,11 +1490,12 @@ define float @global_system_atomic_fsub_ret_f32__offset12b_pos(ptr addrspace(1)
; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB6_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: v_mov_b32_e32 v0, v3
; GFX11-NEXT: s_setpc_b64 s[30:31]
;
@@ -1665,6 +1680,7 @@ define void @global_system_atomic_fsub_noret_f32__offset12b_pos(ptr addrspace(1)
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB7_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -1710,7 +1726,7 @@ define void @global_system_atomic_fsub_noret_f32__offset12b_pos(ptr addrspace(1)
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-NEXT: v_mov_b32_e32 v4, v3
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB7_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -1897,9 +1913,11 @@ define float @global_agent_atomic_fsub_ret_f32__ftz(ptr addrspace(1) %ptr, float
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB8_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: v_mov_b32_e32 v0, v3
; GFX12-NEXT: s_setpc_b64 s[30:31]
;
@@ -1945,11 +1963,12 @@ define float @global_agent_atomic_fsub_ret_f32__ftz(ptr addrspace(1) %ptr, float
; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB8_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: v_mov_b32_e32 v0, v3
; GFX11-NEXT: s_setpc_b64 s[30:31]
;
@@ -2129,9 +2148,11 @@ define float @global_agent_atomic_fsub_ret_f32__offset12b_pos__ftz(ptr addrspace
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB9_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: v_mov_b32_e32 v0, v3
; GFX12-NEXT: s_setpc_b64 s[30:31]
;
@@ -2177,11 +2198,12 @@ define float @global_agent_atomic_fsub_ret_f32__offset12b_pos__ftz(ptr addrspace
; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB9_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: v_mov_b32_e32 v0, v3
; GFX11-NEXT: s_setpc_b64 s[30:31]
;
@@ -2363,9 +2385,11 @@ define float @global_agent_atomic_fsub_ret_f32__offset12b_neg__ftz(ptr addrspace
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB10_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: v_mov_b32_e32 v0, v3
; GFX12-NEXT: s_setpc_b64 s[30:31]
;
@@ -2411,11 +2435,12 @@ define float @global_agent_atomic_fsub_ret_f32__offset12b_neg__ftz(ptr addrspace
; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB10_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: v_mov_b32_e32 v0, v3
; GFX11-NEXT: s_setpc_b64 s[30:31]
;
@@ -2606,6 +2631,7 @@ define void @global_agent_atomic_fsub_noret_f32__ftz(ptr addrspace(1) %ptr, floa
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB11_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -2651,7 +2677,7 @@ define void @global_agent_atomic_fsub_noret_f32__ftz(ptr addrspace(1) %ptr, floa
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-NEXT: v_mov_b32_e32 v4, v3
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB11_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -2827,6 +2853,7 @@ define void @global_agent_atomic_fsub_noret_f32__offset12b_pos__ftz(ptr addrspac
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB12_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -2872,7 +2899,7 @@ define void @global_agent_atomic_fsub_noret_f32__offset12b_pos__ftz(ptr addrspac
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-NEXT: v_mov_b32_e32 v4, v3
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB12_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -3051,6 +3078,7 @@ define void @global_agent_atomic_fsub_noret_f32__offset12b_neg__ftz(ptr addrspac
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB13_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -3096,7 +3124,7 @@ define void @global_agent_atomic_fsub_noret_f32__offset12b_neg__ftz(ptr addrspac
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-NEXT: v_mov_b32_e32 v4, v3
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB13_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -3285,9 +3313,11 @@ define float @global_system_atomic_fsub_ret_f32__offset12b_pos__ftz(ptr addrspac
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB14_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: v_mov_b32_e32 v0, v3
; GFX12-NEXT: s_setpc_b64 s[30:31]
;
@@ -3333,11 +3363,12 @@ define float @global_system_atomic_fsub_ret_f32__offset12b_pos__ftz(ptr addrspac
; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB14_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: v_mov_b32_e32 v0, v3
; GFX11-NEXT: s_setpc_b64 s[30:31]
;
@@ -3522,6 +3553,7 @@ define void @global_system_atomic_fsub_noret_f32__offset12b_pos__ftz(ptr addrspa
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB15_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -3567,7 +3599,7 @@ define void @global_system_atomic_fsub_noret_f32__offset12b_pos__ftz(ptr addrspa
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-NEXT: v_mov_b32_e32 v4, v3
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB15_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -3754,9 +3786,11 @@ define double @global_agent_atomic_fsub_ret_f64(ptr addrspace(1) %ptr, double %v
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB16_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: v_dual_mov_b32 v0, v4 :: v_dual_mov_b32 v1, v5
; GFX12-NEXT: s_setpc_b64 s[30:31]
;
@@ -3803,11 +3837,12 @@ define double @global_agent_atomic_fsub_ret_f64(ptr addrspace(1) %ptr, double %v
; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: v_cmp_eq_u64_e32 vcc_lo, v[4:5], v[6:7]
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB16_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: v_dual_mov_b32 v0, v4 :: v_dual_mov_b32 v1, v5
; GFX11-NEXT: s_setpc_b64 s[30:31]
;
@@ -4006,9 +4041,11 @@ define double @global_agent_atomic_fsub_ret_f64__offset12b_pos(ptr addrspace(1)
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB17_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: v_dual_mov_b32 v0, v4 :: v_dual_mov_b32 v1, v5
; GFX12-NEXT: s_setpc_b64 s[30:31]
;
@@ -4055,11 +4092,12 @@ define double @global_agent_atomic_fsub_ret_f64__offset12b_pos(ptr addrspace(1)
; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: v_cmp_eq_u64_e32 vcc_lo, v[4:5], v[6:7]
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB17_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: v_dual_mov_b32 v0, v4 :: v_dual_mov_b32 v1, v5
; GFX11-NEXT: s_setpc_b64 s[30:31]
;
@@ -4259,9 +4297,11 @@ define double @global_agent_atomic_fsub_ret_f64__offset12b_neg(ptr addrspace(1)
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB18_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: v_dual_mov_b32 v0, v4 :: v_dual_mov_b32 v1, v5
; GFX12-NEXT: s_setpc_b64 s[30:31]
;
@@ -4308,11 +4348,12 @@ define double @global_agent_atomic_fsub_ret_f64__offset12b_neg(ptr addrspace(1)
; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: v_cmp_eq_u64_e32 vcc_lo, v[4:5], v[6:7]
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB18_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: v_dual_mov_b32 v0, v4 :: v_dual_mov_b32 v1, v5
; GFX11-NEXT: s_setpc_b64 s[30:31]
;
@@ -4515,6 +4556,7 @@ define void @global_agent_atomic_fsub_noret_f64(ptr addrspace(1) %ptr, double %v
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB19_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -4560,7 +4602,7 @@ define void @global_agent_atomic_fsub_noret_f64(ptr addrspace(1) %ptr, double %v
; GFX11-NEXT: v_cmp_eq_u64_e32 vcc_lo, v[4:5], v[6:7]
; GFX11-NEXT: v_dual_mov_b32 v7, v5 :: v_dual_mov_b32 v6, v4
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB19_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -4745,6 +4787,7 @@ define void @global_agent_atomic_fsub_noret_f64__offset12b_pos(ptr addrspace(1)
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB20_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -4790,7 +4833,7 @@ define void @global_agent_atomic_fsub_noret_f64__offset12b_pos(ptr addrspace(1)
; GFX11-NEXT: v_cmp_eq_u64_e32 vcc_lo, v[4:5], v[6:7]
; GFX11-NEXT: v_dual_mov_b32 v7, v5 :: v_dual_mov_b32 v6, v4
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB20_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -4978,6 +5021,7 @@ define void @global_agent_atomic_fsub_noret_f64__offset12b_neg(ptr addrspace(1)
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB21_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -5023,7 +5067,7 @@ define void @global_agent_atomic_fsub_noret_f64__offset12b_neg(ptr addrspace(1)
; GFX11-NEXT: v_cmp_eq_u64_e32 vcc_lo, v[4:5], v[6:7]
; GFX11-NEXT: v_dual_mov_b32 v7, v5 :: v_dual_mov_b32 v6, v4
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB21_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -5238,9 +5282,11 @@ define half @global_agent_atomic_fsub_ret_f16(ptr addrspace(1) %ptr, half %val)
; GFX12-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_cbranch_execnz .LBB22_1
; GFX12-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: v_lshrrev_b32_e32 v0, v3, v5
; GFX12-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -5282,9 +5328,11 @@ define half @global_agent_atomic_fsub_ret_f16(ptr addrspace(1) %ptr, half %val)
; GFX12-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: s_cbranch_execnz .LBB22_1
; GFX12-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: v_lshrrev_b32_e32 v0, v3, v5
; GFX12-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -5354,11 +5402,12 @@ define half @global_agent_atomic_fsub_ret_f16(ptr addrspace(1) %ptr, half %val)
; GFX11-TRUE16-NEXT: buffer_gl0_inv
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v5, v6
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB22_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v0, v3, v5
; GFX11-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -5394,11 +5443,12 @@ define half @global_agent_atomic_fsub_ret_f16(ptr addrspace(1) %ptr, half %val)
; GFX11-FAKE16-NEXT: buffer_gl0_inv
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v5, v6
; GFX11-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB22_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v0, v3, v5
; GFX11-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -5658,9 +5708,11 @@ define half @global_agent_atomic_fsub_ret_f16__offset12b_pos(ptr addrspace(1) %p
; GFX12-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_cbranch_execnz .LBB23_1
; GFX12-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: v_lshrrev_b32_e32 v0, v3, v5
; GFX12-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -5703,9 +5755,11 @@ define half @global_agent_atomic_fsub_ret_f16__offset12b_pos(ptr addrspace(1) %p
; GFX12-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: s_cbranch_execnz .LBB23_1
; GFX12-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: v_lshrrev_b32_e32 v0, v3, v5
; GFX12-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -5749,7 +5803,6 @@ define half @global_agent_atomic_fsub_ret_f16__offset12b_pos(ptr addrspace(1) %p
; GFX11-TRUE16: ; %bb.0:
; GFX11-TRUE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-TRUE16-NEXT: v_add_co_u32 v3, vcc_lo, 0x7fe, v0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-TRUE16-NEXT: s_mov_b32 s0, 0
; GFX11-TRUE16-NEXT: v_and_b32_e32 v0, -4, v3
@@ -5778,11 +5831,12 @@ define half @global_agent_atomic_fsub_ret_f16__offset12b_pos(ptr addrspace(1) %p
; GFX11-TRUE16-NEXT: buffer_gl0_inv
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v5, v6
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB23_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v0, v3, v5
; GFX11-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -5790,7 +5844,6 @@ define half @global_agent_atomic_fsub_ret_f16__offset12b_pos(ptr addrspace(1) %p
; GFX11-FAKE16: ; %bb.0:
; GFX11-FAKE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-FAKE16-NEXT: v_add_co_u32 v3, vcc_lo, 0x7fe, v0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-FAKE16-NEXT: s_mov_b32 s0, 0
; GFX11-FAKE16-NEXT: v_and_b32_e32 v0, -4, v3
@@ -5819,11 +5872,12 @@ define half @global_agent_atomic_fsub_ret_f16__offset12b_pos(ptr addrspace(1) %p
; GFX11-FAKE16-NEXT: buffer_gl0_inv
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v5, v6
; GFX11-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB23_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v0, v3, v5
; GFX11-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -6091,9 +6145,11 @@ define half @global_agent_atomic_fsub_ret_f16__offset12b_neg(ptr addrspace(1) %p
; GFX12-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_cbranch_execnz .LBB24_1
; GFX12-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: v_lshrrev_b32_e32 v0, v3, v5
; GFX12-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -6136,9 +6192,11 @@ define half @global_agent_atomic_fsub_ret_f16__offset12b_neg(ptr addrspace(1) %p
; GFX12-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: s_cbranch_execnz .LBB24_1
; GFX12-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: v_lshrrev_b32_e32 v0, v3, v5
; GFX12-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -6183,7 +6241,6 @@ define half @global_agent_atomic_fsub_ret_f16__offset12b_neg(ptr addrspace(1) %p
; GFX11-TRUE16: ; %bb.0:
; GFX11-TRUE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-TRUE16-NEXT: v_add_co_u32 v3, vcc_lo, 0xfffff800, v0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v1, null, -1, v1, vcc_lo
; GFX11-TRUE16-NEXT: s_mov_b32 s0, 0
; GFX11-TRUE16-NEXT: v_and_b32_e32 v0, -4, v3
@@ -6212,11 +6269,12 @@ define half @global_agent_atomic_fsub_ret_f16__offset12b_neg(ptr addrspace(1) %p
; GFX11-TRUE16-NEXT: buffer_gl0_inv
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v5, v6
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB24_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v0, v3, v5
; GFX11-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -6224,7 +6282,6 @@ define half @global_agent_atomic_fsub_ret_f16__offset12b_neg(ptr addrspace(1) %p
; GFX11-FAKE16: ; %bb.0:
; GFX11-FAKE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-FAKE16-NEXT: v_add_co_u32 v3, vcc_lo, 0xfffff800, v0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v1, null, -1, v1, vcc_lo
; GFX11-FAKE16-NEXT: s_mov_b32 s0, 0
; GFX11-FAKE16-NEXT: v_and_b32_e32 v0, -4, v3
@@ -6253,11 +6310,12 @@ define half @global_agent_atomic_fsub_ret_f16__offset12b_neg(ptr addrspace(1) %p
; GFX11-FAKE16-NEXT: buffer_gl0_inv
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v5, v6
; GFX11-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB24_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v0, v3, v5
; GFX11-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -6523,6 +6581,7 @@ define void @global_agent_atomic_fsub_noret_f16(ptr addrspace(1) %ptr, half %val
; GFX12-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_cbranch_execnz .LBB25_1
; GFX12-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -6565,6 +6624,7 @@ define void @global_agent_atomic_fsub_noret_f16(ptr addrspace(1) %ptr, half %val
; GFX12-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: s_cbranch_execnz .LBB25_1
; GFX12-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -6634,7 +6694,7 @@ define void @global_agent_atomic_fsub_noret_f16(ptr addrspace(1) %ptr, half %val
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-TRUE16-NEXT: v_mov_b32_e32 v4, v3
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB25_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -6672,7 +6732,7 @@ define void @global_agent_atomic_fsub_noret_f16(ptr addrspace(1) %ptr, half %val
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v4, v3
; GFX11-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB25_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -6929,6 +6989,7 @@ define void @global_agent_atomic_fsub_noret_f16__offset12b_pos(ptr addrspace(1)
; GFX12-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_cbranch_execnz .LBB26_1
; GFX12-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -6972,6 +7033,7 @@ define void @global_agent_atomic_fsub_noret_f16__offset12b_pos(ptr addrspace(1)
; GFX12-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: s_cbranch_execnz .LBB26_1
; GFX12-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -7016,7 +7078,6 @@ define void @global_agent_atomic_fsub_noret_f16__offset12b_pos(ptr addrspace(1)
; GFX11-TRUE16: ; %bb.0:
; GFX11-TRUE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-TRUE16-NEXT: v_add_co_u32 v3, vcc_lo, 0x7fe, v0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-TRUE16-NEXT: s_mov_b32 s0, 0
; GFX11-TRUE16-NEXT: v_and_b32_e32 v0, -4, v3
@@ -7044,7 +7105,7 @@ define void @global_agent_atomic_fsub_noret_f16__offset12b_pos(ptr addrspace(1)
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-TRUE16-NEXT: v_mov_b32_e32 v4, v3
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB26_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -7055,7 +7116,6 @@ define void @global_agent_atomic_fsub_noret_f16__offset12b_pos(ptr addrspace(1)
; GFX11-FAKE16: ; %bb.0:
; GFX11-FAKE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-FAKE16-NEXT: v_add_co_u32 v3, vcc_lo, 0x7fe, v0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-FAKE16-NEXT: s_mov_b32 s0, 0
; GFX11-FAKE16-NEXT: v_and_b32_e32 v0, -4, v3
@@ -7083,7 +7143,7 @@ define void @global_agent_atomic_fsub_noret_f16__offset12b_pos(ptr addrspace(1)
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v4, v3
; GFX11-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB26_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -7347,6 +7407,7 @@ define void @global_agent_atomic_fsub_noret_f16__offset12b_neg(ptr addrspace(1)
; GFX12-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_cbranch_execnz .LBB27_1
; GFX12-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -7390,6 +7451,7 @@ define void @global_agent_atomic_fsub_noret_f16__offset12b_neg(ptr addrspace(1)
; GFX12-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: s_cbranch_execnz .LBB27_1
; GFX12-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -7435,7 +7497,6 @@ define void @global_agent_atomic_fsub_noret_f16__offset12b_neg(ptr addrspace(1)
; GFX11-TRUE16: ; %bb.0:
; GFX11-TRUE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-TRUE16-NEXT: v_add_co_u32 v3, vcc_lo, 0xfffff800, v0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v1, null, -1, v1, vcc_lo
; GFX11-TRUE16-NEXT: s_mov_b32 s0, 0
; GFX11-TRUE16-NEXT: v_and_b32_e32 v0, -4, v3
@@ -7463,7 +7524,7 @@ define void @global_agent_atomic_fsub_noret_f16__offset12b_neg(ptr addrspace(1)
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-TRUE16-NEXT: v_mov_b32_e32 v4, v3
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB27_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -7474,7 +7535,6 @@ define void @global_agent_atomic_fsub_noret_f16__offset12b_neg(ptr addrspace(1)
; GFX11-FAKE16: ; %bb.0:
; GFX11-FAKE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-FAKE16-NEXT: v_add_co_u32 v3, vcc_lo, 0xfffff800, v0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v1, null, -1, v1, vcc_lo
; GFX11-FAKE16-NEXT: s_mov_b32 s0, 0
; GFX11-FAKE16-NEXT: v_and_b32_e32 v0, -4, v3
@@ -7502,7 +7562,7 @@ define void @global_agent_atomic_fsub_noret_f16__offset12b_neg(ptr addrspace(1)
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v4, v3
; GFX11-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB27_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -7755,9 +7815,11 @@ define half @global_agent_atomic_fsub_ret_f16__offset12b_pos__align4(ptr addrspa
; GFX12-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_cbranch_execnz .LBB28_1
; GFX12-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: v_mov_b16_e32 v0.l, v3.l
; GFX12-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -7788,9 +7850,11 @@ define half @global_agent_atomic_fsub_ret_f16__offset12b_pos__align4(ptr addrspa
; GFX12-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: s_cbranch_execnz .LBB28_1
; GFX12-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: v_mov_b32_e32 v0, v3
; GFX12-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -7841,11 +7905,12 @@ define half @global_agent_atomic_fsub_ret_f16__offset12b_pos__align4(ptr addrspa
; GFX11-TRUE16-NEXT: buffer_gl0_inv
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB28_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v0.l, v3.l
; GFX11-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -7870,11 +7935,12 @@ define half @global_agent_atomic_fsub_ret_f16__offset12b_pos__align4(ptr addrspa
; GFX11-FAKE16-NEXT: buffer_gl0_inv
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB28_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v0, v3
; GFX11-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -8076,6 +8142,7 @@ define void @global_agent_atomic_fsub_noret_f16__offset12b__align4_pos(ptr addrs
; GFX12-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_cbranch_execnz .LBB29_1
; GFX12-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -8107,6 +8174,7 @@ define void @global_agent_atomic_fsub_noret_f16__offset12b__align4_pos(ptr addrs
; GFX12-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: s_cbranch_execnz .LBB29_1
; GFX12-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -8157,7 +8225,7 @@ define void @global_agent_atomic_fsub_noret_f16__offset12b__align4_pos(ptr addrs
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-TRUE16-NEXT: v_mov_b32_e32 v4, v3
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB29_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -8184,7 +8252,7 @@ define void @global_agent_atomic_fsub_noret_f16__offset12b__align4_pos(ptr addrs
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v4, v3
; GFX11-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB29_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -8398,9 +8466,11 @@ define half @global_system_atomic_fsub_ret_f16__offset12b_pos(ptr addrspace(1) %
; GFX12-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_cbranch_execnz .LBB30_1
; GFX12-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: v_lshrrev_b32_e32 v0, v3, v5
; GFX12-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -8444,9 +8514,11 @@ define half @global_system_atomic_fsub_ret_f16__offset12b_pos(ptr addrspace(1) %
; GFX12-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: s_cbranch_execnz .LBB30_1
; GFX12-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: v_lshrrev_b32_e32 v0, v3, v5
; GFX12-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -8490,7 +8562,6 @@ define half @global_system_atomic_fsub_ret_f16__offset12b_pos(ptr addrspace(1) %
; GFX11-TRUE16: ; %bb.0:
; GFX11-TRUE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-TRUE16-NEXT: v_add_co_u32 v3, vcc_lo, 0x7fe, v0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-TRUE16-NEXT: s_mov_b32 s0, 0
; GFX11-TRUE16-NEXT: v_and_b32_e32 v0, -4, v3
@@ -8519,11 +8590,12 @@ define half @global_system_atomic_fsub_ret_f16__offset12b_pos(ptr addrspace(1) %
; GFX11-TRUE16-NEXT: buffer_gl0_inv
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v5, v6
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB30_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v0, v3, v5
; GFX11-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -8531,7 +8603,6 @@ define half @global_system_atomic_fsub_ret_f16__offset12b_pos(ptr addrspace(1) %
; GFX11-FAKE16: ; %bb.0:
; GFX11-FAKE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-FAKE16-NEXT: v_add_co_u32 v3, vcc_lo, 0x7fe, v0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-FAKE16-NEXT: s_mov_b32 s0, 0
; GFX11-FAKE16-NEXT: v_and_b32_e32 v0, -4, v3
@@ -8560,11 +8631,12 @@ define half @global_system_atomic_fsub_ret_f16__offset12b_pos(ptr addrspace(1) %
; GFX11-FAKE16-NEXT: buffer_gl0_inv
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v5, v6
; GFX11-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB30_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v0, v3, v5
; GFX11-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -8835,6 +8907,7 @@ define void @global_system_atomic_fsub_noret_f16__offset12b_pos(ptr addrspace(1)
; GFX12-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_cbranch_execnz .LBB31_1
; GFX12-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -8879,6 +8952,7 @@ define void @global_system_atomic_fsub_noret_f16__offset12b_pos(ptr addrspace(1)
; GFX12-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: s_cbranch_execnz .LBB31_1
; GFX12-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -8923,7 +8997,6 @@ define void @global_system_atomic_fsub_noret_f16__offset12b_pos(ptr addrspace(1)
; GFX11-TRUE16: ; %bb.0:
; GFX11-TRUE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-TRUE16-NEXT: v_add_co_u32 v3, vcc_lo, 0x7fe, v0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-TRUE16-NEXT: s_mov_b32 s0, 0
; GFX11-TRUE16-NEXT: v_and_b32_e32 v0, -4, v3
@@ -8951,7 +9024,7 @@ define void @global_system_atomic_fsub_noret_f16__offset12b_pos(ptr addrspace(1)
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-TRUE16-NEXT: v_mov_b32_e32 v4, v3
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB31_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -8962,7 +9035,6 @@ define void @global_system_atomic_fsub_noret_f16__offset12b_pos(ptr addrspace(1)
; GFX11-FAKE16: ; %bb.0:
; GFX11-FAKE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-FAKE16-NEXT: v_add_co_u32 v3, vcc_lo, 0x7fe, v0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-FAKE16-NEXT: s_mov_b32 s0, 0
; GFX11-FAKE16-NEXT: v_and_b32_e32 v0, -4, v3
@@ -8990,7 +9062,7 @@ define void @global_system_atomic_fsub_noret_f16__offset12b_pos(ptr addrspace(1)
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v4, v3
; GFX11-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB31_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -9270,9 +9342,11 @@ define bfloat @global_agent_atomic_fsub_ret_bf16(ptr addrspace(1) %ptr, bfloat %
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB32_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: v_lshrrev_b32_e32 v0, v3, v5
; GFX12-NEXT: s_setpc_b64 s[30:31]
;
@@ -9360,11 +9434,12 @@ define bfloat @global_agent_atomic_fsub_ret_bf16(ptr addrspace(1) %ptr, bfloat %
; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v5, v6
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB32_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: v_lshrrev_b32_e32 v0, v3, v5
; GFX11-NEXT: s_setpc_b64 s[30:31]
;
@@ -9666,9 +9741,11 @@ define bfloat @global_agent_atomic_fsub_ret_bf16__offset12b_pos(ptr addrspace(1)
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB33_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: v_lshrrev_b32_e32 v0, v3, v5
; GFX12-NEXT: s_setpc_b64 s[30:31]
;
@@ -9721,16 +9798,16 @@ define bfloat @global_agent_atomic_fsub_ret_bf16__offset12b_pos(ptr addrspace(1)
; GFX11: ; %bb.0:
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: v_add_co_u32 v3, vcc_lo, 0x7fe, v0
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_3)
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-NEXT: v_lshlrev_b32_e32 v2, 16, v2
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_4) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_and_b32_e32 v0, -4, v3
; GFX11-NEXT: v_and_b32_e32 v3, 3, v3
; GFX11-NEXT: s_mov_b32 s0, 0
; GFX11-NEXT: global_load_b32 v5, v[0:1], off
; GFX11-NEXT: v_lshlrev_b32_e32 v3, 3, v3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_lshlrev_b32_e64 v4, v3, 0xffff
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_not_b32_e32 v4, v4
; GFX11-NEXT: .p2align 6
; GFX11-NEXT: .LBB33_1: ; %atomicrmw.start
@@ -9760,11 +9837,12 @@ define bfloat @global_agent_atomic_fsub_ret_bf16__offset12b_pos(ptr addrspace(1)
; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v5, v6
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB33_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: v_lshrrev_b32_e32 v0, v3, v5
; GFX11-NEXT: s_setpc_b64 s[30:31]
;
@@ -10074,9 +10152,11 @@ define bfloat @global_agent_atomic_fsub_ret_bf16__offset12b_neg(ptr addrspace(1)
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB34_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: v_lshrrev_b32_e32 v0, v3, v5
; GFX12-NEXT: s_setpc_b64 s[30:31]
;
@@ -10130,16 +10210,16 @@ define bfloat @global_agent_atomic_fsub_ret_bf16__offset12b_neg(ptr addrspace(1)
; GFX11: ; %bb.0:
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: v_add_co_u32 v3, vcc_lo, 0xfffff800, v0
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_3)
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, -1, v1, vcc_lo
; GFX11-NEXT: v_lshlrev_b32_e32 v2, 16, v2
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_4) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_and_b32_e32 v0, -4, v3
; GFX11-NEXT: v_and_b32_e32 v3, 3, v3
; GFX11-NEXT: s_mov_b32 s0, 0
; GFX11-NEXT: global_load_b32 v5, v[0:1], off
; GFX11-NEXT: v_lshlrev_b32_e32 v3, 3, v3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_lshlrev_b32_e64 v4, v3, 0xffff
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_not_b32_e32 v4, v4
; GFX11-NEXT: .p2align 6
; GFX11-NEXT: .LBB34_1: ; %atomicrmw.start
@@ -10169,11 +10249,12 @@ define bfloat @global_agent_atomic_fsub_ret_bf16__offset12b_neg(ptr addrspace(1)
; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v5, v6
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB34_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: v_lshrrev_b32_e32 v0, v3, v5
; GFX11-NEXT: s_setpc_b64 s[30:31]
;
@@ -10479,6 +10560,7 @@ define void @global_agent_atomic_fsub_noret_bf16(ptr addrspace(1) %ptr, bfloat %
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB35_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -10566,7 +10648,7 @@ define void @global_agent_atomic_fsub_noret_bf16(ptr addrspace(1) %ptr, bfloat %
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-NEXT: v_mov_b32_e32 v4, v3
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB35_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -10865,6 +10947,7 @@ define void @global_agent_atomic_fsub_noret_bf16__offset12b_pos(ptr addrspace(1)
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB36_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -10918,16 +11001,16 @@ define void @global_agent_atomic_fsub_noret_bf16__offset12b_pos(ptr addrspace(1)
; GFX11: ; %bb.0:
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: v_add_co_u32 v4, vcc_lo, 0x7fe, v0
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_3)
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-NEXT: v_lshlrev_b32_e32 v6, 16, v2
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_4) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_and_b32_e32 v0, -4, v4
; GFX11-NEXT: v_and_b32_e32 v4, 3, v4
; GFX11-NEXT: s_mov_b32 s0, 0
; GFX11-NEXT: global_load_b32 v3, v[0:1], off
; GFX11-NEXT: v_lshlrev_b32_e32 v4, 3, v4
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_lshlrev_b32_e64 v5, v4, 0xffff
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_not_b32_e32 v5, v5
; GFX11-NEXT: .p2align 6
; GFX11-NEXT: .LBB36_1: ; %atomicrmw.start
@@ -10956,7 +11039,7 @@ define void @global_agent_atomic_fsub_noret_bf16__offset12b_pos(ptr addrspace(1)
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v3
; GFX11-NEXT: v_mov_b32_e32 v3, v2
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB36_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -11262,6 +11345,7 @@ define void @global_agent_atomic_fsub_noret_bf16__offset12b_neg(ptr addrspace(1)
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB37_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -11316,16 +11400,16 @@ define void @global_agent_atomic_fsub_noret_bf16__offset12b_neg(ptr addrspace(1)
; GFX11: ; %bb.0:
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: v_add_co_u32 v4, vcc_lo, 0xfffff800, v0
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_3)
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, -1, v1, vcc_lo
; GFX11-NEXT: v_lshlrev_b32_e32 v6, 16, v2
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_4) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_and_b32_e32 v0, -4, v4
; GFX11-NEXT: v_and_b32_e32 v4, 3, v4
; GFX11-NEXT: s_mov_b32 s0, 0
; GFX11-NEXT: global_load_b32 v3, v[0:1], off
; GFX11-NEXT: v_lshlrev_b32_e32 v4, 3, v4
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_lshlrev_b32_e64 v5, v4, 0xffff
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_not_b32_e32 v5, v5
; GFX11-NEXT: .p2align 6
; GFX11-NEXT: .LBB37_1: ; %atomicrmw.start
@@ -11354,7 +11438,7 @@ define void @global_agent_atomic_fsub_noret_bf16__offset12b_neg(ptr addrspace(1)
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v3
; GFX11-NEXT: v_mov_b32_e32 v3, v2
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB37_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -11648,9 +11732,11 @@ define bfloat @global_agent_atomic_fsub_ret_bf16__offset12b_pos__align4(ptr addr
; GFX12-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_cbranch_execnz .LBB38_1
; GFX12-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: v_mov_b16_e32 v0.l, v3.l
; GFX12-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -11691,9 +11777,11 @@ define bfloat @global_agent_atomic_fsub_ret_bf16__offset12b_pos__align4(ptr addr
; GFX12-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: s_cbranch_execnz .LBB38_1
; GFX12-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: v_mov_b32_e32 v0, v3
; GFX12-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -11764,11 +11852,12 @@ define bfloat @global_agent_atomic_fsub_ret_bf16__offset12b_pos__align4(ptr addr
; GFX11-TRUE16-NEXT: buffer_gl0_inv
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB38_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v0.l, v3.l
; GFX11-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -11803,11 +11892,12 @@ define bfloat @global_agent_atomic_fsub_ret_bf16__offset12b_pos__align4(ptr addr
; GFX11-FAKE16-NEXT: buffer_gl0_inv
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB38_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v0, v3
; GFX11-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -12056,6 +12146,7 @@ define void @global_agent_atomic_fsub_noret_bf16__offset12b__align4_pos(ptr addr
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB39_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -12126,7 +12217,7 @@ define void @global_agent_atomic_fsub_noret_bf16__offset12b__align4_pos(ptr addr
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v3
; GFX11-NEXT: v_mov_b32_e32 v3, v2
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB39_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -12388,9 +12479,11 @@ define bfloat @global_system_atomic_fsub_ret_bf16__offset12b_pos(ptr addrspace(1
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB40_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: v_lshrrev_b32_e32 v0, v3, v5
; GFX12-NEXT: s_setpc_b64 s[30:31]
;
@@ -12443,16 +12536,16 @@ define bfloat @global_system_atomic_fsub_ret_bf16__offset12b_pos(ptr addrspace(1
; GFX11: ; %bb.0:
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: v_add_co_u32 v3, vcc_lo, 0x7fe, v0
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_3)
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-NEXT: v_lshlrev_b32_e32 v2, 16, v2
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_4) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_and_b32_e32 v0, -4, v3
; GFX11-NEXT: v_and_b32_e32 v3, 3, v3
; GFX11-NEXT: s_mov_b32 s0, 0
; GFX11-NEXT: global_load_b32 v5, v[0:1], off
; GFX11-NEXT: v_lshlrev_b32_e32 v3, 3, v3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_lshlrev_b32_e64 v4, v3, 0xffff
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_not_b32_e32 v4, v4
; GFX11-NEXT: .p2align 6
; GFX11-NEXT: .LBB40_1: ; %atomicrmw.start
@@ -12482,11 +12575,12 @@ define bfloat @global_system_atomic_fsub_ret_bf16__offset12b_pos(ptr addrspace(1
; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v5, v6
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB40_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: v_lshrrev_b32_e32 v0, v3, v5
; GFX11-NEXT: s_setpc_b64 s[30:31]
;
@@ -12799,6 +12893,7 @@ define void @global_system_atomic_fsub_noret_bf16__offset12b_pos(ptr addrspace(1
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB41_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -12852,16 +12947,16 @@ define void @global_system_atomic_fsub_noret_bf16__offset12b_pos(ptr addrspace(1
; GFX11: ; %bb.0:
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: v_add_co_u32 v4, vcc_lo, 0x7fe, v0
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_3)
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-NEXT: v_lshlrev_b32_e32 v6, 16, v2
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_4) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_and_b32_e32 v0, -4, v4
; GFX11-NEXT: v_and_b32_e32 v4, 3, v4
; GFX11-NEXT: s_mov_b32 s0, 0
; GFX11-NEXT: global_load_b32 v3, v[0:1], off
; GFX11-NEXT: v_lshlrev_b32_e32 v4, 3, v4
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_lshlrev_b32_e64 v5, v4, 0xffff
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_not_b32_e32 v5, v5
; GFX11-NEXT: .p2align 6
; GFX11-NEXT: .LBB41_1: ; %atomicrmw.start
@@ -12890,7 +12985,7 @@ define void @global_system_atomic_fsub_noret_bf16__offset12b_pos(ptr addrspace(1
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v3
; GFX11-NEXT: v_mov_b32_e32 v3, v2
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB41_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -13178,9 +13273,11 @@ define <2 x half> @global_agent_atomic_fsub_ret_v2f16(ptr addrspace(1) %ptr, <2
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB42_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: v_mov_b32_e32 v0, v3
; GFX12-NEXT: s_setpc_b64 s[30:31]
;
@@ -13226,11 +13323,12 @@ define <2 x half> @global_agent_atomic_fsub_ret_v2f16(ptr addrspace(1) %ptr, <2
; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB42_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: v_mov_b32_e32 v0, v3
; GFX11-NEXT: s_setpc_b64 s[30:31]
;
@@ -13443,9 +13541,11 @@ define <2 x half> @global_agent_atomic_fsub_ret_v2f16__offset12b_pos(ptr addrspa
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB43_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: v_mov_b32_e32 v0, v3
; GFX12-NEXT: s_setpc_b64 s[30:31]
;
@@ -13491,11 +13591,12 @@ define <2 x half> @global_agent_atomic_fsub_ret_v2f16__offset12b_pos(ptr addrspa
; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB43_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: v_mov_b32_e32 v0, v3
; GFX11-NEXT: s_setpc_b64 s[30:31]
;
@@ -13710,9 +13811,11 @@ define <2 x half> @global_agent_atomic_fsub_ret_v2f16__offset12b_neg(ptr addrspa
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB44_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: v_mov_b32_e32 v0, v3
; GFX12-NEXT: s_setpc_b64 s[30:31]
;
@@ -13758,11 +13861,12 @@ define <2 x half> @global_agent_atomic_fsub_ret_v2f16__offset12b_neg(ptr addrspa
; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB44_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: v_mov_b32_e32 v0, v3
; GFX11-NEXT: s_setpc_b64 s[30:31]
;
@@ -13984,6 +14088,7 @@ define void @global_agent_atomic_fsub_noret_v2f16(ptr addrspace(1) %ptr, <2 x ha
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB45_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -14029,7 +14134,7 @@ define void @global_agent_atomic_fsub_noret_v2f16(ptr addrspace(1) %ptr, <2 x ha
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-NEXT: v_mov_b32_e32 v4, v3
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB45_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -14234,6 +14339,7 @@ define void @global_agent_atomic_fsub_noret_v2f16__offset12b_pos(ptr addrspace(1
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB46_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -14279,7 +14385,7 @@ define void @global_agent_atomic_fsub_noret_v2f16__offset12b_pos(ptr addrspace(1
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-NEXT: v_mov_b32_e32 v4, v3
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB46_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -14487,6 +14593,7 @@ define void @global_agent_atomic_fsub_noret_v2f16__offset12b_neg(ptr addrspace(1
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB47_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -14532,7 +14639,7 @@ define void @global_agent_atomic_fsub_noret_v2f16__offset12b_neg(ptr addrspace(1
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-NEXT: v_mov_b32_e32 v4, v3
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB47_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -14750,9 +14857,11 @@ define <2 x half> @global_system_atomic_fsub_ret_v2f16__offset12b_pos(ptr addrsp
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB48_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: v_mov_b32_e32 v0, v3
; GFX12-NEXT: s_setpc_b64 s[30:31]
;
@@ -14798,11 +14907,12 @@ define <2 x half> @global_system_atomic_fsub_ret_v2f16__offset12b_pos(ptr addrsp
; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB48_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: v_mov_b32_e32 v0, v3
; GFX11-NEXT: s_setpc_b64 s[30:31]
;
@@ -15020,6 +15130,7 @@ define void @global_system_atomic_fsub_noret_v2f16__offset12b_pos(ptr addrspace(
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB49_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -15065,7 +15176,7 @@ define void @global_system_atomic_fsub_noret_v2f16__offset12b_pos(ptr addrspace(
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-NEXT: v_mov_b32_e32 v4, v3
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB49_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -15306,9 +15417,11 @@ define <2 x bfloat> @global_agent_atomic_fsub_ret_v2bf16(ptr addrspace(1) %ptr,
; GFX12-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_cbranch_execnz .LBB50_1
; GFX12-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: v_mov_b32_e32 v0, v3
; GFX12-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -15357,9 +15470,11 @@ define <2 x bfloat> @global_agent_atomic_fsub_ret_v2bf16(ptr addrspace(1) %ptr,
; GFX12-FAKE16-NEXT: s_or_b32 s1, vcc_lo, s1
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: s_cbranch_execnz .LBB50_1
; GFX12-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s1
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: v_mov_b32_e32 v0, v3
; GFX12-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -15449,12 +15564,13 @@ define <2 x bfloat> @global_agent_atomic_fsub_ret_v2bf16(ptr addrspace(1) %ptr,
; GFX11-TRUE16-NEXT: buffer_gl0_inv
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v6
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB50_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-TRUE16-NEXT: s_set_inst_prefetch_distance 0x2
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_mov_b32_e32 v0, v3
; GFX11-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -15498,12 +15614,13 @@ define <2 x bfloat> @global_agent_atomic_fsub_ret_v2bf16(ptr addrspace(1) %ptr,
; GFX11-FAKE16-NEXT: buffer_gl0_inv
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v6
; GFX11-FAKE16-NEXT: s_or_b32 s1, vcc_lo, s1
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB50_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-FAKE16-NEXT: s_set_inst_prefetch_distance 0x2
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s1
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v0, v3
; GFX11-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -15821,9 +15938,11 @@ define <2 x bfloat> @global_agent_atomic_fsub_ret_v2bf16__offset12b_pos(ptr addr
; GFX12-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_cbranch_execnz .LBB51_1
; GFX12-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: v_mov_b32_e32 v0, v3
; GFX12-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -15872,9 +15991,11 @@ define <2 x bfloat> @global_agent_atomic_fsub_ret_v2bf16__offset12b_pos(ptr addr
; GFX12-FAKE16-NEXT: s_or_b32 s1, vcc_lo, s1
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: s_cbranch_execnz .LBB51_1
; GFX12-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s1
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: v_mov_b32_e32 v0, v3
; GFX12-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -15964,12 +16085,13 @@ define <2 x bfloat> @global_agent_atomic_fsub_ret_v2bf16__offset12b_pos(ptr addr
; GFX11-TRUE16-NEXT: buffer_gl0_inv
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v6
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB51_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-TRUE16-NEXT: s_set_inst_prefetch_distance 0x2
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_mov_b32_e32 v0, v3
; GFX11-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -16013,12 +16135,13 @@ define <2 x bfloat> @global_agent_atomic_fsub_ret_v2bf16__offset12b_pos(ptr addr
; GFX11-FAKE16-NEXT: buffer_gl0_inv
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v6
; GFX11-FAKE16-NEXT: s_or_b32 s1, vcc_lo, s1
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB51_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-FAKE16-NEXT: s_set_inst_prefetch_distance 0x2
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s1
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v0, v3
; GFX11-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -16338,9 +16461,11 @@ define <2 x bfloat> @global_agent_atomic_fsub_ret_v2bf16__offset12b_neg(ptr addr
; GFX12-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_cbranch_execnz .LBB52_1
; GFX12-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: v_mov_b32_e32 v0, v3
; GFX12-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -16389,9 +16514,11 @@ define <2 x bfloat> @global_agent_atomic_fsub_ret_v2bf16__offset12b_neg(ptr addr
; GFX12-FAKE16-NEXT: s_or_b32 s1, vcc_lo, s1
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: s_cbranch_execnz .LBB52_1
; GFX12-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s1
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: v_mov_b32_e32 v0, v3
; GFX12-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -16481,12 +16608,13 @@ define <2 x bfloat> @global_agent_atomic_fsub_ret_v2bf16__offset12b_neg(ptr addr
; GFX11-TRUE16-NEXT: buffer_gl0_inv
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v6
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB52_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-TRUE16-NEXT: s_set_inst_prefetch_distance 0x2
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_mov_b32_e32 v0, v3
; GFX11-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -16530,12 +16658,13 @@ define <2 x bfloat> @global_agent_atomic_fsub_ret_v2bf16__offset12b_neg(ptr addr
; GFX11-FAKE16-NEXT: buffer_gl0_inv
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v6
; GFX11-FAKE16-NEXT: s_or_b32 s1, vcc_lo, s1
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB52_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-FAKE16-NEXT: s_set_inst_prefetch_distance 0x2
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s1
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v0, v3
; GFX11-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -16861,6 +16990,7 @@ define void @global_agent_atomic_fsub_noret_v2bf16(ptr addrspace(1) %ptr, <2 x b
; GFX12-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_cbranch_execnz .LBB53_1
; GFX12-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -16895,11 +17025,10 @@ define void @global_agent_atomic_fsub_noret_v2bf16(ptr addrspace(1) %ptr, <2 x b
; GFX12-FAKE16-NEXT: v_add3_u32 v8, v8, v6, 0x7fff
; GFX12-FAKE16-NEXT: v_cmp_u_f32_e64 s0, v2, v2
; GFX12-FAKE16-NEXT: s_wait_alu depctr_va_vcc(0)
-; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_2)
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_2) | instid1(VALU_DEP_1)
; GFX12-FAKE16-NEXT: v_cndmask_b32_e32 v6, v8, v10, vcc_lo
; GFX12-FAKE16-NEXT: s_wait_alu depctr_va_sdst(0)
; GFX12-FAKE16-NEXT: v_cndmask_b32_e64 v2, v7, v9, s0
-; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-FAKE16-NEXT: v_perm_b32 v2, v6, v2, 0x7060302
; GFX12-FAKE16-NEXT: s_wait_storecnt 0x0
; GFX12-FAKE16-NEXT: global_atomic_cmpswap_b32 v2, v[0:1], v[2:3], off th:TH_ATOMIC_RETURN scope:SCOPE_DEV
@@ -16911,6 +17040,7 @@ define void @global_agent_atomic_fsub_noret_v2bf16(ptr addrspace(1) %ptr, <2 x b
; GFX12-FAKE16-NEXT: s_or_b32 s1, vcc_lo, s1
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: s_cbranch_execnz .LBB53_1
; GFX12-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s1
@@ -17000,7 +17130,7 @@ define void @global_agent_atomic_fsub_noret_v2bf16(ptr addrspace(1) %ptr, <2 x b
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v3
; GFX11-TRUE16-NEXT: v_mov_b32_e32 v3, v2
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB53_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -17034,7 +17164,7 @@ define void @global_agent_atomic_fsub_noret_v2bf16(ptr addrspace(1) %ptr, <2 x b
; GFX11-FAKE16-NEXT: v_add3_u32 v7, v7, v2, 0x7fff
; GFX11-FAKE16-NEXT: v_add3_u32 v8, v8, v6, 0x7fff
; GFX11-FAKE16-NEXT: v_cmp_u_f32_e64 s0, v2, v2
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_4)
; GFX11-FAKE16-NEXT: v_cndmask_b32_e32 v6, v8, v10, vcc_lo
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v2, v7, v9, s0
; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
@@ -17047,7 +17177,7 @@ define void @global_agent_atomic_fsub_noret_v2bf16(ptr addrspace(1) %ptr, <2 x b
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v3
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v3, v2
; GFX11-FAKE16-NEXT: s_or_b32 s1, vcc_lo, s1
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB53_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -17355,6 +17485,7 @@ define void @global_agent_atomic_fsub_noret_v2bf16__offset12b_pos(ptr addrspace(
; GFX12-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_cbranch_execnz .LBB54_1
; GFX12-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -17389,11 +17520,10 @@ define void @global_agent_atomic_fsub_noret_v2bf16__offset12b_pos(ptr addrspace(
; GFX12-FAKE16-NEXT: v_add3_u32 v8, v8, v6, 0x7fff
; GFX12-FAKE16-NEXT: v_cmp_u_f32_e64 s0, v2, v2
; GFX12-FAKE16-NEXT: s_wait_alu depctr_va_vcc(0)
-; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_2)
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_2) | instid1(VALU_DEP_1)
; GFX12-FAKE16-NEXT: v_cndmask_b32_e32 v6, v8, v10, vcc_lo
; GFX12-FAKE16-NEXT: s_wait_alu depctr_va_sdst(0)
; GFX12-FAKE16-NEXT: v_cndmask_b32_e64 v2, v7, v9, s0
-; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-FAKE16-NEXT: v_perm_b32 v2, v6, v2, 0x7060302
; GFX12-FAKE16-NEXT: s_wait_storecnt 0x0
; GFX12-FAKE16-NEXT: global_atomic_cmpswap_b32 v2, v[0:1], v[2:3], off offset:2044 th:TH_ATOMIC_RETURN scope:SCOPE_DEV
@@ -17405,6 +17535,7 @@ define void @global_agent_atomic_fsub_noret_v2bf16__offset12b_pos(ptr addrspace(
; GFX12-FAKE16-NEXT: s_or_b32 s1, vcc_lo, s1
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: s_cbranch_execnz .LBB54_1
; GFX12-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s1
@@ -17494,7 +17625,7 @@ define void @global_agent_atomic_fsub_noret_v2bf16__offset12b_pos(ptr addrspace(
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v3
; GFX11-TRUE16-NEXT: v_mov_b32_e32 v3, v2
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB54_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -17528,7 +17659,7 @@ define void @global_agent_atomic_fsub_noret_v2bf16__offset12b_pos(ptr addrspace(
; GFX11-FAKE16-NEXT: v_add3_u32 v7, v7, v2, 0x7fff
; GFX11-FAKE16-NEXT: v_add3_u32 v8, v8, v6, 0x7fff
; GFX11-FAKE16-NEXT: v_cmp_u_f32_e64 s0, v2, v2
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_4)
; GFX11-FAKE16-NEXT: v_cndmask_b32_e32 v6, v8, v10, vcc_lo
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v2, v7, v9, s0
; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
@@ -17541,7 +17672,7 @@ define void @global_agent_atomic_fsub_noret_v2bf16__offset12b_pos(ptr addrspace(
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v3
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v3, v2
; GFX11-FAKE16-NEXT: s_or_b32 s1, vcc_lo, s1
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB54_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -17852,6 +17983,7 @@ define void @global_agent_atomic_fsub_noret_v2bf16__offset12b_neg(ptr addrspace(
; GFX12-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_cbranch_execnz .LBB55_1
; GFX12-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -17886,11 +18018,10 @@ define void @global_agent_atomic_fsub_noret_v2bf16__offset12b_neg(ptr addrspace(
; GFX12-FAKE16-NEXT: v_add3_u32 v8, v8, v6, 0x7fff
; GFX12-FAKE16-NEXT: v_cmp_u_f32_e64 s0, v2, v2
; GFX12-FAKE16-NEXT: s_wait_alu depctr_va_vcc(0)
-; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_2)
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_2) | instid1(VALU_DEP_1)
; GFX12-FAKE16-NEXT: v_cndmask_b32_e32 v6, v8, v10, vcc_lo
; GFX12-FAKE16-NEXT: s_wait_alu depctr_va_sdst(0)
; GFX12-FAKE16-NEXT: v_cndmask_b32_e64 v2, v7, v9, s0
-; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-FAKE16-NEXT: v_perm_b32 v2, v6, v2, 0x7060302
; GFX12-FAKE16-NEXT: s_wait_storecnt 0x0
; GFX12-FAKE16-NEXT: global_atomic_cmpswap_b32 v2, v[0:1], v[2:3], off offset:-2048 th:TH_ATOMIC_RETURN scope:SCOPE_DEV
@@ -17902,6 +18033,7 @@ define void @global_agent_atomic_fsub_noret_v2bf16__offset12b_neg(ptr addrspace(
; GFX12-FAKE16-NEXT: s_or_b32 s1, vcc_lo, s1
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: s_cbranch_execnz .LBB55_1
; GFX12-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s1
@@ -17991,7 +18123,7 @@ define void @global_agent_atomic_fsub_noret_v2bf16__offset12b_neg(ptr addrspace(
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v3
; GFX11-TRUE16-NEXT: v_mov_b32_e32 v3, v2
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB55_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -18025,7 +18157,7 @@ define void @global_agent_atomic_fsub_noret_v2bf16__offset12b_neg(ptr addrspace(
; GFX11-FAKE16-NEXT: v_add3_u32 v7, v7, v2, 0x7fff
; GFX11-FAKE16-NEXT: v_add3_u32 v8, v8, v6, 0x7fff
; GFX11-FAKE16-NEXT: v_cmp_u_f32_e64 s0, v2, v2
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_4)
; GFX11-FAKE16-NEXT: v_cndmask_b32_e32 v6, v8, v10, vcc_lo
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v2, v7, v9, s0
; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
@@ -18038,7 +18170,7 @@ define void @global_agent_atomic_fsub_noret_v2bf16__offset12b_neg(ptr addrspace(
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v3
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v3, v2
; GFX11-FAKE16-NEXT: s_or_b32 s1, vcc_lo, s1
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB55_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -18360,9 +18492,11 @@ define <2 x bfloat> @global_system_atomic_fsub_ret_v2bf16__offset12b_pos(ptr add
; GFX12-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_cbranch_execnz .LBB56_1
; GFX12-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: v_mov_b32_e32 v0, v3
; GFX12-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -18412,9 +18546,11 @@ define <2 x bfloat> @global_system_atomic_fsub_ret_v2bf16__offset12b_pos(ptr add
; GFX12-FAKE16-NEXT: s_or_b32 s1, vcc_lo, s1
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: s_cbranch_execnz .LBB56_1
; GFX12-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s1
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: v_mov_b32_e32 v0, v3
; GFX12-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -18504,12 +18640,13 @@ define <2 x bfloat> @global_system_atomic_fsub_ret_v2bf16__offset12b_pos(ptr add
; GFX11-TRUE16-NEXT: buffer_gl0_inv
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v6
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB56_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-TRUE16-NEXT: s_set_inst_prefetch_distance 0x2
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_mov_b32_e32 v0, v3
; GFX11-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -18553,12 +18690,13 @@ define <2 x bfloat> @global_system_atomic_fsub_ret_v2bf16__offset12b_pos(ptr add
; GFX11-FAKE16-NEXT: buffer_gl0_inv
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v6
; GFX11-FAKE16-NEXT: s_or_b32 s1, vcc_lo, s1
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB56_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-FAKE16-NEXT: s_set_inst_prefetch_distance 0x2
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s1
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v0, v3
; GFX11-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -18880,6 +19018,7 @@ define void @global_system_atomic_fsub_noret_v2bf16__offset12b_pos(ptr addrspace
; GFX12-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_cbranch_execnz .LBB57_1
; GFX12-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -18914,11 +19053,10 @@ define void @global_system_atomic_fsub_noret_v2bf16__offset12b_pos(ptr addrspace
; GFX12-FAKE16-NEXT: v_add3_u32 v8, v8, v6, 0x7fff
; GFX12-FAKE16-NEXT: v_cmp_u_f32_e64 s0, v2, v2
; GFX12-FAKE16-NEXT: s_wait_alu depctr_va_vcc(0)
-; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_2)
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_2) | instid1(VALU_DEP_1)
; GFX12-FAKE16-NEXT: v_cndmask_b32_e32 v6, v8, v10, vcc_lo
; GFX12-FAKE16-NEXT: s_wait_alu depctr_va_sdst(0)
; GFX12-FAKE16-NEXT: v_cndmask_b32_e64 v2, v7, v9, s0
-; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-FAKE16-NEXT: v_perm_b32 v2, v6, v2, 0x7060302
; GFX12-FAKE16-NEXT: global_wb scope:SCOPE_SYS
; GFX12-FAKE16-NEXT: s_wait_storecnt 0x0
@@ -18931,6 +19069,7 @@ define void @global_system_atomic_fsub_noret_v2bf16__offset12b_pos(ptr addrspace
; GFX12-FAKE16-NEXT: s_or_b32 s1, vcc_lo, s1
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: s_cbranch_execnz .LBB57_1
; GFX12-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s1
@@ -19020,7 +19159,7 @@ define void @global_system_atomic_fsub_noret_v2bf16__offset12b_pos(ptr addrspace
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v3
; GFX11-TRUE16-NEXT: v_mov_b32_e32 v3, v2
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB57_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -19054,7 +19193,7 @@ define void @global_system_atomic_fsub_noret_v2bf16__offset12b_pos(ptr addrspace
; GFX11-FAKE16-NEXT: v_add3_u32 v7, v7, v2, 0x7fff
; GFX11-FAKE16-NEXT: v_add3_u32 v8, v8, v6, 0x7fff
; GFX11-FAKE16-NEXT: v_cmp_u_f32_e64 s0, v2, v2
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_4)
; GFX11-FAKE16-NEXT: v_cndmask_b32_e32 v6, v8, v10, vcc_lo
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v2, v7, v9, s0
; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
@@ -19067,7 +19206,7 @@ define void @global_system_atomic_fsub_noret_v2bf16__offset12b_pos(ptr addrspace
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v3
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v3, v2
; GFX11-FAKE16-NEXT: s_or_b32 s1, vcc_lo, s1
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB57_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
diff --git a/llvm/test/CodeGen/AMDGPU/global-saddr-atomics-min-max-system.ll b/llvm/test/CodeGen/AMDGPU/global-saddr-atomics-min-max-system.ll
index 7b7a8f0ed101e7..cdcc342199abb0 100644
--- a/llvm/test/CodeGen/AMDGPU/global-saddr-atomics-min-max-system.ll
+++ b/llvm/test/CodeGen/AMDGPU/global-saddr-atomics-min-max-system.ll
@@ -64,7 +64,6 @@ define amdgpu_ps float @global_max_saddr_i32_rtn(ptr addrspace(1) inreg %sbase,
; GFX11-NEXT: v_mov_b32_e32 v2, v0
; GFX11-NEXT: global_load_b32 v0, v0, s[2:3]
; GFX11-NEXT: v_add_co_u32 v2, s[0:1], s2, v2
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v3, null, s3, 0, s[0:1]
; GFX11-NEXT: s_mov_b64 s[0:1], 0
; GFX11-NEXT: s_waitcnt_depctr depctr_sa_sdst(0)
@@ -79,9 +78,10 @@ define amdgpu_ps float @global_max_saddr_i32_rtn(ptr addrspace(1) inreg %sbase,
; GFX11-NEXT: buffer_gl1_inv
; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc, v0, v5
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_or_b64 s[0:1], vcc, s[0:1]
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b64 exec, exec, s[0:1]
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_execnz .LBB0_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-NEXT: s_or_b64 exec, exec, s[0:1]
@@ -92,7 +92,6 @@ define amdgpu_ps float @global_max_saddr_i32_rtn(ptr addrspace(1) inreg %sbase,
; GFX12-NEXT: v_mov_b32_e32 v2, v0
; GFX12-NEXT: global_load_b32 v0, v0, s[2:3]
; GFX12-NEXT: v_add_co_u32 v2, s[0:1], s2, v2
-; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-NEXT: v_add_co_ci_u32_e64 v3, null, s3, 0, s[0:1]
; GFX12-NEXT: s_mov_b64 s[0:1], 0
; GFX12-NEXT: .LBB0_1: ; %atomicrmw.start
@@ -108,6 +107,7 @@ define amdgpu_ps float @global_max_saddr_i32_rtn(ptr addrspace(1) inreg %sbase,
; GFX12-NEXT: global_inv scope:SCOPE_SYS
; GFX12-NEXT: v_cmp_eq_u32_e32 vcc, v0, v5
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
+; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_or_b64 s[0:1], vcc, s[0:1]
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b64 exec, exec, s[0:1]
@@ -176,7 +176,6 @@ define amdgpu_ps float @global_max_saddr_i32_rtn_neg128(ptr addrspace(1) inreg %
; GFX11-NEXT: v_mov_b32_e32 v2, v0
; GFX11-NEXT: global_load_b32 v0, v0, s[2:3] offset:-128
; GFX11-NEXT: v_add_co_u32 v2, s[0:1], s2, v2
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v3, null, s3, 0, s[0:1]
; GFX11-NEXT: s_mov_b64 s[0:1], 0
; GFX11-NEXT: s_waitcnt_depctr depctr_sa_sdst(0)
@@ -191,9 +190,10 @@ define amdgpu_ps float @global_max_saddr_i32_rtn_neg128(ptr addrspace(1) inreg %
; GFX11-NEXT: buffer_gl1_inv
; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc, v0, v5
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_or_b64 s[0:1], vcc, s[0:1]
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b64 exec, exec, s[0:1]
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_execnz .LBB1_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-NEXT: s_or_b64 exec, exec, s[0:1]
@@ -204,7 +204,6 @@ define amdgpu_ps float @global_max_saddr_i32_rtn_neg128(ptr addrspace(1) inreg %
; GFX12-NEXT: v_mov_b32_e32 v2, v0
; GFX12-NEXT: global_load_b32 v0, v0, s[2:3] offset:-128
; GFX12-NEXT: v_add_co_u32 v2, s[0:1], s2, v2
-; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-NEXT: v_add_co_ci_u32_e64 v3, null, s3, 0, s[0:1]
; GFX12-NEXT: s_mov_b64 s[0:1], 0
; GFX12-NEXT: .LBB1_1: ; %atomicrmw.start
@@ -220,6 +219,7 @@ define amdgpu_ps float @global_max_saddr_i32_rtn_neg128(ptr addrspace(1) inreg %
; GFX12-NEXT: global_inv scope:SCOPE_SYS
; GFX12-NEXT: v_cmp_eq_u32_e32 vcc, v0, v5
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
+; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_or_b64 s[0:1], vcc, s[0:1]
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b64 exec, exec, s[0:1]
@@ -284,7 +284,6 @@ define amdgpu_ps void @global_max_saddr_i32_nortn(ptr addrspace(1) inreg %sbase,
; GFX11: ; %bb.0:
; GFX11-NEXT: global_load_b32 v5, v0, s[2:3]
; GFX11-NEXT: v_add_co_u32 v2, s[0:1], s2, v0
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v3, null, s3, 0, s[0:1]
; GFX11-NEXT: s_mov_b64 s[0:1], 0
; GFX11-NEXT: s_waitcnt_depctr depctr_sa_sdst(0)
@@ -298,9 +297,10 @@ define amdgpu_ps void @global_max_saddr_i32_nortn(ptr addrspace(1) inreg %sbase,
; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc, v0, v5
; GFX11-NEXT: v_mov_b32_e32 v5, v0
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_or_b64 s[0:1], vcc, s[0:1]
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b64 exec, exec, s[0:1]
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_execnz .LBB2_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-NEXT: s_endpgm
@@ -309,7 +309,6 @@ define amdgpu_ps void @global_max_saddr_i32_nortn(ptr addrspace(1) inreg %sbase,
; GFX12: ; %bb.0:
; GFX12-NEXT: global_load_b32 v5, v0, s[2:3]
; GFX12-NEXT: v_add_co_u32 v2, s[0:1], s2, v0
-; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-NEXT: v_add_co_ci_u32_e64 v3, null, s3, 0, s[0:1]
; GFX12-NEXT: s_mov_b64 s[0:1], 0
; GFX12-NEXT: .LBB2_1: ; %atomicrmw.start
@@ -324,6 +323,7 @@ define amdgpu_ps void @global_max_saddr_i32_nortn(ptr addrspace(1) inreg %sbase,
; GFX12-NEXT: v_cmp_eq_u32_e32 vcc, v0, v5
; GFX12-NEXT: v_mov_b32_e32 v5, v0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
+; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_or_b64 s[0:1], vcc, s[0:1]
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b64 exec, exec, s[0:1]
@@ -385,7 +385,6 @@ define amdgpu_ps void @global_max_saddr_i32_nortn_neg128(ptr addrspace(1) inreg
; GFX11: ; %bb.0:
; GFX11-NEXT: global_load_b32 v5, v0, s[2:3] offset:-128
; GFX11-NEXT: v_add_co_u32 v2, s[0:1], s2, v0
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v3, null, s3, 0, s[0:1]
; GFX11-NEXT: s_mov_b64 s[0:1], 0
; GFX11-NEXT: s_waitcnt_depctr depctr_sa_sdst(0)
@@ -399,9 +398,10 @@ define amdgpu_ps void @global_max_saddr_i32_nortn_neg128(ptr addrspace(1) inreg
; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc, v0, v5
; GFX11-NEXT: v_mov_b32_e32 v5, v0
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_or_b64 s[0:1], vcc, s[0:1]
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b64 exec, exec, s[0:1]
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_execnz .LBB3_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-NEXT: s_endpgm
@@ -410,7 +410,6 @@ define amdgpu_ps void @global_max_saddr_i32_nortn_neg128(ptr addrspace(1) inreg
; GFX12: ; %bb.0:
; GFX12-NEXT: global_load_b32 v5, v0, s[2:3] offset:-128
; GFX12-NEXT: v_add_co_u32 v2, s[0:1], s2, v0
-; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-NEXT: v_add_co_ci_u32_e64 v3, null, s3, 0, s[0:1]
; GFX12-NEXT: s_mov_b64 s[0:1], 0
; GFX12-NEXT: .LBB3_1: ; %atomicrmw.start
@@ -425,6 +424,7 @@ define amdgpu_ps void @global_max_saddr_i32_nortn_neg128(ptr addrspace(1) inreg
; GFX12-NEXT: v_cmp_eq_u32_e32 vcc, v0, v5
; GFX12-NEXT: v_mov_b32_e32 v5, v0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
+; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_or_b64 s[0:1], vcc, s[0:1]
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b64 exec, exec, s[0:1]
@@ -499,7 +499,6 @@ define amdgpu_ps <2 x float> @global_max_saddr_i64_rtn(ptr addrspace(1) inreg %s
; GFX11: ; %bb.0:
; GFX11-NEXT: global_load_b64 v[3:4], v0, s[2:3]
; GFX11-NEXT: v_add_co_u32 v5, s[0:1], s2, v0
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v6, null, s3, 0, s[0:1]
; GFX11-NEXT: s_mov_b64 s[0:1], 0
; GFX11-NEXT: s_waitcnt_depctr depctr_sa_sdst(0)
@@ -518,12 +517,14 @@ define amdgpu_ps <2 x float> @global_max_saddr_i64_rtn(ptr addrspace(1) inreg %s
; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: v_cmp_eq_u64_e32 vcc, v[3:4], v[9:10]
; GFX11-NEXT: s_waitcnt_depctr depctr_va_vcc(0)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_or_b64 s[0:1], vcc, s[0:1]
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b64 exec, exec, s[0:1]
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_execnz .LBB4_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-NEXT: s_or_b64 exec, exec, s[0:1]
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: v_mov_b32_e32 v0, v3
; GFX11-NEXT: v_mov_b32_e32 v1, v4
; GFX11-NEXT: ; return to shader part epilog
@@ -532,7 +533,6 @@ define amdgpu_ps <2 x float> @global_max_saddr_i64_rtn(ptr addrspace(1) inreg %s
; GFX12: ; %bb.0:
; GFX12-NEXT: global_load_b64 v[3:4], v0, s[2:3]
; GFX12-NEXT: v_add_co_u32 v5, s[0:1], s2, v0
-; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-NEXT: v_add_co_ci_u32_e64 v6, null, s3, 0, s[0:1]
; GFX12-NEXT: s_mov_b64 s[0:1], 0
; GFX12-NEXT: .LBB4_1: ; %atomicrmw.start
@@ -552,12 +552,14 @@ define amdgpu_ps <2 x float> @global_max_saddr_i64_rtn(ptr addrspace(1) inreg %s
; GFX12-NEXT: global_inv scope:SCOPE_SYS
; GFX12-NEXT: v_cmp_eq_u64_e32 vcc, v[3:4], v[9:10]
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
+; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_or_b64 s[0:1], vcc, s[0:1]
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b64 exec, exec, s[0:1]
; GFX12-NEXT: s_cbranch_execnz .LBB4_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b64 exec, exec, s[0:1]
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: v_mov_b32_e32 v0, v3
; GFX12-NEXT: v_mov_b32_e32 v1, v4
; GFX12-NEXT: ; return to shader part epilog
@@ -629,7 +631,6 @@ define amdgpu_ps <2 x float> @global_max_saddr_i64_rtn_neg128(ptr addrspace(1) i
; GFX11: ; %bb.0:
; GFX11-NEXT: global_load_b64 v[3:4], v0, s[2:3] offset:-128
; GFX11-NEXT: v_add_co_u32 v5, s[0:1], s2, v0
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v6, null, s3, 0, s[0:1]
; GFX11-NEXT: s_mov_b64 s[0:1], 0
; GFX11-NEXT: s_waitcnt_depctr depctr_sa_sdst(0)
@@ -648,12 +649,14 @@ define amdgpu_ps <2 x float> @global_max_saddr_i64_rtn_neg128(ptr addrspace(1) i
; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: v_cmp_eq_u64_e32 vcc, v[3:4], v[9:10]
; GFX11-NEXT: s_waitcnt_depctr depctr_va_vcc(0)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_or_b64 s[0:1], vcc, s[0:1]
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b64 exec, exec, s[0:1]
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_execnz .LBB5_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-NEXT: s_or_b64 exec, exec, s[0:1]
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: v_mov_b32_e32 v0, v3
; GFX11-NEXT: v_mov_b32_e32 v1, v4
; GFX11-NEXT: ; return to shader part epilog
@@ -662,7 +665,6 @@ define amdgpu_ps <2 x float> @global_max_saddr_i64_rtn_neg128(ptr addrspace(1) i
; GFX12: ; %bb.0:
; GFX12-NEXT: global_load_b64 v[3:4], v0, s[2:3] offset:-128
; GFX12-NEXT: v_add_co_u32 v5, s[0:1], s2, v0
-; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-NEXT: v_add_co_ci_u32_e64 v6, null, s3, 0, s[0:1]
; GFX12-NEXT: s_mov_b64 s[0:1], 0
; GFX12-NEXT: .LBB5_1: ; %atomicrmw.start
@@ -682,12 +684,14 @@ define amdgpu_ps <2 x float> @global_max_saddr_i64_rtn_neg128(ptr addrspace(1) i
; GFX12-NEXT: global_inv scope:SCOPE_SYS
; GFX12-NEXT: v_cmp_eq_u64_e32 vcc, v[3:4], v[9:10]
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
+; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_or_b64 s[0:1], vcc, s[0:1]
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b64 exec, exec, s[0:1]
; GFX12-NEXT: s_cbranch_execnz .LBB5_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b64 exec, exec, s[0:1]
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: v_mov_b32_e32 v0, v3
; GFX12-NEXT: v_mov_b32_e32 v1, v4
; GFX12-NEXT: ; return to shader part epilog
@@ -754,7 +758,6 @@ define amdgpu_ps void @global_max_saddr_i64_nortn(ptr addrspace(1) inreg %sbase,
; GFX11: ; %bb.0:
; GFX11-NEXT: global_load_b64 v[5:6], v0, s[2:3]
; GFX11-NEXT: v_add_co_u32 v7, s[0:1], s2, v0
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v8, null, s3, 0, s[0:1]
; GFX11-NEXT: s_mov_b64 s[0:1], 0
; GFX11-NEXT: s_waitcnt_depctr depctr_sa_sdst(0)
@@ -772,9 +775,10 @@ define amdgpu_ps void @global_max_saddr_i64_nortn(ptr addrspace(1) inreg %sbase,
; GFX11-NEXT: v_mov_b32_e32 v6, v4
; GFX11-NEXT: v_mov_b32_e32 v5, v3
; GFX11-NEXT: s_waitcnt_depctr depctr_va_vcc(0)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_or_b64 s[0:1], vcc, s[0:1]
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b64 exec, exec, s[0:1]
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_execnz .LBB6_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-NEXT: s_endpgm
@@ -783,7 +787,6 @@ define amdgpu_ps void @global_max_saddr_i64_nortn(ptr addrspace(1) inreg %sbase,
; GFX12: ; %bb.0:
; GFX12-NEXT: global_load_b64 v[5:6], v0, s[2:3]
; GFX12-NEXT: v_add_co_u32 v7, s[0:1], s2, v0
-; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-NEXT: v_add_co_ci_u32_e64 v8, null, s3, 0, s[0:1]
; GFX12-NEXT: s_mov_b64 s[0:1], 0
; GFX12-NEXT: .LBB6_1: ; %atomicrmw.start
@@ -802,6 +805,7 @@ define amdgpu_ps void @global_max_saddr_i64_nortn(ptr addrspace(1) inreg %sbase,
; GFX12-NEXT: v_mov_b32_e32 v6, v4
; GFX12-NEXT: v_mov_b32_e32 v5, v3
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
+; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_or_b64 s[0:1], vcc, s[0:1]
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b64 exec, exec, s[0:1]
@@ -869,7 +873,6 @@ define amdgpu_ps void @global_max_saddr_i64_nortn_neg128(ptr addrspace(1) inreg
; GFX11: ; %bb.0:
; GFX11-NEXT: global_load_b64 v[5:6], v0, s[2:3] offset:-128
; GFX11-NEXT: v_add_co_u32 v7, s[0:1], s2, v0
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v8, null, s3, 0, s[0:1]
; GFX11-NEXT: s_mov_b64 s[0:1], 0
; GFX11-NEXT: s_waitcnt_depctr depctr_sa_sdst(0)
@@ -887,9 +890,10 @@ define amdgpu_ps void @global_max_saddr_i64_nortn_neg128(ptr addrspace(1) inreg
; GFX11-NEXT: v_mov_b32_e32 v6, v4
; GFX11-NEXT: v_mov_b32_e32 v5, v3
; GFX11-NEXT: s_waitcnt_depctr depctr_va_vcc(0)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_or_b64 s[0:1], vcc, s[0:1]
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b64 exec, exec, s[0:1]
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_execnz .LBB7_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-NEXT: s_endpgm
@@ -898,7 +902,6 @@ define amdgpu_ps void @global_max_saddr_i64_nortn_neg128(ptr addrspace(1) inreg
; GFX12: ; %bb.0:
; GFX12-NEXT: global_load_b64 v[5:6], v0, s[2:3] offset:-128
; GFX12-NEXT: v_add_co_u32 v7, s[0:1], s2, v0
-; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-NEXT: v_add_co_ci_u32_e64 v8, null, s3, 0, s[0:1]
; GFX12-NEXT: s_mov_b64 s[0:1], 0
; GFX12-NEXT: .LBB7_1: ; %atomicrmw.start
@@ -917,6 +920,7 @@ define amdgpu_ps void @global_max_saddr_i64_nortn_neg128(ptr addrspace(1) inreg
; GFX12-NEXT: v_mov_b32_e32 v6, v4
; GFX12-NEXT: v_mov_b32_e32 v5, v3
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
+; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_or_b64 s[0:1], vcc, s[0:1]
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b64 exec, exec, s[0:1]
@@ -988,7 +992,6 @@ define amdgpu_ps float @global_min_saddr_i32_rtn(ptr addrspace(1) inreg %sbase,
; GFX11-NEXT: v_mov_b32_e32 v2, v0
; GFX11-NEXT: global_load_b32 v0, v0, s[2:3]
; GFX11-NEXT: v_add_co_u32 v2, s[0:1], s2, v2
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v3, null, s3, 0, s[0:1]
; GFX11-NEXT: s_mov_b64 s[0:1], 0
; GFX11-NEXT: s_waitcnt_depctr depctr_sa_sdst(0)
@@ -1003,9 +1006,10 @@ define amdgpu_ps float @global_min_saddr_i32_rtn(ptr addrspace(1) inreg %sbase,
; GFX11-NEXT: buffer_gl1_inv
; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc, v0, v5
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_or_b64 s[0:1], vcc, s[0:1]
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b64 exec, exec, s[0:1]
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_execnz .LBB8_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-NEXT: s_or_b64 exec, exec, s[0:1]
@@ -1016,7 +1020,6 @@ define amdgpu_ps float @global_min_saddr_i32_rtn(ptr addrspace(1) inreg %sbase,
; GFX12-NEXT: v_mov_b32_e32 v2, v0
; GFX12-NEXT: global_load_b32 v0, v0, s[2:3]
; GFX12-NEXT: v_add_co_u32 v2, s[0:1], s2, v2
-; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-NEXT: v_add_co_ci_u32_e64 v3, null, s3, 0, s[0:1]
; GFX12-NEXT: s_mov_b64 s[0:1], 0
; GFX12-NEXT: .LBB8_1: ; %atomicrmw.start
@@ -1032,6 +1035,7 @@ define amdgpu_ps float @global_min_saddr_i32_rtn(ptr addrspace(1) inreg %sbase,
; GFX12-NEXT: global_inv scope:SCOPE_SYS
; GFX12-NEXT: v_cmp_eq_u32_e32 vcc, v0, v5
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
+; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_or_b64 s[0:1], vcc, s[0:1]
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b64 exec, exec, s[0:1]
@@ -1100,7 +1104,6 @@ define amdgpu_ps float @global_min_saddr_i32_rtn_neg128(ptr addrspace(1) inreg %
; GFX11-NEXT: v_mov_b32_e32 v2, v0
; GFX11-NEXT: global_load_b32 v0, v0, s[2:3] offset:-128
; GFX11-NEXT: v_add_co_u32 v2, s[0:1], s2, v2
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v3, null, s3, 0, s[0:1]
; GFX11-NEXT: s_mov_b64 s[0:1], 0
; GFX11-NEXT: s_waitcnt_depctr depctr_sa_sdst(0)
@@ -1115,9 +1118,10 @@ define amdgpu_ps float @global_min_saddr_i32_rtn_neg128(ptr addrspace(1) inreg %
; GFX11-NEXT: buffer_gl1_inv
; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc, v0, v5
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_or_b64 s[0:1], vcc, s[0:1]
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b64 exec, exec, s[0:1]
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_execnz .LBB9_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-NEXT: s_or_b64 exec, exec, s[0:1]
@@ -1128,7 +1132,6 @@ define amdgpu_ps float @global_min_saddr_i32_rtn_neg128(ptr addrspace(1) inreg %
; GFX12-NEXT: v_mov_b32_e32 v2, v0
; GFX12-NEXT: global_load_b32 v0, v0, s[2:3] offset:-128
; GFX12-NEXT: v_add_co_u32 v2, s[0:1], s2, v2
-; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-NEXT: v_add_co_ci_u32_e64 v3, null, s3, 0, s[0:1]
; GFX12-NEXT: s_mov_b64 s[0:1], 0
; GFX12-NEXT: .LBB9_1: ; %atomicrmw.start
@@ -1144,6 +1147,7 @@ define amdgpu_ps float @global_min_saddr_i32_rtn_neg128(ptr addrspace(1) inreg %
; GFX12-NEXT: global_inv scope:SCOPE_SYS
; GFX12-NEXT: v_cmp_eq_u32_e32 vcc, v0, v5
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
+; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_or_b64 s[0:1], vcc, s[0:1]
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b64 exec, exec, s[0:1]
@@ -1208,7 +1212,6 @@ define amdgpu_ps void @global_min_saddr_i32_nortn(ptr addrspace(1) inreg %sbase,
; GFX11: ; %bb.0:
; GFX11-NEXT: global_load_b32 v5, v0, s[2:3]
; GFX11-NEXT: v_add_co_u32 v2, s[0:1], s2, v0
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v3, null, s3, 0, s[0:1]
; GFX11-NEXT: s_mov_b64 s[0:1], 0
; GFX11-NEXT: s_waitcnt_depctr depctr_sa_sdst(0)
@@ -1222,9 +1225,10 @@ define amdgpu_ps void @global_min_saddr_i32_nortn(ptr addrspace(1) inreg %sbase,
; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc, v0, v5
; GFX11-NEXT: v_mov_b32_e32 v5, v0
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_or_b64 s[0:1], vcc, s[0:1]
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b64 exec, exec, s[0:1]
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_execnz .LBB10_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-NEXT: s_endpgm
@@ -1233,7 +1237,6 @@ define amdgpu_ps void @global_min_saddr_i32_nortn(ptr addrspace(1) inreg %sbase,
; GFX12: ; %bb.0:
; GFX12-NEXT: global_load_b32 v5, v0, s[2:3]
; GFX12-NEXT: v_add_co_u32 v2, s[0:1], s2, v0
-; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-NEXT: v_add_co_ci_u32_e64 v3, null, s3, 0, s[0:1]
; GFX12-NEXT: s_mov_b64 s[0:1], 0
; GFX12-NEXT: .LBB10_1: ; %atomicrmw.start
@@ -1248,6 +1251,7 @@ define amdgpu_ps void @global_min_saddr_i32_nortn(ptr addrspace(1) inreg %sbase,
; GFX12-NEXT: v_cmp_eq_u32_e32 vcc, v0, v5
; GFX12-NEXT: v_mov_b32_e32 v5, v0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
+; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_or_b64 s[0:1], vcc, s[0:1]
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b64 exec, exec, s[0:1]
@@ -1309,7 +1313,6 @@ define amdgpu_ps void @global_min_saddr_i32_nortn_neg128(ptr addrspace(1) inreg
; GFX11: ; %bb.0:
; GFX11-NEXT: global_load_b32 v5, v0, s[2:3] offset:-128
; GFX11-NEXT: v_add_co_u32 v2, s[0:1], s2, v0
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v3, null, s3, 0, s[0:1]
; GFX11-NEXT: s_mov_b64 s[0:1], 0
; GFX11-NEXT: s_waitcnt_depctr depctr_sa_sdst(0)
@@ -1323,9 +1326,10 @@ define amdgpu_ps void @global_min_saddr_i32_nortn_neg128(ptr addrspace(1) inreg
; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc, v0, v5
; GFX11-NEXT: v_mov_b32_e32 v5, v0
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_or_b64 s[0:1], vcc, s[0:1]
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b64 exec, exec, s[0:1]
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_execnz .LBB11_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-NEXT: s_endpgm
@@ -1334,7 +1338,6 @@ define amdgpu_ps void @global_min_saddr_i32_nortn_neg128(ptr addrspace(1) inreg
; GFX12: ; %bb.0:
; GFX12-NEXT: global_load_b32 v5, v0, s[2:3] offset:-128
; GFX12-NEXT: v_add_co_u32 v2, s[0:1], s2, v0
-; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-NEXT: v_add_co_ci_u32_e64 v3, null, s3, 0, s[0:1]
; GFX12-NEXT: s_mov_b64 s[0:1], 0
; GFX12-NEXT: .LBB11_1: ; %atomicrmw.start
@@ -1349,6 +1352,7 @@ define amdgpu_ps void @global_min_saddr_i32_nortn_neg128(ptr addrspace(1) inreg
; GFX12-NEXT: v_cmp_eq_u32_e32 vcc, v0, v5
; GFX12-NEXT: v_mov_b32_e32 v5, v0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
+; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_or_b64 s[0:1], vcc, s[0:1]
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b64 exec, exec, s[0:1]
@@ -1423,7 +1427,6 @@ define amdgpu_ps <2 x float> @global_min_saddr_i64_rtn(ptr addrspace(1) inreg %s
; GFX11: ; %bb.0:
; GFX11-NEXT: global_load_b64 v[3:4], v0, s[2:3]
; GFX11-NEXT: v_add_co_u32 v5, s[0:1], s2, v0
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v6, null, s3, 0, s[0:1]
; GFX11-NEXT: s_mov_b64 s[0:1], 0
; GFX11-NEXT: s_waitcnt_depctr depctr_sa_sdst(0)
@@ -1442,12 +1445,14 @@ define amdgpu_ps <2 x float> @global_min_saddr_i64_rtn(ptr addrspace(1) inreg %s
; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: v_cmp_eq_u64_e32 vcc, v[3:4], v[9:10]
; GFX11-NEXT: s_waitcnt_depctr depctr_va_vcc(0)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_or_b64 s[0:1], vcc, s[0:1]
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b64 exec, exec, s[0:1]
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_execnz .LBB12_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-NEXT: s_or_b64 exec, exec, s[0:1]
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: v_mov_b32_e32 v0, v3
; GFX11-NEXT: v_mov_b32_e32 v1, v4
; GFX11-NEXT: ; return to shader part epilog
@@ -1456,7 +1461,6 @@ define amdgpu_ps <2 x float> @global_min_saddr_i64_rtn(ptr addrspace(1) inreg %s
; GFX12: ; %bb.0:
; GFX12-NEXT: global_load_b64 v[3:4], v0, s[2:3]
; GFX12-NEXT: v_add_co_u32 v5, s[0:1], s2, v0
-; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-NEXT: v_add_co_ci_u32_e64 v6, null, s3, 0, s[0:1]
; GFX12-NEXT: s_mov_b64 s[0:1], 0
; GFX12-NEXT: .LBB12_1: ; %atomicrmw.start
@@ -1476,12 +1480,14 @@ define amdgpu_ps <2 x float> @global_min_saddr_i64_rtn(ptr addrspace(1) inreg %s
; GFX12-NEXT: global_inv scope:SCOPE_SYS
; GFX12-NEXT: v_cmp_eq_u64_e32 vcc, v[3:4], v[9:10]
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
+; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_or_b64 s[0:1], vcc, s[0:1]
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b64 exec, exec, s[0:1]
; GFX12-NEXT: s_cbranch_execnz .LBB12_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b64 exec, exec, s[0:1]
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: v_mov_b32_e32 v0, v3
; GFX12-NEXT: v_mov_b32_e32 v1, v4
; GFX12-NEXT: ; return to shader part epilog
@@ -1553,7 +1559,6 @@ define amdgpu_ps <2 x float> @global_min_saddr_i64_rtn_neg128(ptr addrspace(1) i
; GFX11: ; %bb.0:
; GFX11-NEXT: global_load_b64 v[3:4], v0, s[2:3] offset:-128
; GFX11-NEXT: v_add_co_u32 v5, s[0:1], s2, v0
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v6, null, s3, 0, s[0:1]
; GFX11-NEXT: s_mov_b64 s[0:1], 0
; GFX11-NEXT: s_waitcnt_depctr depctr_sa_sdst(0)
@@ -1572,12 +1577,14 @@ define amdgpu_ps <2 x float> @global_min_saddr_i64_rtn_neg128(ptr addrspace(1) i
; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: v_cmp_eq_u64_e32 vcc, v[3:4], v[9:10]
; GFX11-NEXT: s_waitcnt_depctr depctr_va_vcc(0)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_or_b64 s[0:1], vcc, s[0:1]
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b64 exec, exec, s[0:1]
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_execnz .LBB13_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-NEXT: s_or_b64 exec, exec, s[0:1]
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: v_mov_b32_e32 v0, v3
; GFX11-NEXT: v_mov_b32_e32 v1, v4
; GFX11-NEXT: ; return to shader part epilog
@@ -1586,7 +1593,6 @@ define amdgpu_ps <2 x float> @global_min_saddr_i64_rtn_neg128(ptr addrspace(1) i
; GFX12: ; %bb.0:
; GFX12-NEXT: global_load_b64 v[3:4], v0, s[2:3] offset:-128
; GFX12-NEXT: v_add_co_u32 v5, s[0:1], s2, v0
-; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-NEXT: v_add_co_ci_u32_e64 v6, null, s3, 0, s[0:1]
; GFX12-NEXT: s_mov_b64 s[0:1], 0
; GFX12-NEXT: .LBB13_1: ; %atomicrmw.start
@@ -1606,12 +1612,14 @@ define amdgpu_ps <2 x float> @global_min_saddr_i64_rtn_neg128(ptr addrspace(1) i
; GFX12-NEXT: global_inv scope:SCOPE_SYS
; GFX12-NEXT: v_cmp_eq_u64_e32 vcc, v[3:4], v[9:10]
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
+; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_or_b64 s[0:1], vcc, s[0:1]
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b64 exec, exec, s[0:1]
; GFX12-NEXT: s_cbranch_execnz .LBB13_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b64 exec, exec, s[0:1]
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: v_mov_b32_e32 v0, v3
; GFX12-NEXT: v_mov_b32_e32 v1, v4
; GFX12-NEXT: ; return to shader part epilog
@@ -1678,7 +1686,6 @@ define amdgpu_ps void @global_min_saddr_i64_nortn(ptr addrspace(1) inreg %sbase,
; GFX11: ; %bb.0:
; GFX11-NEXT: global_load_b64 v[5:6], v0, s[2:3]
; GFX11-NEXT: v_add_co_u32 v7, s[0:1], s2, v0
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v8, null, s3, 0, s[0:1]
; GFX11-NEXT: s_mov_b64 s[0:1], 0
; GFX11-NEXT: s_waitcnt_depctr depctr_sa_sdst(0)
@@ -1696,9 +1703,10 @@ define amdgpu_ps void @global_min_saddr_i64_nortn(ptr addrspace(1) inreg %sbase,
; GFX11-NEXT: v_mov_b32_e32 v6, v4
; GFX11-NEXT: v_mov_b32_e32 v5, v3
; GFX11-NEXT: s_waitcnt_depctr depctr_va_vcc(0)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_or_b64 s[0:1], vcc, s[0:1]
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b64 exec, exec, s[0:1]
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_execnz .LBB14_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-NEXT: s_endpgm
@@ -1707,7 +1715,6 @@ define amdgpu_ps void @global_min_saddr_i64_nortn(ptr addrspace(1) inreg %sbase,
; GFX12: ; %bb.0:
; GFX12-NEXT: global_load_b64 v[5:6], v0, s[2:3]
; GFX12-NEXT: v_add_co_u32 v7, s[0:1], s2, v0
-; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-NEXT: v_add_co_ci_u32_e64 v8, null, s3, 0, s[0:1]
; GFX12-NEXT: s_mov_b64 s[0:1], 0
; GFX12-NEXT: .LBB14_1: ; %atomicrmw.start
@@ -1726,6 +1733,7 @@ define amdgpu_ps void @global_min_saddr_i64_nortn(ptr addrspace(1) inreg %sbase,
; GFX12-NEXT: v_mov_b32_e32 v6, v4
; GFX12-NEXT: v_mov_b32_e32 v5, v3
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
+; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_or_b64 s[0:1], vcc, s[0:1]
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b64 exec, exec, s[0:1]
@@ -1793,7 +1801,6 @@ define amdgpu_ps void @global_min_saddr_i64_nortn_neg128(ptr addrspace(1) inreg
; GFX11: ; %bb.0:
; GFX11-NEXT: global_load_b64 v[5:6], v0, s[2:3] offset:-128
; GFX11-NEXT: v_add_co_u32 v7, s[0:1], s2, v0
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v8, null, s3, 0, s[0:1]
; GFX11-NEXT: s_mov_b64 s[0:1], 0
; GFX11-NEXT: s_waitcnt_depctr depctr_sa_sdst(0)
@@ -1811,9 +1818,10 @@ define amdgpu_ps void @global_min_saddr_i64_nortn_neg128(ptr addrspace(1) inreg
; GFX11-NEXT: v_mov_b32_e32 v6, v4
; GFX11-NEXT: v_mov_b32_e32 v5, v3
; GFX11-NEXT: s_waitcnt_depctr depctr_va_vcc(0)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_or_b64 s[0:1], vcc, s[0:1]
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b64 exec, exec, s[0:1]
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_execnz .LBB15_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-NEXT: s_endpgm
@@ -1822,7 +1830,6 @@ define amdgpu_ps void @global_min_saddr_i64_nortn_neg128(ptr addrspace(1) inreg
; GFX12: ; %bb.0:
; GFX12-NEXT: global_load_b64 v[5:6], v0, s[2:3] offset:-128
; GFX12-NEXT: v_add_co_u32 v7, s[0:1], s2, v0
-; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-NEXT: v_add_co_ci_u32_e64 v8, null, s3, 0, s[0:1]
; GFX12-NEXT: s_mov_b64 s[0:1], 0
; GFX12-NEXT: .LBB15_1: ; %atomicrmw.start
@@ -1841,6 +1848,7 @@ define amdgpu_ps void @global_min_saddr_i64_nortn_neg128(ptr addrspace(1) inreg
; GFX12-NEXT: v_mov_b32_e32 v6, v4
; GFX12-NEXT: v_mov_b32_e32 v5, v3
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
+; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_or_b64 s[0:1], vcc, s[0:1]
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b64 exec, exec, s[0:1]
@@ -1912,7 +1920,6 @@ define amdgpu_ps float @global_umax_saddr_i32_rtn(ptr addrspace(1) inreg %sbase,
; GFX11-NEXT: v_mov_b32_e32 v2, v0
; GFX11-NEXT: global_load_b32 v0, v0, s[2:3]
; GFX11-NEXT: v_add_co_u32 v2, s[0:1], s2, v2
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v3, null, s3, 0, s[0:1]
; GFX11-NEXT: s_mov_b64 s[0:1], 0
; GFX11-NEXT: s_waitcnt_depctr depctr_sa_sdst(0)
@@ -1927,9 +1934,10 @@ define amdgpu_ps float @global_umax_saddr_i32_rtn(ptr addrspace(1) inreg %sbase,
; GFX11-NEXT: buffer_gl1_inv
; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc, v0, v5
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_or_b64 s[0:1], vcc, s[0:1]
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b64 exec, exec, s[0:1]
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_execnz .LBB16_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-NEXT: s_or_b64 exec, exec, s[0:1]
@@ -1940,7 +1948,6 @@ define amdgpu_ps float @global_umax_saddr_i32_rtn(ptr addrspace(1) inreg %sbase,
; GFX12-NEXT: v_mov_b32_e32 v2, v0
; GFX12-NEXT: global_load_b32 v0, v0, s[2:3]
; GFX12-NEXT: v_add_co_u32 v2, s[0:1], s2, v2
-; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-NEXT: v_add_co_ci_u32_e64 v3, null, s3, 0, s[0:1]
; GFX12-NEXT: s_mov_b64 s[0:1], 0
; GFX12-NEXT: .LBB16_1: ; %atomicrmw.start
@@ -1956,6 +1963,7 @@ define amdgpu_ps float @global_umax_saddr_i32_rtn(ptr addrspace(1) inreg %sbase,
; GFX12-NEXT: global_inv scope:SCOPE_SYS
; GFX12-NEXT: v_cmp_eq_u32_e32 vcc, v0, v5
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
+; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_or_b64 s[0:1], vcc, s[0:1]
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b64 exec, exec, s[0:1]
@@ -2024,7 +2032,6 @@ define amdgpu_ps float @global_umax_saddr_i32_rtn_neg128(ptr addrspace(1) inreg
; GFX11-NEXT: v_mov_b32_e32 v2, v0
; GFX11-NEXT: global_load_b32 v0, v0, s[2:3] offset:-128
; GFX11-NEXT: v_add_co_u32 v2, s[0:1], s2, v2
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v3, null, s3, 0, s[0:1]
; GFX11-NEXT: s_mov_b64 s[0:1], 0
; GFX11-NEXT: s_waitcnt_depctr depctr_sa_sdst(0)
@@ -2039,9 +2046,10 @@ define amdgpu_ps float @global_umax_saddr_i32_rtn_neg128(ptr addrspace(1) inreg
; GFX11-NEXT: buffer_gl1_inv
; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc, v0, v5
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_or_b64 s[0:1], vcc, s[0:1]
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b64 exec, exec, s[0:1]
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_execnz .LBB17_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-NEXT: s_or_b64 exec, exec, s[0:1]
@@ -2052,7 +2060,6 @@ define amdgpu_ps float @global_umax_saddr_i32_rtn_neg128(ptr addrspace(1) inreg
; GFX12-NEXT: v_mov_b32_e32 v2, v0
; GFX12-NEXT: global_load_b32 v0, v0, s[2:3] offset:-128
; GFX12-NEXT: v_add_co_u32 v2, s[0:1], s2, v2
-; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-NEXT: v_add_co_ci_u32_e64 v3, null, s3, 0, s[0:1]
; GFX12-NEXT: s_mov_b64 s[0:1], 0
; GFX12-NEXT: .LBB17_1: ; %atomicrmw.start
@@ -2068,6 +2075,7 @@ define amdgpu_ps float @global_umax_saddr_i32_rtn_neg128(ptr addrspace(1) inreg
; GFX12-NEXT: global_inv scope:SCOPE_SYS
; GFX12-NEXT: v_cmp_eq_u32_e32 vcc, v0, v5
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
+; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_or_b64 s[0:1], vcc, s[0:1]
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b64 exec, exec, s[0:1]
@@ -2132,7 +2140,6 @@ define amdgpu_ps void @global_umax_saddr_i32_nortn(ptr addrspace(1) inreg %sbase
; GFX11: ; %bb.0:
; GFX11-NEXT: global_load_b32 v5, v0, s[2:3]
; GFX11-NEXT: v_add_co_u32 v2, s[0:1], s2, v0
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v3, null, s3, 0, s[0:1]
; GFX11-NEXT: s_mov_b64 s[0:1], 0
; GFX11-NEXT: s_waitcnt_depctr depctr_sa_sdst(0)
@@ -2146,9 +2153,10 @@ define amdgpu_ps void @global_umax_saddr_i32_nortn(ptr addrspace(1) inreg %sbase
; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc, v0, v5
; GFX11-NEXT: v_mov_b32_e32 v5, v0
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_or_b64 s[0:1], vcc, s[0:1]
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b64 exec, exec, s[0:1]
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_execnz .LBB18_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-NEXT: s_endpgm
@@ -2157,7 +2165,6 @@ define amdgpu_ps void @global_umax_saddr_i32_nortn(ptr addrspace(1) inreg %sbase
; GFX12: ; %bb.0:
; GFX12-NEXT: global_load_b32 v5, v0, s[2:3]
; GFX12-NEXT: v_add_co_u32 v2, s[0:1], s2, v0
-; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-NEXT: v_add_co_ci_u32_e64 v3, null, s3, 0, s[0:1]
; GFX12-NEXT: s_mov_b64 s[0:1], 0
; GFX12-NEXT: .LBB18_1: ; %atomicrmw.start
@@ -2172,6 +2179,7 @@ define amdgpu_ps void @global_umax_saddr_i32_nortn(ptr addrspace(1) inreg %sbase
; GFX12-NEXT: v_cmp_eq_u32_e32 vcc, v0, v5
; GFX12-NEXT: v_mov_b32_e32 v5, v0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
+; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_or_b64 s[0:1], vcc, s[0:1]
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b64 exec, exec, s[0:1]
@@ -2233,7 +2241,6 @@ define amdgpu_ps void @global_umax_saddr_i32_nortn_neg128(ptr addrspace(1) inreg
; GFX11: ; %bb.0:
; GFX11-NEXT: global_load_b32 v5, v0, s[2:3] offset:-128
; GFX11-NEXT: v_add_co_u32 v2, s[0:1], s2, v0
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v3, null, s3, 0, s[0:1]
; GFX11-NEXT: s_mov_b64 s[0:1], 0
; GFX11-NEXT: s_waitcnt_depctr depctr_sa_sdst(0)
@@ -2247,9 +2254,10 @@ define amdgpu_ps void @global_umax_saddr_i32_nortn_neg128(ptr addrspace(1) inreg
; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc, v0, v5
; GFX11-NEXT: v_mov_b32_e32 v5, v0
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_or_b64 s[0:1], vcc, s[0:1]
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b64 exec, exec, s[0:1]
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_execnz .LBB19_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-NEXT: s_endpgm
@@ -2258,7 +2266,6 @@ define amdgpu_ps void @global_umax_saddr_i32_nortn_neg128(ptr addrspace(1) inreg
; GFX12: ; %bb.0:
; GFX12-NEXT: global_load_b32 v5, v0, s[2:3] offset:-128
; GFX12-NEXT: v_add_co_u32 v2, s[0:1], s2, v0
-; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-NEXT: v_add_co_ci_u32_e64 v3, null, s3, 0, s[0:1]
; GFX12-NEXT: s_mov_b64 s[0:1], 0
; GFX12-NEXT: .LBB19_1: ; %atomicrmw.start
@@ -2273,6 +2280,7 @@ define amdgpu_ps void @global_umax_saddr_i32_nortn_neg128(ptr addrspace(1) inreg
; GFX12-NEXT: v_cmp_eq_u32_e32 vcc, v0, v5
; GFX12-NEXT: v_mov_b32_e32 v5, v0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
+; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_or_b64 s[0:1], vcc, s[0:1]
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b64 exec, exec, s[0:1]
@@ -2347,7 +2355,6 @@ define amdgpu_ps <2 x float> @global_umax_saddr_i64_rtn(ptr addrspace(1) inreg %
; GFX11: ; %bb.0:
; GFX11-NEXT: global_load_b64 v[3:4], v0, s[2:3]
; GFX11-NEXT: v_add_co_u32 v5, s[0:1], s2, v0
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v6, null, s3, 0, s[0:1]
; GFX11-NEXT: s_mov_b64 s[0:1], 0
; GFX11-NEXT: s_waitcnt_depctr depctr_sa_sdst(0)
@@ -2366,12 +2373,14 @@ define amdgpu_ps <2 x float> @global_umax_saddr_i64_rtn(ptr addrspace(1) inreg %
; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: v_cmp_eq_u64_e32 vcc, v[3:4], v[9:10]
; GFX11-NEXT: s_waitcnt_depctr depctr_va_vcc(0)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_or_b64 s[0:1], vcc, s[0:1]
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b64 exec, exec, s[0:1]
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_execnz .LBB20_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-NEXT: s_or_b64 exec, exec, s[0:1]
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: v_mov_b32_e32 v0, v3
; GFX11-NEXT: v_mov_b32_e32 v1, v4
; GFX11-NEXT: ; return to shader part epilog
@@ -2380,7 +2389,6 @@ define amdgpu_ps <2 x float> @global_umax_saddr_i64_rtn(ptr addrspace(1) inreg %
; GFX12: ; %bb.0:
; GFX12-NEXT: global_load_b64 v[3:4], v0, s[2:3]
; GFX12-NEXT: v_add_co_u32 v5, s[0:1], s2, v0
-; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-NEXT: v_add_co_ci_u32_e64 v6, null, s3, 0, s[0:1]
; GFX12-NEXT: s_mov_b64 s[0:1], 0
; GFX12-NEXT: .LBB20_1: ; %atomicrmw.start
@@ -2400,12 +2408,14 @@ define amdgpu_ps <2 x float> @global_umax_saddr_i64_rtn(ptr addrspace(1) inreg %
; GFX12-NEXT: global_inv scope:SCOPE_SYS
; GFX12-NEXT: v_cmp_eq_u64_e32 vcc, v[3:4], v[9:10]
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
+; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_or_b64 s[0:1], vcc, s[0:1]
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b64 exec, exec, s[0:1]
; GFX12-NEXT: s_cbranch_execnz .LBB20_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b64 exec, exec, s[0:1]
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: v_mov_b32_e32 v0, v3
; GFX12-NEXT: v_mov_b32_e32 v1, v4
; GFX12-NEXT: ; return to shader part epilog
@@ -2477,7 +2487,6 @@ define amdgpu_ps <2 x float> @global_umax_saddr_i64_rtn_neg128(ptr addrspace(1)
; GFX11: ; %bb.0:
; GFX11-NEXT: global_load_b64 v[3:4], v0, s[2:3] offset:-128
; GFX11-NEXT: v_add_co_u32 v5, s[0:1], s2, v0
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v6, null, s3, 0, s[0:1]
; GFX11-NEXT: s_mov_b64 s[0:1], 0
; GFX11-NEXT: s_waitcnt_depctr depctr_sa_sdst(0)
@@ -2496,12 +2505,14 @@ define amdgpu_ps <2 x float> @global_umax_saddr_i64_rtn_neg128(ptr addrspace(1)
; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: v_cmp_eq_u64_e32 vcc, v[3:4], v[9:10]
; GFX11-NEXT: s_waitcnt_depctr depctr_va_vcc(0)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_or_b64 s[0:1], vcc, s[0:1]
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b64 exec, exec, s[0:1]
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_execnz .LBB21_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-NEXT: s_or_b64 exec, exec, s[0:1]
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: v_mov_b32_e32 v0, v3
; GFX11-NEXT: v_mov_b32_e32 v1, v4
; GFX11-NEXT: ; return to shader part epilog
@@ -2510,7 +2521,6 @@ define amdgpu_ps <2 x float> @global_umax_saddr_i64_rtn_neg128(ptr addrspace(1)
; GFX12: ; %bb.0:
; GFX12-NEXT: global_load_b64 v[3:4], v0, s[2:3] offset:-128
; GFX12-NEXT: v_add_co_u32 v5, s[0:1], s2, v0
-; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-NEXT: v_add_co_ci_u32_e64 v6, null, s3, 0, s[0:1]
; GFX12-NEXT: s_mov_b64 s[0:1], 0
; GFX12-NEXT: .LBB21_1: ; %atomicrmw.start
@@ -2530,12 +2540,14 @@ define amdgpu_ps <2 x float> @global_umax_saddr_i64_rtn_neg128(ptr addrspace(1)
; GFX12-NEXT: global_inv scope:SCOPE_SYS
; GFX12-NEXT: v_cmp_eq_u64_e32 vcc, v[3:4], v[9:10]
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
+; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_or_b64 s[0:1], vcc, s[0:1]
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b64 exec, exec, s[0:1]
; GFX12-NEXT: s_cbranch_execnz .LBB21_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b64 exec, exec, s[0:1]
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: v_mov_b32_e32 v0, v3
; GFX12-NEXT: v_mov_b32_e32 v1, v4
; GFX12-NEXT: ; return to shader part epilog
@@ -2602,7 +2614,6 @@ define amdgpu_ps void @global_umax_saddr_i64_nortn(ptr addrspace(1) inreg %sbase
; GFX11: ; %bb.0:
; GFX11-NEXT: global_load_b64 v[5:6], v0, s[2:3]
; GFX11-NEXT: v_add_co_u32 v7, s[0:1], s2, v0
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v8, null, s3, 0, s[0:1]
; GFX11-NEXT: s_mov_b64 s[0:1], 0
; GFX11-NEXT: s_waitcnt_depctr depctr_sa_sdst(0)
@@ -2620,9 +2631,10 @@ define amdgpu_ps void @global_umax_saddr_i64_nortn(ptr addrspace(1) inreg %sbase
; GFX11-NEXT: v_mov_b32_e32 v6, v4
; GFX11-NEXT: v_mov_b32_e32 v5, v3
; GFX11-NEXT: s_waitcnt_depctr depctr_va_vcc(0)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_or_b64 s[0:1], vcc, s[0:1]
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b64 exec, exec, s[0:1]
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_execnz .LBB22_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-NEXT: s_endpgm
@@ -2631,7 +2643,6 @@ define amdgpu_ps void @global_umax_saddr_i64_nortn(ptr addrspace(1) inreg %sbase
; GFX12: ; %bb.0:
; GFX12-NEXT: global_load_b64 v[5:6], v0, s[2:3]
; GFX12-NEXT: v_add_co_u32 v7, s[0:1], s2, v0
-; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-NEXT: v_add_co_ci_u32_e64 v8, null, s3, 0, s[0:1]
; GFX12-NEXT: s_mov_b64 s[0:1], 0
; GFX12-NEXT: .LBB22_1: ; %atomicrmw.start
@@ -2650,6 +2661,7 @@ define amdgpu_ps void @global_umax_saddr_i64_nortn(ptr addrspace(1) inreg %sbase
; GFX12-NEXT: v_mov_b32_e32 v6, v4
; GFX12-NEXT: v_mov_b32_e32 v5, v3
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
+; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_or_b64 s[0:1], vcc, s[0:1]
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b64 exec, exec, s[0:1]
@@ -2717,7 +2729,6 @@ define amdgpu_ps void @global_umax_saddr_i64_nortn_neg128(ptr addrspace(1) inreg
; GFX11: ; %bb.0:
; GFX11-NEXT: global_load_b64 v[5:6], v0, s[2:3] offset:-128
; GFX11-NEXT: v_add_co_u32 v7, s[0:1], s2, v0
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v8, null, s3, 0, s[0:1]
; GFX11-NEXT: s_mov_b64 s[0:1], 0
; GFX11-NEXT: s_waitcnt_depctr depctr_sa_sdst(0)
@@ -2735,9 +2746,10 @@ define amdgpu_ps void @global_umax_saddr_i64_nortn_neg128(ptr addrspace(1) inreg
; GFX11-NEXT: v_mov_b32_e32 v6, v4
; GFX11-NEXT: v_mov_b32_e32 v5, v3
; GFX11-NEXT: s_waitcnt_depctr depctr_va_vcc(0)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_or_b64 s[0:1], vcc, s[0:1]
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b64 exec, exec, s[0:1]
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_execnz .LBB23_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-NEXT: s_endpgm
@@ -2746,7 +2758,6 @@ define amdgpu_ps void @global_umax_saddr_i64_nortn_neg128(ptr addrspace(1) inreg
; GFX12: ; %bb.0:
; GFX12-NEXT: global_load_b64 v[5:6], v0, s[2:3] offset:-128
; GFX12-NEXT: v_add_co_u32 v7, s[0:1], s2, v0
-; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-NEXT: v_add_co_ci_u32_e64 v8, null, s3, 0, s[0:1]
; GFX12-NEXT: s_mov_b64 s[0:1], 0
; GFX12-NEXT: .LBB23_1: ; %atomicrmw.start
@@ -2765,6 +2776,7 @@ define amdgpu_ps void @global_umax_saddr_i64_nortn_neg128(ptr addrspace(1) inreg
; GFX12-NEXT: v_mov_b32_e32 v6, v4
; GFX12-NEXT: v_mov_b32_e32 v5, v3
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
+; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_or_b64 s[0:1], vcc, s[0:1]
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b64 exec, exec, s[0:1]
@@ -2836,7 +2848,6 @@ define amdgpu_ps float @global_umin_saddr_i32_rtn(ptr addrspace(1) inreg %sbase,
; GFX11-NEXT: v_mov_b32_e32 v2, v0
; GFX11-NEXT: global_load_b32 v0, v0, s[2:3]
; GFX11-NEXT: v_add_co_u32 v2, s[0:1], s2, v2
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v3, null, s3, 0, s[0:1]
; GFX11-NEXT: s_mov_b64 s[0:1], 0
; GFX11-NEXT: s_waitcnt_depctr depctr_sa_sdst(0)
@@ -2851,9 +2862,10 @@ define amdgpu_ps float @global_umin_saddr_i32_rtn(ptr addrspace(1) inreg %sbase,
; GFX11-NEXT: buffer_gl1_inv
; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc, v0, v5
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_or_b64 s[0:1], vcc, s[0:1]
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b64 exec, exec, s[0:1]
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_execnz .LBB24_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-NEXT: s_or_b64 exec, exec, s[0:1]
@@ -2864,7 +2876,6 @@ define amdgpu_ps float @global_umin_saddr_i32_rtn(ptr addrspace(1) inreg %sbase,
; GFX12-NEXT: v_mov_b32_e32 v2, v0
; GFX12-NEXT: global_load_b32 v0, v0, s[2:3]
; GFX12-NEXT: v_add_co_u32 v2, s[0:1], s2, v2
-; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-NEXT: v_add_co_ci_u32_e64 v3, null, s3, 0, s[0:1]
; GFX12-NEXT: s_mov_b64 s[0:1], 0
; GFX12-NEXT: .LBB24_1: ; %atomicrmw.start
@@ -2880,6 +2891,7 @@ define amdgpu_ps float @global_umin_saddr_i32_rtn(ptr addrspace(1) inreg %sbase,
; GFX12-NEXT: global_inv scope:SCOPE_SYS
; GFX12-NEXT: v_cmp_eq_u32_e32 vcc, v0, v5
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
+; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_or_b64 s[0:1], vcc, s[0:1]
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b64 exec, exec, s[0:1]
@@ -2948,7 +2960,6 @@ define amdgpu_ps float @global_umin_saddr_i32_rtn_neg128(ptr addrspace(1) inreg
; GFX11-NEXT: v_mov_b32_e32 v2, v0
; GFX11-NEXT: global_load_b32 v0, v0, s[2:3] offset:-128
; GFX11-NEXT: v_add_co_u32 v2, s[0:1], s2, v2
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v3, null, s3, 0, s[0:1]
; GFX11-NEXT: s_mov_b64 s[0:1], 0
; GFX11-NEXT: s_waitcnt_depctr depctr_sa_sdst(0)
@@ -2963,9 +2974,10 @@ define amdgpu_ps float @global_umin_saddr_i32_rtn_neg128(ptr addrspace(1) inreg
; GFX11-NEXT: buffer_gl1_inv
; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc, v0, v5
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_or_b64 s[0:1], vcc, s[0:1]
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b64 exec, exec, s[0:1]
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_execnz .LBB25_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-NEXT: s_or_b64 exec, exec, s[0:1]
@@ -2976,7 +2988,6 @@ define amdgpu_ps float @global_umin_saddr_i32_rtn_neg128(ptr addrspace(1) inreg
; GFX12-NEXT: v_mov_b32_e32 v2, v0
; GFX12-NEXT: global_load_b32 v0, v0, s[2:3] offset:-128
; GFX12-NEXT: v_add_co_u32 v2, s[0:1], s2, v2
-; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-NEXT: v_add_co_ci_u32_e64 v3, null, s3, 0, s[0:1]
; GFX12-NEXT: s_mov_b64 s[0:1], 0
; GFX12-NEXT: .LBB25_1: ; %atomicrmw.start
@@ -2992,6 +3003,7 @@ define amdgpu_ps float @global_umin_saddr_i32_rtn_neg128(ptr addrspace(1) inreg
; GFX12-NEXT: global_inv scope:SCOPE_SYS
; GFX12-NEXT: v_cmp_eq_u32_e32 vcc, v0, v5
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
+; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_or_b64 s[0:1], vcc, s[0:1]
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b64 exec, exec, s[0:1]
@@ -3056,7 +3068,6 @@ define amdgpu_ps void @global_umin_saddr_i32_nortn(ptr addrspace(1) inreg %sbase
; GFX11: ; %bb.0:
; GFX11-NEXT: global_load_b32 v5, v0, s[2:3]
; GFX11-NEXT: v_add_co_u32 v2, s[0:1], s2, v0
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v3, null, s3, 0, s[0:1]
; GFX11-NEXT: s_mov_b64 s[0:1], 0
; GFX11-NEXT: s_waitcnt_depctr depctr_sa_sdst(0)
@@ -3070,9 +3081,10 @@ define amdgpu_ps void @global_umin_saddr_i32_nortn(ptr addrspace(1) inreg %sbase
; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc, v0, v5
; GFX11-NEXT: v_mov_b32_e32 v5, v0
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_or_b64 s[0:1], vcc, s[0:1]
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b64 exec, exec, s[0:1]
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_execnz .LBB26_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-NEXT: s_endpgm
@@ -3081,7 +3093,6 @@ define amdgpu_ps void @global_umin_saddr_i32_nortn(ptr addrspace(1) inreg %sbase
; GFX12: ; %bb.0:
; GFX12-NEXT: global_load_b32 v5, v0, s[2:3]
; GFX12-NEXT: v_add_co_u32 v2, s[0:1], s2, v0
-; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-NEXT: v_add_co_ci_u32_e64 v3, null, s3, 0, s[0:1]
; GFX12-NEXT: s_mov_b64 s[0:1], 0
; GFX12-NEXT: .LBB26_1: ; %atomicrmw.start
@@ -3096,6 +3107,7 @@ define amdgpu_ps void @global_umin_saddr_i32_nortn(ptr addrspace(1) inreg %sbase
; GFX12-NEXT: v_cmp_eq_u32_e32 vcc, v0, v5
; GFX12-NEXT: v_mov_b32_e32 v5, v0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
+; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_or_b64 s[0:1], vcc, s[0:1]
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b64 exec, exec, s[0:1]
@@ -3157,7 +3169,6 @@ define amdgpu_ps void @global_umin_saddr_i32_nortn_neg128(ptr addrspace(1) inreg
; GFX11: ; %bb.0:
; GFX11-NEXT: global_load_b32 v5, v0, s[2:3] offset:-128
; GFX11-NEXT: v_add_co_u32 v2, s[0:1], s2, v0
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v3, null, s3, 0, s[0:1]
; GFX11-NEXT: s_mov_b64 s[0:1], 0
; GFX11-NEXT: s_waitcnt_depctr depctr_sa_sdst(0)
@@ -3171,9 +3182,10 @@ define amdgpu_ps void @global_umin_saddr_i32_nortn_neg128(ptr addrspace(1) inreg
; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc, v0, v5
; GFX11-NEXT: v_mov_b32_e32 v5, v0
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_or_b64 s[0:1], vcc, s[0:1]
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b64 exec, exec, s[0:1]
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_execnz .LBB27_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-NEXT: s_endpgm
@@ -3182,7 +3194,6 @@ define amdgpu_ps void @global_umin_saddr_i32_nortn_neg128(ptr addrspace(1) inreg
; GFX12: ; %bb.0:
; GFX12-NEXT: global_load_b32 v5, v0, s[2:3] offset:-128
; GFX12-NEXT: v_add_co_u32 v2, s[0:1], s2, v0
-; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-NEXT: v_add_co_ci_u32_e64 v3, null, s3, 0, s[0:1]
; GFX12-NEXT: s_mov_b64 s[0:1], 0
; GFX12-NEXT: .LBB27_1: ; %atomicrmw.start
@@ -3197,6 +3208,7 @@ define amdgpu_ps void @global_umin_saddr_i32_nortn_neg128(ptr addrspace(1) inreg
; GFX12-NEXT: v_cmp_eq_u32_e32 vcc, v0, v5
; GFX12-NEXT: v_mov_b32_e32 v5, v0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
+; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_or_b64 s[0:1], vcc, s[0:1]
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b64 exec, exec, s[0:1]
@@ -3271,7 +3283,6 @@ define amdgpu_ps <2 x float> @global_umin_saddr_i64_rtn(ptr addrspace(1) inreg %
; GFX11: ; %bb.0:
; GFX11-NEXT: global_load_b64 v[3:4], v0, s[2:3]
; GFX11-NEXT: v_add_co_u32 v5, s[0:1], s2, v0
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v6, null, s3, 0, s[0:1]
; GFX11-NEXT: s_mov_b64 s[0:1], 0
; GFX11-NEXT: s_waitcnt_depctr depctr_sa_sdst(0)
@@ -3290,12 +3301,14 @@ define amdgpu_ps <2 x float> @global_umin_saddr_i64_rtn(ptr addrspace(1) inreg %
; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: v_cmp_eq_u64_e32 vcc, v[3:4], v[9:10]
; GFX11-NEXT: s_waitcnt_depctr depctr_va_vcc(0)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_or_b64 s[0:1], vcc, s[0:1]
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b64 exec, exec, s[0:1]
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_execnz .LBB28_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-NEXT: s_or_b64 exec, exec, s[0:1]
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: v_mov_b32_e32 v0, v3
; GFX11-NEXT: v_mov_b32_e32 v1, v4
; GFX11-NEXT: ; return to shader part epilog
@@ -3304,7 +3317,6 @@ define amdgpu_ps <2 x float> @global_umin_saddr_i64_rtn(ptr addrspace(1) inreg %
; GFX12: ; %bb.0:
; GFX12-NEXT: global_load_b64 v[3:4], v0, s[2:3]
; GFX12-NEXT: v_add_co_u32 v5, s[0:1], s2, v0
-; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-NEXT: v_add_co_ci_u32_e64 v6, null, s3, 0, s[0:1]
; GFX12-NEXT: s_mov_b64 s[0:1], 0
; GFX12-NEXT: .LBB28_1: ; %atomicrmw.start
@@ -3324,12 +3336,14 @@ define amdgpu_ps <2 x float> @global_umin_saddr_i64_rtn(ptr addrspace(1) inreg %
; GFX12-NEXT: global_inv scope:SCOPE_SYS
; GFX12-NEXT: v_cmp_eq_u64_e32 vcc, v[3:4], v[9:10]
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
+; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_or_b64 s[0:1], vcc, s[0:1]
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b64 exec, exec, s[0:1]
; GFX12-NEXT: s_cbranch_execnz .LBB28_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b64 exec, exec, s[0:1]
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: v_mov_b32_e32 v0, v3
; GFX12-NEXT: v_mov_b32_e32 v1, v4
; GFX12-NEXT: ; return to shader part epilog
@@ -3401,7 +3415,6 @@ define amdgpu_ps <2 x float> @global_umin_saddr_i64_rtn_neg128(ptr addrspace(1)
; GFX11: ; %bb.0:
; GFX11-NEXT: global_load_b64 v[3:4], v0, s[2:3] offset:-128
; GFX11-NEXT: v_add_co_u32 v5, s[0:1], s2, v0
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v6, null, s3, 0, s[0:1]
; GFX11-NEXT: s_mov_b64 s[0:1], 0
; GFX11-NEXT: s_waitcnt_depctr depctr_sa_sdst(0)
@@ -3420,12 +3433,14 @@ define amdgpu_ps <2 x float> @global_umin_saddr_i64_rtn_neg128(ptr addrspace(1)
; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: v_cmp_eq_u64_e32 vcc, v[3:4], v[9:10]
; GFX11-NEXT: s_waitcnt_depctr depctr_va_vcc(0)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_or_b64 s[0:1], vcc, s[0:1]
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b64 exec, exec, s[0:1]
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_execnz .LBB29_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-NEXT: s_or_b64 exec, exec, s[0:1]
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: v_mov_b32_e32 v0, v3
; GFX11-NEXT: v_mov_b32_e32 v1, v4
; GFX11-NEXT: ; return to shader part epilog
@@ -3434,7 +3449,6 @@ define amdgpu_ps <2 x float> @global_umin_saddr_i64_rtn_neg128(ptr addrspace(1)
; GFX12: ; %bb.0:
; GFX12-NEXT: global_load_b64 v[3:4], v0, s[2:3] offset:-128
; GFX12-NEXT: v_add_co_u32 v5, s[0:1], s2, v0
-; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-NEXT: v_add_co_ci_u32_e64 v6, null, s3, 0, s[0:1]
; GFX12-NEXT: s_mov_b64 s[0:1], 0
; GFX12-NEXT: .LBB29_1: ; %atomicrmw.start
@@ -3454,12 +3468,14 @@ define amdgpu_ps <2 x float> @global_umin_saddr_i64_rtn_neg128(ptr addrspace(1)
; GFX12-NEXT: global_inv scope:SCOPE_SYS
; GFX12-NEXT: v_cmp_eq_u64_e32 vcc, v[3:4], v[9:10]
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
+; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_or_b64 s[0:1], vcc, s[0:1]
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b64 exec, exec, s[0:1]
; GFX12-NEXT: s_cbranch_execnz .LBB29_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b64 exec, exec, s[0:1]
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: v_mov_b32_e32 v0, v3
; GFX12-NEXT: v_mov_b32_e32 v1, v4
; GFX12-NEXT: ; return to shader part epilog
@@ -3526,7 +3542,6 @@ define amdgpu_ps void @global_umin_saddr_i64_nortn(ptr addrspace(1) inreg %sbase
; GFX11: ; %bb.0:
; GFX11-NEXT: global_load_b64 v[5:6], v0, s[2:3]
; GFX11-NEXT: v_add_co_u32 v7, s[0:1], s2, v0
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v8, null, s3, 0, s[0:1]
; GFX11-NEXT: s_mov_b64 s[0:1], 0
; GFX11-NEXT: s_waitcnt_depctr depctr_sa_sdst(0)
@@ -3544,9 +3559,10 @@ define amdgpu_ps void @global_umin_saddr_i64_nortn(ptr addrspace(1) inreg %sbase
; GFX11-NEXT: v_mov_b32_e32 v6, v4
; GFX11-NEXT: v_mov_b32_e32 v5, v3
; GFX11-NEXT: s_waitcnt_depctr depctr_va_vcc(0)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_or_b64 s[0:1], vcc, s[0:1]
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b64 exec, exec, s[0:1]
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_execnz .LBB30_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-NEXT: s_endpgm
@@ -3555,7 +3571,6 @@ define amdgpu_ps void @global_umin_saddr_i64_nortn(ptr addrspace(1) inreg %sbase
; GFX12: ; %bb.0:
; GFX12-NEXT: global_load_b64 v[5:6], v0, s[2:3]
; GFX12-NEXT: v_add_co_u32 v7, s[0:1], s2, v0
-; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-NEXT: v_add_co_ci_u32_e64 v8, null, s3, 0, s[0:1]
; GFX12-NEXT: s_mov_b64 s[0:1], 0
; GFX12-NEXT: .LBB30_1: ; %atomicrmw.start
@@ -3574,6 +3589,7 @@ define amdgpu_ps void @global_umin_saddr_i64_nortn(ptr addrspace(1) inreg %sbase
; GFX12-NEXT: v_mov_b32_e32 v6, v4
; GFX12-NEXT: v_mov_b32_e32 v5, v3
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
+; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_or_b64 s[0:1], vcc, s[0:1]
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b64 exec, exec, s[0:1]
@@ -3641,7 +3657,6 @@ define amdgpu_ps void @global_umin_saddr_i64_nortn_neg128(ptr addrspace(1) inreg
; GFX11: ; %bb.0:
; GFX11-NEXT: global_load_b64 v[5:6], v0, s[2:3] offset:-128
; GFX11-NEXT: v_add_co_u32 v7, s[0:1], s2, v0
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v8, null, s3, 0, s[0:1]
; GFX11-NEXT: s_mov_b64 s[0:1], 0
; GFX11-NEXT: s_waitcnt_depctr depctr_sa_sdst(0)
@@ -3659,9 +3674,10 @@ define amdgpu_ps void @global_umin_saddr_i64_nortn_neg128(ptr addrspace(1) inreg
; GFX11-NEXT: v_mov_b32_e32 v6, v4
; GFX11-NEXT: v_mov_b32_e32 v5, v3
; GFX11-NEXT: s_waitcnt_depctr depctr_va_vcc(0)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_or_b64 s[0:1], vcc, s[0:1]
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b64 exec, exec, s[0:1]
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_execnz .LBB31_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-NEXT: s_endpgm
@@ -3670,7 +3686,6 @@ define amdgpu_ps void @global_umin_saddr_i64_nortn_neg128(ptr addrspace(1) inreg
; GFX12: ; %bb.0:
; GFX12-NEXT: global_load_b64 v[5:6], v0, s[2:3] offset:-128
; GFX12-NEXT: v_add_co_u32 v7, s[0:1], s2, v0
-; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-NEXT: v_add_co_ci_u32_e64 v8, null, s3, 0, s[0:1]
; GFX12-NEXT: s_mov_b64 s[0:1], 0
; GFX12-NEXT: .LBB31_1: ; %atomicrmw.start
@@ -3689,6 +3704,7 @@ define amdgpu_ps void @global_umin_saddr_i64_nortn_neg128(ptr addrspace(1) inreg
; GFX12-NEXT: v_mov_b32_e32 v6, v4
; GFX12-NEXT: v_mov_b32_e32 v5, v3
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
+; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_or_b64 s[0:1], vcc, s[0:1]
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b64 exec, exec, s[0:1]
diff --git a/llvm/test/CodeGen/AMDGPU/global-saddr-load.ll b/llvm/test/CodeGen/AMDGPU/global-saddr-load.ll
index a1130abf89a7a5..4667f4a8ba5335 100644
--- a/llvm/test/CodeGen/AMDGPU/global-saddr-load.ll
+++ b/llvm/test/CodeGen/AMDGPU/global-saddr-load.ll
@@ -208,7 +208,6 @@ define amdgpu_ps float @global_load_saddr_i8_offset_neg4097(ptr addrspace(1) inr
; GFX11-LABEL: global_load_saddr_i8_offset_neg4097:
; GFX11: ; %bb.0:
; GFX11-NEXT: v_add_co_u32 v0, s[0:1], 0xfffff000, s2
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, -1, s3, s[0:1]
; GFX11-NEXT: global_load_u8 v0, v[0:1], off offset:-1
; GFX11-NEXT: s_waitcnt vmcnt(0)
@@ -262,7 +261,6 @@ define amdgpu_ps float @global_load_saddr_i8_offset_neg4098(ptr addrspace(1) inr
; GFX11-LABEL: global_load_saddr_i8_offset_neg4098:
; GFX11: ; %bb.0:
; GFX11-NEXT: v_add_co_u32 v0, s[0:1], 0xfffff000, s2
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, -1, s3, s[0:1]
; GFX11-NEXT: global_load_u8 v0, v[0:1], off offset:-2
; GFX11-NEXT: s_waitcnt vmcnt(0)
@@ -600,7 +598,6 @@ define amdgpu_ps float @global_load_saddr_i8_offset_0xFFFFFF(ptr addrspace(1) in
; GFX11-LABEL: global_load_saddr_i8_offset_0xFFFFFF:
; GFX11: ; %bb.0:
; GFX11-NEXT: v_add_co_u32 v0, s[0:1], 0xff800000, s2
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, -1, s3, s[0:1]
; GFX11-NEXT: global_load_u8 v0, v[0:1], off
; GFX11-NEXT: s_waitcnt vmcnt(0)
@@ -737,7 +734,6 @@ define amdgpu_ps float @global_load_saddr_i8_offset_0x100000001(ptr addrspace(1)
; GFX11-LABEL: global_load_saddr_i8_offset_0x100000001:
; GFX11: ; %bb.0:
; GFX11-NEXT: v_add_co_u32 v0, s[0:1], 0, s2
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, 1, s3, s[0:1]
; GFX11-NEXT: global_load_u8 v0, v[0:1], off offset:1
; GFX11-NEXT: s_waitcnt vmcnt(0)
@@ -790,7 +786,6 @@ define amdgpu_ps float @global_load_saddr_i8_offset_0x100000FFF(ptr addrspace(1)
; GFX11-LABEL: global_load_saddr_i8_offset_0x100000FFF:
; GFX11: ; %bb.0:
; GFX11-NEXT: v_add_co_u32 v0, s[0:1], 0, s2
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, 1, s3, s[0:1]
; GFX11-NEXT: global_load_u8 v0, v[0:1], off offset:4095
; GFX11-NEXT: s_waitcnt vmcnt(0)
@@ -843,7 +838,6 @@ define amdgpu_ps float @global_load_saddr_i8_offset_0x100001000(ptr addrspace(1)
; GFX11-LABEL: global_load_saddr_i8_offset_0x100001000:
; GFX11: ; %bb.0:
; GFX11-NEXT: v_add_co_u32 v0, s[0:1], 0x1000, s2
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, 1, s3, s[0:1]
; GFX11-NEXT: global_load_u8 v0, v[0:1], off
; GFX11-NEXT: s_waitcnt vmcnt(0)
@@ -897,7 +891,6 @@ define amdgpu_ps float @global_load_saddr_i8_offset_neg0xFFFFFFFF(ptr addrspace(
; GFX11-LABEL: global_load_saddr_i8_offset_neg0xFFFFFFFF:
; GFX11: ; %bb.0:
; GFX11-NEXT: v_add_co_u32 v0, s[0:1], 0x1000, s2
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, -1, s3, s[0:1]
; GFX11-NEXT: global_load_u8 v0, v[0:1], off offset:-4095
; GFX11-NEXT: s_waitcnt vmcnt(0)
@@ -998,7 +991,6 @@ define amdgpu_ps float @global_load_saddr_i8_offset_neg0x100000001(ptr addrspace
; GFX11-LABEL: global_load_saddr_i8_offset_neg0x100000001:
; GFX11: ; %bb.0:
; GFX11-NEXT: v_add_co_u32 v0, s[0:1], 0, s2
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, -1, s3, s[0:1]
; GFX11-NEXT: global_load_u8 v0, v[0:1], off offset:-1
; GFX11-NEXT: s_waitcnt vmcnt(0)
@@ -1675,7 +1667,6 @@ define amdgpu_ps float @global_load_saddr_uniform_ptr_in_vgprs(i32 %voffset) {
; GFX12-GISEL-NEXT: ds_load_b64 v[1:2], v1
; GFX12-GISEL-NEXT: s_wait_dscnt 0x0
; GFX12-GISEL-NEXT: v_add_co_u32 v0, vcc, v1, v0
-; GFX12-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v2, vcc
; GFX12-GISEL-NEXT: global_load_u8 v0, v[0:1], off
; GFX12-GISEL-NEXT: s_wait_loadcnt 0x0
@@ -1742,7 +1733,6 @@ define amdgpu_ps float @global_load_saddr_uniform_ptr_in_vgprs_immoffset(i32 %vo
; GFX12-GISEL-NEXT: ds_load_b64 v[1:2], v1
; GFX12-GISEL-NEXT: s_wait_dscnt 0x0
; GFX12-GISEL-NEXT: v_add_co_u32 v0, vcc, v1, v0
-; GFX12-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v2, vcc
; GFX12-GISEL-NEXT: global_load_u8 v0, v[0:1], off offset:42
; GFX12-GISEL-NEXT: s_wait_loadcnt 0x0
@@ -1921,7 +1911,6 @@ define amdgpu_ps float @global_load_i8_vgpr64_sgpr32(ptr addrspace(1) %vbase, i3
; GFX11-LABEL: global_load_i8_vgpr64_sgpr32:
; GFX11: ; %bb.0:
; GFX11-NEXT: v_add_co_u32 v0, vcc, v0, s2
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc
; GFX11-NEXT: global_load_u8 v0, v[0:1], off
; GFX11-NEXT: s_waitcnt vmcnt(0)
@@ -1930,7 +1919,6 @@ define amdgpu_ps float @global_load_i8_vgpr64_sgpr32(ptr addrspace(1) %vbase, i3
; GFX12-SDAG-LABEL: global_load_i8_vgpr64_sgpr32:
; GFX12-SDAG: ; %bb.0:
; GFX12-SDAG-NEXT: v_add_co_u32 v0, vcc, v0, s2
-; GFX12-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-SDAG-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc
; GFX12-SDAG-NEXT: global_load_u8 v0, v[0:1], off
; GFX12-SDAG-NEXT: s_wait_loadcnt 0x0
@@ -1941,7 +1929,7 @@ define amdgpu_ps float @global_load_i8_vgpr64_sgpr32(ptr addrspace(1) %vbase, i3
; GFX12-GISEL-NEXT: s_mov_b32 s3, 0
; GFX12-GISEL-NEXT: v_mov_b32_e32 v2, s2
; GFX12-GISEL-NEXT: v_mov_b32_e32 v3, s3
-; GFX12-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX12-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX12-GISEL-NEXT: v_add_co_u32 v0, vcc, v0, v2
; GFX12-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, v1, v3, vcc
; GFX12-GISEL-NEXT: global_load_u8 v0, v[0:1], off
@@ -1978,7 +1966,6 @@ define amdgpu_ps float @global_load_i8_vgpr64_sgpr32_offset_4095(ptr addrspace(1
; GFX11-LABEL: global_load_i8_vgpr64_sgpr32_offset_4095:
; GFX11: ; %bb.0:
; GFX11-NEXT: v_add_co_u32 v0, vcc, v0, s2
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc
; GFX11-NEXT: global_load_u8 v0, v[0:1], off offset:4095
; GFX11-NEXT: s_waitcnt vmcnt(0)
@@ -1987,7 +1974,6 @@ define amdgpu_ps float @global_load_i8_vgpr64_sgpr32_offset_4095(ptr addrspace(1
; GFX12-SDAG-LABEL: global_load_i8_vgpr64_sgpr32_offset_4095:
; GFX12-SDAG: ; %bb.0:
; GFX12-SDAG-NEXT: v_add_co_u32 v0, vcc, v0, s2
-; GFX12-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-SDAG-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc
; GFX12-SDAG-NEXT: global_load_u8 v0, v[0:1], off offset:4095
; GFX12-SDAG-NEXT: s_wait_loadcnt 0x0
@@ -1998,7 +1984,7 @@ define amdgpu_ps float @global_load_i8_vgpr64_sgpr32_offset_4095(ptr addrspace(1
; GFX12-GISEL-NEXT: s_mov_b32 s3, 0
; GFX12-GISEL-NEXT: v_mov_b32_e32 v2, s2
; GFX12-GISEL-NEXT: v_mov_b32_e32 v3, s3
-; GFX12-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX12-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX12-GISEL-NEXT: v_add_co_u32 v0, vcc, v0, v2
; GFX12-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, v1, v3, vcc
; GFX12-GISEL-NEXT: global_load_u8 v0, v[0:1], off offset:4095
@@ -2052,7 +2038,7 @@ define amdgpu_ps float @global_load_saddr_f32_natural_addressing(ptr addrspace(1
; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_lshlrev_b64 v[0:1], 2, v[0:1]
; GFX11-NEXT: v_add_co_u32 v0, vcc, s2, v0
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, s3, v1, vcc
; GFX11-NEXT: global_load_b32 v0, v[0:1], off
; GFX11-NEXT: s_waitcnt vmcnt(0)
@@ -2066,7 +2052,7 @@ define amdgpu_ps float @global_load_saddr_f32_natural_addressing(ptr addrspace(1
; GFX12-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX12-SDAG-NEXT: v_lshlrev_b64_e32 v[0:1], 2, v[0:1]
; GFX12-SDAG-NEXT: v_add_co_u32 v0, vcc, s2, v0
-; GFX12-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX12-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX12-SDAG-NEXT: v_add_co_ci_u32_e64 v1, null, s3, v1, vcc
; GFX12-SDAG-NEXT: global_load_b32 v0, v[0:1], off
; GFX12-SDAG-NEXT: s_wait_loadcnt 0x0
@@ -2082,7 +2068,7 @@ define amdgpu_ps float @global_load_saddr_f32_natural_addressing(ptr addrspace(1
; GFX12-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX12-GISEL-NEXT: v_lshlrev_b64_e32 v[0:1], 2, v[0:1]
; GFX12-GISEL-NEXT: v_add_co_u32 v0, vcc, v2, v0
-; GFX12-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX12-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX12-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, v3, v1, vcc
; GFX12-GISEL-NEXT: global_load_b32 v0, v[0:1], off
; GFX12-GISEL-NEXT: s_wait_loadcnt 0x0
@@ -2233,7 +2219,7 @@ define amdgpu_ps float @global_load_f32_saddr_zext_vgpr_range_too_large(ptr addr
; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_lshlrev_b64 v[0:1], 2, v[0:1]
; GFX11-NEXT: v_add_co_u32 v0, vcc, s2, v0
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, s3, v1, vcc
; GFX11-NEXT: global_load_b32 v0, v[0:1], off
; GFX11-NEXT: s_waitcnt vmcnt(0)
@@ -2247,7 +2233,7 @@ define amdgpu_ps float @global_load_f32_saddr_zext_vgpr_range_too_large(ptr addr
; GFX12-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX12-SDAG-NEXT: v_lshlrev_b64_e32 v[0:1], 2, v[0:1]
; GFX12-SDAG-NEXT: v_add_co_u32 v0, vcc, s2, v0
-; GFX12-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX12-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX12-SDAG-NEXT: v_add_co_ci_u32_e64 v1, null, s3, v1, vcc
; GFX12-SDAG-NEXT: global_load_b32 v0, v[0:1], off
; GFX12-SDAG-NEXT: s_wait_loadcnt 0x0
@@ -2263,7 +2249,7 @@ define amdgpu_ps float @global_load_f32_saddr_zext_vgpr_range_too_large(ptr addr
; GFX12-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX12-GISEL-NEXT: v_lshlrev_b64_e32 v[0:1], 2, v[0:1]
; GFX12-GISEL-NEXT: v_add_co_u32 v0, vcc, v2, v0
-; GFX12-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX12-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX12-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, v3, v1, vcc
; GFX12-GISEL-NEXT: global_load_b32 v0, v[0:1], off
; GFX12-GISEL-NEXT: s_wait_loadcnt 0x0
@@ -4775,6 +4761,7 @@ define amdgpu_ps void @global_addr_64bit_lsr_iv(ptr addrspace(1) inreg %arg) {
; GFX11-NEXT: s_add_u32 s2, s2, 4
; GFX11-NEXT: s_addc_u32 s3, s3, 0
; GFX11-NEXT: s_cmp_eq_u32 s0, 0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc0 .LBB132_1
; GFX11-NEXT: ; %bb.2: ; %bb2
; GFX11-NEXT: s_endpgm
@@ -4790,6 +4777,7 @@ define amdgpu_ps void @global_addr_64bit_lsr_iv(ptr addrspace(1) inreg %arg) {
; GFX12-SDAG-NEXT: s_add_co_i32 s0, s0, -1
; GFX12-SDAG-NEXT: s_add_nc_u64 s[2:3], s[2:3], 4
; GFX12-SDAG-NEXT: s_cmp_eq_u32 s0, 0
+; GFX12-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-SDAG-NEXT: s_cbranch_scc0 .LBB132_1
; GFX12-SDAG-NEXT: ; %bb.2: ; %bb2
; GFX12-SDAG-NEXT: s_endpgm
@@ -4806,6 +4794,7 @@ define amdgpu_ps void @global_addr_64bit_lsr_iv(ptr addrspace(1) inreg %arg) {
; GFX12-GISEL-NEXT: s_add_co_u32 s2, s2, 4
; GFX12-GISEL-NEXT: s_add_co_ci_u32 s3, s3, 0
; GFX12-GISEL-NEXT: s_cmp_eq_u32 s0, 0
+; GFX12-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-GISEL-NEXT: s_cbranch_scc0 .LBB132_1
; GFX12-GISEL-NEXT: ; %bb.2: ; %bb2
; GFX12-GISEL-NEXT: s_endpgm
@@ -4879,6 +4868,7 @@ define amdgpu_ps void @global_addr_64bit_lsr_iv_multiload(ptr addrspace(1) inreg
; GFX11-NEXT: s_add_u32 s2, s2, 4
; GFX11-NEXT: s_addc_u32 s3, s3, 0
; GFX11-NEXT: s_cmp_eq_u32 s0, 0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc0 .LBB133_1
; GFX11-NEXT: ; %bb.2: ; %bb2
; GFX11-NEXT: s_endpgm
@@ -4896,6 +4886,7 @@ define amdgpu_ps void @global_addr_64bit_lsr_iv_multiload(ptr addrspace(1) inreg
; GFX12-SDAG-NEXT: s_add_co_i32 s0, s0, -1
; GFX12-SDAG-NEXT: s_add_nc_u64 s[2:3], s[2:3], 4
; GFX12-SDAG-NEXT: s_cmp_eq_u32 s0, 0
+; GFX12-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-SDAG-NEXT: s_cbranch_scc0 .LBB133_1
; GFX12-SDAG-NEXT: ; %bb.2: ; %bb2
; GFX12-SDAG-NEXT: s_endpgm
@@ -4914,6 +4905,7 @@ define amdgpu_ps void @global_addr_64bit_lsr_iv_multiload(ptr addrspace(1) inreg
; GFX12-GISEL-NEXT: s_add_co_u32 s2, s2, 4
; GFX12-GISEL-NEXT: s_add_co_ci_u32 s3, s3, 0
; GFX12-GISEL-NEXT: s_cmp_eq_u32 s0, 0
+; GFX12-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-GISEL-NEXT: s_cbranch_scc0 .LBB133_1
; GFX12-GISEL-NEXT: ; %bb.2: ; %bb2
; GFX12-GISEL-NEXT: s_endpgm
diff --git a/llvm/test/CodeGen/AMDGPU/global_atomics_scan_fadd.ll b/llvm/test/CodeGen/AMDGPU/global_atomics_scan_fadd.ll
index f8b49da87b7e86..7da65ee5f492b4 100644
--- a/llvm/test/CodeGen/AMDGPU/global_atomics_scan_fadd.ll
+++ b/llvm/test/CodeGen/AMDGPU/global_atomics_scan_fadd.ll
@@ -151,6 +151,7 @@ define amdgpu_kernel void @global_atomic_fadd_uni_address_uni_value_agent_scope_
; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164-NEXT: v_mbcnt_hi_u32_b32 v0, s1, v0
; GFX1164-NEXT: v_cmpx_eq_u32_e32 0, v0
+; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1164-NEXT: s_cbranch_execz .LBB0_2
; GFX1164-NEXT: ; %bb.1:
; GFX1164-NEXT: s_load_b64 s[2:3], s[4:5], 0x24
@@ -169,7 +170,7 @@ define amdgpu_kernel void @global_atomic_fadd_uni_address_uni_value_agent_scope_
; GFX1132-NEXT: s_mov_b32 s0, exec_lo
; GFX1132-NEXT: s_mov_b32 s1, exec_lo
; GFX1132-NEXT: v_mbcnt_lo_u32_b32 v0, s0, 0
-; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX1132-NEXT: s_cbranch_execz .LBB0_2
; GFX1132-NEXT: ; %bb.1:
@@ -318,6 +319,7 @@ define amdgpu_kernel void @global_atomic_fadd_uni_address_uni_value_agent_scope_
; GFX1164-DPP-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164-DPP-NEXT: v_mbcnt_hi_u32_b32 v0, s1, v0
; GFX1164-DPP-NEXT: v_cmpx_eq_u32_e32 0, v0
+; GFX1164-DPP-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1164-DPP-NEXT: s_cbranch_execz .LBB0_2
; GFX1164-DPP-NEXT: ; %bb.1:
; GFX1164-DPP-NEXT: s_load_b64 s[2:3], s[4:5], 0x24
@@ -336,7 +338,7 @@ define amdgpu_kernel void @global_atomic_fadd_uni_address_uni_value_agent_scope_
; GFX1132-DPP-NEXT: s_mov_b32 s0, exec_lo
; GFX1132-DPP-NEXT: s_mov_b32 s1, exec_lo
; GFX1132-DPP-NEXT: v_mbcnt_lo_u32_b32 v0, s0, 0
-; GFX1132-DPP-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1132-DPP-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132-DPP-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX1132-DPP-NEXT: s_cbranch_execz .LBB0_2
; GFX1132-DPP-NEXT: ; %bb.1:
@@ -666,6 +668,7 @@ define amdgpu_kernel void @global_atomic_fadd_uni_address_div_value_agent_scope_
; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
; GFX1164-NEXT: v_cmpx_eq_u32_e32 0, v0
+; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1164-NEXT: s_xor_b64 s[0:1], exec, s[0:1]
; GFX1164-NEXT: s_cbranch_execz .LBB1_4
; GFX1164-NEXT: ; %bb.3:
@@ -712,9 +715,10 @@ define amdgpu_kernel void @global_atomic_fadd_uni_address_div_value_agent_scope_
; GFX1132-NEXT: ; %bb.2: ; %ComputeEnd
; GFX1132-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
; GFX1132-NEXT: s_mov_b32 s0, exec_lo
-; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX1132-NEXT: s_xor_b32 s0, exec_lo, s0
+; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1132-NEXT: s_cbranch_execz .LBB1_4
; GFX1132-NEXT: ; %bb.3:
; GFX1132-NEXT: s_load_b64 s[0:1], s[34:35], 0x24
@@ -1064,6 +1068,7 @@ define amdgpu_kernel void @global_atomic_fadd_uni_address_div_value_agent_scope_
; GFX1164-DPP-NEXT: v_mov_b32_e32 v0, v1
; GFX1164-DPP-NEXT: s_mov_b64 s[0:1], exec
; GFX1164-DPP-NEXT: v_cmpx_eq_u32_e32 0, v4
+; GFX1164-DPP-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1164-DPP-NEXT: s_cbranch_execz .LBB1_2
; GFX1164-DPP-NEXT: ; %bb.1:
; GFX1164-DPP-NEXT: s_load_b64 s[0:1], s[34:35], 0x24
@@ -1120,7 +1125,7 @@ define amdgpu_kernel void @global_atomic_fadd_uni_address_div_value_agent_scope_
; GFX1132-DPP-NEXT: v_mbcnt_lo_u32_b32 v4, exec_lo, 0
; GFX1132-DPP-NEXT: v_mov_b32_e32 v0, v1
; GFX1132-DPP-NEXT: s_mov_b32 s0, exec_lo
-; GFX1132-DPP-NEXT: s_delay_alu instid0(VALU_DEP_2)
+; GFX1132-DPP-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132-DPP-NEXT: v_cmpx_eq_u32_e32 0, v4
; GFX1132-DPP-NEXT: s_cbranch_execz .LBB1_2
; GFX1132-DPP-NEXT: ; %bb.1:
@@ -1315,7 +1320,7 @@ define amdgpu_kernel void @global_atomic_fadd_uni_address_uni_value_one_as_scope
; GFX1164-NEXT: scratch_store_b32 off, v1, off
; GFX1164-NEXT: scratch_load_b64 v[0:1], off, off
; GFX1164-NEXT: v_mbcnt_hi_u32_b32 v2, exec_hi, v2
-; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164-NEXT: v_cmpx_eq_u32_e32 0, v2
; GFX1164-NEXT: s_cbranch_execz .LBB2_2
; GFX1164-NEXT: ; %bb.1:
@@ -1343,6 +1348,7 @@ define amdgpu_kernel void @global_atomic_fadd_uni_address_uni_value_one_as_scope
; GFX1132-NEXT: scratch_store_b32 off, v1, off
; GFX1132-NEXT: scratch_load_b64 v[0:1], off, off
; GFX1132-NEXT: v_cmpx_eq_u32_e32 0, v2
+; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1132-NEXT: s_cbranch_execz .LBB2_2
; GFX1132-NEXT: ; %bb.1:
; GFX1132-NEXT: s_waitcnt vmcnt(0)
@@ -1535,7 +1541,7 @@ define amdgpu_kernel void @global_atomic_fadd_uni_address_uni_value_one_as_scope
; GFX1164-DPP-NEXT: scratch_store_b32 off, v1, off
; GFX1164-DPP-NEXT: scratch_load_b64 v[0:1], off, off
; GFX1164-DPP-NEXT: v_mbcnt_hi_u32_b32 v2, exec_hi, v2
-; GFX1164-DPP-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1164-DPP-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164-DPP-NEXT: v_cmpx_eq_u32_e32 0, v2
; GFX1164-DPP-NEXT: s_cbranch_execz .LBB2_2
; GFX1164-DPP-NEXT: ; %bb.1:
@@ -1563,6 +1569,7 @@ define amdgpu_kernel void @global_atomic_fadd_uni_address_uni_value_one_as_scope
; GFX1132-DPP-NEXT: scratch_store_b32 off, v1, off
; GFX1132-DPP-NEXT: scratch_load_b64 v[0:1], off, off
; GFX1132-DPP-NEXT: v_cmpx_eq_u32_e32 0, v2
+; GFX1132-DPP-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1132-DPP-NEXT: s_cbranch_execz .LBB2_2
; GFX1132-DPP-NEXT: ; %bb.1:
; GFX1132-DPP-NEXT: s_waitcnt vmcnt(0)
@@ -1893,6 +1900,7 @@ define amdgpu_kernel void @global_atomic_fadd_uni_address_div_value_one_as_scope
; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
; GFX1164-NEXT: v_cmpx_eq_u32_e32 0, v0
+; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1164-NEXT: s_xor_b64 s[0:1], exec, s[0:1]
; GFX1164-NEXT: s_cbranch_execz .LBB3_4
; GFX1164-NEXT: ; %bb.3:
@@ -1939,9 +1947,10 @@ define amdgpu_kernel void @global_atomic_fadd_uni_address_div_value_one_as_scope
; GFX1132-NEXT: ; %bb.2: ; %ComputeEnd
; GFX1132-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
; GFX1132-NEXT: s_mov_b32 s0, exec_lo
-; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX1132-NEXT: s_xor_b32 s0, exec_lo, s0
+; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1132-NEXT: s_cbranch_execz .LBB3_4
; GFX1132-NEXT: ; %bb.3:
; GFX1132-NEXT: s_load_b64 s[0:1], s[34:35], 0x24
@@ -2291,6 +2300,7 @@ define amdgpu_kernel void @global_atomic_fadd_uni_address_div_value_one_as_scope
; GFX1164-DPP-NEXT: v_mov_b32_e32 v0, v1
; GFX1164-DPP-NEXT: s_mov_b64 s[0:1], exec
; GFX1164-DPP-NEXT: v_cmpx_eq_u32_e32 0, v4
+; GFX1164-DPP-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1164-DPP-NEXT: s_cbranch_execz .LBB3_2
; GFX1164-DPP-NEXT: ; %bb.1:
; GFX1164-DPP-NEXT: s_load_b64 s[0:1], s[34:35], 0x24
@@ -2347,7 +2357,7 @@ define amdgpu_kernel void @global_atomic_fadd_uni_address_div_value_one_as_scope
; GFX1132-DPP-NEXT: v_mbcnt_lo_u32_b32 v4, exec_lo, 0
; GFX1132-DPP-NEXT: v_mov_b32_e32 v0, v1
; GFX1132-DPP-NEXT: s_mov_b32 s0, exec_lo
-; GFX1132-DPP-NEXT: s_delay_alu instid0(VALU_DEP_2)
+; GFX1132-DPP-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132-DPP-NEXT: v_cmpx_eq_u32_e32 0, v4
; GFX1132-DPP-NEXT: s_cbranch_execz .LBB3_2
; GFX1132-DPP-NEXT: ; %bb.1:
@@ -2542,7 +2552,7 @@ define amdgpu_kernel void @global_atomic_fadd_uni_address_uni_value_agent_scope_
; GFX1164-NEXT: scratch_store_b32 off, v1, off
; GFX1164-NEXT: scratch_load_b64 v[0:1], off, off
; GFX1164-NEXT: v_mbcnt_hi_u32_b32 v2, exec_hi, v2
-; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164-NEXT: v_cmpx_eq_u32_e32 0, v2
; GFX1164-NEXT: s_cbranch_execz .LBB4_3
; GFX1164-NEXT: ; %bb.1:
@@ -2560,14 +2570,14 @@ define amdgpu_kernel void @global_atomic_fadd_uni_address_uni_value_agent_scope_
; GFX1164-NEXT: v_mul_f32_e32 v2, 4.0, v0
; GFX1164-NEXT: .LBB4_2: ; %atomicrmw.start
; GFX1164-NEXT: ; =>This Inner Loop Header: Depth=1
-; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_4) | instid1(VALU_DEP_2)
; GFX1164-NEXT: v_add_f32_e32 v0, v1, v2
; GFX1164-NEXT: global_atomic_cmpswap_b32 v0, v3, v[0:1], s[0:1] glc
; GFX1164-NEXT: s_waitcnt vmcnt(0)
; GFX1164-NEXT: v_cmp_eq_u32_e32 vcc, v0, v1
; GFX1164-NEXT: v_mov_b32_e32 v1, v0
; GFX1164-NEXT: s_or_b64 s[2:3], vcc, s[2:3]
-; GFX1164-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1164-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1164-NEXT: s_and_not1_b64 exec, exec, s[2:3]
; GFX1164-NEXT: s_cbranch_execnz .LBB4_2
; GFX1164-NEXT: .LBB4_3:
@@ -2586,6 +2596,7 @@ define amdgpu_kernel void @global_atomic_fadd_uni_address_uni_value_agent_scope_
; GFX1132-NEXT: scratch_store_b32 off, v1, off
; GFX1132-NEXT: scratch_load_b64 v[0:1], off, off
; GFX1132-NEXT: v_cmpx_eq_u32_e32 0, v2
+; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1132-NEXT: s_cbranch_execz .LBB4_3
; GFX1132-NEXT: ; %bb.1:
; GFX1132-NEXT: s_waitcnt vmcnt(0)
@@ -2607,7 +2618,7 @@ define amdgpu_kernel void @global_atomic_fadd_uni_address_uni_value_agent_scope_
; GFX1132-NEXT: v_cmp_eq_u32_e32 vcc_lo, v0, v1
; GFX1132-NEXT: v_mov_b32_e32 v1, v0
; GFX1132-NEXT: s_or_b32 s2, vcc_lo, s2
-; GFX1132-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1132-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1132-NEXT: s_and_not1_b32 exec_lo, exec_lo, s2
; GFX1132-NEXT: s_cbranch_execnz .LBB4_2
; GFX1132-NEXT: .LBB4_3:
@@ -2792,7 +2803,7 @@ define amdgpu_kernel void @global_atomic_fadd_uni_address_uni_value_agent_scope_
; GFX1164-DPP-NEXT: scratch_store_b32 off, v1, off
; GFX1164-DPP-NEXT: scratch_load_b64 v[0:1], off, off
; GFX1164-DPP-NEXT: v_mbcnt_hi_u32_b32 v2, exec_hi, v2
-; GFX1164-DPP-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1164-DPP-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164-DPP-NEXT: v_cmpx_eq_u32_e32 0, v2
; GFX1164-DPP-NEXT: s_cbranch_execz .LBB4_3
; GFX1164-DPP-NEXT: ; %bb.1:
@@ -2810,14 +2821,14 @@ define amdgpu_kernel void @global_atomic_fadd_uni_address_uni_value_agent_scope_
; GFX1164-DPP-NEXT: v_mul_f32_e32 v2, 4.0, v0
; GFX1164-DPP-NEXT: .LBB4_2: ; %atomicrmw.start
; GFX1164-DPP-NEXT: ; =>This Inner Loop Header: Depth=1
-; GFX1164-DPP-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1164-DPP-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_4) | instid1(VALU_DEP_2)
; GFX1164-DPP-NEXT: v_add_f32_e32 v0, v1, v2
; GFX1164-DPP-NEXT: global_atomic_cmpswap_b32 v0, v3, v[0:1], s[0:1] glc
; GFX1164-DPP-NEXT: s_waitcnt vmcnt(0)
; GFX1164-DPP-NEXT: v_cmp_eq_u32_e32 vcc, v0, v1
; GFX1164-DPP-NEXT: v_mov_b32_e32 v1, v0
; GFX1164-DPP-NEXT: s_or_b64 s[2:3], vcc, s[2:3]
-; GFX1164-DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1164-DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1164-DPP-NEXT: s_and_not1_b64 exec, exec, s[2:3]
; GFX1164-DPP-NEXT: s_cbranch_execnz .LBB4_2
; GFX1164-DPP-NEXT: .LBB4_3:
@@ -2836,6 +2847,7 @@ define amdgpu_kernel void @global_atomic_fadd_uni_address_uni_value_agent_scope_
; GFX1132-DPP-NEXT: scratch_store_b32 off, v1, off
; GFX1132-DPP-NEXT: scratch_load_b64 v[0:1], off, off
; GFX1132-DPP-NEXT: v_cmpx_eq_u32_e32 0, v2
+; GFX1132-DPP-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1132-DPP-NEXT: s_cbranch_execz .LBB4_3
; GFX1132-DPP-NEXT: ; %bb.1:
; GFX1132-DPP-NEXT: s_waitcnt vmcnt(0)
@@ -2857,7 +2869,7 @@ define amdgpu_kernel void @global_atomic_fadd_uni_address_uni_value_agent_scope_
; GFX1132-DPP-NEXT: v_cmp_eq_u32_e32 vcc_lo, v0, v1
; GFX1132-DPP-NEXT: v_mov_b32_e32 v1, v0
; GFX1132-DPP-NEXT: s_or_b32 s2, vcc_lo, s2
-; GFX1132-DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1132-DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1132-DPP-NEXT: s_and_not1_b32 exec_lo, exec_lo, s2
; GFX1132-DPP-NEXT: s_cbranch_execnz .LBB4_2
; GFX1132-DPP-NEXT: .LBB4_3:
@@ -3180,6 +3192,7 @@ define amdgpu_kernel void @global_atomic_fadd_uni_address_div_value_agent_scope_
; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
; GFX1164-NEXT: v_cmpx_eq_u32_e32 0, v0
+; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1164-NEXT: s_xor_b64 s[0:1], exec, s[0:1]
; GFX1164-NEXT: s_cbranch_execz .LBB5_4
; GFX1164-NEXT: ; %bb.3:
@@ -3226,9 +3239,10 @@ define amdgpu_kernel void @global_atomic_fadd_uni_address_div_value_agent_scope_
; GFX1132-NEXT: ; %bb.2: ; %ComputeEnd
; GFX1132-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
; GFX1132-NEXT: s_mov_b32 s0, exec_lo
-; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX1132-NEXT: s_xor_b32 s0, exec_lo, s0
+; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1132-NEXT: s_cbranch_execz .LBB5_4
; GFX1132-NEXT: ; %bb.3:
; GFX1132-NEXT: s_load_b64 s[0:1], s[34:35], 0x24
@@ -3578,6 +3592,7 @@ define amdgpu_kernel void @global_atomic_fadd_uni_address_div_value_agent_scope_
; GFX1164-DPP-NEXT: v_mov_b32_e32 v0, v1
; GFX1164-DPP-NEXT: s_mov_b64 s[0:1], exec
; GFX1164-DPP-NEXT: v_cmpx_eq_u32_e32 0, v4
+; GFX1164-DPP-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1164-DPP-NEXT: s_cbranch_execz .LBB5_2
; GFX1164-DPP-NEXT: ; %bb.1:
; GFX1164-DPP-NEXT: s_load_b64 s[0:1], s[34:35], 0x24
@@ -3634,7 +3649,7 @@ define amdgpu_kernel void @global_atomic_fadd_uni_address_div_value_agent_scope_
; GFX1132-DPP-NEXT: v_mbcnt_lo_u32_b32 v4, exec_lo, 0
; GFX1132-DPP-NEXT: v_mov_b32_e32 v0, v1
; GFX1132-DPP-NEXT: s_mov_b32 s0, exec_lo
-; GFX1132-DPP-NEXT: s_delay_alu instid0(VALU_DEP_2)
+; GFX1132-DPP-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132-DPP-NEXT: v_cmpx_eq_u32_e32 0, v4
; GFX1132-DPP-NEXT: s_cbranch_execz .LBB5_2
; GFX1132-DPP-NEXT: ; %bb.1:
@@ -3963,6 +3978,7 @@ define amdgpu_kernel void @global_atomic_fadd_uni_address_div_value_agent_scope_
; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
; GFX1164-NEXT: v_cmpx_eq_u32_e32 0, v0
+; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1164-NEXT: s_xor_b64 s[0:1], exec, s[0:1]
; GFX1164-NEXT: s_cbranch_execz .LBB6_4
; GFX1164-NEXT: ; %bb.3:
@@ -4009,9 +4025,10 @@ define amdgpu_kernel void @global_atomic_fadd_uni_address_div_value_agent_scope_
; GFX1132-NEXT: ; %bb.2: ; %ComputeEnd
; GFX1132-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
; GFX1132-NEXT: s_mov_b32 s0, exec_lo
-; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX1132-NEXT: s_xor_b32 s0, exec_lo, s0
+; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1132-NEXT: s_cbranch_execz .LBB6_4
; GFX1132-NEXT: ; %bb.3:
; GFX1132-NEXT: s_load_b64 s[0:1], s[34:35], 0x24
@@ -4361,6 +4378,7 @@ define amdgpu_kernel void @global_atomic_fadd_uni_address_div_value_agent_scope_
; GFX1164-DPP-NEXT: v_mov_b32_e32 v0, v1
; GFX1164-DPP-NEXT: s_mov_b64 s[0:1], exec
; GFX1164-DPP-NEXT: v_cmpx_eq_u32_e32 0, v4
+; GFX1164-DPP-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1164-DPP-NEXT: s_cbranch_execz .LBB6_2
; GFX1164-DPP-NEXT: ; %bb.1:
; GFX1164-DPP-NEXT: s_load_b64 s[0:1], s[34:35], 0x24
@@ -4417,7 +4435,7 @@ define amdgpu_kernel void @global_atomic_fadd_uni_address_div_value_agent_scope_
; GFX1132-DPP-NEXT: v_mbcnt_lo_u32_b32 v4, exec_lo, 0
; GFX1132-DPP-NEXT: v_mov_b32_e32 v0, v1
; GFX1132-DPP-NEXT: s_mov_b32 s0, exec_lo
-; GFX1132-DPP-NEXT: s_delay_alu instid0(VALU_DEP_2)
+; GFX1132-DPP-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132-DPP-NEXT: v_cmpx_eq_u32_e32 0, v4
; GFX1132-DPP-NEXT: s_cbranch_execz .LBB6_2
; GFX1132-DPP-NEXT: ; %bb.1:
@@ -4612,7 +4630,7 @@ define amdgpu_kernel void @global_atomic_fadd_uni_address_uni_value_default_scop
; GFX1164-NEXT: scratch_store_b32 off, v1, off
; GFX1164-NEXT: scratch_load_b64 v[0:1], off, off
; GFX1164-NEXT: v_mbcnt_hi_u32_b32 v2, exec_hi, v2
-; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164-NEXT: v_cmpx_eq_u32_e32 0, v2
; GFX1164-NEXT: s_cbranch_execz .LBB7_3
; GFX1164-NEXT: ; %bb.1:
@@ -4630,14 +4648,14 @@ define amdgpu_kernel void @global_atomic_fadd_uni_address_uni_value_default_scop
; GFX1164-NEXT: v_mul_f32_e32 v2, 4.0, v0
; GFX1164-NEXT: .LBB7_2: ; %atomicrmw.start
; GFX1164-NEXT: ; =>This Inner Loop Header: Depth=1
-; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_4) | instid1(VALU_DEP_2)
; GFX1164-NEXT: v_add_f32_e32 v0, v1, v2
; GFX1164-NEXT: global_atomic_cmpswap_b32 v0, v3, v[0:1], s[0:1] glc
; GFX1164-NEXT: s_waitcnt vmcnt(0)
; GFX1164-NEXT: v_cmp_eq_u32_e32 vcc, v0, v1
; GFX1164-NEXT: v_mov_b32_e32 v1, v0
; GFX1164-NEXT: s_or_b64 s[2:3], vcc, s[2:3]
-; GFX1164-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1164-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1164-NEXT: s_and_not1_b64 exec, exec, s[2:3]
; GFX1164-NEXT: s_cbranch_execnz .LBB7_2
; GFX1164-NEXT: .LBB7_3:
@@ -4656,6 +4674,7 @@ define amdgpu_kernel void @global_atomic_fadd_uni_address_uni_value_default_scop
; GFX1132-NEXT: scratch_store_b32 off, v1, off
; GFX1132-NEXT: scratch_load_b64 v[0:1], off, off
; GFX1132-NEXT: v_cmpx_eq_u32_e32 0, v2
+; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1132-NEXT: s_cbranch_execz .LBB7_3
; GFX1132-NEXT: ; %bb.1:
; GFX1132-NEXT: s_waitcnt vmcnt(0)
@@ -4677,7 +4696,7 @@ define amdgpu_kernel void @global_atomic_fadd_uni_address_uni_value_default_scop
; GFX1132-NEXT: v_cmp_eq_u32_e32 vcc_lo, v0, v1
; GFX1132-NEXT: v_mov_b32_e32 v1, v0
; GFX1132-NEXT: s_or_b32 s2, vcc_lo, s2
-; GFX1132-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1132-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1132-NEXT: s_and_not1_b32 exec_lo, exec_lo, s2
; GFX1132-NEXT: s_cbranch_execnz .LBB7_2
; GFX1132-NEXT: .LBB7_3:
@@ -4862,7 +4881,7 @@ define amdgpu_kernel void @global_atomic_fadd_uni_address_uni_value_default_scop
; GFX1164-DPP-NEXT: scratch_store_b32 off, v1, off
; GFX1164-DPP-NEXT: scratch_load_b64 v[0:1], off, off
; GFX1164-DPP-NEXT: v_mbcnt_hi_u32_b32 v2, exec_hi, v2
-; GFX1164-DPP-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1164-DPP-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164-DPP-NEXT: v_cmpx_eq_u32_e32 0, v2
; GFX1164-DPP-NEXT: s_cbranch_execz .LBB7_3
; GFX1164-DPP-NEXT: ; %bb.1:
@@ -4880,14 +4899,14 @@ define amdgpu_kernel void @global_atomic_fadd_uni_address_uni_value_default_scop
; GFX1164-DPP-NEXT: v_mul_f32_e32 v2, 4.0, v0
; GFX1164-DPP-NEXT: .LBB7_2: ; %atomicrmw.start
; GFX1164-DPP-NEXT: ; =>This Inner Loop Header: Depth=1
-; GFX1164-DPP-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1164-DPP-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_4) | instid1(VALU_DEP_2)
; GFX1164-DPP-NEXT: v_add_f32_e32 v0, v1, v2
; GFX1164-DPP-NEXT: global_atomic_cmpswap_b32 v0, v3, v[0:1], s[0:1] glc
; GFX1164-DPP-NEXT: s_waitcnt vmcnt(0)
; GFX1164-DPP-NEXT: v_cmp_eq_u32_e32 vcc, v0, v1
; GFX1164-DPP-NEXT: v_mov_b32_e32 v1, v0
; GFX1164-DPP-NEXT: s_or_b64 s[2:3], vcc, s[2:3]
-; GFX1164-DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1164-DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1164-DPP-NEXT: s_and_not1_b64 exec, exec, s[2:3]
; GFX1164-DPP-NEXT: s_cbranch_execnz .LBB7_2
; GFX1164-DPP-NEXT: .LBB7_3:
@@ -4906,6 +4925,7 @@ define amdgpu_kernel void @global_atomic_fadd_uni_address_uni_value_default_scop
; GFX1132-DPP-NEXT: scratch_store_b32 off, v1, off
; GFX1132-DPP-NEXT: scratch_load_b64 v[0:1], off, off
; GFX1132-DPP-NEXT: v_cmpx_eq_u32_e32 0, v2
+; GFX1132-DPP-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1132-DPP-NEXT: s_cbranch_execz .LBB7_3
; GFX1132-DPP-NEXT: ; %bb.1:
; GFX1132-DPP-NEXT: s_waitcnt vmcnt(0)
@@ -4927,7 +4947,7 @@ define amdgpu_kernel void @global_atomic_fadd_uni_address_uni_value_default_scop
; GFX1132-DPP-NEXT: v_cmp_eq_u32_e32 vcc_lo, v0, v1
; GFX1132-DPP-NEXT: v_mov_b32_e32 v1, v0
; GFX1132-DPP-NEXT: s_or_b32 s2, vcc_lo, s2
-; GFX1132-DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1132-DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1132-DPP-NEXT: s_and_not1_b32 exec_lo, exec_lo, s2
; GFX1132-DPP-NEXT: s_cbranch_execnz .LBB7_2
; GFX1132-DPP-NEXT: .LBB7_3:
@@ -5249,6 +5269,7 @@ define amdgpu_kernel void @global_atomic_fadd_uni_address_div_value_default_scop
; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
; GFX1164-NEXT: v_cmpx_eq_u32_e32 0, v0
+; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1164-NEXT: s_xor_b64 s[0:1], exec, s[0:1]
; GFX1164-NEXT: s_cbranch_execz .LBB8_5
; GFX1164-NEXT: ; %bb.3:
@@ -5265,9 +5286,10 @@ define amdgpu_kernel void @global_atomic_fadd_uni_address_div_value_default_scop
; GFX1164-NEXT: s_waitcnt vmcnt(0)
; GFX1164-NEXT: v_cmp_eq_u32_e32 vcc, v0, v1
; GFX1164-NEXT: v_mov_b32_e32 v1, v0
+; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1164-NEXT: s_or_b64 s[2:3], vcc, s[2:3]
-; GFX1164-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164-NEXT: s_and_not1_b64 exec, exec, s[2:3]
+; GFX1164-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164-NEXT: s_cbranch_execnz .LBB8_4
; GFX1164-NEXT: .LBB8_5:
; GFX1164-NEXT: s_endpgm
@@ -5309,9 +5331,10 @@ define amdgpu_kernel void @global_atomic_fadd_uni_address_div_value_default_scop
; GFX1132-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
; GFX1132-NEXT: s_mov_b32 s2, 0
; GFX1132-NEXT: s_mov_b32 s0, exec_lo
-; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX1132-NEXT: s_xor_b32 s0, exec_lo, s0
+; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1132-NEXT: s_cbranch_execz .LBB8_5
; GFX1132-NEXT: ; %bb.3:
; GFX1132-NEXT: s_load_b64 s[0:1], s[34:35], 0x24
@@ -5327,7 +5350,7 @@ define amdgpu_kernel void @global_atomic_fadd_uni_address_div_value_default_scop
; GFX1132-NEXT: v_cmp_eq_u32_e32 vcc_lo, v0, v1
; GFX1132-NEXT: v_mov_b32_e32 v1, v0
; GFX1132-NEXT: s_or_b32 s2, vcc_lo, s2
-; GFX1132-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1132-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1132-NEXT: s_and_not1_b32 exec_lo, exec_lo, s2
; GFX1132-NEXT: s_cbranch_execnz .LBB8_4
; GFX1132-NEXT: .LBB8_5:
@@ -5673,6 +5696,7 @@ define amdgpu_kernel void @global_atomic_fadd_uni_address_div_value_default_scop
; GFX1164-DPP-NEXT: v_mov_b32_e32 v0, v1
; GFX1164-DPP-NEXT: s_mov_b64 s[0:1], exec
; GFX1164-DPP-NEXT: v_cmpx_eq_u32_e32 0, v4
+; GFX1164-DPP-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1164-DPP-NEXT: s_cbranch_execz .LBB8_3
; GFX1164-DPP-NEXT: ; %bb.1:
; GFX1164-DPP-NEXT: s_load_b64 s[0:1], s[34:35], 0x24
@@ -5688,9 +5712,10 @@ define amdgpu_kernel void @global_atomic_fadd_uni_address_div_value_default_scop
; GFX1164-DPP-NEXT: s_waitcnt vmcnt(0)
; GFX1164-DPP-NEXT: v_cmp_eq_u32_e32 vcc, v4, v5
; GFX1164-DPP-NEXT: v_mov_b32_e32 v5, v4
+; GFX1164-DPP-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1164-DPP-NEXT: s_or_b64 s[2:3], vcc, s[2:3]
-; GFX1164-DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164-DPP-NEXT: s_and_not1_b64 exec, exec, s[2:3]
+; GFX1164-DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164-DPP-NEXT: s_cbranch_execnz .LBB8_2
; GFX1164-DPP-NEXT: .LBB8_3:
; GFX1164-DPP-NEXT: s_endpgm
@@ -5743,7 +5768,7 @@ define amdgpu_kernel void @global_atomic_fadd_uni_address_div_value_default_scop
; GFX1132-DPP-NEXT: v_mov_b32_e32 v0, v1
; GFX1132-DPP-NEXT: s_mov_b32 s2, 0
; GFX1132-DPP-NEXT: s_mov_b32 s0, exec_lo
-; GFX1132-DPP-NEXT: s_delay_alu instid0(VALU_DEP_2)
+; GFX1132-DPP-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132-DPP-NEXT: v_cmpx_eq_u32_e32 0, v4
; GFX1132-DPP-NEXT: s_cbranch_execz .LBB8_3
; GFX1132-DPP-NEXT: ; %bb.1:
@@ -5760,7 +5785,7 @@ define amdgpu_kernel void @global_atomic_fadd_uni_address_div_value_default_scop
; GFX1132-DPP-NEXT: v_cmp_eq_u32_e32 vcc_lo, v4, v5
; GFX1132-DPP-NEXT: v_mov_b32_e32 v5, v4
; GFX1132-DPP-NEXT: s_or_b32 s2, vcc_lo, s2
-; GFX1132-DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1132-DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1132-DPP-NEXT: s_and_not1_b32 exec_lo, exec_lo, s2
; GFX1132-DPP-NEXT: s_cbranch_execnz .LBB8_2
; GFX1132-DPP-NEXT: .LBB8_3:
@@ -5920,6 +5945,7 @@ define amdgpu_kernel void @global_atomic_fadd_double_uni_address_uni_value_agent
; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164-NEXT: v_mbcnt_hi_u32_b32 v0, s1, v0
; GFX1164-NEXT: v_cmpx_eq_u32_e32 0, v0
+; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1164-NEXT: s_cbranch_execz .LBB9_3
; GFX1164-NEXT: ; %bb.1:
; GFX1164-NEXT: s_bcnt1_i32_b64 s0, s[0:1]
@@ -5943,9 +5969,10 @@ define amdgpu_kernel void @global_atomic_fadd_double_uni_address_uni_value_agent
; GFX1164-NEXT: v_cmp_eq_u64_e32 vcc, v[0:1], v[2:3]
; GFX1164-NEXT: v_mov_b32_e32 v3, v1
; GFX1164-NEXT: v_mov_b32_e32 v2, v0
+; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1164-NEXT: s_or_b64 s[2:3], vcc, s[2:3]
-; GFX1164-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164-NEXT: s_and_not1_b64 exec, exec, s[2:3]
+; GFX1164-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164-NEXT: s_cbranch_execnz .LBB9_2
; GFX1164-NEXT: .LBB9_3:
; GFX1164-NEXT: s_endpgm
@@ -5956,7 +5983,7 @@ define amdgpu_kernel void @global_atomic_fadd_double_uni_address_uni_value_agent
; GFX1132-NEXT: s_mov_b32 s2, 0
; GFX1132-NEXT: v_mbcnt_lo_u32_b32 v0, s0, 0
; GFX1132-NEXT: s_mov_b32 s1, exec_lo
-; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX1132-NEXT: s_cbranch_execz .LBB9_3
; GFX1132-NEXT: ; %bb.1:
@@ -5979,7 +6006,7 @@ define amdgpu_kernel void @global_atomic_fadd_double_uni_address_uni_value_agent
; GFX1132-NEXT: v_cmp_eq_u64_e32 vcc_lo, v[0:1], v[2:3]
; GFX1132-NEXT: v_dual_mov_b32 v3, v1 :: v_dual_mov_b32 v2, v0
; GFX1132-NEXT: s_or_b32 s2, vcc_lo, s2
-; GFX1132-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1132-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1132-NEXT: s_and_not1_b32 exec_lo, exec_lo, s2
; GFX1132-NEXT: s_cbranch_execnz .LBB9_2
; GFX1132-NEXT: .LBB9_3:
@@ -6134,6 +6161,7 @@ define amdgpu_kernel void @global_atomic_fadd_double_uni_address_uni_value_agent
; GFX1164-DPP-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164-DPP-NEXT: v_mbcnt_hi_u32_b32 v0, s1, v0
; GFX1164-DPP-NEXT: v_cmpx_eq_u32_e32 0, v0
+; GFX1164-DPP-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1164-DPP-NEXT: s_cbranch_execz .LBB9_3
; GFX1164-DPP-NEXT: ; %bb.1:
; GFX1164-DPP-NEXT: s_bcnt1_i32_b64 s0, s[0:1]
@@ -6157,9 +6185,10 @@ define amdgpu_kernel void @global_atomic_fadd_double_uni_address_uni_value_agent
; GFX1164-DPP-NEXT: v_cmp_eq_u64_e32 vcc, v[0:1], v[2:3]
; GFX1164-DPP-NEXT: v_mov_b32_e32 v3, v1
; GFX1164-DPP-NEXT: v_mov_b32_e32 v2, v0
+; GFX1164-DPP-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1164-DPP-NEXT: s_or_b64 s[2:3], vcc, s[2:3]
-; GFX1164-DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164-DPP-NEXT: s_and_not1_b64 exec, exec, s[2:3]
+; GFX1164-DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164-DPP-NEXT: s_cbranch_execnz .LBB9_2
; GFX1164-DPP-NEXT: .LBB9_3:
; GFX1164-DPP-NEXT: s_endpgm
@@ -6170,7 +6199,7 @@ define amdgpu_kernel void @global_atomic_fadd_double_uni_address_uni_value_agent
; GFX1132-DPP-NEXT: s_mov_b32 s2, 0
; GFX1132-DPP-NEXT: v_mbcnt_lo_u32_b32 v0, s0, 0
; GFX1132-DPP-NEXT: s_mov_b32 s1, exec_lo
-; GFX1132-DPP-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1132-DPP-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132-DPP-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX1132-DPP-NEXT: s_cbranch_execz .LBB9_3
; GFX1132-DPP-NEXT: ; %bb.1:
@@ -6193,7 +6222,7 @@ define amdgpu_kernel void @global_atomic_fadd_double_uni_address_uni_value_agent
; GFX1132-DPP-NEXT: v_cmp_eq_u64_e32 vcc_lo, v[0:1], v[2:3]
; GFX1132-DPP-NEXT: v_dual_mov_b32 v3, v1 :: v_dual_mov_b32 v2, v0
; GFX1132-DPP-NEXT: s_or_b32 s2, vcc_lo, s2
-; GFX1132-DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1132-DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1132-DPP-NEXT: s_and_not1_b32 exec_lo, exec_lo, s2
; GFX1132-DPP-NEXT: s_cbranch_execnz .LBB9_2
; GFX1132-DPP-NEXT: .LBB9_3:
@@ -6531,6 +6560,7 @@ define amdgpu_kernel void @global_atomic_fadd_double_uni_address_div_value_agent
; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
; GFX1164-NEXT: v_cmpx_eq_u32_e32 0, v0
+; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1164-NEXT: s_xor_b64 s[0:1], exec, s[0:1]
; GFX1164-NEXT: s_cbranch_execz .LBB10_5
; GFX1164-NEXT: ; %bb.3:
@@ -6548,9 +6578,10 @@ define amdgpu_kernel void @global_atomic_fadd_double_uni_address_div_value_agent
; GFX1164-NEXT: v_cmp_eq_u64_e32 vcc, v[0:1], v[2:3]
; GFX1164-NEXT: v_mov_b32_e32 v3, v1
; GFX1164-NEXT: v_mov_b32_e32 v2, v0
+; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1164-NEXT: s_or_b64 s[2:3], vcc, s[2:3]
-; GFX1164-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164-NEXT: s_and_not1_b64 exec, exec, s[2:3]
+; GFX1164-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164-NEXT: s_cbranch_execnz .LBB10_4
; GFX1164-NEXT: .LBB10_5:
; GFX1164-NEXT: s_endpgm
@@ -6594,9 +6625,10 @@ define amdgpu_kernel void @global_atomic_fadd_double_uni_address_div_value_agent
; GFX1132-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
; GFX1132-NEXT: s_mov_b32 s2, 0
; GFX1132-NEXT: s_mov_b32 s0, exec_lo
-; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX1132-NEXT: s_xor_b32 s0, exec_lo, s0
+; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1132-NEXT: s_cbranch_execz .LBB10_5
; GFX1132-NEXT: ; %bb.3:
; GFX1132-NEXT: s_load_b64 s[0:1], s[34:35], 0x24
@@ -6612,7 +6644,7 @@ define amdgpu_kernel void @global_atomic_fadd_double_uni_address_div_value_agent
; GFX1132-NEXT: v_cmp_eq_u64_e32 vcc_lo, v[0:1], v[2:3]
; GFX1132-NEXT: v_dual_mov_b32 v3, v1 :: v_dual_mov_b32 v2, v0
; GFX1132-NEXT: s_or_b32 s2, vcc_lo, s2
-; GFX1132-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1132-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1132-NEXT: s_and_not1_b32 exec_lo, exec_lo, s2
; GFX1132-NEXT: s_cbranch_execnz .LBB10_4
; GFX1132-NEXT: .LBB10_5:
@@ -7016,6 +7048,7 @@ define amdgpu_kernel void @global_atomic_fadd_double_uni_address_div_value_agent
; GFX1164-DPP-NEXT: v_mov_b32_e32 v0, v2
; GFX1164-DPP-NEXT: v_cmpx_eq_u32_e32 0, v6
; GFX1164-DPP-NEXT: s_waitcnt_depctr depctr_sa_sdst(0)
+; GFX1164-DPP-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1164-DPP-NEXT: s_cbranch_execz .LBB10_3
; GFX1164-DPP-NEXT: ; %bb.1:
; GFX1164-DPP-NEXT: s_load_b64 s[0:1], s[34:35], 0x24
@@ -7031,9 +7064,10 @@ define amdgpu_kernel void @global_atomic_fadd_double_uni_address_div_value_agent
; GFX1164-DPP-NEXT: v_cmp_eq_u64_e32 vcc, v[6:7], v[8:9]
; GFX1164-DPP-NEXT: v_mov_b32_e32 v9, v7
; GFX1164-DPP-NEXT: v_mov_b32_e32 v8, v6
+; GFX1164-DPP-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1164-DPP-NEXT: s_or_b64 s[2:3], vcc, s[2:3]
-; GFX1164-DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164-DPP-NEXT: s_and_not1_b64 exec, exec, s[2:3]
+; GFX1164-DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164-DPP-NEXT: s_cbranch_execnz .LBB10_2
; GFX1164-DPP-NEXT: .LBB10_3:
; GFX1164-DPP-NEXT: s_endpgm
@@ -7101,6 +7135,7 @@ define amdgpu_kernel void @global_atomic_fadd_double_uni_address_div_value_agent
; GFX1132-DPP-NEXT: s_mov_b32 s2, 0
; GFX1132-DPP-NEXT: s_mov_b32 s0, exec_lo
; GFX1132-DPP-NEXT: v_cmpx_eq_u32_e32 0, v6
+; GFX1132-DPP-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1132-DPP-NEXT: s_cbranch_execz .LBB10_3
; GFX1132-DPP-NEXT: ; %bb.1:
; GFX1132-DPP-NEXT: s_load_b64 s[0:1], s[34:35], 0x24
@@ -7115,7 +7150,7 @@ define amdgpu_kernel void @global_atomic_fadd_double_uni_address_div_value_agent
; GFX1132-DPP-NEXT: v_cmp_eq_u64_e32 vcc_lo, v[6:7], v[8:9]
; GFX1132-DPP-NEXT: v_dual_mov_b32 v9, v7 :: v_dual_mov_b32 v8, v6
; GFX1132-DPP-NEXT: s_or_b32 s2, vcc_lo, s2
-; GFX1132-DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1132-DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1132-DPP-NEXT: s_and_not1_b32 exec_lo, exec_lo, s2
; GFX1132-DPP-NEXT: s_cbranch_execnz .LBB10_2
; GFX1132-DPP-NEXT: .LBB10_3:
@@ -7311,7 +7346,7 @@ define amdgpu_kernel void @global_atomic_fadd_double_uni_address_uni_value_one_a
; GFX1164-NEXT: scratch_store_b32 off, v1, off
; GFX1164-NEXT: scratch_load_b64 v[0:1], off, off
; GFX1164-NEXT: v_mbcnt_hi_u32_b32 v2, exec_hi, v2
-; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164-NEXT: v_cmpx_eq_u32_e32 0, v2
; GFX1164-NEXT: s_cbranch_execz .LBB11_3
; GFX1164-NEXT: ; %bb.1:
@@ -7336,9 +7371,10 @@ define amdgpu_kernel void @global_atomic_fadd_double_uni_address_uni_value_one_a
; GFX1164-NEXT: v_cmp_eq_u64_e32 vcc, v[0:1], v[2:3]
; GFX1164-NEXT: v_mov_b32_e32 v3, v1
; GFX1164-NEXT: v_mov_b32_e32 v2, v0
+; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1164-NEXT: s_or_b64 s[2:3], vcc, s[2:3]
-; GFX1164-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164-NEXT: s_and_not1_b64 exec, exec, s[2:3]
+; GFX1164-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164-NEXT: s_cbranch_execnz .LBB11_2
; GFX1164-NEXT: .LBB11_3:
; GFX1164-NEXT: s_endpgm
@@ -7356,6 +7392,7 @@ define amdgpu_kernel void @global_atomic_fadd_double_uni_address_uni_value_one_a
; GFX1132-NEXT: scratch_store_b32 off, v1, off
; GFX1132-NEXT: scratch_load_b64 v[0:1], off, off
; GFX1132-NEXT: v_cmpx_eq_u32_e32 0, v2
+; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1132-NEXT: s_cbranch_execz .LBB11_3
; GFX1132-NEXT: ; %bb.1:
; GFX1132-NEXT: s_waitcnt vmcnt(0)
@@ -7377,7 +7414,7 @@ define amdgpu_kernel void @global_atomic_fadd_double_uni_address_uni_value_one_a
; GFX1132-NEXT: v_cmp_eq_u64_e32 vcc_lo, v[0:1], v[2:3]
; GFX1132-NEXT: v_dual_mov_b32 v3, v1 :: v_dual_mov_b32 v2, v0
; GFX1132-NEXT: s_or_b32 s2, vcc_lo, s2
-; GFX1132-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1132-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1132-NEXT: s_and_not1_b32 exec_lo, exec_lo, s2
; GFX1132-NEXT: s_cbranch_execnz .LBB11_2
; GFX1132-NEXT: .LBB11_3:
@@ -7568,7 +7605,7 @@ define amdgpu_kernel void @global_atomic_fadd_double_uni_address_uni_value_one_a
; GFX1164-DPP-NEXT: scratch_store_b32 off, v1, off
; GFX1164-DPP-NEXT: scratch_load_b64 v[0:1], off, off
; GFX1164-DPP-NEXT: v_mbcnt_hi_u32_b32 v2, exec_hi, v2
-; GFX1164-DPP-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1164-DPP-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164-DPP-NEXT: v_cmpx_eq_u32_e32 0, v2
; GFX1164-DPP-NEXT: s_cbranch_execz .LBB11_3
; GFX1164-DPP-NEXT: ; %bb.1:
@@ -7593,9 +7630,10 @@ define amdgpu_kernel void @global_atomic_fadd_double_uni_address_uni_value_one_a
; GFX1164-DPP-NEXT: v_cmp_eq_u64_e32 vcc, v[0:1], v[2:3]
; GFX1164-DPP-NEXT: v_mov_b32_e32 v3, v1
; GFX1164-DPP-NEXT: v_mov_b32_e32 v2, v0
+; GFX1164-DPP-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1164-DPP-NEXT: s_or_b64 s[2:3], vcc, s[2:3]
-; GFX1164-DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164-DPP-NEXT: s_and_not1_b64 exec, exec, s[2:3]
+; GFX1164-DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164-DPP-NEXT: s_cbranch_execnz .LBB11_2
; GFX1164-DPP-NEXT: .LBB11_3:
; GFX1164-DPP-NEXT: s_endpgm
@@ -7613,6 +7651,7 @@ define amdgpu_kernel void @global_atomic_fadd_double_uni_address_uni_value_one_a
; GFX1132-DPP-NEXT: scratch_store_b32 off, v1, off
; GFX1132-DPP-NEXT: scratch_load_b64 v[0:1], off, off
; GFX1132-DPP-NEXT: v_cmpx_eq_u32_e32 0, v2
+; GFX1132-DPP-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1132-DPP-NEXT: s_cbranch_execz .LBB11_3
; GFX1132-DPP-NEXT: ; %bb.1:
; GFX1132-DPP-NEXT: s_waitcnt vmcnt(0)
@@ -7634,7 +7673,7 @@ define amdgpu_kernel void @global_atomic_fadd_double_uni_address_uni_value_one_a
; GFX1132-DPP-NEXT: v_cmp_eq_u64_e32 vcc_lo, v[0:1], v[2:3]
; GFX1132-DPP-NEXT: v_dual_mov_b32 v3, v1 :: v_dual_mov_b32 v2, v0
; GFX1132-DPP-NEXT: s_or_b32 s2, vcc_lo, s2
-; GFX1132-DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1132-DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1132-DPP-NEXT: s_and_not1_b32 exec_lo, exec_lo, s2
; GFX1132-DPP-NEXT: s_cbranch_execnz .LBB11_2
; GFX1132-DPP-NEXT: .LBB11_3:
@@ -7972,6 +8011,7 @@ define amdgpu_kernel void @global_atomic_fadd_double_uni_address_div_value_one_a
; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
; GFX1164-NEXT: v_cmpx_eq_u32_e32 0, v0
+; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1164-NEXT: s_xor_b64 s[0:1], exec, s[0:1]
; GFX1164-NEXT: s_cbranch_execz .LBB12_5
; GFX1164-NEXT: ; %bb.3:
@@ -7989,9 +8029,10 @@ define amdgpu_kernel void @global_atomic_fadd_double_uni_address_div_value_one_a
; GFX1164-NEXT: v_cmp_eq_u64_e32 vcc, v[0:1], v[2:3]
; GFX1164-NEXT: v_mov_b32_e32 v3, v1
; GFX1164-NEXT: v_mov_b32_e32 v2, v0
+; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1164-NEXT: s_or_b64 s[2:3], vcc, s[2:3]
-; GFX1164-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164-NEXT: s_and_not1_b64 exec, exec, s[2:3]
+; GFX1164-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164-NEXT: s_cbranch_execnz .LBB12_4
; GFX1164-NEXT: .LBB12_5:
; GFX1164-NEXT: s_endpgm
@@ -8035,9 +8076,10 @@ define amdgpu_kernel void @global_atomic_fadd_double_uni_address_div_value_one_a
; GFX1132-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
; GFX1132-NEXT: s_mov_b32 s2, 0
; GFX1132-NEXT: s_mov_b32 s0, exec_lo
-; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX1132-NEXT: s_xor_b32 s0, exec_lo, s0
+; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1132-NEXT: s_cbranch_execz .LBB12_5
; GFX1132-NEXT: ; %bb.3:
; GFX1132-NEXT: s_load_b64 s[0:1], s[34:35], 0x24
@@ -8053,7 +8095,7 @@ define amdgpu_kernel void @global_atomic_fadd_double_uni_address_div_value_one_a
; GFX1132-NEXT: v_cmp_eq_u64_e32 vcc_lo, v[0:1], v[2:3]
; GFX1132-NEXT: v_dual_mov_b32 v3, v1 :: v_dual_mov_b32 v2, v0
; GFX1132-NEXT: s_or_b32 s2, vcc_lo, s2
-; GFX1132-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1132-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1132-NEXT: s_and_not1_b32 exec_lo, exec_lo, s2
; GFX1132-NEXT: s_cbranch_execnz .LBB12_4
; GFX1132-NEXT: .LBB12_5:
@@ -8457,6 +8499,7 @@ define amdgpu_kernel void @global_atomic_fadd_double_uni_address_div_value_one_a
; GFX1164-DPP-NEXT: v_mov_b32_e32 v0, v2
; GFX1164-DPP-NEXT: v_cmpx_eq_u32_e32 0, v6
; GFX1164-DPP-NEXT: s_waitcnt_depctr depctr_sa_sdst(0)
+; GFX1164-DPP-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1164-DPP-NEXT: s_cbranch_execz .LBB12_3
; GFX1164-DPP-NEXT: ; %bb.1:
; GFX1164-DPP-NEXT: s_load_b64 s[0:1], s[34:35], 0x24
@@ -8472,9 +8515,10 @@ define amdgpu_kernel void @global_atomic_fadd_double_uni_address_div_value_one_a
; GFX1164-DPP-NEXT: v_cmp_eq_u64_e32 vcc, v[6:7], v[8:9]
; GFX1164-DPP-NEXT: v_mov_b32_e32 v9, v7
; GFX1164-DPP-NEXT: v_mov_b32_e32 v8, v6
+; GFX1164-DPP-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1164-DPP-NEXT: s_or_b64 s[2:3], vcc, s[2:3]
-; GFX1164-DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164-DPP-NEXT: s_and_not1_b64 exec, exec, s[2:3]
+; GFX1164-DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164-DPP-NEXT: s_cbranch_execnz .LBB12_2
; GFX1164-DPP-NEXT: .LBB12_3:
; GFX1164-DPP-NEXT: s_endpgm
@@ -8542,6 +8586,7 @@ define amdgpu_kernel void @global_atomic_fadd_double_uni_address_div_value_one_a
; GFX1132-DPP-NEXT: s_mov_b32 s2, 0
; GFX1132-DPP-NEXT: s_mov_b32 s0, exec_lo
; GFX1132-DPP-NEXT: v_cmpx_eq_u32_e32 0, v6
+; GFX1132-DPP-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1132-DPP-NEXT: s_cbranch_execz .LBB12_3
; GFX1132-DPP-NEXT: ; %bb.1:
; GFX1132-DPP-NEXT: s_load_b64 s[0:1], s[34:35], 0x24
@@ -8556,7 +8601,7 @@ define amdgpu_kernel void @global_atomic_fadd_double_uni_address_div_value_one_a
; GFX1132-DPP-NEXT: v_cmp_eq_u64_e32 vcc_lo, v[6:7], v[8:9]
; GFX1132-DPP-NEXT: v_dual_mov_b32 v9, v7 :: v_dual_mov_b32 v8, v6
; GFX1132-DPP-NEXT: s_or_b32 s2, vcc_lo, s2
-; GFX1132-DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1132-DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1132-DPP-NEXT: s_and_not1_b32 exec_lo, exec_lo, s2
; GFX1132-DPP-NEXT: s_cbranch_execnz .LBB12_2
; GFX1132-DPP-NEXT: .LBB12_3:
@@ -8752,7 +8797,7 @@ define amdgpu_kernel void @global_atomic_fadd_double_uni_address_uni_value_agent
; GFX1164-NEXT: scratch_store_b32 off, v1, off
; GFX1164-NEXT: scratch_load_b64 v[0:1], off, off
; GFX1164-NEXT: v_mbcnt_hi_u32_b32 v2, exec_hi, v2
-; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164-NEXT: v_cmpx_eq_u32_e32 0, v2
; GFX1164-NEXT: s_cbranch_execz .LBB13_3
; GFX1164-NEXT: ; %bb.1:
@@ -8777,9 +8822,10 @@ define amdgpu_kernel void @global_atomic_fadd_double_uni_address_uni_value_agent
; GFX1164-NEXT: v_cmp_eq_u64_e32 vcc, v[0:1], v[2:3]
; GFX1164-NEXT: v_mov_b32_e32 v3, v1
; GFX1164-NEXT: v_mov_b32_e32 v2, v0
+; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1164-NEXT: s_or_b64 s[2:3], vcc, s[2:3]
-; GFX1164-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164-NEXT: s_and_not1_b64 exec, exec, s[2:3]
+; GFX1164-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164-NEXT: s_cbranch_execnz .LBB13_2
; GFX1164-NEXT: .LBB13_3:
; GFX1164-NEXT: s_endpgm
@@ -8797,6 +8843,7 @@ define amdgpu_kernel void @global_atomic_fadd_double_uni_address_uni_value_agent
; GFX1132-NEXT: scratch_store_b32 off, v1, off
; GFX1132-NEXT: scratch_load_b64 v[0:1], off, off
; GFX1132-NEXT: v_cmpx_eq_u32_e32 0, v2
+; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1132-NEXT: s_cbranch_execz .LBB13_3
; GFX1132-NEXT: ; %bb.1:
; GFX1132-NEXT: s_waitcnt vmcnt(0)
@@ -8818,7 +8865,7 @@ define amdgpu_kernel void @global_atomic_fadd_double_uni_address_uni_value_agent
; GFX1132-NEXT: v_cmp_eq_u64_e32 vcc_lo, v[0:1], v[2:3]
; GFX1132-NEXT: v_dual_mov_b32 v3, v1 :: v_dual_mov_b32 v2, v0
; GFX1132-NEXT: s_or_b32 s2, vcc_lo, s2
-; GFX1132-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1132-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1132-NEXT: s_and_not1_b32 exec_lo, exec_lo, s2
; GFX1132-NEXT: s_cbranch_execnz .LBB13_2
; GFX1132-NEXT: .LBB13_3:
@@ -9009,7 +9056,7 @@ define amdgpu_kernel void @global_atomic_fadd_double_uni_address_uni_value_agent
; GFX1164-DPP-NEXT: scratch_store_b32 off, v1, off
; GFX1164-DPP-NEXT: scratch_load_b64 v[0:1], off, off
; GFX1164-DPP-NEXT: v_mbcnt_hi_u32_b32 v2, exec_hi, v2
-; GFX1164-DPP-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1164-DPP-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164-DPP-NEXT: v_cmpx_eq_u32_e32 0, v2
; GFX1164-DPP-NEXT: s_cbranch_execz .LBB13_3
; GFX1164-DPP-NEXT: ; %bb.1:
@@ -9034,9 +9081,10 @@ define amdgpu_kernel void @global_atomic_fadd_double_uni_address_uni_value_agent
; GFX1164-DPP-NEXT: v_cmp_eq_u64_e32 vcc, v[0:1], v[2:3]
; GFX1164-DPP-NEXT: v_mov_b32_e32 v3, v1
; GFX1164-DPP-NEXT: v_mov_b32_e32 v2, v0
+; GFX1164-DPP-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1164-DPP-NEXT: s_or_b64 s[2:3], vcc, s[2:3]
-; GFX1164-DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164-DPP-NEXT: s_and_not1_b64 exec, exec, s[2:3]
+; GFX1164-DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164-DPP-NEXT: s_cbranch_execnz .LBB13_2
; GFX1164-DPP-NEXT: .LBB13_3:
; GFX1164-DPP-NEXT: s_endpgm
@@ -9054,6 +9102,7 @@ define amdgpu_kernel void @global_atomic_fadd_double_uni_address_uni_value_agent
; GFX1132-DPP-NEXT: scratch_store_b32 off, v1, off
; GFX1132-DPP-NEXT: scratch_load_b64 v[0:1], off, off
; GFX1132-DPP-NEXT: v_cmpx_eq_u32_e32 0, v2
+; GFX1132-DPP-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1132-DPP-NEXT: s_cbranch_execz .LBB13_3
; GFX1132-DPP-NEXT: ; %bb.1:
; GFX1132-DPP-NEXT: s_waitcnt vmcnt(0)
@@ -9075,7 +9124,7 @@ define amdgpu_kernel void @global_atomic_fadd_double_uni_address_uni_value_agent
; GFX1132-DPP-NEXT: v_cmp_eq_u64_e32 vcc_lo, v[0:1], v[2:3]
; GFX1132-DPP-NEXT: v_dual_mov_b32 v3, v1 :: v_dual_mov_b32 v2, v0
; GFX1132-DPP-NEXT: s_or_b32 s2, vcc_lo, s2
-; GFX1132-DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1132-DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1132-DPP-NEXT: s_and_not1_b32 exec_lo, exec_lo, s2
; GFX1132-DPP-NEXT: s_cbranch_execnz .LBB13_2
; GFX1132-DPP-NEXT: .LBB13_3:
@@ -9413,6 +9462,7 @@ define amdgpu_kernel void @global_atomic_fadd_double_uni_address_div_value_agent
; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
; GFX1164-NEXT: v_cmpx_eq_u32_e32 0, v0
+; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1164-NEXT: s_xor_b64 s[0:1], exec, s[0:1]
; GFX1164-NEXT: s_cbranch_execz .LBB14_5
; GFX1164-NEXT: ; %bb.3:
@@ -9430,9 +9480,10 @@ define amdgpu_kernel void @global_atomic_fadd_double_uni_address_div_value_agent
; GFX1164-NEXT: v_cmp_eq_u64_e32 vcc, v[0:1], v[2:3]
; GFX1164-NEXT: v_mov_b32_e32 v3, v1
; GFX1164-NEXT: v_mov_b32_e32 v2, v0
+; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1164-NEXT: s_or_b64 s[2:3], vcc, s[2:3]
-; GFX1164-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164-NEXT: s_and_not1_b64 exec, exec, s[2:3]
+; GFX1164-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164-NEXT: s_cbranch_execnz .LBB14_4
; GFX1164-NEXT: .LBB14_5:
; GFX1164-NEXT: s_endpgm
@@ -9476,9 +9527,10 @@ define amdgpu_kernel void @global_atomic_fadd_double_uni_address_div_value_agent
; GFX1132-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
; GFX1132-NEXT: s_mov_b32 s2, 0
; GFX1132-NEXT: s_mov_b32 s0, exec_lo
-; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX1132-NEXT: s_xor_b32 s0, exec_lo, s0
+; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1132-NEXT: s_cbranch_execz .LBB14_5
; GFX1132-NEXT: ; %bb.3:
; GFX1132-NEXT: s_load_b64 s[0:1], s[34:35], 0x24
@@ -9494,7 +9546,7 @@ define amdgpu_kernel void @global_atomic_fadd_double_uni_address_div_value_agent
; GFX1132-NEXT: v_cmp_eq_u64_e32 vcc_lo, v[0:1], v[2:3]
; GFX1132-NEXT: v_dual_mov_b32 v3, v1 :: v_dual_mov_b32 v2, v0
; GFX1132-NEXT: s_or_b32 s2, vcc_lo, s2
-; GFX1132-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1132-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1132-NEXT: s_and_not1_b32 exec_lo, exec_lo, s2
; GFX1132-NEXT: s_cbranch_execnz .LBB14_4
; GFX1132-NEXT: .LBB14_5:
@@ -9898,6 +9950,7 @@ define amdgpu_kernel void @global_atomic_fadd_double_uni_address_div_value_agent
; GFX1164-DPP-NEXT: v_mov_b32_e32 v0, v2
; GFX1164-DPP-NEXT: v_cmpx_eq_u32_e32 0, v6
; GFX1164-DPP-NEXT: s_waitcnt_depctr depctr_sa_sdst(0)
+; GFX1164-DPP-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1164-DPP-NEXT: s_cbranch_execz .LBB14_3
; GFX1164-DPP-NEXT: ; %bb.1:
; GFX1164-DPP-NEXT: s_load_b64 s[0:1], s[34:35], 0x24
@@ -9913,9 +9966,10 @@ define amdgpu_kernel void @global_atomic_fadd_double_uni_address_div_value_agent
; GFX1164-DPP-NEXT: v_cmp_eq_u64_e32 vcc, v[6:7], v[8:9]
; GFX1164-DPP-NEXT: v_mov_b32_e32 v9, v7
; GFX1164-DPP-NEXT: v_mov_b32_e32 v8, v6
+; GFX1164-DPP-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1164-DPP-NEXT: s_or_b64 s[2:3], vcc, s[2:3]
-; GFX1164-DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164-DPP-NEXT: s_and_not1_b64 exec, exec, s[2:3]
+; GFX1164-DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164-DPP-NEXT: s_cbranch_execnz .LBB14_2
; GFX1164-DPP-NEXT: .LBB14_3:
; GFX1164-DPP-NEXT: s_endpgm
@@ -9983,6 +10037,7 @@ define amdgpu_kernel void @global_atomic_fadd_double_uni_address_div_value_agent
; GFX1132-DPP-NEXT: s_mov_b32 s2, 0
; GFX1132-DPP-NEXT: s_mov_b32 s0, exec_lo
; GFX1132-DPP-NEXT: v_cmpx_eq_u32_e32 0, v6
+; GFX1132-DPP-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1132-DPP-NEXT: s_cbranch_execz .LBB14_3
; GFX1132-DPP-NEXT: ; %bb.1:
; GFX1132-DPP-NEXT: s_load_b64 s[0:1], s[34:35], 0x24
@@ -9997,7 +10052,7 @@ define amdgpu_kernel void @global_atomic_fadd_double_uni_address_div_value_agent
; GFX1132-DPP-NEXT: v_cmp_eq_u64_e32 vcc_lo, v[6:7], v[8:9]
; GFX1132-DPP-NEXT: v_dual_mov_b32 v9, v7 :: v_dual_mov_b32 v8, v6
; GFX1132-DPP-NEXT: s_or_b32 s2, vcc_lo, s2
-; GFX1132-DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1132-DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1132-DPP-NEXT: s_and_not1_b32 exec_lo, exec_lo, s2
; GFX1132-DPP-NEXT: s_cbranch_execnz .LBB14_2
; GFX1132-DPP-NEXT: .LBB14_3:
@@ -10336,6 +10391,7 @@ define amdgpu_kernel void @global_atomic_fadd_double_uni_address_div_value_agent
; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
; GFX1164-NEXT: v_cmpx_eq_u32_e32 0, v0
+; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1164-NEXT: s_xor_b64 s[0:1], exec, s[0:1]
; GFX1164-NEXT: s_cbranch_execz .LBB15_5
; GFX1164-NEXT: ; %bb.3:
@@ -10353,9 +10409,10 @@ define amdgpu_kernel void @global_atomic_fadd_double_uni_address_div_value_agent
; GFX1164-NEXT: v_cmp_eq_u64_e32 vcc, v[0:1], v[2:3]
; GFX1164-NEXT: v_mov_b32_e32 v3, v1
; GFX1164-NEXT: v_mov_b32_e32 v2, v0
+; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1164-NEXT: s_or_b64 s[2:3], vcc, s[2:3]
-; GFX1164-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164-NEXT: s_and_not1_b64 exec, exec, s[2:3]
+; GFX1164-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164-NEXT: s_cbranch_execnz .LBB15_4
; GFX1164-NEXT: .LBB15_5:
; GFX1164-NEXT: s_endpgm
@@ -10399,9 +10456,10 @@ define amdgpu_kernel void @global_atomic_fadd_double_uni_address_div_value_agent
; GFX1132-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
; GFX1132-NEXT: s_mov_b32 s2, 0
; GFX1132-NEXT: s_mov_b32 s0, exec_lo
-; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX1132-NEXT: s_xor_b32 s0, exec_lo, s0
+; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1132-NEXT: s_cbranch_execz .LBB15_5
; GFX1132-NEXT: ; %bb.3:
; GFX1132-NEXT: s_load_b64 s[0:1], s[34:35], 0x24
@@ -10417,7 +10475,7 @@ define amdgpu_kernel void @global_atomic_fadd_double_uni_address_div_value_agent
; GFX1132-NEXT: v_cmp_eq_u64_e32 vcc_lo, v[0:1], v[2:3]
; GFX1132-NEXT: v_dual_mov_b32 v3, v1 :: v_dual_mov_b32 v2, v0
; GFX1132-NEXT: s_or_b32 s2, vcc_lo, s2
-; GFX1132-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1132-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1132-NEXT: s_and_not1_b32 exec_lo, exec_lo, s2
; GFX1132-NEXT: s_cbranch_execnz .LBB15_4
; GFX1132-NEXT: .LBB15_5:
@@ -10821,6 +10879,7 @@ define amdgpu_kernel void @global_atomic_fadd_double_uni_address_div_value_agent
; GFX1164-DPP-NEXT: v_mov_b32_e32 v0, v2
; GFX1164-DPP-NEXT: v_cmpx_eq_u32_e32 0, v6
; GFX1164-DPP-NEXT: s_waitcnt_depctr depctr_sa_sdst(0)
+; GFX1164-DPP-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1164-DPP-NEXT: s_cbranch_execz .LBB15_3
; GFX1164-DPP-NEXT: ; %bb.1:
; GFX1164-DPP-NEXT: s_load_b64 s[0:1], s[34:35], 0x24
@@ -10836,9 +10895,10 @@ define amdgpu_kernel void @global_atomic_fadd_double_uni_address_div_value_agent
; GFX1164-DPP-NEXT: v_cmp_eq_u64_e32 vcc, v[6:7], v[8:9]
; GFX1164-DPP-NEXT: v_mov_b32_e32 v9, v7
; GFX1164-DPP-NEXT: v_mov_b32_e32 v8, v6
+; GFX1164-DPP-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1164-DPP-NEXT: s_or_b64 s[2:3], vcc, s[2:3]
-; GFX1164-DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164-DPP-NEXT: s_and_not1_b64 exec, exec, s[2:3]
+; GFX1164-DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164-DPP-NEXT: s_cbranch_execnz .LBB15_2
; GFX1164-DPP-NEXT: .LBB15_3:
; GFX1164-DPP-NEXT: s_endpgm
@@ -10906,6 +10966,7 @@ define amdgpu_kernel void @global_atomic_fadd_double_uni_address_div_value_agent
; GFX1132-DPP-NEXT: s_mov_b32 s2, 0
; GFX1132-DPP-NEXT: s_mov_b32 s0, exec_lo
; GFX1132-DPP-NEXT: v_cmpx_eq_u32_e32 0, v6
+; GFX1132-DPP-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1132-DPP-NEXT: s_cbranch_execz .LBB15_3
; GFX1132-DPP-NEXT: ; %bb.1:
; GFX1132-DPP-NEXT: s_load_b64 s[0:1], s[34:35], 0x24
@@ -10920,7 +10981,7 @@ define amdgpu_kernel void @global_atomic_fadd_double_uni_address_div_value_agent
; GFX1132-DPP-NEXT: v_cmp_eq_u64_e32 vcc_lo, v[6:7], v[8:9]
; GFX1132-DPP-NEXT: v_dual_mov_b32 v9, v7 :: v_dual_mov_b32 v8, v6
; GFX1132-DPP-NEXT: s_or_b32 s2, vcc_lo, s2
-; GFX1132-DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1132-DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1132-DPP-NEXT: s_and_not1_b32 exec_lo, exec_lo, s2
; GFX1132-DPP-NEXT: s_cbranch_execnz .LBB15_2
; GFX1132-DPP-NEXT: .LBB15_3:
@@ -11116,7 +11177,7 @@ define amdgpu_kernel void @global_atomic_fadd_double_uni_address_uni_value_defau
; GFX1164-NEXT: scratch_store_b32 off, v1, off
; GFX1164-NEXT: scratch_load_b64 v[0:1], off, off
; GFX1164-NEXT: v_mbcnt_hi_u32_b32 v2, exec_hi, v2
-; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164-NEXT: v_cmpx_eq_u32_e32 0, v2
; GFX1164-NEXT: s_cbranch_execz .LBB16_3
; GFX1164-NEXT: ; %bb.1:
@@ -11141,9 +11202,10 @@ define amdgpu_kernel void @global_atomic_fadd_double_uni_address_uni_value_defau
; GFX1164-NEXT: v_cmp_eq_u64_e32 vcc, v[0:1], v[2:3]
; GFX1164-NEXT: v_mov_b32_e32 v3, v1
; GFX1164-NEXT: v_mov_b32_e32 v2, v0
+; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1164-NEXT: s_or_b64 s[2:3], vcc, s[2:3]
-; GFX1164-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164-NEXT: s_and_not1_b64 exec, exec, s[2:3]
+; GFX1164-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164-NEXT: s_cbranch_execnz .LBB16_2
; GFX1164-NEXT: .LBB16_3:
; GFX1164-NEXT: s_endpgm
@@ -11161,6 +11223,7 @@ define amdgpu_kernel void @global_atomic_fadd_double_uni_address_uni_value_defau
; GFX1132-NEXT: scratch_store_b32 off, v1, off
; GFX1132-NEXT: scratch_load_b64 v[0:1], off, off
; GFX1132-NEXT: v_cmpx_eq_u32_e32 0, v2
+; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1132-NEXT: s_cbranch_execz .LBB16_3
; GFX1132-NEXT: ; %bb.1:
; GFX1132-NEXT: s_waitcnt vmcnt(0)
@@ -11182,7 +11245,7 @@ define amdgpu_kernel void @global_atomic_fadd_double_uni_address_uni_value_defau
; GFX1132-NEXT: v_cmp_eq_u64_e32 vcc_lo, v[0:1], v[2:3]
; GFX1132-NEXT: v_dual_mov_b32 v3, v1 :: v_dual_mov_b32 v2, v0
; GFX1132-NEXT: s_or_b32 s2, vcc_lo, s2
-; GFX1132-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1132-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1132-NEXT: s_and_not1_b32 exec_lo, exec_lo, s2
; GFX1132-NEXT: s_cbranch_execnz .LBB16_2
; GFX1132-NEXT: .LBB16_3:
@@ -11373,7 +11436,7 @@ define amdgpu_kernel void @global_atomic_fadd_double_uni_address_uni_value_defau
; GFX1164-DPP-NEXT: scratch_store_b32 off, v1, off
; GFX1164-DPP-NEXT: scratch_load_b64 v[0:1], off, off
; GFX1164-DPP-NEXT: v_mbcnt_hi_u32_b32 v2, exec_hi, v2
-; GFX1164-DPP-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1164-DPP-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164-DPP-NEXT: v_cmpx_eq_u32_e32 0, v2
; GFX1164-DPP-NEXT: s_cbranch_execz .LBB16_3
; GFX1164-DPP-NEXT: ; %bb.1:
@@ -11398,9 +11461,10 @@ define amdgpu_kernel void @global_atomic_fadd_double_uni_address_uni_value_defau
; GFX1164-DPP-NEXT: v_cmp_eq_u64_e32 vcc, v[0:1], v[2:3]
; GFX1164-DPP-NEXT: v_mov_b32_e32 v3, v1
; GFX1164-DPP-NEXT: v_mov_b32_e32 v2, v0
+; GFX1164-DPP-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1164-DPP-NEXT: s_or_b64 s[2:3], vcc, s[2:3]
-; GFX1164-DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164-DPP-NEXT: s_and_not1_b64 exec, exec, s[2:3]
+; GFX1164-DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164-DPP-NEXT: s_cbranch_execnz .LBB16_2
; GFX1164-DPP-NEXT: .LBB16_3:
; GFX1164-DPP-NEXT: s_endpgm
@@ -11418,6 +11482,7 @@ define amdgpu_kernel void @global_atomic_fadd_double_uni_address_uni_value_defau
; GFX1132-DPP-NEXT: scratch_store_b32 off, v1, off
; GFX1132-DPP-NEXT: scratch_load_b64 v[0:1], off, off
; GFX1132-DPP-NEXT: v_cmpx_eq_u32_e32 0, v2
+; GFX1132-DPP-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1132-DPP-NEXT: s_cbranch_execz .LBB16_3
; GFX1132-DPP-NEXT: ; %bb.1:
; GFX1132-DPP-NEXT: s_waitcnt vmcnt(0)
@@ -11439,7 +11504,7 @@ define amdgpu_kernel void @global_atomic_fadd_double_uni_address_uni_value_defau
; GFX1132-DPP-NEXT: v_cmp_eq_u64_e32 vcc_lo, v[0:1], v[2:3]
; GFX1132-DPP-NEXT: v_dual_mov_b32 v3, v1 :: v_dual_mov_b32 v2, v0
; GFX1132-DPP-NEXT: s_or_b32 s2, vcc_lo, s2
-; GFX1132-DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1132-DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1132-DPP-NEXT: s_and_not1_b32 exec_lo, exec_lo, s2
; GFX1132-DPP-NEXT: s_cbranch_execnz .LBB16_2
; GFX1132-DPP-NEXT: .LBB16_3:
@@ -11777,6 +11842,7 @@ define amdgpu_kernel void @global_atomic_fadd_double_uni_address_div_value_defau
; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
; GFX1164-NEXT: v_cmpx_eq_u32_e32 0, v0
+; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1164-NEXT: s_xor_b64 s[0:1], exec, s[0:1]
; GFX1164-NEXT: s_cbranch_execz .LBB17_5
; GFX1164-NEXT: ; %bb.3:
@@ -11794,9 +11860,10 @@ define amdgpu_kernel void @global_atomic_fadd_double_uni_address_div_value_defau
; GFX1164-NEXT: v_cmp_eq_u64_e32 vcc, v[0:1], v[2:3]
; GFX1164-NEXT: v_mov_b32_e32 v3, v1
; GFX1164-NEXT: v_mov_b32_e32 v2, v0
+; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1164-NEXT: s_or_b64 s[2:3], vcc, s[2:3]
-; GFX1164-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164-NEXT: s_and_not1_b64 exec, exec, s[2:3]
+; GFX1164-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164-NEXT: s_cbranch_execnz .LBB17_4
; GFX1164-NEXT: .LBB17_5:
; GFX1164-NEXT: s_endpgm
@@ -11840,9 +11907,10 @@ define amdgpu_kernel void @global_atomic_fadd_double_uni_address_div_value_defau
; GFX1132-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
; GFX1132-NEXT: s_mov_b32 s2, 0
; GFX1132-NEXT: s_mov_b32 s0, exec_lo
-; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX1132-NEXT: s_xor_b32 s0, exec_lo, s0
+; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1132-NEXT: s_cbranch_execz .LBB17_5
; GFX1132-NEXT: ; %bb.3:
; GFX1132-NEXT: s_load_b64 s[0:1], s[34:35], 0x24
@@ -11858,7 +11926,7 @@ define amdgpu_kernel void @global_atomic_fadd_double_uni_address_div_value_defau
; GFX1132-NEXT: v_cmp_eq_u64_e32 vcc_lo, v[0:1], v[2:3]
; GFX1132-NEXT: v_dual_mov_b32 v3, v1 :: v_dual_mov_b32 v2, v0
; GFX1132-NEXT: s_or_b32 s2, vcc_lo, s2
-; GFX1132-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1132-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1132-NEXT: s_and_not1_b32 exec_lo, exec_lo, s2
; GFX1132-NEXT: s_cbranch_execnz .LBB17_4
; GFX1132-NEXT: .LBB17_5:
@@ -12262,6 +12330,7 @@ define amdgpu_kernel void @global_atomic_fadd_double_uni_address_div_value_defau
; GFX1164-DPP-NEXT: v_mov_b32_e32 v0, v2
; GFX1164-DPP-NEXT: v_cmpx_eq_u32_e32 0, v6
; GFX1164-DPP-NEXT: s_waitcnt_depctr depctr_sa_sdst(0)
+; GFX1164-DPP-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1164-DPP-NEXT: s_cbranch_execz .LBB17_3
; GFX1164-DPP-NEXT: ; %bb.1:
; GFX1164-DPP-NEXT: s_load_b64 s[0:1], s[34:35], 0x24
@@ -12277,9 +12346,10 @@ define amdgpu_kernel void @global_atomic_fadd_double_uni_address_div_value_defau
; GFX1164-DPP-NEXT: v_cmp_eq_u64_e32 vcc, v[6:7], v[8:9]
; GFX1164-DPP-NEXT: v_mov_b32_e32 v9, v7
; GFX1164-DPP-NEXT: v_mov_b32_e32 v8, v6
+; GFX1164-DPP-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1164-DPP-NEXT: s_or_b64 s[2:3], vcc, s[2:3]
-; GFX1164-DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164-DPP-NEXT: s_and_not1_b64 exec, exec, s[2:3]
+; GFX1164-DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164-DPP-NEXT: s_cbranch_execnz .LBB17_2
; GFX1164-DPP-NEXT: .LBB17_3:
; GFX1164-DPP-NEXT: s_endpgm
@@ -12347,6 +12417,7 @@ define amdgpu_kernel void @global_atomic_fadd_double_uni_address_div_value_defau
; GFX1132-DPP-NEXT: s_mov_b32 s2, 0
; GFX1132-DPP-NEXT: s_mov_b32 s0, exec_lo
; GFX1132-DPP-NEXT: v_cmpx_eq_u32_e32 0, v6
+; GFX1132-DPP-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1132-DPP-NEXT: s_cbranch_execz .LBB17_3
; GFX1132-DPP-NEXT: ; %bb.1:
; GFX1132-DPP-NEXT: s_load_b64 s[0:1], s[34:35], 0x24
@@ -12361,7 +12432,7 @@ define amdgpu_kernel void @global_atomic_fadd_double_uni_address_div_value_defau
; GFX1132-DPP-NEXT: v_cmp_eq_u64_e32 vcc_lo, v[6:7], v[8:9]
; GFX1132-DPP-NEXT: v_dual_mov_b32 v9, v7 :: v_dual_mov_b32 v8, v6
; GFX1132-DPP-NEXT: s_or_b32 s2, vcc_lo, s2
-; GFX1132-DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1132-DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1132-DPP-NEXT: s_and_not1_b32 exec_lo, exec_lo, s2
; GFX1132-DPP-NEXT: s_cbranch_execnz .LBB17_2
; GFX1132-DPP-NEXT: .LBB17_3:
@@ -12511,6 +12582,7 @@ define amdgpu_kernel void @global_atomic_fadd_uni_address_uni_value_system_scope
; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164-NEXT: v_mbcnt_hi_u32_b32 v0, s1, v0
; GFX1164-NEXT: v_cmpx_eq_u32_e32 0, v0
+; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1164-NEXT: s_cbranch_execz .LBB18_2
; GFX1164-NEXT: ; %bb.1:
; GFX1164-NEXT: s_load_b64 s[2:3], s[4:5], 0x24
@@ -12529,7 +12601,7 @@ define amdgpu_kernel void @global_atomic_fadd_uni_address_uni_value_system_scope
; GFX1132-NEXT: s_mov_b32 s0, exec_lo
; GFX1132-NEXT: s_mov_b32 s1, exec_lo
; GFX1132-NEXT: v_mbcnt_lo_u32_b32 v0, s0, 0
-; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX1132-NEXT: s_cbranch_execz .LBB18_2
; GFX1132-NEXT: ; %bb.1:
@@ -12682,6 +12754,7 @@ define amdgpu_kernel void @global_atomic_fadd_uni_address_uni_value_system_scope
; GFX1164-DPP-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164-DPP-NEXT: v_mbcnt_hi_u32_b32 v0, s1, v0
; GFX1164-DPP-NEXT: v_cmpx_eq_u32_e32 0, v0
+; GFX1164-DPP-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1164-DPP-NEXT: s_cbranch_execz .LBB18_2
; GFX1164-DPP-NEXT: ; %bb.1:
; GFX1164-DPP-NEXT: s_load_b64 s[2:3], s[4:5], 0x24
@@ -12700,7 +12773,7 @@ define amdgpu_kernel void @global_atomic_fadd_uni_address_uni_value_system_scope
; GFX1132-DPP-NEXT: s_mov_b32 s0, exec_lo
; GFX1132-DPP-NEXT: s_mov_b32 s1, exec_lo
; GFX1132-DPP-NEXT: v_mbcnt_lo_u32_b32 v0, s0, 0
-; GFX1132-DPP-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1132-DPP-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132-DPP-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX1132-DPP-NEXT: s_cbranch_execz .LBB18_2
; GFX1132-DPP-NEXT: ; %bb.1:
@@ -12857,6 +12930,7 @@ define amdgpu_kernel void @global_atomic_fadd_uni_address_uni_value_system_scope
; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164-NEXT: v_mbcnt_hi_u32_b32 v0, s1, v0
; GFX1164-NEXT: v_cmpx_eq_u32_e32 0, v0
+; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1164-NEXT: s_cbranch_execz .LBB19_2
; GFX1164-NEXT: ; %bb.1:
; GFX1164-NEXT: s_load_b64 s[2:3], s[4:5], 0x24
@@ -12875,7 +12949,7 @@ define amdgpu_kernel void @global_atomic_fadd_uni_address_uni_value_system_scope
; GFX1132-NEXT: s_mov_b32 s0, exec_lo
; GFX1132-NEXT: s_mov_b32 s1, exec_lo
; GFX1132-NEXT: v_mbcnt_lo_u32_b32 v0, s0, 0
-; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX1132-NEXT: s_cbranch_execz .LBB19_2
; GFX1132-NEXT: ; %bb.1:
@@ -13028,6 +13102,7 @@ define amdgpu_kernel void @global_atomic_fadd_uni_address_uni_value_system_scope
; GFX1164-DPP-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164-DPP-NEXT: v_mbcnt_hi_u32_b32 v0, s1, v0
; GFX1164-DPP-NEXT: v_cmpx_eq_u32_e32 0, v0
+; GFX1164-DPP-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1164-DPP-NEXT: s_cbranch_execz .LBB19_2
; GFX1164-DPP-NEXT: ; %bb.1:
; GFX1164-DPP-NEXT: s_load_b64 s[2:3], s[4:5], 0x24
@@ -13046,7 +13121,7 @@ define amdgpu_kernel void @global_atomic_fadd_uni_address_uni_value_system_scope
; GFX1132-DPP-NEXT: s_mov_b32 s0, exec_lo
; GFX1132-DPP-NEXT: s_mov_b32 s1, exec_lo
; GFX1132-DPP-NEXT: v_mbcnt_lo_u32_b32 v0, s0, 0
-; GFX1132-DPP-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1132-DPP-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132-DPP-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX1132-DPP-NEXT: s_cbranch_execz .LBB19_2
; GFX1132-DPP-NEXT: ; %bb.1:
diff --git a/llvm/test/CodeGen/AMDGPU/global_atomics_scan_fmax.ll b/llvm/test/CodeGen/AMDGPU/global_atomics_scan_fmax.ll
index c1e5a18ec6aa29..763bc7afd423ce 100644
--- a/llvm/test/CodeGen/AMDGPU/global_atomics_scan_fmax.ll
+++ b/llvm/test/CodeGen/AMDGPU/global_atomics_scan_fmax.ll
@@ -100,6 +100,7 @@ define amdgpu_kernel void @global_atomic_fmax_uni_address_uni_value_agent_scope_
; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
; GFX1164-NEXT: v_cmpx_eq_u32_e32 0, v0
+; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1164-NEXT: s_cbranch_execz .LBB0_2
; GFX1164-NEXT: ; %bb.1:
; GFX1164-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
@@ -114,7 +115,7 @@ define amdgpu_kernel void @global_atomic_fmax_uni_address_uni_value_agent_scope_
; GFX1132: ; %bb.0:
; GFX1132-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
; GFX1132-NEXT: s_mov_b32 s0, exec_lo
-; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX1132-NEXT: s_cbranch_execz .LBB0_2
; GFX1132-NEXT: ; %bb.1:
@@ -209,6 +210,7 @@ define amdgpu_kernel void @global_atomic_fmax_uni_address_uni_value_agent_scope_
; GFX1164-DPP-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164-DPP-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
; GFX1164-DPP-NEXT: v_cmpx_eq_u32_e32 0, v0
+; GFX1164-DPP-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1164-DPP-NEXT: s_cbranch_execz .LBB0_2
; GFX1164-DPP-NEXT: ; %bb.1:
; GFX1164-DPP-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
@@ -223,7 +225,7 @@ define amdgpu_kernel void @global_atomic_fmax_uni_address_uni_value_agent_scope_
; GFX1132-DPP: ; %bb.0:
; GFX1132-DPP-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
; GFX1132-DPP-NEXT: s_mov_b32 s0, exec_lo
-; GFX1132-DPP-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1132-DPP-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132-DPP-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX1132-DPP-NEXT: s_cbranch_execz .LBB0_2
; GFX1132-DPP-NEXT: ; %bb.1:
@@ -524,6 +526,7 @@ define amdgpu_kernel void @global_atomic_fmax_uni_address_div_value_agent_scope_
; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
; GFX1164-NEXT: v_cmpx_eq_u32_e32 0, v0
+; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1164-NEXT: s_xor_b64 s[0:1], exec, s[0:1]
; GFX1164-NEXT: s_cbranch_execz .LBB1_4
; GFX1164-NEXT: ; %bb.3:
@@ -572,9 +575,10 @@ define amdgpu_kernel void @global_atomic_fmax_uni_address_div_value_agent_scope_
; GFX1132-NEXT: ; %bb.2: ; %ComputeEnd
; GFX1132-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
; GFX1132-NEXT: s_mov_b32 s0, exec_lo
-; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX1132-NEXT: s_xor_b32 s0, exec_lo, s0
+; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1132-NEXT: s_cbranch_execz .LBB1_4
; GFX1132-NEXT: ; %bb.3:
; GFX1132-NEXT: s_load_b64 s[0:1], s[34:35], 0x24
@@ -916,6 +920,7 @@ define amdgpu_kernel void @global_atomic_fmax_uni_address_div_value_agent_scope_
; GFX1164-DPP-NEXT: v_mov_b32_e32 v0, v1
; GFX1164-DPP-NEXT: s_mov_b64 s[0:1], exec
; GFX1164-DPP-NEXT: v_cmpx_eq_u32_e32 0, v4
+; GFX1164-DPP-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1164-DPP-NEXT: s_cbranch_execz .LBB1_2
; GFX1164-DPP-NEXT: ; %bb.1:
; GFX1164-DPP-NEXT: s_load_b64 s[0:1], s[34:35], 0x24
@@ -978,7 +983,7 @@ define amdgpu_kernel void @global_atomic_fmax_uni_address_div_value_agent_scope_
; GFX1132-DPP-NEXT: v_mbcnt_lo_u32_b32 v4, exec_lo, 0
; GFX1132-DPP-NEXT: v_mov_b32_e32 v0, v1
; GFX1132-DPP-NEXT: s_mov_b32 s0, exec_lo
-; GFX1132-DPP-NEXT: s_delay_alu instid0(VALU_DEP_2)
+; GFX1132-DPP-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132-DPP-NEXT: v_cmpx_eq_u32_e32 0, v4
; GFX1132-DPP-NEXT: s_cbranch_execz .LBB1_2
; GFX1132-DPP-NEXT: ; %bb.1:
@@ -1078,6 +1083,7 @@ define amdgpu_kernel void @global_atomic_fmax_uni_address_uni_value_one_as_scope
; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
; GFX1164-NEXT: v_cmpx_eq_u32_e32 0, v0
+; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1164-NEXT: s_cbranch_execz .LBB2_2
; GFX1164-NEXT: ; %bb.1:
; GFX1164-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
@@ -1092,7 +1098,7 @@ define amdgpu_kernel void @global_atomic_fmax_uni_address_uni_value_one_as_scope
; GFX1132: ; %bb.0:
; GFX1132-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
; GFX1132-NEXT: s_mov_b32 s0, exec_lo
-; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX1132-NEXT: s_cbranch_execz .LBB2_2
; GFX1132-NEXT: ; %bb.1:
@@ -1187,6 +1193,7 @@ define amdgpu_kernel void @global_atomic_fmax_uni_address_uni_value_one_as_scope
; GFX1164-DPP-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164-DPP-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
; GFX1164-DPP-NEXT: v_cmpx_eq_u32_e32 0, v0
+; GFX1164-DPP-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1164-DPP-NEXT: s_cbranch_execz .LBB2_2
; GFX1164-DPP-NEXT: ; %bb.1:
; GFX1164-DPP-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
@@ -1201,7 +1208,7 @@ define amdgpu_kernel void @global_atomic_fmax_uni_address_uni_value_one_as_scope
; GFX1132-DPP: ; %bb.0:
; GFX1132-DPP-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
; GFX1132-DPP-NEXT: s_mov_b32 s0, exec_lo
-; GFX1132-DPP-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1132-DPP-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132-DPP-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX1132-DPP-NEXT: s_cbranch_execz .LBB2_2
; GFX1132-DPP-NEXT: ; %bb.1:
@@ -1503,6 +1510,7 @@ define amdgpu_kernel void @global_atomic_fmax_uni_address_div_value_one_as_scope
; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
; GFX1164-NEXT: v_cmpx_eq_u32_e32 0, v0
+; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1164-NEXT: s_xor_b64 s[0:1], exec, s[0:1]
; GFX1164-NEXT: s_cbranch_execz .LBB3_4
; GFX1164-NEXT: ; %bb.3:
@@ -1551,9 +1559,10 @@ define amdgpu_kernel void @global_atomic_fmax_uni_address_div_value_one_as_scope
; GFX1132-NEXT: ; %bb.2: ; %ComputeEnd
; GFX1132-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
; GFX1132-NEXT: s_mov_b32 s0, exec_lo
-; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX1132-NEXT: s_xor_b32 s0, exec_lo, s0
+; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1132-NEXT: s_cbranch_execz .LBB3_4
; GFX1132-NEXT: ; %bb.3:
; GFX1132-NEXT: s_load_b64 s[0:1], s[34:35], 0x24
@@ -1895,6 +1904,7 @@ define amdgpu_kernel void @global_atomic_fmax_uni_address_div_value_one_as_scope
; GFX1164-DPP-NEXT: v_mov_b32_e32 v0, v1
; GFX1164-DPP-NEXT: s_mov_b64 s[0:1], exec
; GFX1164-DPP-NEXT: v_cmpx_eq_u32_e32 0, v4
+; GFX1164-DPP-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1164-DPP-NEXT: s_cbranch_execz .LBB3_2
; GFX1164-DPP-NEXT: ; %bb.1:
; GFX1164-DPP-NEXT: s_load_b64 s[0:1], s[34:35], 0x24
@@ -1957,7 +1967,7 @@ define amdgpu_kernel void @global_atomic_fmax_uni_address_div_value_one_as_scope
; GFX1132-DPP-NEXT: v_mbcnt_lo_u32_b32 v4, exec_lo, 0
; GFX1132-DPP-NEXT: v_mov_b32_e32 v0, v1
; GFX1132-DPP-NEXT: s_mov_b32 s0, exec_lo
-; GFX1132-DPP-NEXT: s_delay_alu instid0(VALU_DEP_2)
+; GFX1132-DPP-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132-DPP-NEXT: v_cmpx_eq_u32_e32 0, v4
; GFX1132-DPP-NEXT: s_cbranch_execz .LBB3_2
; GFX1132-DPP-NEXT: ; %bb.1:
@@ -2058,6 +2068,7 @@ define amdgpu_kernel void @global_atomic_fmax_uni_address_uni_value_default_scop
; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
; GFX1164-NEXT: v_cmpx_eq_u32_e32 0, v0
+; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1164-NEXT: s_cbranch_execz .LBB4_2
; GFX1164-NEXT: ; %bb.1:
; GFX1164-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
@@ -2072,7 +2083,7 @@ define amdgpu_kernel void @global_atomic_fmax_uni_address_uni_value_default_scop
; GFX1132: ; %bb.0:
; GFX1132-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
; GFX1132-NEXT: s_mov_b32 s0, exec_lo
-; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX1132-NEXT: s_cbranch_execz .LBB4_2
; GFX1132-NEXT: ; %bb.1:
@@ -2167,6 +2178,7 @@ define amdgpu_kernel void @global_atomic_fmax_uni_address_uni_value_default_scop
; GFX1164-DPP-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164-DPP-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
; GFX1164-DPP-NEXT: v_cmpx_eq_u32_e32 0, v0
+; GFX1164-DPP-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1164-DPP-NEXT: s_cbranch_execz .LBB4_2
; GFX1164-DPP-NEXT: ; %bb.1:
; GFX1164-DPP-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
@@ -2181,7 +2193,7 @@ define amdgpu_kernel void @global_atomic_fmax_uni_address_uni_value_default_scop
; GFX1132-DPP: ; %bb.0:
; GFX1132-DPP-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
; GFX1132-DPP-NEXT: s_mov_b32 s0, exec_lo
-; GFX1132-DPP-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1132-DPP-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132-DPP-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX1132-DPP-NEXT: s_cbranch_execz .LBB4_2
; GFX1132-DPP-NEXT: ; %bb.1:
@@ -2482,6 +2494,7 @@ define amdgpu_kernel void @global_atomic_fmax_uni_address_div_value_default_scop
; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
; GFX1164-NEXT: v_cmpx_eq_u32_e32 0, v0
+; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1164-NEXT: s_xor_b64 s[0:1], exec, s[0:1]
; GFX1164-NEXT: s_cbranch_execz .LBB5_4
; GFX1164-NEXT: ; %bb.3:
@@ -2530,9 +2543,10 @@ define amdgpu_kernel void @global_atomic_fmax_uni_address_div_value_default_scop
; GFX1132-NEXT: ; %bb.2: ; %ComputeEnd
; GFX1132-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
; GFX1132-NEXT: s_mov_b32 s0, exec_lo
-; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX1132-NEXT: s_xor_b32 s0, exec_lo, s0
+; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1132-NEXT: s_cbranch_execz .LBB5_4
; GFX1132-NEXT: ; %bb.3:
; GFX1132-NEXT: s_load_b64 s[0:1], s[34:35], 0x24
@@ -2874,6 +2888,7 @@ define amdgpu_kernel void @global_atomic_fmax_uni_address_div_value_default_scop
; GFX1164-DPP-NEXT: v_mov_b32_e32 v0, v1
; GFX1164-DPP-NEXT: s_mov_b64 s[0:1], exec
; GFX1164-DPP-NEXT: v_cmpx_eq_u32_e32 0, v4
+; GFX1164-DPP-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1164-DPP-NEXT: s_cbranch_execz .LBB5_2
; GFX1164-DPP-NEXT: ; %bb.1:
; GFX1164-DPP-NEXT: s_load_b64 s[0:1], s[34:35], 0x24
@@ -2936,7 +2951,7 @@ define amdgpu_kernel void @global_atomic_fmax_uni_address_div_value_default_scop
; GFX1132-DPP-NEXT: v_mbcnt_lo_u32_b32 v4, exec_lo, 0
; GFX1132-DPP-NEXT: v_mov_b32_e32 v0, v1
; GFX1132-DPP-NEXT: s_mov_b32 s0, exec_lo
-; GFX1132-DPP-NEXT: s_delay_alu instid0(VALU_DEP_2)
+; GFX1132-DPP-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132-DPP-NEXT: v_cmpx_eq_u32_e32 0, v4
; GFX1132-DPP-NEXT: s_cbranch_execz .LBB5_2
; GFX1132-DPP-NEXT: ; %bb.1:
@@ -3041,6 +3056,7 @@ define amdgpu_kernel void @global_atomic_fmax_double_uni_address_uni_value_agent
; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
; GFX1164-NEXT: v_cmpx_eq_u32_e32 0, v0
+; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1164-NEXT: s_cbranch_execz .LBB6_3
; GFX1164-NEXT: ; %bb.1:
; GFX1164-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
@@ -3061,9 +3077,10 @@ define amdgpu_kernel void @global_atomic_fmax_double_uni_address_uni_value_agent
; GFX1164-NEXT: v_cmp_eq_u64_e32 vcc, v[0:1], v[2:3]
; GFX1164-NEXT: v_mov_b32_e32 v3, v1
; GFX1164-NEXT: v_mov_b32_e32 v2, v0
+; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1164-NEXT: s_or_b64 s[2:3], vcc, s[2:3]
-; GFX1164-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164-NEXT: s_and_not1_b64 exec, exec, s[2:3]
+; GFX1164-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164-NEXT: s_cbranch_execnz .LBB6_2
; GFX1164-NEXT: .LBB6_3:
; GFX1164-NEXT: s_endpgm
@@ -3073,7 +3090,7 @@ define amdgpu_kernel void @global_atomic_fmax_double_uni_address_uni_value_agent
; GFX1132-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
; GFX1132-NEXT: s_mov_b32 s2, 0
; GFX1132-NEXT: s_mov_b32 s0, exec_lo
-; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX1132-NEXT: s_cbranch_execz .LBB6_3
; GFX1132-NEXT: ; %bb.1:
@@ -3093,7 +3110,7 @@ define amdgpu_kernel void @global_atomic_fmax_double_uni_address_uni_value_agent
; GFX1132-NEXT: v_cmp_eq_u64_e32 vcc_lo, v[0:1], v[2:3]
; GFX1132-NEXT: v_dual_mov_b32 v3, v1 :: v_dual_mov_b32 v2, v0
; GFX1132-NEXT: s_or_b32 s2, vcc_lo, s2
-; GFX1132-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1132-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1132-NEXT: s_and_not1_b32 exec_lo, exec_lo, s2
; GFX1132-NEXT: s_cbranch_execnz .LBB6_2
; GFX1132-NEXT: .LBB6_3:
@@ -3188,6 +3205,7 @@ define amdgpu_kernel void @global_atomic_fmax_double_uni_address_uni_value_agent
; GFX1164-DPP-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164-DPP-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
; GFX1164-DPP-NEXT: v_cmpx_eq_u32_e32 0, v0
+; GFX1164-DPP-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1164-DPP-NEXT: s_cbranch_execz .LBB6_3
; GFX1164-DPP-NEXT: ; %bb.1:
; GFX1164-DPP-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
@@ -3208,9 +3226,10 @@ define amdgpu_kernel void @global_atomic_fmax_double_uni_address_uni_value_agent
; GFX1164-DPP-NEXT: v_cmp_eq_u64_e32 vcc, v[0:1], v[2:3]
; GFX1164-DPP-NEXT: v_mov_b32_e32 v3, v1
; GFX1164-DPP-NEXT: v_mov_b32_e32 v2, v0
+; GFX1164-DPP-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1164-DPP-NEXT: s_or_b64 s[2:3], vcc, s[2:3]
-; GFX1164-DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164-DPP-NEXT: s_and_not1_b64 exec, exec, s[2:3]
+; GFX1164-DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164-DPP-NEXT: s_cbranch_execnz .LBB6_2
; GFX1164-DPP-NEXT: .LBB6_3:
; GFX1164-DPP-NEXT: s_endpgm
@@ -3220,7 +3239,7 @@ define amdgpu_kernel void @global_atomic_fmax_double_uni_address_uni_value_agent
; GFX1132-DPP-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
; GFX1132-DPP-NEXT: s_mov_b32 s2, 0
; GFX1132-DPP-NEXT: s_mov_b32 s0, exec_lo
-; GFX1132-DPP-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1132-DPP-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132-DPP-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX1132-DPP-NEXT: s_cbranch_execz .LBB6_3
; GFX1132-DPP-NEXT: ; %bb.1:
@@ -3240,7 +3259,7 @@ define amdgpu_kernel void @global_atomic_fmax_double_uni_address_uni_value_agent
; GFX1132-DPP-NEXT: v_cmp_eq_u64_e32 vcc_lo, v[0:1], v[2:3]
; GFX1132-DPP-NEXT: v_dual_mov_b32 v3, v1 :: v_dual_mov_b32 v2, v0
; GFX1132-DPP-NEXT: s_or_b32 s2, vcc_lo, s2
-; GFX1132-DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1132-DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1132-DPP-NEXT: s_and_not1_b32 exec_lo, exec_lo, s2
; GFX1132-DPP-NEXT: s_cbranch_execnz .LBB6_2
; GFX1132-DPP-NEXT: .LBB6_3:
@@ -3547,6 +3566,7 @@ define amdgpu_kernel void @global_atomic_fmax_double_uni_address_div_value_agent
; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
; GFX1164-NEXT: v_cmpx_eq_u32_e32 0, v0
+; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1164-NEXT: s_xor_b64 s[0:1], exec, s[0:1]
; GFX1164-NEXT: s_cbranch_execz .LBB7_5
; GFX1164-NEXT: ; %bb.3:
@@ -3567,9 +3587,10 @@ define amdgpu_kernel void @global_atomic_fmax_double_uni_address_div_value_agent
; GFX1164-NEXT: v_cmp_eq_u64_e32 vcc, v[0:1], v[2:3]
; GFX1164-NEXT: v_mov_b32_e32 v3, v1
; GFX1164-NEXT: v_mov_b32_e32 v2, v0
+; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1164-NEXT: s_or_b64 s[2:3], vcc, s[2:3]
-; GFX1164-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164-NEXT: s_and_not1_b64 exec, exec, s[2:3]
+; GFX1164-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164-NEXT: s_cbranch_execnz .LBB7_4
; GFX1164-NEXT: .LBB7_5:
; GFX1164-NEXT: s_endpgm
@@ -3614,9 +3635,10 @@ define amdgpu_kernel void @global_atomic_fmax_double_uni_address_div_value_agent
; GFX1132-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
; GFX1132-NEXT: s_mov_b32 s2, 0
; GFX1132-NEXT: s_mov_b32 s0, exec_lo
-; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX1132-NEXT: s_xor_b32 s0, exec_lo, s0
+; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1132-NEXT: s_cbranch_execz .LBB7_5
; GFX1132-NEXT: ; %bb.3:
; GFX1132-NEXT: s_load_b64 s[0:1], s[34:35], 0x24
@@ -3636,7 +3658,7 @@ define amdgpu_kernel void @global_atomic_fmax_double_uni_address_div_value_agent
; GFX1132-NEXT: v_cmp_eq_u64_e32 vcc_lo, v[0:1], v[2:3]
; GFX1132-NEXT: v_dual_mov_b32 v3, v1 :: v_dual_mov_b32 v2, v0
; GFX1132-NEXT: s_or_b32 s2, vcc_lo, s2
-; GFX1132-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1132-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1132-NEXT: s_and_not1_b32 exec_lo, exec_lo, s2
; GFX1132-NEXT: s_cbranch_execnz .LBB7_4
; GFX1132-NEXT: .LBB7_5:
@@ -4027,6 +4049,7 @@ define amdgpu_kernel void @global_atomic_fmax_double_uni_address_div_value_agent
; GFX1164-DPP-NEXT: v_mov_b32_e32 v0, v2
; GFX1164-DPP-NEXT: v_cmpx_eq_u32_e32 0, v6
; GFX1164-DPP-NEXT: s_waitcnt_depctr depctr_sa_sdst(0)
+; GFX1164-DPP-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1164-DPP-NEXT: s_cbranch_execz .LBB7_3
; GFX1164-DPP-NEXT: ; %bb.1:
; GFX1164-DPP-NEXT: s_load_b64 s[0:1], s[34:35], 0x24
@@ -4045,9 +4068,10 @@ define amdgpu_kernel void @global_atomic_fmax_double_uni_address_div_value_agent
; GFX1164-DPP-NEXT: v_cmp_eq_u64_e32 vcc, v[6:7], v[8:9]
; GFX1164-DPP-NEXT: v_mov_b32_e32 v9, v7
; GFX1164-DPP-NEXT: v_mov_b32_e32 v8, v6
+; GFX1164-DPP-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1164-DPP-NEXT: s_or_b64 s[2:3], vcc, s[2:3]
-; GFX1164-DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164-DPP-NEXT: s_and_not1_b64 exec, exec, s[2:3]
+; GFX1164-DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164-DPP-NEXT: s_cbranch_execnz .LBB7_2
; GFX1164-DPP-NEXT: .LBB7_3:
; GFX1164-DPP-NEXT: s_endpgm
@@ -4115,7 +4139,7 @@ define amdgpu_kernel void @global_atomic_fmax_double_uni_address_div_value_agent
; GFX1132-DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1132-DPP-NEXT: v_mbcnt_lo_u32_b32 v6, exec_lo, 0
; GFX1132-DPP-NEXT: v_dual_mov_b32 v10, 0 :: v_dual_mov_b32 v1, v3
-; GFX1132-DPP-NEXT: s_delay_alu instid0(VALU_DEP_3)
+; GFX1132-DPP-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_3) | instid1(VALU_DEP_1)
; GFX1132-DPP-NEXT: v_mov_b32_e32 v0, v2
; GFX1132-DPP-NEXT: s_mov_b32 s2, 0
; GFX1132-DPP-NEXT: s_mov_b32 s0, exec_lo
@@ -4137,7 +4161,7 @@ define amdgpu_kernel void @global_atomic_fmax_double_uni_address_div_value_agent
; GFX1132-DPP-NEXT: v_cmp_eq_u64_e32 vcc_lo, v[6:7], v[8:9]
; GFX1132-DPP-NEXT: v_dual_mov_b32 v9, v7 :: v_dual_mov_b32 v8, v6
; GFX1132-DPP-NEXT: s_or_b32 s2, vcc_lo, s2
-; GFX1132-DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1132-DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1132-DPP-NEXT: s_and_not1_b32 exec_lo, exec_lo, s2
; GFX1132-DPP-NEXT: s_cbranch_execnz .LBB7_2
; GFX1132-DPP-NEXT: .LBB7_3:
@@ -4237,6 +4261,7 @@ define amdgpu_kernel void @global_atomic_fmax_double_uni_address_uni_value_one_a
; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
; GFX1164-NEXT: v_cmpx_eq_u32_e32 0, v0
+; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1164-NEXT: s_cbranch_execz .LBB8_3
; GFX1164-NEXT: ; %bb.1:
; GFX1164-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
@@ -4257,9 +4282,10 @@ define amdgpu_kernel void @global_atomic_fmax_double_uni_address_uni_value_one_a
; GFX1164-NEXT: v_cmp_eq_u64_e32 vcc, v[0:1], v[2:3]
; GFX1164-NEXT: v_mov_b32_e32 v3, v1
; GFX1164-NEXT: v_mov_b32_e32 v2, v0
+; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1164-NEXT: s_or_b64 s[2:3], vcc, s[2:3]
-; GFX1164-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164-NEXT: s_and_not1_b64 exec, exec, s[2:3]
+; GFX1164-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164-NEXT: s_cbranch_execnz .LBB8_2
; GFX1164-NEXT: .LBB8_3:
; GFX1164-NEXT: s_endpgm
@@ -4269,7 +4295,7 @@ define amdgpu_kernel void @global_atomic_fmax_double_uni_address_uni_value_one_a
; GFX1132-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
; GFX1132-NEXT: s_mov_b32 s2, 0
; GFX1132-NEXT: s_mov_b32 s0, exec_lo
-; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX1132-NEXT: s_cbranch_execz .LBB8_3
; GFX1132-NEXT: ; %bb.1:
@@ -4289,7 +4315,7 @@ define amdgpu_kernel void @global_atomic_fmax_double_uni_address_uni_value_one_a
; GFX1132-NEXT: v_cmp_eq_u64_e32 vcc_lo, v[0:1], v[2:3]
; GFX1132-NEXT: v_dual_mov_b32 v3, v1 :: v_dual_mov_b32 v2, v0
; GFX1132-NEXT: s_or_b32 s2, vcc_lo, s2
-; GFX1132-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1132-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1132-NEXT: s_and_not1_b32 exec_lo, exec_lo, s2
; GFX1132-NEXT: s_cbranch_execnz .LBB8_2
; GFX1132-NEXT: .LBB8_3:
@@ -4384,6 +4410,7 @@ define amdgpu_kernel void @global_atomic_fmax_double_uni_address_uni_value_one_a
; GFX1164-DPP-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164-DPP-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
; GFX1164-DPP-NEXT: v_cmpx_eq_u32_e32 0, v0
+; GFX1164-DPP-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1164-DPP-NEXT: s_cbranch_execz .LBB8_3
; GFX1164-DPP-NEXT: ; %bb.1:
; GFX1164-DPP-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
@@ -4404,9 +4431,10 @@ define amdgpu_kernel void @global_atomic_fmax_double_uni_address_uni_value_one_a
; GFX1164-DPP-NEXT: v_cmp_eq_u64_e32 vcc, v[0:1], v[2:3]
; GFX1164-DPP-NEXT: v_mov_b32_e32 v3, v1
; GFX1164-DPP-NEXT: v_mov_b32_e32 v2, v0
+; GFX1164-DPP-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1164-DPP-NEXT: s_or_b64 s[2:3], vcc, s[2:3]
-; GFX1164-DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164-DPP-NEXT: s_and_not1_b64 exec, exec, s[2:3]
+; GFX1164-DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164-DPP-NEXT: s_cbranch_execnz .LBB8_2
; GFX1164-DPP-NEXT: .LBB8_3:
; GFX1164-DPP-NEXT: s_endpgm
@@ -4416,7 +4444,7 @@ define amdgpu_kernel void @global_atomic_fmax_double_uni_address_uni_value_one_a
; GFX1132-DPP-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
; GFX1132-DPP-NEXT: s_mov_b32 s2, 0
; GFX1132-DPP-NEXT: s_mov_b32 s0, exec_lo
-; GFX1132-DPP-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1132-DPP-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132-DPP-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX1132-DPP-NEXT: s_cbranch_execz .LBB8_3
; GFX1132-DPP-NEXT: ; %bb.1:
@@ -4436,7 +4464,7 @@ define amdgpu_kernel void @global_atomic_fmax_double_uni_address_uni_value_one_a
; GFX1132-DPP-NEXT: v_cmp_eq_u64_e32 vcc_lo, v[0:1], v[2:3]
; GFX1132-DPP-NEXT: v_dual_mov_b32 v3, v1 :: v_dual_mov_b32 v2, v0
; GFX1132-DPP-NEXT: s_or_b32 s2, vcc_lo, s2
-; GFX1132-DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1132-DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1132-DPP-NEXT: s_and_not1_b32 exec_lo, exec_lo, s2
; GFX1132-DPP-NEXT: s_cbranch_execnz .LBB8_2
; GFX1132-DPP-NEXT: .LBB8_3:
@@ -4743,6 +4771,7 @@ define amdgpu_kernel void @global_atomic_fmax_double_uni_address_div_value_one_a
; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
; GFX1164-NEXT: v_cmpx_eq_u32_e32 0, v0
+; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1164-NEXT: s_xor_b64 s[0:1], exec, s[0:1]
; GFX1164-NEXT: s_cbranch_execz .LBB9_5
; GFX1164-NEXT: ; %bb.3:
@@ -4763,9 +4792,10 @@ define amdgpu_kernel void @global_atomic_fmax_double_uni_address_div_value_one_a
; GFX1164-NEXT: v_cmp_eq_u64_e32 vcc, v[0:1], v[2:3]
; GFX1164-NEXT: v_mov_b32_e32 v3, v1
; GFX1164-NEXT: v_mov_b32_e32 v2, v0
+; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1164-NEXT: s_or_b64 s[2:3], vcc, s[2:3]
-; GFX1164-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164-NEXT: s_and_not1_b64 exec, exec, s[2:3]
+; GFX1164-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164-NEXT: s_cbranch_execnz .LBB9_4
; GFX1164-NEXT: .LBB9_5:
; GFX1164-NEXT: s_endpgm
@@ -4810,9 +4840,10 @@ define amdgpu_kernel void @global_atomic_fmax_double_uni_address_div_value_one_a
; GFX1132-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
; GFX1132-NEXT: s_mov_b32 s2, 0
; GFX1132-NEXT: s_mov_b32 s0, exec_lo
-; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX1132-NEXT: s_xor_b32 s0, exec_lo, s0
+; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1132-NEXT: s_cbranch_execz .LBB9_5
; GFX1132-NEXT: ; %bb.3:
; GFX1132-NEXT: s_load_b64 s[0:1], s[34:35], 0x24
@@ -4832,7 +4863,7 @@ define amdgpu_kernel void @global_atomic_fmax_double_uni_address_div_value_one_a
; GFX1132-NEXT: v_cmp_eq_u64_e32 vcc_lo, v[0:1], v[2:3]
; GFX1132-NEXT: v_dual_mov_b32 v3, v1 :: v_dual_mov_b32 v2, v0
; GFX1132-NEXT: s_or_b32 s2, vcc_lo, s2
-; GFX1132-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1132-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1132-NEXT: s_and_not1_b32 exec_lo, exec_lo, s2
; GFX1132-NEXT: s_cbranch_execnz .LBB9_4
; GFX1132-NEXT: .LBB9_5:
@@ -5223,6 +5254,7 @@ define amdgpu_kernel void @global_atomic_fmax_double_uni_address_div_value_one_a
; GFX1164-DPP-NEXT: v_mov_b32_e32 v0, v2
; GFX1164-DPP-NEXT: v_cmpx_eq_u32_e32 0, v6
; GFX1164-DPP-NEXT: s_waitcnt_depctr depctr_sa_sdst(0)
+; GFX1164-DPP-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1164-DPP-NEXT: s_cbranch_execz .LBB9_3
; GFX1164-DPP-NEXT: ; %bb.1:
; GFX1164-DPP-NEXT: s_load_b64 s[0:1], s[34:35], 0x24
@@ -5241,9 +5273,10 @@ define amdgpu_kernel void @global_atomic_fmax_double_uni_address_div_value_one_a
; GFX1164-DPP-NEXT: v_cmp_eq_u64_e32 vcc, v[6:7], v[8:9]
; GFX1164-DPP-NEXT: v_mov_b32_e32 v9, v7
; GFX1164-DPP-NEXT: v_mov_b32_e32 v8, v6
+; GFX1164-DPP-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1164-DPP-NEXT: s_or_b64 s[2:3], vcc, s[2:3]
-; GFX1164-DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164-DPP-NEXT: s_and_not1_b64 exec, exec, s[2:3]
+; GFX1164-DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164-DPP-NEXT: s_cbranch_execnz .LBB9_2
; GFX1164-DPP-NEXT: .LBB9_3:
; GFX1164-DPP-NEXT: s_endpgm
@@ -5311,7 +5344,7 @@ define amdgpu_kernel void @global_atomic_fmax_double_uni_address_div_value_one_a
; GFX1132-DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1132-DPP-NEXT: v_mbcnt_lo_u32_b32 v6, exec_lo, 0
; GFX1132-DPP-NEXT: v_dual_mov_b32 v10, 0 :: v_dual_mov_b32 v1, v3
-; GFX1132-DPP-NEXT: s_delay_alu instid0(VALU_DEP_3)
+; GFX1132-DPP-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_3) | instid1(VALU_DEP_1)
; GFX1132-DPP-NEXT: v_mov_b32_e32 v0, v2
; GFX1132-DPP-NEXT: s_mov_b32 s2, 0
; GFX1132-DPP-NEXT: s_mov_b32 s0, exec_lo
@@ -5333,7 +5366,7 @@ define amdgpu_kernel void @global_atomic_fmax_double_uni_address_div_value_one_a
; GFX1132-DPP-NEXT: v_cmp_eq_u64_e32 vcc_lo, v[6:7], v[8:9]
; GFX1132-DPP-NEXT: v_dual_mov_b32 v9, v7 :: v_dual_mov_b32 v8, v6
; GFX1132-DPP-NEXT: s_or_b32 s2, vcc_lo, s2
-; GFX1132-DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1132-DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1132-DPP-NEXT: s_and_not1_b32 exec_lo, exec_lo, s2
; GFX1132-DPP-NEXT: s_cbranch_execnz .LBB9_2
; GFX1132-DPP-NEXT: .LBB9_3:
@@ -5433,6 +5466,7 @@ define amdgpu_kernel void @global_atomic_fmax_double_uni_address_uni_value_defau
; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
; GFX1164-NEXT: v_cmpx_eq_u32_e32 0, v0
+; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1164-NEXT: s_cbranch_execz .LBB10_3
; GFX1164-NEXT: ; %bb.1:
; GFX1164-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
@@ -5453,9 +5487,10 @@ define amdgpu_kernel void @global_atomic_fmax_double_uni_address_uni_value_defau
; GFX1164-NEXT: v_cmp_eq_u64_e32 vcc, v[0:1], v[2:3]
; GFX1164-NEXT: v_mov_b32_e32 v3, v1
; GFX1164-NEXT: v_mov_b32_e32 v2, v0
+; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1164-NEXT: s_or_b64 s[2:3], vcc, s[2:3]
-; GFX1164-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164-NEXT: s_and_not1_b64 exec, exec, s[2:3]
+; GFX1164-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164-NEXT: s_cbranch_execnz .LBB10_2
; GFX1164-NEXT: .LBB10_3:
; GFX1164-NEXT: s_endpgm
@@ -5465,7 +5500,7 @@ define amdgpu_kernel void @global_atomic_fmax_double_uni_address_uni_value_defau
; GFX1132-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
; GFX1132-NEXT: s_mov_b32 s2, 0
; GFX1132-NEXT: s_mov_b32 s0, exec_lo
-; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX1132-NEXT: s_cbranch_execz .LBB10_3
; GFX1132-NEXT: ; %bb.1:
@@ -5485,7 +5520,7 @@ define amdgpu_kernel void @global_atomic_fmax_double_uni_address_uni_value_defau
; GFX1132-NEXT: v_cmp_eq_u64_e32 vcc_lo, v[0:1], v[2:3]
; GFX1132-NEXT: v_dual_mov_b32 v3, v1 :: v_dual_mov_b32 v2, v0
; GFX1132-NEXT: s_or_b32 s2, vcc_lo, s2
-; GFX1132-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1132-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1132-NEXT: s_and_not1_b32 exec_lo, exec_lo, s2
; GFX1132-NEXT: s_cbranch_execnz .LBB10_2
; GFX1132-NEXT: .LBB10_3:
@@ -5580,6 +5615,7 @@ define amdgpu_kernel void @global_atomic_fmax_double_uni_address_uni_value_defau
; GFX1164-DPP-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164-DPP-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
; GFX1164-DPP-NEXT: v_cmpx_eq_u32_e32 0, v0
+; GFX1164-DPP-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1164-DPP-NEXT: s_cbranch_execz .LBB10_3
; GFX1164-DPP-NEXT: ; %bb.1:
; GFX1164-DPP-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
@@ -5600,9 +5636,10 @@ define amdgpu_kernel void @global_atomic_fmax_double_uni_address_uni_value_defau
; GFX1164-DPP-NEXT: v_cmp_eq_u64_e32 vcc, v[0:1], v[2:3]
; GFX1164-DPP-NEXT: v_mov_b32_e32 v3, v1
; GFX1164-DPP-NEXT: v_mov_b32_e32 v2, v0
+; GFX1164-DPP-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1164-DPP-NEXT: s_or_b64 s[2:3], vcc, s[2:3]
-; GFX1164-DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164-DPP-NEXT: s_and_not1_b64 exec, exec, s[2:3]
+; GFX1164-DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164-DPP-NEXT: s_cbranch_execnz .LBB10_2
; GFX1164-DPP-NEXT: .LBB10_3:
; GFX1164-DPP-NEXT: s_endpgm
@@ -5612,7 +5649,7 @@ define amdgpu_kernel void @global_atomic_fmax_double_uni_address_uni_value_defau
; GFX1132-DPP-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
; GFX1132-DPP-NEXT: s_mov_b32 s2, 0
; GFX1132-DPP-NEXT: s_mov_b32 s0, exec_lo
-; GFX1132-DPP-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1132-DPP-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132-DPP-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX1132-DPP-NEXT: s_cbranch_execz .LBB10_3
; GFX1132-DPP-NEXT: ; %bb.1:
@@ -5632,7 +5669,7 @@ define amdgpu_kernel void @global_atomic_fmax_double_uni_address_uni_value_defau
; GFX1132-DPP-NEXT: v_cmp_eq_u64_e32 vcc_lo, v[0:1], v[2:3]
; GFX1132-DPP-NEXT: v_dual_mov_b32 v3, v1 :: v_dual_mov_b32 v2, v0
; GFX1132-DPP-NEXT: s_or_b32 s2, vcc_lo, s2
-; GFX1132-DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1132-DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1132-DPP-NEXT: s_and_not1_b32 exec_lo, exec_lo, s2
; GFX1132-DPP-NEXT: s_cbranch_execnz .LBB10_2
; GFX1132-DPP-NEXT: .LBB10_3:
@@ -5939,6 +5976,7 @@ define amdgpu_kernel void @global_atomic_fmax_double_uni_address_div_value_defau
; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
; GFX1164-NEXT: v_cmpx_eq_u32_e32 0, v0
+; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1164-NEXT: s_xor_b64 s[0:1], exec, s[0:1]
; GFX1164-NEXT: s_cbranch_execz .LBB11_5
; GFX1164-NEXT: ; %bb.3:
@@ -5959,9 +5997,10 @@ define amdgpu_kernel void @global_atomic_fmax_double_uni_address_div_value_defau
; GFX1164-NEXT: v_cmp_eq_u64_e32 vcc, v[0:1], v[2:3]
; GFX1164-NEXT: v_mov_b32_e32 v3, v1
; GFX1164-NEXT: v_mov_b32_e32 v2, v0
+; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1164-NEXT: s_or_b64 s[2:3], vcc, s[2:3]
-; GFX1164-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164-NEXT: s_and_not1_b64 exec, exec, s[2:3]
+; GFX1164-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164-NEXT: s_cbranch_execnz .LBB11_4
; GFX1164-NEXT: .LBB11_5:
; GFX1164-NEXT: s_endpgm
@@ -6006,9 +6045,10 @@ define amdgpu_kernel void @global_atomic_fmax_double_uni_address_div_value_defau
; GFX1132-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
; GFX1132-NEXT: s_mov_b32 s2, 0
; GFX1132-NEXT: s_mov_b32 s0, exec_lo
-; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX1132-NEXT: s_xor_b32 s0, exec_lo, s0
+; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1132-NEXT: s_cbranch_execz .LBB11_5
; GFX1132-NEXT: ; %bb.3:
; GFX1132-NEXT: s_load_b64 s[0:1], s[34:35], 0x24
@@ -6028,7 +6068,7 @@ define amdgpu_kernel void @global_atomic_fmax_double_uni_address_div_value_defau
; GFX1132-NEXT: v_cmp_eq_u64_e32 vcc_lo, v[0:1], v[2:3]
; GFX1132-NEXT: v_dual_mov_b32 v3, v1 :: v_dual_mov_b32 v2, v0
; GFX1132-NEXT: s_or_b32 s2, vcc_lo, s2
-; GFX1132-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1132-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1132-NEXT: s_and_not1_b32 exec_lo, exec_lo, s2
; GFX1132-NEXT: s_cbranch_execnz .LBB11_4
; GFX1132-NEXT: .LBB11_5:
@@ -6419,6 +6459,7 @@ define amdgpu_kernel void @global_atomic_fmax_double_uni_address_div_value_defau
; GFX1164-DPP-NEXT: v_mov_b32_e32 v0, v2
; GFX1164-DPP-NEXT: v_cmpx_eq_u32_e32 0, v6
; GFX1164-DPP-NEXT: s_waitcnt_depctr depctr_sa_sdst(0)
+; GFX1164-DPP-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1164-DPP-NEXT: s_cbranch_execz .LBB11_3
; GFX1164-DPP-NEXT: ; %bb.1:
; GFX1164-DPP-NEXT: s_load_b64 s[0:1], s[34:35], 0x24
@@ -6437,9 +6478,10 @@ define amdgpu_kernel void @global_atomic_fmax_double_uni_address_div_value_defau
; GFX1164-DPP-NEXT: v_cmp_eq_u64_e32 vcc, v[6:7], v[8:9]
; GFX1164-DPP-NEXT: v_mov_b32_e32 v9, v7
; GFX1164-DPP-NEXT: v_mov_b32_e32 v8, v6
+; GFX1164-DPP-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1164-DPP-NEXT: s_or_b64 s[2:3], vcc, s[2:3]
-; GFX1164-DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164-DPP-NEXT: s_and_not1_b64 exec, exec, s[2:3]
+; GFX1164-DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164-DPP-NEXT: s_cbranch_execnz .LBB11_2
; GFX1164-DPP-NEXT: .LBB11_3:
; GFX1164-DPP-NEXT: s_endpgm
@@ -6507,7 +6549,7 @@ define amdgpu_kernel void @global_atomic_fmax_double_uni_address_div_value_defau
; GFX1132-DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1132-DPP-NEXT: v_mbcnt_lo_u32_b32 v6, exec_lo, 0
; GFX1132-DPP-NEXT: v_dual_mov_b32 v10, 0 :: v_dual_mov_b32 v1, v3
-; GFX1132-DPP-NEXT: s_delay_alu instid0(VALU_DEP_3)
+; GFX1132-DPP-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_3) | instid1(VALU_DEP_1)
; GFX1132-DPP-NEXT: v_mov_b32_e32 v0, v2
; GFX1132-DPP-NEXT: s_mov_b32 s2, 0
; GFX1132-DPP-NEXT: s_mov_b32 s0, exec_lo
@@ -6529,7 +6571,7 @@ define amdgpu_kernel void @global_atomic_fmax_double_uni_address_div_value_defau
; GFX1132-DPP-NEXT: v_cmp_eq_u64_e32 vcc_lo, v[6:7], v[8:9]
; GFX1132-DPP-NEXT: v_dual_mov_b32 v9, v7 :: v_dual_mov_b32 v8, v6
; GFX1132-DPP-NEXT: s_or_b32 s2, vcc_lo, s2
-; GFX1132-DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1132-DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1132-DPP-NEXT: s_and_not1_b32 exec_lo, exec_lo, s2
; GFX1132-DPP-NEXT: s_cbranch_execnz .LBB11_2
; GFX1132-DPP-NEXT: .LBB11_3:
@@ -6624,6 +6666,7 @@ define amdgpu_kernel void @global_atomic_fmax_uni_address_uni_value_system_scope
; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
; GFX1164-NEXT: v_cmpx_eq_u32_e32 0, v0
+; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1164-NEXT: s_cbranch_execz .LBB12_2
; GFX1164-NEXT: ; %bb.1:
; GFX1164-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
@@ -6638,7 +6681,7 @@ define amdgpu_kernel void @global_atomic_fmax_uni_address_uni_value_system_scope
; GFX1132: ; %bb.0:
; GFX1132-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
; GFX1132-NEXT: s_mov_b32 s0, exec_lo
-; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX1132-NEXT: s_cbranch_execz .LBB12_2
; GFX1132-NEXT: ; %bb.1:
@@ -6733,6 +6776,7 @@ define amdgpu_kernel void @global_atomic_fmax_uni_address_uni_value_system_scope
; GFX1164-DPP-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164-DPP-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
; GFX1164-DPP-NEXT: v_cmpx_eq_u32_e32 0, v0
+; GFX1164-DPP-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1164-DPP-NEXT: s_cbranch_execz .LBB12_2
; GFX1164-DPP-NEXT: ; %bb.1:
; GFX1164-DPP-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
@@ -6747,7 +6791,7 @@ define amdgpu_kernel void @global_atomic_fmax_uni_address_uni_value_system_scope
; GFX1132-DPP: ; %bb.0:
; GFX1132-DPP-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
; GFX1132-DPP-NEXT: s_mov_b32 s0, exec_lo
-; GFX1132-DPP-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1132-DPP-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132-DPP-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX1132-DPP-NEXT: s_cbranch_execz .LBB12_2
; GFX1132-DPP-NEXT: ; %bb.1:
@@ -6846,6 +6890,7 @@ define amdgpu_kernel void @global_atomic_fmax_uni_address_uni_value_system_scope
; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
; GFX1164-NEXT: v_cmpx_eq_u32_e32 0, v0
+; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1164-NEXT: s_cbranch_execz .LBB13_2
; GFX1164-NEXT: ; %bb.1:
; GFX1164-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
@@ -6860,7 +6905,7 @@ define amdgpu_kernel void @global_atomic_fmax_uni_address_uni_value_system_scope
; GFX1132: ; %bb.0:
; GFX1132-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
; GFX1132-NEXT: s_mov_b32 s0, exec_lo
-; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX1132-NEXT: s_cbranch_execz .LBB13_2
; GFX1132-NEXT: ; %bb.1:
@@ -6955,6 +7000,7 @@ define amdgpu_kernel void @global_atomic_fmax_uni_address_uni_value_system_scope
; GFX1164-DPP-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164-DPP-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
; GFX1164-DPP-NEXT: v_cmpx_eq_u32_e32 0, v0
+; GFX1164-DPP-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1164-DPP-NEXT: s_cbranch_execz .LBB13_2
; GFX1164-DPP-NEXT: ; %bb.1:
; GFX1164-DPP-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
@@ -6969,7 +7015,7 @@ define amdgpu_kernel void @global_atomic_fmax_uni_address_uni_value_system_scope
; GFX1132-DPP: ; %bb.0:
; GFX1132-DPP-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
; GFX1132-DPP-NEXT: s_mov_b32 s0, exec_lo
-; GFX1132-DPP-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1132-DPP-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132-DPP-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX1132-DPP-NEXT: s_cbranch_execz .LBB13_2
; GFX1132-DPP-NEXT: ; %bb.1:
diff --git a/llvm/test/CodeGen/AMDGPU/global_atomics_scan_fmin.ll b/llvm/test/CodeGen/AMDGPU/global_atomics_scan_fmin.ll
index 85bcfd88181b95..8d93f4650d9182 100644
--- a/llvm/test/CodeGen/AMDGPU/global_atomics_scan_fmin.ll
+++ b/llvm/test/CodeGen/AMDGPU/global_atomics_scan_fmin.ll
@@ -100,6 +100,7 @@ define amdgpu_kernel void @global_atomic_fmin_uni_address_uni_value_agent_scope_
; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
; GFX1164-NEXT: v_cmpx_eq_u32_e32 0, v0
+; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1164-NEXT: s_cbranch_execz .LBB0_2
; GFX1164-NEXT: ; %bb.1:
; GFX1164-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
@@ -114,7 +115,7 @@ define amdgpu_kernel void @global_atomic_fmin_uni_address_uni_value_agent_scope_
; GFX1132: ; %bb.0:
; GFX1132-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
; GFX1132-NEXT: s_mov_b32 s0, exec_lo
-; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX1132-NEXT: s_cbranch_execz .LBB0_2
; GFX1132-NEXT: ; %bb.1:
@@ -209,6 +210,7 @@ define amdgpu_kernel void @global_atomic_fmin_uni_address_uni_value_agent_scope_
; GFX1164-DPP-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164-DPP-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
; GFX1164-DPP-NEXT: v_cmpx_eq_u32_e32 0, v0
+; GFX1164-DPP-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1164-DPP-NEXT: s_cbranch_execz .LBB0_2
; GFX1164-DPP-NEXT: ; %bb.1:
; GFX1164-DPP-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
@@ -223,7 +225,7 @@ define amdgpu_kernel void @global_atomic_fmin_uni_address_uni_value_agent_scope_
; GFX1132-DPP: ; %bb.0:
; GFX1132-DPP-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
; GFX1132-DPP-NEXT: s_mov_b32 s0, exec_lo
-; GFX1132-DPP-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1132-DPP-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132-DPP-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX1132-DPP-NEXT: s_cbranch_execz .LBB0_2
; GFX1132-DPP-NEXT: ; %bb.1:
@@ -524,6 +526,7 @@ define amdgpu_kernel void @global_atomic_fmin_uni_address_div_value_agent_scope_
; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
; GFX1164-NEXT: v_cmpx_eq_u32_e32 0, v0
+; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1164-NEXT: s_xor_b64 s[0:1], exec, s[0:1]
; GFX1164-NEXT: s_cbranch_execz .LBB1_4
; GFX1164-NEXT: ; %bb.3:
@@ -572,9 +575,10 @@ define amdgpu_kernel void @global_atomic_fmin_uni_address_div_value_agent_scope_
; GFX1132-NEXT: ; %bb.2: ; %ComputeEnd
; GFX1132-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
; GFX1132-NEXT: s_mov_b32 s0, exec_lo
-; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX1132-NEXT: s_xor_b32 s0, exec_lo, s0
+; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1132-NEXT: s_cbranch_execz .LBB1_4
; GFX1132-NEXT: ; %bb.3:
; GFX1132-NEXT: s_load_b64 s[0:1], s[34:35], 0x24
@@ -916,6 +920,7 @@ define amdgpu_kernel void @global_atomic_fmin_uni_address_div_value_agent_scope_
; GFX1164-DPP-NEXT: v_mov_b32_e32 v0, v1
; GFX1164-DPP-NEXT: s_mov_b64 s[0:1], exec
; GFX1164-DPP-NEXT: v_cmpx_eq_u32_e32 0, v4
+; GFX1164-DPP-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1164-DPP-NEXT: s_cbranch_execz .LBB1_2
; GFX1164-DPP-NEXT: ; %bb.1:
; GFX1164-DPP-NEXT: s_load_b64 s[0:1], s[34:35], 0x24
@@ -978,7 +983,7 @@ define amdgpu_kernel void @global_atomic_fmin_uni_address_div_value_agent_scope_
; GFX1132-DPP-NEXT: v_mbcnt_lo_u32_b32 v4, exec_lo, 0
; GFX1132-DPP-NEXT: v_mov_b32_e32 v0, v1
; GFX1132-DPP-NEXT: s_mov_b32 s0, exec_lo
-; GFX1132-DPP-NEXT: s_delay_alu instid0(VALU_DEP_2)
+; GFX1132-DPP-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132-DPP-NEXT: v_cmpx_eq_u32_e32 0, v4
; GFX1132-DPP-NEXT: s_cbranch_execz .LBB1_2
; GFX1132-DPP-NEXT: ; %bb.1:
@@ -1078,6 +1083,7 @@ define amdgpu_kernel void @global_atomic_fmin_uni_address_uni_value_one_as_scope
; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
; GFX1164-NEXT: v_cmpx_eq_u32_e32 0, v0
+; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1164-NEXT: s_cbranch_execz .LBB2_2
; GFX1164-NEXT: ; %bb.1:
; GFX1164-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
@@ -1092,7 +1098,7 @@ define amdgpu_kernel void @global_atomic_fmin_uni_address_uni_value_one_as_scope
; GFX1132: ; %bb.0:
; GFX1132-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
; GFX1132-NEXT: s_mov_b32 s0, exec_lo
-; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX1132-NEXT: s_cbranch_execz .LBB2_2
; GFX1132-NEXT: ; %bb.1:
@@ -1187,6 +1193,7 @@ define amdgpu_kernel void @global_atomic_fmin_uni_address_uni_value_one_as_scope
; GFX1164-DPP-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164-DPP-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
; GFX1164-DPP-NEXT: v_cmpx_eq_u32_e32 0, v0
+; GFX1164-DPP-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1164-DPP-NEXT: s_cbranch_execz .LBB2_2
; GFX1164-DPP-NEXT: ; %bb.1:
; GFX1164-DPP-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
@@ -1201,7 +1208,7 @@ define amdgpu_kernel void @global_atomic_fmin_uni_address_uni_value_one_as_scope
; GFX1132-DPP: ; %bb.0:
; GFX1132-DPP-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
; GFX1132-DPP-NEXT: s_mov_b32 s0, exec_lo
-; GFX1132-DPP-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1132-DPP-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132-DPP-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX1132-DPP-NEXT: s_cbranch_execz .LBB2_2
; GFX1132-DPP-NEXT: ; %bb.1:
@@ -1503,6 +1510,7 @@ define amdgpu_kernel void @global_atomic_fmin_uni_address_div_value_one_as_scope
; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
; GFX1164-NEXT: v_cmpx_eq_u32_e32 0, v0
+; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1164-NEXT: s_xor_b64 s[0:1], exec, s[0:1]
; GFX1164-NEXT: s_cbranch_execz .LBB3_4
; GFX1164-NEXT: ; %bb.3:
@@ -1551,9 +1559,10 @@ define amdgpu_kernel void @global_atomic_fmin_uni_address_div_value_one_as_scope
; GFX1132-NEXT: ; %bb.2: ; %ComputeEnd
; GFX1132-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
; GFX1132-NEXT: s_mov_b32 s0, exec_lo
-; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX1132-NEXT: s_xor_b32 s0, exec_lo, s0
+; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1132-NEXT: s_cbranch_execz .LBB3_4
; GFX1132-NEXT: ; %bb.3:
; GFX1132-NEXT: s_load_b64 s[0:1], s[34:35], 0x24
@@ -1895,6 +1904,7 @@ define amdgpu_kernel void @global_atomic_fmin_uni_address_div_value_one_as_scope
; GFX1164-DPP-NEXT: v_mov_b32_e32 v0, v1
; GFX1164-DPP-NEXT: s_mov_b64 s[0:1], exec
; GFX1164-DPP-NEXT: v_cmpx_eq_u32_e32 0, v4
+; GFX1164-DPP-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1164-DPP-NEXT: s_cbranch_execz .LBB3_2
; GFX1164-DPP-NEXT: ; %bb.1:
; GFX1164-DPP-NEXT: s_load_b64 s[0:1], s[34:35], 0x24
@@ -1957,7 +1967,7 @@ define amdgpu_kernel void @global_atomic_fmin_uni_address_div_value_one_as_scope
; GFX1132-DPP-NEXT: v_mbcnt_lo_u32_b32 v4, exec_lo, 0
; GFX1132-DPP-NEXT: v_mov_b32_e32 v0, v1
; GFX1132-DPP-NEXT: s_mov_b32 s0, exec_lo
-; GFX1132-DPP-NEXT: s_delay_alu instid0(VALU_DEP_2)
+; GFX1132-DPP-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132-DPP-NEXT: v_cmpx_eq_u32_e32 0, v4
; GFX1132-DPP-NEXT: s_cbranch_execz .LBB3_2
; GFX1132-DPP-NEXT: ; %bb.1:
@@ -2058,6 +2068,7 @@ define amdgpu_kernel void @global_atomic_fmin_uni_address_uni_value_default_scop
; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
; GFX1164-NEXT: v_cmpx_eq_u32_e32 0, v0
+; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1164-NEXT: s_cbranch_execz .LBB4_2
; GFX1164-NEXT: ; %bb.1:
; GFX1164-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
@@ -2072,7 +2083,7 @@ define amdgpu_kernel void @global_atomic_fmin_uni_address_uni_value_default_scop
; GFX1132: ; %bb.0:
; GFX1132-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
; GFX1132-NEXT: s_mov_b32 s0, exec_lo
-; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX1132-NEXT: s_cbranch_execz .LBB4_2
; GFX1132-NEXT: ; %bb.1:
@@ -2167,6 +2178,7 @@ define amdgpu_kernel void @global_atomic_fmin_uni_address_uni_value_default_scop
; GFX1164-DPP-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164-DPP-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
; GFX1164-DPP-NEXT: v_cmpx_eq_u32_e32 0, v0
+; GFX1164-DPP-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1164-DPP-NEXT: s_cbranch_execz .LBB4_2
; GFX1164-DPP-NEXT: ; %bb.1:
; GFX1164-DPP-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
@@ -2181,7 +2193,7 @@ define amdgpu_kernel void @global_atomic_fmin_uni_address_uni_value_default_scop
; GFX1132-DPP: ; %bb.0:
; GFX1132-DPP-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
; GFX1132-DPP-NEXT: s_mov_b32 s0, exec_lo
-; GFX1132-DPP-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1132-DPP-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132-DPP-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX1132-DPP-NEXT: s_cbranch_execz .LBB4_2
; GFX1132-DPP-NEXT: ; %bb.1:
@@ -2482,6 +2494,7 @@ define amdgpu_kernel void @global_atomic_fmin_uni_address_div_value_default_scop
; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
; GFX1164-NEXT: v_cmpx_eq_u32_e32 0, v0
+; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1164-NEXT: s_xor_b64 s[0:1], exec, s[0:1]
; GFX1164-NEXT: s_cbranch_execz .LBB5_4
; GFX1164-NEXT: ; %bb.3:
@@ -2530,9 +2543,10 @@ define amdgpu_kernel void @global_atomic_fmin_uni_address_div_value_default_scop
; GFX1132-NEXT: ; %bb.2: ; %ComputeEnd
; GFX1132-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
; GFX1132-NEXT: s_mov_b32 s0, exec_lo
-; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX1132-NEXT: s_xor_b32 s0, exec_lo, s0
+; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1132-NEXT: s_cbranch_execz .LBB5_4
; GFX1132-NEXT: ; %bb.3:
; GFX1132-NEXT: s_load_b64 s[0:1], s[34:35], 0x24
@@ -2874,6 +2888,7 @@ define amdgpu_kernel void @global_atomic_fmin_uni_address_div_value_default_scop
; GFX1164-DPP-NEXT: v_mov_b32_e32 v0, v1
; GFX1164-DPP-NEXT: s_mov_b64 s[0:1], exec
; GFX1164-DPP-NEXT: v_cmpx_eq_u32_e32 0, v4
+; GFX1164-DPP-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1164-DPP-NEXT: s_cbranch_execz .LBB5_2
; GFX1164-DPP-NEXT: ; %bb.1:
; GFX1164-DPP-NEXT: s_load_b64 s[0:1], s[34:35], 0x24
@@ -2936,7 +2951,7 @@ define amdgpu_kernel void @global_atomic_fmin_uni_address_div_value_default_scop
; GFX1132-DPP-NEXT: v_mbcnt_lo_u32_b32 v4, exec_lo, 0
; GFX1132-DPP-NEXT: v_mov_b32_e32 v0, v1
; GFX1132-DPP-NEXT: s_mov_b32 s0, exec_lo
-; GFX1132-DPP-NEXT: s_delay_alu instid0(VALU_DEP_2)
+; GFX1132-DPP-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132-DPP-NEXT: v_cmpx_eq_u32_e32 0, v4
; GFX1132-DPP-NEXT: s_cbranch_execz .LBB5_2
; GFX1132-DPP-NEXT: ; %bb.1:
@@ -3041,6 +3056,7 @@ define amdgpu_kernel void @global_atomic_fmin_double_uni_address_uni_value_agent
; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
; GFX1164-NEXT: v_cmpx_eq_u32_e32 0, v0
+; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1164-NEXT: s_cbranch_execz .LBB6_3
; GFX1164-NEXT: ; %bb.1:
; GFX1164-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
@@ -3061,9 +3077,10 @@ define amdgpu_kernel void @global_atomic_fmin_double_uni_address_uni_value_agent
; GFX1164-NEXT: v_cmp_eq_u64_e32 vcc, v[0:1], v[2:3]
; GFX1164-NEXT: v_mov_b32_e32 v3, v1
; GFX1164-NEXT: v_mov_b32_e32 v2, v0
+; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1164-NEXT: s_or_b64 s[2:3], vcc, s[2:3]
-; GFX1164-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164-NEXT: s_and_not1_b64 exec, exec, s[2:3]
+; GFX1164-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164-NEXT: s_cbranch_execnz .LBB6_2
; GFX1164-NEXT: .LBB6_3:
; GFX1164-NEXT: s_endpgm
@@ -3073,7 +3090,7 @@ define amdgpu_kernel void @global_atomic_fmin_double_uni_address_uni_value_agent
; GFX1132-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
; GFX1132-NEXT: s_mov_b32 s2, 0
; GFX1132-NEXT: s_mov_b32 s0, exec_lo
-; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX1132-NEXT: s_cbranch_execz .LBB6_3
; GFX1132-NEXT: ; %bb.1:
@@ -3093,7 +3110,7 @@ define amdgpu_kernel void @global_atomic_fmin_double_uni_address_uni_value_agent
; GFX1132-NEXT: v_cmp_eq_u64_e32 vcc_lo, v[0:1], v[2:3]
; GFX1132-NEXT: v_dual_mov_b32 v3, v1 :: v_dual_mov_b32 v2, v0
; GFX1132-NEXT: s_or_b32 s2, vcc_lo, s2
-; GFX1132-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1132-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1132-NEXT: s_and_not1_b32 exec_lo, exec_lo, s2
; GFX1132-NEXT: s_cbranch_execnz .LBB6_2
; GFX1132-NEXT: .LBB6_3:
@@ -3188,6 +3205,7 @@ define amdgpu_kernel void @global_atomic_fmin_double_uni_address_uni_value_agent
; GFX1164-DPP-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164-DPP-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
; GFX1164-DPP-NEXT: v_cmpx_eq_u32_e32 0, v0
+; GFX1164-DPP-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1164-DPP-NEXT: s_cbranch_execz .LBB6_3
; GFX1164-DPP-NEXT: ; %bb.1:
; GFX1164-DPP-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
@@ -3208,9 +3226,10 @@ define amdgpu_kernel void @global_atomic_fmin_double_uni_address_uni_value_agent
; GFX1164-DPP-NEXT: v_cmp_eq_u64_e32 vcc, v[0:1], v[2:3]
; GFX1164-DPP-NEXT: v_mov_b32_e32 v3, v1
; GFX1164-DPP-NEXT: v_mov_b32_e32 v2, v0
+; GFX1164-DPP-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1164-DPP-NEXT: s_or_b64 s[2:3], vcc, s[2:3]
-; GFX1164-DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164-DPP-NEXT: s_and_not1_b64 exec, exec, s[2:3]
+; GFX1164-DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164-DPP-NEXT: s_cbranch_execnz .LBB6_2
; GFX1164-DPP-NEXT: .LBB6_3:
; GFX1164-DPP-NEXT: s_endpgm
@@ -3220,7 +3239,7 @@ define amdgpu_kernel void @global_atomic_fmin_double_uni_address_uni_value_agent
; GFX1132-DPP-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
; GFX1132-DPP-NEXT: s_mov_b32 s2, 0
; GFX1132-DPP-NEXT: s_mov_b32 s0, exec_lo
-; GFX1132-DPP-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1132-DPP-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132-DPP-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX1132-DPP-NEXT: s_cbranch_execz .LBB6_3
; GFX1132-DPP-NEXT: ; %bb.1:
@@ -3240,7 +3259,7 @@ define amdgpu_kernel void @global_atomic_fmin_double_uni_address_uni_value_agent
; GFX1132-DPP-NEXT: v_cmp_eq_u64_e32 vcc_lo, v[0:1], v[2:3]
; GFX1132-DPP-NEXT: v_dual_mov_b32 v3, v1 :: v_dual_mov_b32 v2, v0
; GFX1132-DPP-NEXT: s_or_b32 s2, vcc_lo, s2
-; GFX1132-DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1132-DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1132-DPP-NEXT: s_and_not1_b32 exec_lo, exec_lo, s2
; GFX1132-DPP-NEXT: s_cbranch_execnz .LBB6_2
; GFX1132-DPP-NEXT: .LBB6_3:
@@ -3547,6 +3566,7 @@ define amdgpu_kernel void @global_atomic_fmin_double_uni_address_div_value_agent
; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
; GFX1164-NEXT: v_cmpx_eq_u32_e32 0, v0
+; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1164-NEXT: s_xor_b64 s[0:1], exec, s[0:1]
; GFX1164-NEXT: s_cbranch_execz .LBB7_5
; GFX1164-NEXT: ; %bb.3:
@@ -3567,9 +3587,10 @@ define amdgpu_kernel void @global_atomic_fmin_double_uni_address_div_value_agent
; GFX1164-NEXT: v_cmp_eq_u64_e32 vcc, v[0:1], v[2:3]
; GFX1164-NEXT: v_mov_b32_e32 v3, v1
; GFX1164-NEXT: v_mov_b32_e32 v2, v0
+; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1164-NEXT: s_or_b64 s[2:3], vcc, s[2:3]
-; GFX1164-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164-NEXT: s_and_not1_b64 exec, exec, s[2:3]
+; GFX1164-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164-NEXT: s_cbranch_execnz .LBB7_4
; GFX1164-NEXT: .LBB7_5:
; GFX1164-NEXT: s_endpgm
@@ -3614,9 +3635,10 @@ define amdgpu_kernel void @global_atomic_fmin_double_uni_address_div_value_agent
; GFX1132-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
; GFX1132-NEXT: s_mov_b32 s2, 0
; GFX1132-NEXT: s_mov_b32 s0, exec_lo
-; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX1132-NEXT: s_xor_b32 s0, exec_lo, s0
+; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1132-NEXT: s_cbranch_execz .LBB7_5
; GFX1132-NEXT: ; %bb.3:
; GFX1132-NEXT: s_load_b64 s[0:1], s[34:35], 0x24
@@ -3636,7 +3658,7 @@ define amdgpu_kernel void @global_atomic_fmin_double_uni_address_div_value_agent
; GFX1132-NEXT: v_cmp_eq_u64_e32 vcc_lo, v[0:1], v[2:3]
; GFX1132-NEXT: v_dual_mov_b32 v3, v1 :: v_dual_mov_b32 v2, v0
; GFX1132-NEXT: s_or_b32 s2, vcc_lo, s2
-; GFX1132-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1132-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1132-NEXT: s_and_not1_b32 exec_lo, exec_lo, s2
; GFX1132-NEXT: s_cbranch_execnz .LBB7_4
; GFX1132-NEXT: .LBB7_5:
@@ -4027,6 +4049,7 @@ define amdgpu_kernel void @global_atomic_fmin_double_uni_address_div_value_agent
; GFX1164-DPP-NEXT: v_mov_b32_e32 v0, v2
; GFX1164-DPP-NEXT: v_cmpx_eq_u32_e32 0, v6
; GFX1164-DPP-NEXT: s_waitcnt_depctr depctr_sa_sdst(0)
+; GFX1164-DPP-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1164-DPP-NEXT: s_cbranch_execz .LBB7_3
; GFX1164-DPP-NEXT: ; %bb.1:
; GFX1164-DPP-NEXT: s_load_b64 s[0:1], s[34:35], 0x24
@@ -4045,9 +4068,10 @@ define amdgpu_kernel void @global_atomic_fmin_double_uni_address_div_value_agent
; GFX1164-DPP-NEXT: v_cmp_eq_u64_e32 vcc, v[6:7], v[8:9]
; GFX1164-DPP-NEXT: v_mov_b32_e32 v9, v7
; GFX1164-DPP-NEXT: v_mov_b32_e32 v8, v6
+; GFX1164-DPP-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1164-DPP-NEXT: s_or_b64 s[2:3], vcc, s[2:3]
-; GFX1164-DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164-DPP-NEXT: s_and_not1_b64 exec, exec, s[2:3]
+; GFX1164-DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164-DPP-NEXT: s_cbranch_execnz .LBB7_2
; GFX1164-DPP-NEXT: .LBB7_3:
; GFX1164-DPP-NEXT: s_endpgm
@@ -4115,7 +4139,7 @@ define amdgpu_kernel void @global_atomic_fmin_double_uni_address_div_value_agent
; GFX1132-DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1132-DPP-NEXT: v_mbcnt_lo_u32_b32 v6, exec_lo, 0
; GFX1132-DPP-NEXT: v_dual_mov_b32 v10, 0 :: v_dual_mov_b32 v1, v3
-; GFX1132-DPP-NEXT: s_delay_alu instid0(VALU_DEP_3)
+; GFX1132-DPP-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_3) | instid1(VALU_DEP_1)
; GFX1132-DPP-NEXT: v_mov_b32_e32 v0, v2
; GFX1132-DPP-NEXT: s_mov_b32 s2, 0
; GFX1132-DPP-NEXT: s_mov_b32 s0, exec_lo
@@ -4137,7 +4161,7 @@ define amdgpu_kernel void @global_atomic_fmin_double_uni_address_div_value_agent
; GFX1132-DPP-NEXT: v_cmp_eq_u64_e32 vcc_lo, v[6:7], v[8:9]
; GFX1132-DPP-NEXT: v_dual_mov_b32 v9, v7 :: v_dual_mov_b32 v8, v6
; GFX1132-DPP-NEXT: s_or_b32 s2, vcc_lo, s2
-; GFX1132-DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1132-DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1132-DPP-NEXT: s_and_not1_b32 exec_lo, exec_lo, s2
; GFX1132-DPP-NEXT: s_cbranch_execnz .LBB7_2
; GFX1132-DPP-NEXT: .LBB7_3:
@@ -4237,6 +4261,7 @@ define amdgpu_kernel void @global_atomic_fmin_double_uni_address_uni_value_one_a
; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
; GFX1164-NEXT: v_cmpx_eq_u32_e32 0, v0
+; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1164-NEXT: s_cbranch_execz .LBB8_3
; GFX1164-NEXT: ; %bb.1:
; GFX1164-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
@@ -4257,9 +4282,10 @@ define amdgpu_kernel void @global_atomic_fmin_double_uni_address_uni_value_one_a
; GFX1164-NEXT: v_cmp_eq_u64_e32 vcc, v[0:1], v[2:3]
; GFX1164-NEXT: v_mov_b32_e32 v3, v1
; GFX1164-NEXT: v_mov_b32_e32 v2, v0
+; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1164-NEXT: s_or_b64 s[2:3], vcc, s[2:3]
-; GFX1164-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164-NEXT: s_and_not1_b64 exec, exec, s[2:3]
+; GFX1164-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164-NEXT: s_cbranch_execnz .LBB8_2
; GFX1164-NEXT: .LBB8_3:
; GFX1164-NEXT: s_endpgm
@@ -4269,7 +4295,7 @@ define amdgpu_kernel void @global_atomic_fmin_double_uni_address_uni_value_one_a
; GFX1132-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
; GFX1132-NEXT: s_mov_b32 s2, 0
; GFX1132-NEXT: s_mov_b32 s0, exec_lo
-; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX1132-NEXT: s_cbranch_execz .LBB8_3
; GFX1132-NEXT: ; %bb.1:
@@ -4289,7 +4315,7 @@ define amdgpu_kernel void @global_atomic_fmin_double_uni_address_uni_value_one_a
; GFX1132-NEXT: v_cmp_eq_u64_e32 vcc_lo, v[0:1], v[2:3]
; GFX1132-NEXT: v_dual_mov_b32 v3, v1 :: v_dual_mov_b32 v2, v0
; GFX1132-NEXT: s_or_b32 s2, vcc_lo, s2
-; GFX1132-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1132-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1132-NEXT: s_and_not1_b32 exec_lo, exec_lo, s2
; GFX1132-NEXT: s_cbranch_execnz .LBB8_2
; GFX1132-NEXT: .LBB8_3:
@@ -4384,6 +4410,7 @@ define amdgpu_kernel void @global_atomic_fmin_double_uni_address_uni_value_one_a
; GFX1164-DPP-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164-DPP-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
; GFX1164-DPP-NEXT: v_cmpx_eq_u32_e32 0, v0
+; GFX1164-DPP-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1164-DPP-NEXT: s_cbranch_execz .LBB8_3
; GFX1164-DPP-NEXT: ; %bb.1:
; GFX1164-DPP-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
@@ -4404,9 +4431,10 @@ define amdgpu_kernel void @global_atomic_fmin_double_uni_address_uni_value_one_a
; GFX1164-DPP-NEXT: v_cmp_eq_u64_e32 vcc, v[0:1], v[2:3]
; GFX1164-DPP-NEXT: v_mov_b32_e32 v3, v1
; GFX1164-DPP-NEXT: v_mov_b32_e32 v2, v0
+; GFX1164-DPP-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1164-DPP-NEXT: s_or_b64 s[2:3], vcc, s[2:3]
-; GFX1164-DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164-DPP-NEXT: s_and_not1_b64 exec, exec, s[2:3]
+; GFX1164-DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164-DPP-NEXT: s_cbranch_execnz .LBB8_2
; GFX1164-DPP-NEXT: .LBB8_3:
; GFX1164-DPP-NEXT: s_endpgm
@@ -4416,7 +4444,7 @@ define amdgpu_kernel void @global_atomic_fmin_double_uni_address_uni_value_one_a
; GFX1132-DPP-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
; GFX1132-DPP-NEXT: s_mov_b32 s2, 0
; GFX1132-DPP-NEXT: s_mov_b32 s0, exec_lo
-; GFX1132-DPP-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1132-DPP-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132-DPP-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX1132-DPP-NEXT: s_cbranch_execz .LBB8_3
; GFX1132-DPP-NEXT: ; %bb.1:
@@ -4436,7 +4464,7 @@ define amdgpu_kernel void @global_atomic_fmin_double_uni_address_uni_value_one_a
; GFX1132-DPP-NEXT: v_cmp_eq_u64_e32 vcc_lo, v[0:1], v[2:3]
; GFX1132-DPP-NEXT: v_dual_mov_b32 v3, v1 :: v_dual_mov_b32 v2, v0
; GFX1132-DPP-NEXT: s_or_b32 s2, vcc_lo, s2
-; GFX1132-DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1132-DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1132-DPP-NEXT: s_and_not1_b32 exec_lo, exec_lo, s2
; GFX1132-DPP-NEXT: s_cbranch_execnz .LBB8_2
; GFX1132-DPP-NEXT: .LBB8_3:
@@ -4743,6 +4771,7 @@ define amdgpu_kernel void @global_atomic_fmin_double_uni_address_div_value_one_a
; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
; GFX1164-NEXT: v_cmpx_eq_u32_e32 0, v0
+; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1164-NEXT: s_xor_b64 s[0:1], exec, s[0:1]
; GFX1164-NEXT: s_cbranch_execz .LBB9_5
; GFX1164-NEXT: ; %bb.3:
@@ -4763,9 +4792,10 @@ define amdgpu_kernel void @global_atomic_fmin_double_uni_address_div_value_one_a
; GFX1164-NEXT: v_cmp_eq_u64_e32 vcc, v[0:1], v[2:3]
; GFX1164-NEXT: v_mov_b32_e32 v3, v1
; GFX1164-NEXT: v_mov_b32_e32 v2, v0
+; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1164-NEXT: s_or_b64 s[2:3], vcc, s[2:3]
-; GFX1164-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164-NEXT: s_and_not1_b64 exec, exec, s[2:3]
+; GFX1164-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164-NEXT: s_cbranch_execnz .LBB9_4
; GFX1164-NEXT: .LBB9_5:
; GFX1164-NEXT: s_endpgm
@@ -4810,9 +4840,10 @@ define amdgpu_kernel void @global_atomic_fmin_double_uni_address_div_value_one_a
; GFX1132-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
; GFX1132-NEXT: s_mov_b32 s2, 0
; GFX1132-NEXT: s_mov_b32 s0, exec_lo
-; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX1132-NEXT: s_xor_b32 s0, exec_lo, s0
+; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1132-NEXT: s_cbranch_execz .LBB9_5
; GFX1132-NEXT: ; %bb.3:
; GFX1132-NEXT: s_load_b64 s[0:1], s[34:35], 0x24
@@ -4832,7 +4863,7 @@ define amdgpu_kernel void @global_atomic_fmin_double_uni_address_div_value_one_a
; GFX1132-NEXT: v_cmp_eq_u64_e32 vcc_lo, v[0:1], v[2:3]
; GFX1132-NEXT: v_dual_mov_b32 v3, v1 :: v_dual_mov_b32 v2, v0
; GFX1132-NEXT: s_or_b32 s2, vcc_lo, s2
-; GFX1132-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1132-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1132-NEXT: s_and_not1_b32 exec_lo, exec_lo, s2
; GFX1132-NEXT: s_cbranch_execnz .LBB9_4
; GFX1132-NEXT: .LBB9_5:
@@ -5223,6 +5254,7 @@ define amdgpu_kernel void @global_atomic_fmin_double_uni_address_div_value_one_a
; GFX1164-DPP-NEXT: v_mov_b32_e32 v0, v2
; GFX1164-DPP-NEXT: v_cmpx_eq_u32_e32 0, v6
; GFX1164-DPP-NEXT: s_waitcnt_depctr depctr_sa_sdst(0)
+; GFX1164-DPP-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1164-DPP-NEXT: s_cbranch_execz .LBB9_3
; GFX1164-DPP-NEXT: ; %bb.1:
; GFX1164-DPP-NEXT: s_load_b64 s[0:1], s[34:35], 0x24
@@ -5241,9 +5273,10 @@ define amdgpu_kernel void @global_atomic_fmin_double_uni_address_div_value_one_a
; GFX1164-DPP-NEXT: v_cmp_eq_u64_e32 vcc, v[6:7], v[8:9]
; GFX1164-DPP-NEXT: v_mov_b32_e32 v9, v7
; GFX1164-DPP-NEXT: v_mov_b32_e32 v8, v6
+; GFX1164-DPP-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1164-DPP-NEXT: s_or_b64 s[2:3], vcc, s[2:3]
-; GFX1164-DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164-DPP-NEXT: s_and_not1_b64 exec, exec, s[2:3]
+; GFX1164-DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164-DPP-NEXT: s_cbranch_execnz .LBB9_2
; GFX1164-DPP-NEXT: .LBB9_3:
; GFX1164-DPP-NEXT: s_endpgm
@@ -5311,7 +5344,7 @@ define amdgpu_kernel void @global_atomic_fmin_double_uni_address_div_value_one_a
; GFX1132-DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1132-DPP-NEXT: v_mbcnt_lo_u32_b32 v6, exec_lo, 0
; GFX1132-DPP-NEXT: v_dual_mov_b32 v10, 0 :: v_dual_mov_b32 v1, v3
-; GFX1132-DPP-NEXT: s_delay_alu instid0(VALU_DEP_3)
+; GFX1132-DPP-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_3) | instid1(VALU_DEP_1)
; GFX1132-DPP-NEXT: v_mov_b32_e32 v0, v2
; GFX1132-DPP-NEXT: s_mov_b32 s2, 0
; GFX1132-DPP-NEXT: s_mov_b32 s0, exec_lo
@@ -5333,7 +5366,7 @@ define amdgpu_kernel void @global_atomic_fmin_double_uni_address_div_value_one_a
; GFX1132-DPP-NEXT: v_cmp_eq_u64_e32 vcc_lo, v[6:7], v[8:9]
; GFX1132-DPP-NEXT: v_dual_mov_b32 v9, v7 :: v_dual_mov_b32 v8, v6
; GFX1132-DPP-NEXT: s_or_b32 s2, vcc_lo, s2
-; GFX1132-DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1132-DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1132-DPP-NEXT: s_and_not1_b32 exec_lo, exec_lo, s2
; GFX1132-DPP-NEXT: s_cbranch_execnz .LBB9_2
; GFX1132-DPP-NEXT: .LBB9_3:
@@ -5433,6 +5466,7 @@ define amdgpu_kernel void @global_atomic_fmin_double_uni_address_uni_value_defau
; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
; GFX1164-NEXT: v_cmpx_eq_u32_e32 0, v0
+; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1164-NEXT: s_cbranch_execz .LBB10_3
; GFX1164-NEXT: ; %bb.1:
; GFX1164-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
@@ -5453,9 +5487,10 @@ define amdgpu_kernel void @global_atomic_fmin_double_uni_address_uni_value_defau
; GFX1164-NEXT: v_cmp_eq_u64_e32 vcc, v[0:1], v[2:3]
; GFX1164-NEXT: v_mov_b32_e32 v3, v1
; GFX1164-NEXT: v_mov_b32_e32 v2, v0
+; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1164-NEXT: s_or_b64 s[2:3], vcc, s[2:3]
-; GFX1164-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164-NEXT: s_and_not1_b64 exec, exec, s[2:3]
+; GFX1164-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164-NEXT: s_cbranch_execnz .LBB10_2
; GFX1164-NEXT: .LBB10_3:
; GFX1164-NEXT: s_endpgm
@@ -5465,7 +5500,7 @@ define amdgpu_kernel void @global_atomic_fmin_double_uni_address_uni_value_defau
; GFX1132-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
; GFX1132-NEXT: s_mov_b32 s2, 0
; GFX1132-NEXT: s_mov_b32 s0, exec_lo
-; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX1132-NEXT: s_cbranch_execz .LBB10_3
; GFX1132-NEXT: ; %bb.1:
@@ -5485,7 +5520,7 @@ define amdgpu_kernel void @global_atomic_fmin_double_uni_address_uni_value_defau
; GFX1132-NEXT: v_cmp_eq_u64_e32 vcc_lo, v[0:1], v[2:3]
; GFX1132-NEXT: v_dual_mov_b32 v3, v1 :: v_dual_mov_b32 v2, v0
; GFX1132-NEXT: s_or_b32 s2, vcc_lo, s2
-; GFX1132-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1132-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1132-NEXT: s_and_not1_b32 exec_lo, exec_lo, s2
; GFX1132-NEXT: s_cbranch_execnz .LBB10_2
; GFX1132-NEXT: .LBB10_3:
@@ -5580,6 +5615,7 @@ define amdgpu_kernel void @global_atomic_fmin_double_uni_address_uni_value_defau
; GFX1164-DPP-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164-DPP-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
; GFX1164-DPP-NEXT: v_cmpx_eq_u32_e32 0, v0
+; GFX1164-DPP-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1164-DPP-NEXT: s_cbranch_execz .LBB10_3
; GFX1164-DPP-NEXT: ; %bb.1:
; GFX1164-DPP-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
@@ -5600,9 +5636,10 @@ define amdgpu_kernel void @global_atomic_fmin_double_uni_address_uni_value_defau
; GFX1164-DPP-NEXT: v_cmp_eq_u64_e32 vcc, v[0:1], v[2:3]
; GFX1164-DPP-NEXT: v_mov_b32_e32 v3, v1
; GFX1164-DPP-NEXT: v_mov_b32_e32 v2, v0
+; GFX1164-DPP-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1164-DPP-NEXT: s_or_b64 s[2:3], vcc, s[2:3]
-; GFX1164-DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164-DPP-NEXT: s_and_not1_b64 exec, exec, s[2:3]
+; GFX1164-DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164-DPP-NEXT: s_cbranch_execnz .LBB10_2
; GFX1164-DPP-NEXT: .LBB10_3:
; GFX1164-DPP-NEXT: s_endpgm
@@ -5612,7 +5649,7 @@ define amdgpu_kernel void @global_atomic_fmin_double_uni_address_uni_value_defau
; GFX1132-DPP-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
; GFX1132-DPP-NEXT: s_mov_b32 s2, 0
; GFX1132-DPP-NEXT: s_mov_b32 s0, exec_lo
-; GFX1132-DPP-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1132-DPP-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132-DPP-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX1132-DPP-NEXT: s_cbranch_execz .LBB10_3
; GFX1132-DPP-NEXT: ; %bb.1:
@@ -5632,7 +5669,7 @@ define amdgpu_kernel void @global_atomic_fmin_double_uni_address_uni_value_defau
; GFX1132-DPP-NEXT: v_cmp_eq_u64_e32 vcc_lo, v[0:1], v[2:3]
; GFX1132-DPP-NEXT: v_dual_mov_b32 v3, v1 :: v_dual_mov_b32 v2, v0
; GFX1132-DPP-NEXT: s_or_b32 s2, vcc_lo, s2
-; GFX1132-DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1132-DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1132-DPP-NEXT: s_and_not1_b32 exec_lo, exec_lo, s2
; GFX1132-DPP-NEXT: s_cbranch_execnz .LBB10_2
; GFX1132-DPP-NEXT: .LBB10_3:
@@ -5939,6 +5976,7 @@ define amdgpu_kernel void @global_atomic_fmin_double_uni_address_div_value_defau
; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
; GFX1164-NEXT: v_cmpx_eq_u32_e32 0, v0
+; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1164-NEXT: s_xor_b64 s[0:1], exec, s[0:1]
; GFX1164-NEXT: s_cbranch_execz .LBB11_5
; GFX1164-NEXT: ; %bb.3:
@@ -5959,9 +5997,10 @@ define amdgpu_kernel void @global_atomic_fmin_double_uni_address_div_value_defau
; GFX1164-NEXT: v_cmp_eq_u64_e32 vcc, v[0:1], v[2:3]
; GFX1164-NEXT: v_mov_b32_e32 v3, v1
; GFX1164-NEXT: v_mov_b32_e32 v2, v0
+; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1164-NEXT: s_or_b64 s[2:3], vcc, s[2:3]
-; GFX1164-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164-NEXT: s_and_not1_b64 exec, exec, s[2:3]
+; GFX1164-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164-NEXT: s_cbranch_execnz .LBB11_4
; GFX1164-NEXT: .LBB11_5:
; GFX1164-NEXT: s_endpgm
@@ -6006,9 +6045,10 @@ define amdgpu_kernel void @global_atomic_fmin_double_uni_address_div_value_defau
; GFX1132-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
; GFX1132-NEXT: s_mov_b32 s2, 0
; GFX1132-NEXT: s_mov_b32 s0, exec_lo
-; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX1132-NEXT: s_xor_b32 s0, exec_lo, s0
+; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1132-NEXT: s_cbranch_execz .LBB11_5
; GFX1132-NEXT: ; %bb.3:
; GFX1132-NEXT: s_load_b64 s[0:1], s[34:35], 0x24
@@ -6028,7 +6068,7 @@ define amdgpu_kernel void @global_atomic_fmin_double_uni_address_div_value_defau
; GFX1132-NEXT: v_cmp_eq_u64_e32 vcc_lo, v[0:1], v[2:3]
; GFX1132-NEXT: v_dual_mov_b32 v3, v1 :: v_dual_mov_b32 v2, v0
; GFX1132-NEXT: s_or_b32 s2, vcc_lo, s2
-; GFX1132-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1132-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1132-NEXT: s_and_not1_b32 exec_lo, exec_lo, s2
; GFX1132-NEXT: s_cbranch_execnz .LBB11_4
; GFX1132-NEXT: .LBB11_5:
@@ -6419,6 +6459,7 @@ define amdgpu_kernel void @global_atomic_fmin_double_uni_address_div_value_defau
; GFX1164-DPP-NEXT: v_mov_b32_e32 v0, v2
; GFX1164-DPP-NEXT: v_cmpx_eq_u32_e32 0, v6
; GFX1164-DPP-NEXT: s_waitcnt_depctr depctr_sa_sdst(0)
+; GFX1164-DPP-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1164-DPP-NEXT: s_cbranch_execz .LBB11_3
; GFX1164-DPP-NEXT: ; %bb.1:
; GFX1164-DPP-NEXT: s_load_b64 s[0:1], s[34:35], 0x24
@@ -6437,9 +6478,10 @@ define amdgpu_kernel void @global_atomic_fmin_double_uni_address_div_value_defau
; GFX1164-DPP-NEXT: v_cmp_eq_u64_e32 vcc, v[6:7], v[8:9]
; GFX1164-DPP-NEXT: v_mov_b32_e32 v9, v7
; GFX1164-DPP-NEXT: v_mov_b32_e32 v8, v6
+; GFX1164-DPP-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1164-DPP-NEXT: s_or_b64 s[2:3], vcc, s[2:3]
-; GFX1164-DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164-DPP-NEXT: s_and_not1_b64 exec, exec, s[2:3]
+; GFX1164-DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164-DPP-NEXT: s_cbranch_execnz .LBB11_2
; GFX1164-DPP-NEXT: .LBB11_3:
; GFX1164-DPP-NEXT: s_endpgm
@@ -6507,7 +6549,7 @@ define amdgpu_kernel void @global_atomic_fmin_double_uni_address_div_value_defau
; GFX1132-DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1132-DPP-NEXT: v_mbcnt_lo_u32_b32 v6, exec_lo, 0
; GFX1132-DPP-NEXT: v_dual_mov_b32 v10, 0 :: v_dual_mov_b32 v1, v3
-; GFX1132-DPP-NEXT: s_delay_alu instid0(VALU_DEP_3)
+; GFX1132-DPP-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_3) | instid1(VALU_DEP_1)
; GFX1132-DPP-NEXT: v_mov_b32_e32 v0, v2
; GFX1132-DPP-NEXT: s_mov_b32 s2, 0
; GFX1132-DPP-NEXT: s_mov_b32 s0, exec_lo
@@ -6529,7 +6571,7 @@ define amdgpu_kernel void @global_atomic_fmin_double_uni_address_div_value_defau
; GFX1132-DPP-NEXT: v_cmp_eq_u64_e32 vcc_lo, v[6:7], v[8:9]
; GFX1132-DPP-NEXT: v_dual_mov_b32 v9, v7 :: v_dual_mov_b32 v8, v6
; GFX1132-DPP-NEXT: s_or_b32 s2, vcc_lo, s2
-; GFX1132-DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1132-DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1132-DPP-NEXT: s_and_not1_b32 exec_lo, exec_lo, s2
; GFX1132-DPP-NEXT: s_cbranch_execnz .LBB11_2
; GFX1132-DPP-NEXT: .LBB11_3:
@@ -6624,6 +6666,7 @@ define amdgpu_kernel void @global_atomic_fmin_uni_address_uni_value_system_scope
; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
; GFX1164-NEXT: v_cmpx_eq_u32_e32 0, v0
+; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1164-NEXT: s_cbranch_execz .LBB12_2
; GFX1164-NEXT: ; %bb.1:
; GFX1164-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
@@ -6638,7 +6681,7 @@ define amdgpu_kernel void @global_atomic_fmin_uni_address_uni_value_system_scope
; GFX1132: ; %bb.0:
; GFX1132-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
; GFX1132-NEXT: s_mov_b32 s0, exec_lo
-; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX1132-NEXT: s_cbranch_execz .LBB12_2
; GFX1132-NEXT: ; %bb.1:
@@ -6733,6 +6776,7 @@ define amdgpu_kernel void @global_atomic_fmin_uni_address_uni_value_system_scope
; GFX1164-DPP-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164-DPP-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
; GFX1164-DPP-NEXT: v_cmpx_eq_u32_e32 0, v0
+; GFX1164-DPP-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1164-DPP-NEXT: s_cbranch_execz .LBB12_2
; GFX1164-DPP-NEXT: ; %bb.1:
; GFX1164-DPP-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
@@ -6747,7 +6791,7 @@ define amdgpu_kernel void @global_atomic_fmin_uni_address_uni_value_system_scope
; GFX1132-DPP: ; %bb.0:
; GFX1132-DPP-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
; GFX1132-DPP-NEXT: s_mov_b32 s0, exec_lo
-; GFX1132-DPP-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1132-DPP-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132-DPP-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX1132-DPP-NEXT: s_cbranch_execz .LBB12_2
; GFX1132-DPP-NEXT: ; %bb.1:
@@ -6846,6 +6890,7 @@ define amdgpu_kernel void @global_atomic_fmin_uni_address_uni_value_system_scope
; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
; GFX1164-NEXT: v_cmpx_eq_u32_e32 0, v0
+; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1164-NEXT: s_cbranch_execz .LBB13_2
; GFX1164-NEXT: ; %bb.1:
; GFX1164-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
@@ -6860,7 +6905,7 @@ define amdgpu_kernel void @global_atomic_fmin_uni_address_uni_value_system_scope
; GFX1132: ; %bb.0:
; GFX1132-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
; GFX1132-NEXT: s_mov_b32 s0, exec_lo
-; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX1132-NEXT: s_cbranch_execz .LBB13_2
; GFX1132-NEXT: ; %bb.1:
@@ -6955,6 +7000,7 @@ define amdgpu_kernel void @global_atomic_fmin_uni_address_uni_value_system_scope
; GFX1164-DPP-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164-DPP-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
; GFX1164-DPP-NEXT: v_cmpx_eq_u32_e32 0, v0
+; GFX1164-DPP-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1164-DPP-NEXT: s_cbranch_execz .LBB13_2
; GFX1164-DPP-NEXT: ; %bb.1:
; GFX1164-DPP-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
@@ -6969,7 +7015,7 @@ define amdgpu_kernel void @global_atomic_fmin_uni_address_uni_value_system_scope
; GFX1132-DPP: ; %bb.0:
; GFX1132-DPP-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
; GFX1132-DPP-NEXT: s_mov_b32 s0, exec_lo
-; GFX1132-DPP-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1132-DPP-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132-DPP-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX1132-DPP-NEXT: s_cbranch_execz .LBB13_2
; GFX1132-DPP-NEXT: ; %bb.1:
diff --git a/llvm/test/CodeGen/AMDGPU/global_atomics_scan_fsub.ll b/llvm/test/CodeGen/AMDGPU/global_atomics_scan_fsub.ll
index 94476b93b4f288..2bbd94527db8dd 100644
--- a/llvm/test/CodeGen/AMDGPU/global_atomics_scan_fsub.ll
+++ b/llvm/test/CodeGen/AMDGPU/global_atomics_scan_fsub.ll
@@ -151,6 +151,7 @@ define amdgpu_kernel void @global_atomic_fsub_uni_address_uni_value_agent_scope_
; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164-NEXT: v_mbcnt_hi_u32_b32 v0, s3, v0
; GFX1164-NEXT: v_cmpx_eq_u32_e32 0, v0
+; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1164-NEXT: s_cbranch_execz .LBB0_3
; GFX1164-NEXT: ; %bb.1:
; GFX1164-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
@@ -166,14 +167,14 @@ define amdgpu_kernel void @global_atomic_fsub_uni_address_uni_value_agent_scope_
; GFX1164-NEXT: v_mov_b32_e32 v1, s4
; GFX1164-NEXT: .LBB0_2: ; %atomicrmw.start
; GFX1164-NEXT: ; =>This Inner Loop Header: Depth=1
-; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_4) | instid1(VALU_DEP_2)
; GFX1164-NEXT: v_sub_f32_e32 v0, v1, v2
; GFX1164-NEXT: global_atomic_cmpswap_b32 v0, v3, v[0:1], s[0:1] glc
; GFX1164-NEXT: s_waitcnt vmcnt(0)
; GFX1164-NEXT: v_cmp_eq_u32_e32 vcc, v0, v1
; GFX1164-NEXT: v_mov_b32_e32 v1, v0
; GFX1164-NEXT: s_or_b64 s[2:3], vcc, s[2:3]
-; GFX1164-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1164-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1164-NEXT: s_and_not1_b64 exec, exec, s[2:3]
; GFX1164-NEXT: s_cbranch_execnz .LBB0_2
; GFX1164-NEXT: .LBB0_3:
@@ -185,7 +186,7 @@ define amdgpu_kernel void @global_atomic_fsub_uni_address_uni_value_agent_scope_
; GFX1132-NEXT: s_mov_b32 s2, 0
; GFX1132-NEXT: v_mbcnt_lo_u32_b32 v0, s3, 0
; GFX1132-NEXT: s_mov_b32 s0, exec_lo
-; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX1132-NEXT: s_cbranch_execz .LBB0_3
; GFX1132-NEXT: ; %bb.1:
@@ -207,7 +208,7 @@ define amdgpu_kernel void @global_atomic_fsub_uni_address_uni_value_agent_scope_
; GFX1132-NEXT: v_cmp_eq_u32_e32 vcc_lo, v0, v1
; GFX1132-NEXT: v_mov_b32_e32 v1, v0
; GFX1132-NEXT: s_or_b32 s2, vcc_lo, s2
-; GFX1132-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1132-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1132-NEXT: s_and_not1_b32 exec_lo, exec_lo, s2
; GFX1132-NEXT: s_cbranch_execnz .LBB0_2
; GFX1132-NEXT: .LBB0_3:
@@ -348,6 +349,7 @@ define amdgpu_kernel void @global_atomic_fsub_uni_address_uni_value_agent_scope_
; GFX1164-DPP-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164-DPP-NEXT: v_mbcnt_hi_u32_b32 v0, s3, v0
; GFX1164-DPP-NEXT: v_cmpx_eq_u32_e32 0, v0
+; GFX1164-DPP-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1164-DPP-NEXT: s_cbranch_execz .LBB0_3
; GFX1164-DPP-NEXT: ; %bb.1:
; GFX1164-DPP-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
@@ -363,14 +365,14 @@ define amdgpu_kernel void @global_atomic_fsub_uni_address_uni_value_agent_scope_
; GFX1164-DPP-NEXT: v_mov_b32_e32 v1, s4
; GFX1164-DPP-NEXT: .LBB0_2: ; %atomicrmw.start
; GFX1164-DPP-NEXT: ; =>This Inner Loop Header: Depth=1
-; GFX1164-DPP-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1164-DPP-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_4) | instid1(VALU_DEP_2)
; GFX1164-DPP-NEXT: v_sub_f32_e32 v0, v1, v2
; GFX1164-DPP-NEXT: global_atomic_cmpswap_b32 v0, v3, v[0:1], s[0:1] glc
; GFX1164-DPP-NEXT: s_waitcnt vmcnt(0)
; GFX1164-DPP-NEXT: v_cmp_eq_u32_e32 vcc, v0, v1
; GFX1164-DPP-NEXT: v_mov_b32_e32 v1, v0
; GFX1164-DPP-NEXT: s_or_b64 s[2:3], vcc, s[2:3]
-; GFX1164-DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1164-DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1164-DPP-NEXT: s_and_not1_b64 exec, exec, s[2:3]
; GFX1164-DPP-NEXT: s_cbranch_execnz .LBB0_2
; GFX1164-DPP-NEXT: .LBB0_3:
@@ -382,7 +384,7 @@ define amdgpu_kernel void @global_atomic_fsub_uni_address_uni_value_agent_scope_
; GFX1132-DPP-NEXT: s_mov_b32 s2, 0
; GFX1132-DPP-NEXT: v_mbcnt_lo_u32_b32 v0, s3, 0
; GFX1132-DPP-NEXT: s_mov_b32 s0, exec_lo
-; GFX1132-DPP-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1132-DPP-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132-DPP-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX1132-DPP-NEXT: s_cbranch_execz .LBB0_3
; GFX1132-DPP-NEXT: ; %bb.1:
@@ -404,7 +406,7 @@ define amdgpu_kernel void @global_atomic_fsub_uni_address_uni_value_agent_scope_
; GFX1132-DPP-NEXT: v_cmp_eq_u32_e32 vcc_lo, v0, v1
; GFX1132-DPP-NEXT: v_mov_b32_e32 v1, v0
; GFX1132-DPP-NEXT: s_or_b32 s2, vcc_lo, s2
-; GFX1132-DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1132-DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1132-DPP-NEXT: s_and_not1_b32 exec_lo, exec_lo, s2
; GFX1132-DPP-NEXT: s_cbranch_execnz .LBB0_2
; GFX1132-DPP-NEXT: .LBB0_3:
@@ -726,6 +728,7 @@ define amdgpu_kernel void @global_atomic_fsub_uni_address_div_value_agent_scope_
; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
; GFX1164-NEXT: v_cmpx_eq_u32_e32 0, v0
+; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1164-NEXT: s_xor_b64 s[0:1], exec, s[0:1]
; GFX1164-NEXT: s_cbranch_execz .LBB1_5
; GFX1164-NEXT: ; %bb.3:
@@ -742,9 +745,10 @@ define amdgpu_kernel void @global_atomic_fsub_uni_address_div_value_agent_scope_
; GFX1164-NEXT: s_waitcnt vmcnt(0)
; GFX1164-NEXT: v_cmp_eq_u32_e32 vcc, v0, v1
; GFX1164-NEXT: v_mov_b32_e32 v1, v0
+; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1164-NEXT: s_or_b64 s[2:3], vcc, s[2:3]
-; GFX1164-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164-NEXT: s_and_not1_b64 exec, exec, s[2:3]
+; GFX1164-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164-NEXT: s_cbranch_execnz .LBB1_4
; GFX1164-NEXT: .LBB1_5:
; GFX1164-NEXT: s_endpgm
@@ -786,9 +790,10 @@ define amdgpu_kernel void @global_atomic_fsub_uni_address_div_value_agent_scope_
; GFX1132-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
; GFX1132-NEXT: s_mov_b32 s2, 0
; GFX1132-NEXT: s_mov_b32 s0, exec_lo
-; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX1132-NEXT: s_xor_b32 s0, exec_lo, s0
+; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1132-NEXT: s_cbranch_execz .LBB1_5
; GFX1132-NEXT: ; %bb.3:
; GFX1132-NEXT: s_load_b64 s[0:1], s[34:35], 0x24
@@ -804,7 +809,7 @@ define amdgpu_kernel void @global_atomic_fsub_uni_address_div_value_agent_scope_
; GFX1132-NEXT: v_cmp_eq_u32_e32 vcc_lo, v0, v1
; GFX1132-NEXT: v_mov_b32_e32 v1, v0
; GFX1132-NEXT: s_or_b32 s2, vcc_lo, s2
-; GFX1132-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1132-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1132-NEXT: s_and_not1_b32 exec_lo, exec_lo, s2
; GFX1132-NEXT: s_cbranch_execnz .LBB1_4
; GFX1132-NEXT: .LBB1_5:
@@ -1150,6 +1155,7 @@ define amdgpu_kernel void @global_atomic_fsub_uni_address_div_value_agent_scope_
; GFX1164-DPP-NEXT: v_mov_b32_e32 v0, v1
; GFX1164-DPP-NEXT: s_mov_b64 s[0:1], exec
; GFX1164-DPP-NEXT: v_cmpx_eq_u32_e32 0, v4
+; GFX1164-DPP-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1164-DPP-NEXT: s_cbranch_execz .LBB1_3
; GFX1164-DPP-NEXT: ; %bb.1:
; GFX1164-DPP-NEXT: s_load_b64 s[0:1], s[34:35], 0x24
@@ -1165,9 +1171,10 @@ define amdgpu_kernel void @global_atomic_fsub_uni_address_div_value_agent_scope_
; GFX1164-DPP-NEXT: s_waitcnt vmcnt(0)
; GFX1164-DPP-NEXT: v_cmp_eq_u32_e32 vcc, v4, v5
; GFX1164-DPP-NEXT: v_mov_b32_e32 v5, v4
+; GFX1164-DPP-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1164-DPP-NEXT: s_or_b64 s[2:3], vcc, s[2:3]
-; GFX1164-DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164-DPP-NEXT: s_and_not1_b64 exec, exec, s[2:3]
+; GFX1164-DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164-DPP-NEXT: s_cbranch_execnz .LBB1_2
; GFX1164-DPP-NEXT: .LBB1_3:
; GFX1164-DPP-NEXT: s_endpgm
@@ -1220,7 +1227,7 @@ define amdgpu_kernel void @global_atomic_fsub_uni_address_div_value_agent_scope_
; GFX1132-DPP-NEXT: v_mov_b32_e32 v0, v1
; GFX1132-DPP-NEXT: s_mov_b32 s2, 0
; GFX1132-DPP-NEXT: s_mov_b32 s0, exec_lo
-; GFX1132-DPP-NEXT: s_delay_alu instid0(VALU_DEP_2)
+; GFX1132-DPP-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132-DPP-NEXT: v_cmpx_eq_u32_e32 0, v4
; GFX1132-DPP-NEXT: s_cbranch_execz .LBB1_3
; GFX1132-DPP-NEXT: ; %bb.1:
@@ -1237,7 +1244,7 @@ define amdgpu_kernel void @global_atomic_fsub_uni_address_div_value_agent_scope_
; GFX1132-DPP-NEXT: v_cmp_eq_u32_e32 vcc_lo, v4, v5
; GFX1132-DPP-NEXT: v_mov_b32_e32 v5, v4
; GFX1132-DPP-NEXT: s_or_b32 s2, vcc_lo, s2
-; GFX1132-DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1132-DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1132-DPP-NEXT: s_and_not1_b32 exec_lo, exec_lo, s2
; GFX1132-DPP-NEXT: s_cbranch_execnz .LBB1_2
; GFX1132-DPP-NEXT: .LBB1_3:
@@ -1427,7 +1434,7 @@ define amdgpu_kernel void @global_atomic_fsub_uni_address_uni_value_one_as_scope
; GFX1164-NEXT: scratch_store_b32 off, v1, off
; GFX1164-NEXT: scratch_load_b64 v[0:1], off, off
; GFX1164-NEXT: v_mbcnt_hi_u32_b32 v2, exec_hi, v2
-; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164-NEXT: v_cmpx_eq_u32_e32 0, v2
; GFX1164-NEXT: s_cbranch_execz .LBB2_3
; GFX1164-NEXT: ; %bb.1:
@@ -1445,14 +1452,14 @@ define amdgpu_kernel void @global_atomic_fsub_uni_address_uni_value_one_as_scope
; GFX1164-NEXT: v_mul_f32_e32 v2, 4.0, v0
; GFX1164-NEXT: .LBB2_2: ; %atomicrmw.start
; GFX1164-NEXT: ; =>This Inner Loop Header: Depth=1
-; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_4) | instid1(VALU_DEP_2)
; GFX1164-NEXT: v_sub_f32_e32 v0, v1, v2
; GFX1164-NEXT: global_atomic_cmpswap_b32 v0, v3, v[0:1], s[0:1] glc
; GFX1164-NEXT: s_waitcnt vmcnt(0)
; GFX1164-NEXT: v_cmp_eq_u32_e32 vcc, v0, v1
; GFX1164-NEXT: v_mov_b32_e32 v1, v0
; GFX1164-NEXT: s_or_b64 s[2:3], vcc, s[2:3]
-; GFX1164-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1164-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1164-NEXT: s_and_not1_b64 exec, exec, s[2:3]
; GFX1164-NEXT: s_cbranch_execnz .LBB2_2
; GFX1164-NEXT: .LBB2_3:
@@ -1471,6 +1478,7 @@ define amdgpu_kernel void @global_atomic_fsub_uni_address_uni_value_one_as_scope
; GFX1132-NEXT: scratch_store_b32 off, v1, off
; GFX1132-NEXT: scratch_load_b64 v[0:1], off, off
; GFX1132-NEXT: v_cmpx_eq_u32_e32 0, v2
+; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1132-NEXT: s_cbranch_execz .LBB2_3
; GFX1132-NEXT: ; %bb.1:
; GFX1132-NEXT: s_waitcnt vmcnt(0)
@@ -1492,7 +1500,7 @@ define amdgpu_kernel void @global_atomic_fsub_uni_address_uni_value_one_as_scope
; GFX1132-NEXT: v_cmp_eq_u32_e32 vcc_lo, v0, v1
; GFX1132-NEXT: v_mov_b32_e32 v1, v0
; GFX1132-NEXT: s_or_b32 s2, vcc_lo, s2
-; GFX1132-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1132-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1132-NEXT: s_and_not1_b32 exec_lo, exec_lo, s2
; GFX1132-NEXT: s_cbranch_execnz .LBB2_2
; GFX1132-NEXT: .LBB2_3:
@@ -1677,7 +1685,7 @@ define amdgpu_kernel void @global_atomic_fsub_uni_address_uni_value_one_as_scope
; GFX1164-DPP-NEXT: scratch_store_b32 off, v1, off
; GFX1164-DPP-NEXT: scratch_load_b64 v[0:1], off, off
; GFX1164-DPP-NEXT: v_mbcnt_hi_u32_b32 v2, exec_hi, v2
-; GFX1164-DPP-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1164-DPP-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164-DPP-NEXT: v_cmpx_eq_u32_e32 0, v2
; GFX1164-DPP-NEXT: s_cbranch_execz .LBB2_3
; GFX1164-DPP-NEXT: ; %bb.1:
@@ -1695,14 +1703,14 @@ define amdgpu_kernel void @global_atomic_fsub_uni_address_uni_value_one_as_scope
; GFX1164-DPP-NEXT: v_mul_f32_e32 v2, 4.0, v0
; GFX1164-DPP-NEXT: .LBB2_2: ; %atomicrmw.start
; GFX1164-DPP-NEXT: ; =>This Inner Loop Header: Depth=1
-; GFX1164-DPP-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1164-DPP-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_4) | instid1(VALU_DEP_2)
; GFX1164-DPP-NEXT: v_sub_f32_e32 v0, v1, v2
; GFX1164-DPP-NEXT: global_atomic_cmpswap_b32 v0, v3, v[0:1], s[0:1] glc
; GFX1164-DPP-NEXT: s_waitcnt vmcnt(0)
; GFX1164-DPP-NEXT: v_cmp_eq_u32_e32 vcc, v0, v1
; GFX1164-DPP-NEXT: v_mov_b32_e32 v1, v0
; GFX1164-DPP-NEXT: s_or_b64 s[2:3], vcc, s[2:3]
-; GFX1164-DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1164-DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1164-DPP-NEXT: s_and_not1_b64 exec, exec, s[2:3]
; GFX1164-DPP-NEXT: s_cbranch_execnz .LBB2_2
; GFX1164-DPP-NEXT: .LBB2_3:
@@ -1721,6 +1729,7 @@ define amdgpu_kernel void @global_atomic_fsub_uni_address_uni_value_one_as_scope
; GFX1132-DPP-NEXT: scratch_store_b32 off, v1, off
; GFX1132-DPP-NEXT: scratch_load_b64 v[0:1], off, off
; GFX1132-DPP-NEXT: v_cmpx_eq_u32_e32 0, v2
+; GFX1132-DPP-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1132-DPP-NEXT: s_cbranch_execz .LBB2_3
; GFX1132-DPP-NEXT: ; %bb.1:
; GFX1132-DPP-NEXT: s_waitcnt vmcnt(0)
@@ -1742,7 +1751,7 @@ define amdgpu_kernel void @global_atomic_fsub_uni_address_uni_value_one_as_scope
; GFX1132-DPP-NEXT: v_cmp_eq_u32_e32 vcc_lo, v0, v1
; GFX1132-DPP-NEXT: v_mov_b32_e32 v1, v0
; GFX1132-DPP-NEXT: s_or_b32 s2, vcc_lo, s2
-; GFX1132-DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1132-DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1132-DPP-NEXT: s_and_not1_b32 exec_lo, exec_lo, s2
; GFX1132-DPP-NEXT: s_cbranch_execnz .LBB2_2
; GFX1132-DPP-NEXT: .LBB2_3:
@@ -2065,6 +2074,7 @@ define amdgpu_kernel void @global_atomic_fsub_uni_address_div_value_one_as_scope
; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
; GFX1164-NEXT: v_cmpx_eq_u32_e32 0, v0
+; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1164-NEXT: s_xor_b64 s[0:1], exec, s[0:1]
; GFX1164-NEXT: s_cbranch_execz .LBB3_5
; GFX1164-NEXT: ; %bb.3:
@@ -2081,9 +2091,10 @@ define amdgpu_kernel void @global_atomic_fsub_uni_address_div_value_one_as_scope
; GFX1164-NEXT: s_waitcnt vmcnt(0)
; GFX1164-NEXT: v_cmp_eq_u32_e32 vcc, v0, v1
; GFX1164-NEXT: v_mov_b32_e32 v1, v0
+; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1164-NEXT: s_or_b64 s[2:3], vcc, s[2:3]
-; GFX1164-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164-NEXT: s_and_not1_b64 exec, exec, s[2:3]
+; GFX1164-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164-NEXT: s_cbranch_execnz .LBB3_4
; GFX1164-NEXT: .LBB3_5:
; GFX1164-NEXT: s_endpgm
@@ -2125,9 +2136,10 @@ define amdgpu_kernel void @global_atomic_fsub_uni_address_div_value_one_as_scope
; GFX1132-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
; GFX1132-NEXT: s_mov_b32 s2, 0
; GFX1132-NEXT: s_mov_b32 s0, exec_lo
-; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX1132-NEXT: s_xor_b32 s0, exec_lo, s0
+; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1132-NEXT: s_cbranch_execz .LBB3_5
; GFX1132-NEXT: ; %bb.3:
; GFX1132-NEXT: s_load_b64 s[0:1], s[34:35], 0x24
@@ -2143,7 +2155,7 @@ define amdgpu_kernel void @global_atomic_fsub_uni_address_div_value_one_as_scope
; GFX1132-NEXT: v_cmp_eq_u32_e32 vcc_lo, v0, v1
; GFX1132-NEXT: v_mov_b32_e32 v1, v0
; GFX1132-NEXT: s_or_b32 s2, vcc_lo, s2
-; GFX1132-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1132-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1132-NEXT: s_and_not1_b32 exec_lo, exec_lo, s2
; GFX1132-NEXT: s_cbranch_execnz .LBB3_4
; GFX1132-NEXT: .LBB3_5:
@@ -2489,6 +2501,7 @@ define amdgpu_kernel void @global_atomic_fsub_uni_address_div_value_one_as_scope
; GFX1164-DPP-NEXT: v_mov_b32_e32 v0, v1
; GFX1164-DPP-NEXT: s_mov_b64 s[0:1], exec
; GFX1164-DPP-NEXT: v_cmpx_eq_u32_e32 0, v4
+; GFX1164-DPP-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1164-DPP-NEXT: s_cbranch_execz .LBB3_3
; GFX1164-DPP-NEXT: ; %bb.1:
; GFX1164-DPP-NEXT: s_load_b64 s[0:1], s[34:35], 0x24
@@ -2504,9 +2517,10 @@ define amdgpu_kernel void @global_atomic_fsub_uni_address_div_value_one_as_scope
; GFX1164-DPP-NEXT: s_waitcnt vmcnt(0)
; GFX1164-DPP-NEXT: v_cmp_eq_u32_e32 vcc, v4, v5
; GFX1164-DPP-NEXT: v_mov_b32_e32 v5, v4
+; GFX1164-DPP-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1164-DPP-NEXT: s_or_b64 s[2:3], vcc, s[2:3]
-; GFX1164-DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164-DPP-NEXT: s_and_not1_b64 exec, exec, s[2:3]
+; GFX1164-DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164-DPP-NEXT: s_cbranch_execnz .LBB3_2
; GFX1164-DPP-NEXT: .LBB3_3:
; GFX1164-DPP-NEXT: s_endpgm
@@ -2559,7 +2573,7 @@ define amdgpu_kernel void @global_atomic_fsub_uni_address_div_value_one_as_scope
; GFX1132-DPP-NEXT: v_mov_b32_e32 v0, v1
; GFX1132-DPP-NEXT: s_mov_b32 s2, 0
; GFX1132-DPP-NEXT: s_mov_b32 s0, exec_lo
-; GFX1132-DPP-NEXT: s_delay_alu instid0(VALU_DEP_2)
+; GFX1132-DPP-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132-DPP-NEXT: v_cmpx_eq_u32_e32 0, v4
; GFX1132-DPP-NEXT: s_cbranch_execz .LBB3_3
; GFX1132-DPP-NEXT: ; %bb.1:
@@ -2576,7 +2590,7 @@ define amdgpu_kernel void @global_atomic_fsub_uni_address_div_value_one_as_scope
; GFX1132-DPP-NEXT: v_cmp_eq_u32_e32 vcc_lo, v4, v5
; GFX1132-DPP-NEXT: v_mov_b32_e32 v5, v4
; GFX1132-DPP-NEXT: s_or_b32 s2, vcc_lo, s2
-; GFX1132-DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1132-DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1132-DPP-NEXT: s_and_not1_b32 exec_lo, exec_lo, s2
; GFX1132-DPP-NEXT: s_cbranch_execnz .LBB3_2
; GFX1132-DPP-NEXT: .LBB3_3:
@@ -2766,7 +2780,7 @@ define amdgpu_kernel void @global_atomic_fsub_uni_address_uni_value_agent_scope_
; GFX1164-NEXT: scratch_store_b32 off, v1, off
; GFX1164-NEXT: scratch_load_b64 v[0:1], off, off
; GFX1164-NEXT: v_mbcnt_hi_u32_b32 v2, exec_hi, v2
-; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164-NEXT: v_cmpx_eq_u32_e32 0, v2
; GFX1164-NEXT: s_cbranch_execz .LBB4_3
; GFX1164-NEXT: ; %bb.1:
@@ -2784,14 +2798,14 @@ define amdgpu_kernel void @global_atomic_fsub_uni_address_uni_value_agent_scope_
; GFX1164-NEXT: v_mul_f32_e32 v2, 4.0, v0
; GFX1164-NEXT: .LBB4_2: ; %atomicrmw.start
; GFX1164-NEXT: ; =>This Inner Loop Header: Depth=1
-; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_4) | instid1(VALU_DEP_2)
; GFX1164-NEXT: v_sub_f32_e32 v0, v1, v2
; GFX1164-NEXT: global_atomic_cmpswap_b32 v0, v3, v[0:1], s[0:1] glc
; GFX1164-NEXT: s_waitcnt vmcnt(0)
; GFX1164-NEXT: v_cmp_eq_u32_e32 vcc, v0, v1
; GFX1164-NEXT: v_mov_b32_e32 v1, v0
; GFX1164-NEXT: s_or_b64 s[2:3], vcc, s[2:3]
-; GFX1164-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1164-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1164-NEXT: s_and_not1_b64 exec, exec, s[2:3]
; GFX1164-NEXT: s_cbranch_execnz .LBB4_2
; GFX1164-NEXT: .LBB4_3:
@@ -2810,6 +2824,7 @@ define amdgpu_kernel void @global_atomic_fsub_uni_address_uni_value_agent_scope_
; GFX1132-NEXT: scratch_store_b32 off, v1, off
; GFX1132-NEXT: scratch_load_b64 v[0:1], off, off
; GFX1132-NEXT: v_cmpx_eq_u32_e32 0, v2
+; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1132-NEXT: s_cbranch_execz .LBB4_3
; GFX1132-NEXT: ; %bb.1:
; GFX1132-NEXT: s_waitcnt vmcnt(0)
@@ -2831,7 +2846,7 @@ define amdgpu_kernel void @global_atomic_fsub_uni_address_uni_value_agent_scope_
; GFX1132-NEXT: v_cmp_eq_u32_e32 vcc_lo, v0, v1
; GFX1132-NEXT: v_mov_b32_e32 v1, v0
; GFX1132-NEXT: s_or_b32 s2, vcc_lo, s2
-; GFX1132-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1132-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1132-NEXT: s_and_not1_b32 exec_lo, exec_lo, s2
; GFX1132-NEXT: s_cbranch_execnz .LBB4_2
; GFX1132-NEXT: .LBB4_3:
@@ -3016,7 +3031,7 @@ define amdgpu_kernel void @global_atomic_fsub_uni_address_uni_value_agent_scope_
; GFX1164-DPP-NEXT: scratch_store_b32 off, v1, off
; GFX1164-DPP-NEXT: scratch_load_b64 v[0:1], off, off
; GFX1164-DPP-NEXT: v_mbcnt_hi_u32_b32 v2, exec_hi, v2
-; GFX1164-DPP-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1164-DPP-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164-DPP-NEXT: v_cmpx_eq_u32_e32 0, v2
; GFX1164-DPP-NEXT: s_cbranch_execz .LBB4_3
; GFX1164-DPP-NEXT: ; %bb.1:
@@ -3034,14 +3049,14 @@ define amdgpu_kernel void @global_atomic_fsub_uni_address_uni_value_agent_scope_
; GFX1164-DPP-NEXT: v_mul_f32_e32 v2, 4.0, v0
; GFX1164-DPP-NEXT: .LBB4_2: ; %atomicrmw.start
; GFX1164-DPP-NEXT: ; =>This Inner Loop Header: Depth=1
-; GFX1164-DPP-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1164-DPP-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_4) | instid1(VALU_DEP_2)
; GFX1164-DPP-NEXT: v_sub_f32_e32 v0, v1, v2
; GFX1164-DPP-NEXT: global_atomic_cmpswap_b32 v0, v3, v[0:1], s[0:1] glc
; GFX1164-DPP-NEXT: s_waitcnt vmcnt(0)
; GFX1164-DPP-NEXT: v_cmp_eq_u32_e32 vcc, v0, v1
; GFX1164-DPP-NEXT: v_mov_b32_e32 v1, v0
; GFX1164-DPP-NEXT: s_or_b64 s[2:3], vcc, s[2:3]
-; GFX1164-DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1164-DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1164-DPP-NEXT: s_and_not1_b64 exec, exec, s[2:3]
; GFX1164-DPP-NEXT: s_cbranch_execnz .LBB4_2
; GFX1164-DPP-NEXT: .LBB4_3:
@@ -3060,6 +3075,7 @@ define amdgpu_kernel void @global_atomic_fsub_uni_address_uni_value_agent_scope_
; GFX1132-DPP-NEXT: scratch_store_b32 off, v1, off
; GFX1132-DPP-NEXT: scratch_load_b64 v[0:1], off, off
; GFX1132-DPP-NEXT: v_cmpx_eq_u32_e32 0, v2
+; GFX1132-DPP-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1132-DPP-NEXT: s_cbranch_execz .LBB4_3
; GFX1132-DPP-NEXT: ; %bb.1:
; GFX1132-DPP-NEXT: s_waitcnt vmcnt(0)
@@ -3081,7 +3097,7 @@ define amdgpu_kernel void @global_atomic_fsub_uni_address_uni_value_agent_scope_
; GFX1132-DPP-NEXT: v_cmp_eq_u32_e32 vcc_lo, v0, v1
; GFX1132-DPP-NEXT: v_mov_b32_e32 v1, v0
; GFX1132-DPP-NEXT: s_or_b32 s2, vcc_lo, s2
-; GFX1132-DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1132-DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1132-DPP-NEXT: s_and_not1_b32 exec_lo, exec_lo, s2
; GFX1132-DPP-NEXT: s_cbranch_execnz .LBB4_2
; GFX1132-DPP-NEXT: .LBB4_3:
@@ -3404,6 +3420,7 @@ define amdgpu_kernel void @global_atomic_fsub_uni_address_div_value_agent_scope_
; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
; GFX1164-NEXT: v_cmpx_eq_u32_e32 0, v0
+; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1164-NEXT: s_xor_b64 s[0:1], exec, s[0:1]
; GFX1164-NEXT: s_cbranch_execz .LBB5_5
; GFX1164-NEXT: ; %bb.3:
@@ -3420,9 +3437,10 @@ define amdgpu_kernel void @global_atomic_fsub_uni_address_div_value_agent_scope_
; GFX1164-NEXT: s_waitcnt vmcnt(0)
; GFX1164-NEXT: v_cmp_eq_u32_e32 vcc, v0, v1
; GFX1164-NEXT: v_mov_b32_e32 v1, v0
+; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1164-NEXT: s_or_b64 s[2:3], vcc, s[2:3]
-; GFX1164-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164-NEXT: s_and_not1_b64 exec, exec, s[2:3]
+; GFX1164-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164-NEXT: s_cbranch_execnz .LBB5_4
; GFX1164-NEXT: .LBB5_5:
; GFX1164-NEXT: s_endpgm
@@ -3464,9 +3482,10 @@ define amdgpu_kernel void @global_atomic_fsub_uni_address_div_value_agent_scope_
; GFX1132-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
; GFX1132-NEXT: s_mov_b32 s2, 0
; GFX1132-NEXT: s_mov_b32 s0, exec_lo
-; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX1132-NEXT: s_xor_b32 s0, exec_lo, s0
+; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1132-NEXT: s_cbranch_execz .LBB5_5
; GFX1132-NEXT: ; %bb.3:
; GFX1132-NEXT: s_load_b64 s[0:1], s[34:35], 0x24
@@ -3482,7 +3501,7 @@ define amdgpu_kernel void @global_atomic_fsub_uni_address_div_value_agent_scope_
; GFX1132-NEXT: v_cmp_eq_u32_e32 vcc_lo, v0, v1
; GFX1132-NEXT: v_mov_b32_e32 v1, v0
; GFX1132-NEXT: s_or_b32 s2, vcc_lo, s2
-; GFX1132-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1132-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1132-NEXT: s_and_not1_b32 exec_lo, exec_lo, s2
; GFX1132-NEXT: s_cbranch_execnz .LBB5_4
; GFX1132-NEXT: .LBB5_5:
@@ -3828,6 +3847,7 @@ define amdgpu_kernel void @global_atomic_fsub_uni_address_div_value_agent_scope_
; GFX1164-DPP-NEXT: v_mov_b32_e32 v0, v1
; GFX1164-DPP-NEXT: s_mov_b64 s[0:1], exec
; GFX1164-DPP-NEXT: v_cmpx_eq_u32_e32 0, v4
+; GFX1164-DPP-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1164-DPP-NEXT: s_cbranch_execz .LBB5_3
; GFX1164-DPP-NEXT: ; %bb.1:
; GFX1164-DPP-NEXT: s_load_b64 s[0:1], s[34:35], 0x24
@@ -3843,9 +3863,10 @@ define amdgpu_kernel void @global_atomic_fsub_uni_address_div_value_agent_scope_
; GFX1164-DPP-NEXT: s_waitcnt vmcnt(0)
; GFX1164-DPP-NEXT: v_cmp_eq_u32_e32 vcc, v4, v5
; GFX1164-DPP-NEXT: v_mov_b32_e32 v5, v4
+; GFX1164-DPP-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1164-DPP-NEXT: s_or_b64 s[2:3], vcc, s[2:3]
-; GFX1164-DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164-DPP-NEXT: s_and_not1_b64 exec, exec, s[2:3]
+; GFX1164-DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164-DPP-NEXT: s_cbranch_execnz .LBB5_2
; GFX1164-DPP-NEXT: .LBB5_3:
; GFX1164-DPP-NEXT: s_endpgm
@@ -3898,7 +3919,7 @@ define amdgpu_kernel void @global_atomic_fsub_uni_address_div_value_agent_scope_
; GFX1132-DPP-NEXT: v_mov_b32_e32 v0, v1
; GFX1132-DPP-NEXT: s_mov_b32 s2, 0
; GFX1132-DPP-NEXT: s_mov_b32 s0, exec_lo
-; GFX1132-DPP-NEXT: s_delay_alu instid0(VALU_DEP_2)
+; GFX1132-DPP-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132-DPP-NEXT: v_cmpx_eq_u32_e32 0, v4
; GFX1132-DPP-NEXT: s_cbranch_execz .LBB5_3
; GFX1132-DPP-NEXT: ; %bb.1:
@@ -3915,7 +3936,7 @@ define amdgpu_kernel void @global_atomic_fsub_uni_address_div_value_agent_scope_
; GFX1132-DPP-NEXT: v_cmp_eq_u32_e32 vcc_lo, v4, v5
; GFX1132-DPP-NEXT: v_mov_b32_e32 v5, v4
; GFX1132-DPP-NEXT: s_or_b32 s2, vcc_lo, s2
-; GFX1132-DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1132-DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1132-DPP-NEXT: s_and_not1_b32 exec_lo, exec_lo, s2
; GFX1132-DPP-NEXT: s_cbranch_execnz .LBB5_2
; GFX1132-DPP-NEXT: .LBB5_3:
@@ -4239,6 +4260,7 @@ define amdgpu_kernel void @global_atomic_fsub_uni_address_div_value_agent_scope_
; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
; GFX1164-NEXT: v_cmpx_eq_u32_e32 0, v0
+; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1164-NEXT: s_xor_b64 s[0:1], exec, s[0:1]
; GFX1164-NEXT: s_cbranch_execz .LBB6_5
; GFX1164-NEXT: ; %bb.3:
@@ -4255,9 +4277,10 @@ define amdgpu_kernel void @global_atomic_fsub_uni_address_div_value_agent_scope_
; GFX1164-NEXT: s_waitcnt vmcnt(0)
; GFX1164-NEXT: v_cmp_eq_u32_e32 vcc, v0, v1
; GFX1164-NEXT: v_mov_b32_e32 v1, v0
+; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1164-NEXT: s_or_b64 s[2:3], vcc, s[2:3]
-; GFX1164-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164-NEXT: s_and_not1_b64 exec, exec, s[2:3]
+; GFX1164-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164-NEXT: s_cbranch_execnz .LBB6_4
; GFX1164-NEXT: .LBB6_5:
; GFX1164-NEXT: s_endpgm
@@ -4299,9 +4322,10 @@ define amdgpu_kernel void @global_atomic_fsub_uni_address_div_value_agent_scope_
; GFX1132-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
; GFX1132-NEXT: s_mov_b32 s2, 0
; GFX1132-NEXT: s_mov_b32 s0, exec_lo
-; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX1132-NEXT: s_xor_b32 s0, exec_lo, s0
+; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1132-NEXT: s_cbranch_execz .LBB6_5
; GFX1132-NEXT: ; %bb.3:
; GFX1132-NEXT: s_load_b64 s[0:1], s[34:35], 0x24
@@ -4317,7 +4341,7 @@ define amdgpu_kernel void @global_atomic_fsub_uni_address_div_value_agent_scope_
; GFX1132-NEXT: v_cmp_eq_u32_e32 vcc_lo, v0, v1
; GFX1132-NEXT: v_mov_b32_e32 v1, v0
; GFX1132-NEXT: s_or_b32 s2, vcc_lo, s2
-; GFX1132-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1132-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1132-NEXT: s_and_not1_b32 exec_lo, exec_lo, s2
; GFX1132-NEXT: s_cbranch_execnz .LBB6_4
; GFX1132-NEXT: .LBB6_5:
@@ -4663,6 +4687,7 @@ define amdgpu_kernel void @global_atomic_fsub_uni_address_div_value_agent_scope_
; GFX1164-DPP-NEXT: v_mov_b32_e32 v0, v1
; GFX1164-DPP-NEXT: s_mov_b64 s[0:1], exec
; GFX1164-DPP-NEXT: v_cmpx_eq_u32_e32 0, v4
+; GFX1164-DPP-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1164-DPP-NEXT: s_cbranch_execz .LBB6_3
; GFX1164-DPP-NEXT: ; %bb.1:
; GFX1164-DPP-NEXT: s_load_b64 s[0:1], s[34:35], 0x24
@@ -4678,9 +4703,10 @@ define amdgpu_kernel void @global_atomic_fsub_uni_address_div_value_agent_scope_
; GFX1164-DPP-NEXT: s_waitcnt vmcnt(0)
; GFX1164-DPP-NEXT: v_cmp_eq_u32_e32 vcc, v4, v5
; GFX1164-DPP-NEXT: v_mov_b32_e32 v5, v4
+; GFX1164-DPP-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1164-DPP-NEXT: s_or_b64 s[2:3], vcc, s[2:3]
-; GFX1164-DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164-DPP-NEXT: s_and_not1_b64 exec, exec, s[2:3]
+; GFX1164-DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164-DPP-NEXT: s_cbranch_execnz .LBB6_2
; GFX1164-DPP-NEXT: .LBB6_3:
; GFX1164-DPP-NEXT: s_endpgm
@@ -4733,7 +4759,7 @@ define amdgpu_kernel void @global_atomic_fsub_uni_address_div_value_agent_scope_
; GFX1132-DPP-NEXT: v_mov_b32_e32 v0, v1
; GFX1132-DPP-NEXT: s_mov_b32 s2, 0
; GFX1132-DPP-NEXT: s_mov_b32 s0, exec_lo
-; GFX1132-DPP-NEXT: s_delay_alu instid0(VALU_DEP_2)
+; GFX1132-DPP-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132-DPP-NEXT: v_cmpx_eq_u32_e32 0, v4
; GFX1132-DPP-NEXT: s_cbranch_execz .LBB6_3
; GFX1132-DPP-NEXT: ; %bb.1:
@@ -4750,7 +4776,7 @@ define amdgpu_kernel void @global_atomic_fsub_uni_address_div_value_agent_scope_
; GFX1132-DPP-NEXT: v_cmp_eq_u32_e32 vcc_lo, v4, v5
; GFX1132-DPP-NEXT: v_mov_b32_e32 v5, v4
; GFX1132-DPP-NEXT: s_or_b32 s2, vcc_lo, s2
-; GFX1132-DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1132-DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1132-DPP-NEXT: s_and_not1_b32 exec_lo, exec_lo, s2
; GFX1132-DPP-NEXT: s_cbranch_execnz .LBB6_2
; GFX1132-DPP-NEXT: .LBB6_3:
@@ -4940,7 +4966,7 @@ define amdgpu_kernel void @global_atomic_fsub_uni_address_uni_value_default_scop
; GFX1164-NEXT: scratch_store_b32 off, v1, off
; GFX1164-NEXT: scratch_load_b64 v[0:1], off, off
; GFX1164-NEXT: v_mbcnt_hi_u32_b32 v2, exec_hi, v2
-; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164-NEXT: v_cmpx_eq_u32_e32 0, v2
; GFX1164-NEXT: s_cbranch_execz .LBB7_3
; GFX1164-NEXT: ; %bb.1:
@@ -4958,14 +4984,14 @@ define amdgpu_kernel void @global_atomic_fsub_uni_address_uni_value_default_scop
; GFX1164-NEXT: v_mul_f32_e32 v2, 4.0, v0
; GFX1164-NEXT: .LBB7_2: ; %atomicrmw.start
; GFX1164-NEXT: ; =>This Inner Loop Header: Depth=1
-; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_4) | instid1(VALU_DEP_2)
; GFX1164-NEXT: v_sub_f32_e32 v0, v1, v2
; GFX1164-NEXT: global_atomic_cmpswap_b32 v0, v3, v[0:1], s[0:1] glc
; GFX1164-NEXT: s_waitcnt vmcnt(0)
; GFX1164-NEXT: v_cmp_eq_u32_e32 vcc, v0, v1
; GFX1164-NEXT: v_mov_b32_e32 v1, v0
; GFX1164-NEXT: s_or_b64 s[2:3], vcc, s[2:3]
-; GFX1164-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1164-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1164-NEXT: s_and_not1_b64 exec, exec, s[2:3]
; GFX1164-NEXT: s_cbranch_execnz .LBB7_2
; GFX1164-NEXT: .LBB7_3:
@@ -4984,6 +5010,7 @@ define amdgpu_kernel void @global_atomic_fsub_uni_address_uni_value_default_scop
; GFX1132-NEXT: scratch_store_b32 off, v1, off
; GFX1132-NEXT: scratch_load_b64 v[0:1], off, off
; GFX1132-NEXT: v_cmpx_eq_u32_e32 0, v2
+; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1132-NEXT: s_cbranch_execz .LBB7_3
; GFX1132-NEXT: ; %bb.1:
; GFX1132-NEXT: s_waitcnt vmcnt(0)
@@ -5005,7 +5032,7 @@ define amdgpu_kernel void @global_atomic_fsub_uni_address_uni_value_default_scop
; GFX1132-NEXT: v_cmp_eq_u32_e32 vcc_lo, v0, v1
; GFX1132-NEXT: v_mov_b32_e32 v1, v0
; GFX1132-NEXT: s_or_b32 s2, vcc_lo, s2
-; GFX1132-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1132-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1132-NEXT: s_and_not1_b32 exec_lo, exec_lo, s2
; GFX1132-NEXT: s_cbranch_execnz .LBB7_2
; GFX1132-NEXT: .LBB7_3:
@@ -5190,7 +5217,7 @@ define amdgpu_kernel void @global_atomic_fsub_uni_address_uni_value_default_scop
; GFX1164-DPP-NEXT: scratch_store_b32 off, v1, off
; GFX1164-DPP-NEXT: scratch_load_b64 v[0:1], off, off
; GFX1164-DPP-NEXT: v_mbcnt_hi_u32_b32 v2, exec_hi, v2
-; GFX1164-DPP-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1164-DPP-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164-DPP-NEXT: v_cmpx_eq_u32_e32 0, v2
; GFX1164-DPP-NEXT: s_cbranch_execz .LBB7_3
; GFX1164-DPP-NEXT: ; %bb.1:
@@ -5208,14 +5235,14 @@ define amdgpu_kernel void @global_atomic_fsub_uni_address_uni_value_default_scop
; GFX1164-DPP-NEXT: v_mul_f32_e32 v2, 4.0, v0
; GFX1164-DPP-NEXT: .LBB7_2: ; %atomicrmw.start
; GFX1164-DPP-NEXT: ; =>This Inner Loop Header: Depth=1
-; GFX1164-DPP-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1164-DPP-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_4) | instid1(VALU_DEP_2)
; GFX1164-DPP-NEXT: v_sub_f32_e32 v0, v1, v2
; GFX1164-DPP-NEXT: global_atomic_cmpswap_b32 v0, v3, v[0:1], s[0:1] glc
; GFX1164-DPP-NEXT: s_waitcnt vmcnt(0)
; GFX1164-DPP-NEXT: v_cmp_eq_u32_e32 vcc, v0, v1
; GFX1164-DPP-NEXT: v_mov_b32_e32 v1, v0
; GFX1164-DPP-NEXT: s_or_b64 s[2:3], vcc, s[2:3]
-; GFX1164-DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1164-DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1164-DPP-NEXT: s_and_not1_b64 exec, exec, s[2:3]
; GFX1164-DPP-NEXT: s_cbranch_execnz .LBB7_2
; GFX1164-DPP-NEXT: .LBB7_3:
@@ -5234,6 +5261,7 @@ define amdgpu_kernel void @global_atomic_fsub_uni_address_uni_value_default_scop
; GFX1132-DPP-NEXT: scratch_store_b32 off, v1, off
; GFX1132-DPP-NEXT: scratch_load_b64 v[0:1], off, off
; GFX1132-DPP-NEXT: v_cmpx_eq_u32_e32 0, v2
+; GFX1132-DPP-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1132-DPP-NEXT: s_cbranch_execz .LBB7_3
; GFX1132-DPP-NEXT: ; %bb.1:
; GFX1132-DPP-NEXT: s_waitcnt vmcnt(0)
@@ -5255,7 +5283,7 @@ define amdgpu_kernel void @global_atomic_fsub_uni_address_uni_value_default_scop
; GFX1132-DPP-NEXT: v_cmp_eq_u32_e32 vcc_lo, v0, v1
; GFX1132-DPP-NEXT: v_mov_b32_e32 v1, v0
; GFX1132-DPP-NEXT: s_or_b32 s2, vcc_lo, s2
-; GFX1132-DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1132-DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1132-DPP-NEXT: s_and_not1_b32 exec_lo, exec_lo, s2
; GFX1132-DPP-NEXT: s_cbranch_execnz .LBB7_2
; GFX1132-DPP-NEXT: .LBB7_3:
@@ -5577,6 +5605,7 @@ define amdgpu_kernel void @global_atomic_fsub_uni_address_div_value_default_scop
; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
; GFX1164-NEXT: v_cmpx_eq_u32_e32 0, v0
+; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1164-NEXT: s_xor_b64 s[0:1], exec, s[0:1]
; GFX1164-NEXT: s_cbranch_execz .LBB8_5
; GFX1164-NEXT: ; %bb.3:
@@ -5593,9 +5622,10 @@ define amdgpu_kernel void @global_atomic_fsub_uni_address_div_value_default_scop
; GFX1164-NEXT: s_waitcnt vmcnt(0)
; GFX1164-NEXT: v_cmp_eq_u32_e32 vcc, v0, v1
; GFX1164-NEXT: v_mov_b32_e32 v1, v0
+; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1164-NEXT: s_or_b64 s[2:3], vcc, s[2:3]
-; GFX1164-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164-NEXT: s_and_not1_b64 exec, exec, s[2:3]
+; GFX1164-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164-NEXT: s_cbranch_execnz .LBB8_4
; GFX1164-NEXT: .LBB8_5:
; GFX1164-NEXT: s_endpgm
@@ -5637,9 +5667,10 @@ define amdgpu_kernel void @global_atomic_fsub_uni_address_div_value_default_scop
; GFX1132-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
; GFX1132-NEXT: s_mov_b32 s2, 0
; GFX1132-NEXT: s_mov_b32 s0, exec_lo
-; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX1132-NEXT: s_xor_b32 s0, exec_lo, s0
+; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1132-NEXT: s_cbranch_execz .LBB8_5
; GFX1132-NEXT: ; %bb.3:
; GFX1132-NEXT: s_load_b64 s[0:1], s[34:35], 0x24
@@ -5655,7 +5686,7 @@ define amdgpu_kernel void @global_atomic_fsub_uni_address_div_value_default_scop
; GFX1132-NEXT: v_cmp_eq_u32_e32 vcc_lo, v0, v1
; GFX1132-NEXT: v_mov_b32_e32 v1, v0
; GFX1132-NEXT: s_or_b32 s2, vcc_lo, s2
-; GFX1132-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1132-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1132-NEXT: s_and_not1_b32 exec_lo, exec_lo, s2
; GFX1132-NEXT: s_cbranch_execnz .LBB8_4
; GFX1132-NEXT: .LBB8_5:
@@ -6001,6 +6032,7 @@ define amdgpu_kernel void @global_atomic_fsub_uni_address_div_value_default_scop
; GFX1164-DPP-NEXT: v_mov_b32_e32 v0, v1
; GFX1164-DPP-NEXT: s_mov_b64 s[0:1], exec
; GFX1164-DPP-NEXT: v_cmpx_eq_u32_e32 0, v4
+; GFX1164-DPP-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1164-DPP-NEXT: s_cbranch_execz .LBB8_3
; GFX1164-DPP-NEXT: ; %bb.1:
; GFX1164-DPP-NEXT: s_load_b64 s[0:1], s[34:35], 0x24
@@ -6016,9 +6048,10 @@ define amdgpu_kernel void @global_atomic_fsub_uni_address_div_value_default_scop
; GFX1164-DPP-NEXT: s_waitcnt vmcnt(0)
; GFX1164-DPP-NEXT: v_cmp_eq_u32_e32 vcc, v4, v5
; GFX1164-DPP-NEXT: v_mov_b32_e32 v5, v4
+; GFX1164-DPP-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1164-DPP-NEXT: s_or_b64 s[2:3], vcc, s[2:3]
-; GFX1164-DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164-DPP-NEXT: s_and_not1_b64 exec, exec, s[2:3]
+; GFX1164-DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164-DPP-NEXT: s_cbranch_execnz .LBB8_2
; GFX1164-DPP-NEXT: .LBB8_3:
; GFX1164-DPP-NEXT: s_endpgm
@@ -6071,7 +6104,7 @@ define amdgpu_kernel void @global_atomic_fsub_uni_address_div_value_default_scop
; GFX1132-DPP-NEXT: v_mov_b32_e32 v0, v1
; GFX1132-DPP-NEXT: s_mov_b32 s2, 0
; GFX1132-DPP-NEXT: s_mov_b32 s0, exec_lo
-; GFX1132-DPP-NEXT: s_delay_alu instid0(VALU_DEP_2)
+; GFX1132-DPP-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132-DPP-NEXT: v_cmpx_eq_u32_e32 0, v4
; GFX1132-DPP-NEXT: s_cbranch_execz .LBB8_3
; GFX1132-DPP-NEXT: ; %bb.1:
@@ -6088,7 +6121,7 @@ define amdgpu_kernel void @global_atomic_fsub_uni_address_div_value_default_scop
; GFX1132-DPP-NEXT: v_cmp_eq_u32_e32 vcc_lo, v4, v5
; GFX1132-DPP-NEXT: v_mov_b32_e32 v5, v4
; GFX1132-DPP-NEXT: s_or_b32 s2, vcc_lo, s2
-; GFX1132-DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1132-DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1132-DPP-NEXT: s_and_not1_b32 exec_lo, exec_lo, s2
; GFX1132-DPP-NEXT: s_cbranch_execnz .LBB8_2
; GFX1132-DPP-NEXT: .LBB8_3:
@@ -6248,6 +6281,7 @@ define amdgpu_kernel void @global_atomic_fsub_double_uni_address_uni_value_agent
; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164-NEXT: v_mbcnt_hi_u32_b32 v0, s1, v0
; GFX1164-NEXT: v_cmpx_eq_u32_e32 0, v0
+; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1164-NEXT: s_cbranch_execz .LBB9_3
; GFX1164-NEXT: ; %bb.1:
; GFX1164-NEXT: s_bcnt1_i32_b64 s0, s[0:1]
@@ -6271,9 +6305,10 @@ define amdgpu_kernel void @global_atomic_fsub_double_uni_address_uni_value_agent
; GFX1164-NEXT: v_cmp_eq_u64_e32 vcc, v[0:1], v[2:3]
; GFX1164-NEXT: v_mov_b32_e32 v3, v1
; GFX1164-NEXT: v_mov_b32_e32 v2, v0
+; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1164-NEXT: s_or_b64 s[2:3], vcc, s[2:3]
-; GFX1164-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164-NEXT: s_and_not1_b64 exec, exec, s[2:3]
+; GFX1164-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164-NEXT: s_cbranch_execnz .LBB9_2
; GFX1164-NEXT: .LBB9_3:
; GFX1164-NEXT: s_endpgm
@@ -6284,7 +6319,7 @@ define amdgpu_kernel void @global_atomic_fsub_double_uni_address_uni_value_agent
; GFX1132-NEXT: s_mov_b32 s2, 0
; GFX1132-NEXT: v_mbcnt_lo_u32_b32 v0, s0, 0
; GFX1132-NEXT: s_mov_b32 s1, exec_lo
-; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX1132-NEXT: s_cbranch_execz .LBB9_3
; GFX1132-NEXT: ; %bb.1:
@@ -6307,7 +6342,7 @@ define amdgpu_kernel void @global_atomic_fsub_double_uni_address_uni_value_agent
; GFX1132-NEXT: v_cmp_eq_u64_e32 vcc_lo, v[0:1], v[2:3]
; GFX1132-NEXT: v_dual_mov_b32 v3, v1 :: v_dual_mov_b32 v2, v0
; GFX1132-NEXT: s_or_b32 s2, vcc_lo, s2
-; GFX1132-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1132-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1132-NEXT: s_and_not1_b32 exec_lo, exec_lo, s2
; GFX1132-NEXT: s_cbranch_execnz .LBB9_2
; GFX1132-NEXT: .LBB9_3:
@@ -6462,6 +6497,7 @@ define amdgpu_kernel void @global_atomic_fsub_double_uni_address_uni_value_agent
; GFX1164-DPP-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164-DPP-NEXT: v_mbcnt_hi_u32_b32 v0, s1, v0
; GFX1164-DPP-NEXT: v_cmpx_eq_u32_e32 0, v0
+; GFX1164-DPP-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1164-DPP-NEXT: s_cbranch_execz .LBB9_3
; GFX1164-DPP-NEXT: ; %bb.1:
; GFX1164-DPP-NEXT: s_bcnt1_i32_b64 s0, s[0:1]
@@ -6485,9 +6521,10 @@ define amdgpu_kernel void @global_atomic_fsub_double_uni_address_uni_value_agent
; GFX1164-DPP-NEXT: v_cmp_eq_u64_e32 vcc, v[0:1], v[2:3]
; GFX1164-DPP-NEXT: v_mov_b32_e32 v3, v1
; GFX1164-DPP-NEXT: v_mov_b32_e32 v2, v0
+; GFX1164-DPP-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1164-DPP-NEXT: s_or_b64 s[2:3], vcc, s[2:3]
-; GFX1164-DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164-DPP-NEXT: s_and_not1_b64 exec, exec, s[2:3]
+; GFX1164-DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164-DPP-NEXT: s_cbranch_execnz .LBB9_2
; GFX1164-DPP-NEXT: .LBB9_3:
; GFX1164-DPP-NEXT: s_endpgm
@@ -6498,7 +6535,7 @@ define amdgpu_kernel void @global_atomic_fsub_double_uni_address_uni_value_agent
; GFX1132-DPP-NEXT: s_mov_b32 s2, 0
; GFX1132-DPP-NEXT: v_mbcnt_lo_u32_b32 v0, s0, 0
; GFX1132-DPP-NEXT: s_mov_b32 s1, exec_lo
-; GFX1132-DPP-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1132-DPP-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132-DPP-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX1132-DPP-NEXT: s_cbranch_execz .LBB9_3
; GFX1132-DPP-NEXT: ; %bb.1:
@@ -6521,7 +6558,7 @@ define amdgpu_kernel void @global_atomic_fsub_double_uni_address_uni_value_agent
; GFX1132-DPP-NEXT: v_cmp_eq_u64_e32 vcc_lo, v[0:1], v[2:3]
; GFX1132-DPP-NEXT: v_dual_mov_b32 v3, v1 :: v_dual_mov_b32 v2, v0
; GFX1132-DPP-NEXT: s_or_b32 s2, vcc_lo, s2
-; GFX1132-DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1132-DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1132-DPP-NEXT: s_and_not1_b32 exec_lo, exec_lo, s2
; GFX1132-DPP-NEXT: s_cbranch_execnz .LBB9_2
; GFX1132-DPP-NEXT: .LBB9_3:
@@ -6859,6 +6896,7 @@ define amdgpu_kernel void @global_atomic_fsub_double_uni_address_div_value_agent
; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
; GFX1164-NEXT: v_cmpx_eq_u32_e32 0, v0
+; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1164-NEXT: s_xor_b64 s[0:1], exec, s[0:1]
; GFX1164-NEXT: s_cbranch_execz .LBB10_5
; GFX1164-NEXT: ; %bb.3:
@@ -6876,9 +6914,10 @@ define amdgpu_kernel void @global_atomic_fsub_double_uni_address_div_value_agent
; GFX1164-NEXT: v_cmp_eq_u64_e32 vcc, v[0:1], v[2:3]
; GFX1164-NEXT: v_mov_b32_e32 v3, v1
; GFX1164-NEXT: v_mov_b32_e32 v2, v0
+; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1164-NEXT: s_or_b64 s[2:3], vcc, s[2:3]
-; GFX1164-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164-NEXT: s_and_not1_b64 exec, exec, s[2:3]
+; GFX1164-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164-NEXT: s_cbranch_execnz .LBB10_4
; GFX1164-NEXT: .LBB10_5:
; GFX1164-NEXT: s_endpgm
@@ -6922,9 +6961,10 @@ define amdgpu_kernel void @global_atomic_fsub_double_uni_address_div_value_agent
; GFX1132-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
; GFX1132-NEXT: s_mov_b32 s2, 0
; GFX1132-NEXT: s_mov_b32 s0, exec_lo
-; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX1132-NEXT: s_xor_b32 s0, exec_lo, s0
+; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1132-NEXT: s_cbranch_execz .LBB10_5
; GFX1132-NEXT: ; %bb.3:
; GFX1132-NEXT: s_load_b64 s[0:1], s[34:35], 0x24
@@ -6940,7 +6980,7 @@ define amdgpu_kernel void @global_atomic_fsub_double_uni_address_div_value_agent
; GFX1132-NEXT: v_cmp_eq_u64_e32 vcc_lo, v[0:1], v[2:3]
; GFX1132-NEXT: v_dual_mov_b32 v3, v1 :: v_dual_mov_b32 v2, v0
; GFX1132-NEXT: s_or_b32 s2, vcc_lo, s2
-; GFX1132-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1132-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1132-NEXT: s_and_not1_b32 exec_lo, exec_lo, s2
; GFX1132-NEXT: s_cbranch_execnz .LBB10_4
; GFX1132-NEXT: .LBB10_5:
@@ -7344,6 +7384,7 @@ define amdgpu_kernel void @global_atomic_fsub_double_uni_address_div_value_agent
; GFX1164-DPP-NEXT: v_mov_b32_e32 v0, v2
; GFX1164-DPP-NEXT: v_cmpx_eq_u32_e32 0, v6
; GFX1164-DPP-NEXT: s_waitcnt_depctr depctr_sa_sdst(0)
+; GFX1164-DPP-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1164-DPP-NEXT: s_cbranch_execz .LBB10_3
; GFX1164-DPP-NEXT: ; %bb.1:
; GFX1164-DPP-NEXT: s_load_b64 s[0:1], s[34:35], 0x24
@@ -7359,9 +7400,10 @@ define amdgpu_kernel void @global_atomic_fsub_double_uni_address_div_value_agent
; GFX1164-DPP-NEXT: v_cmp_eq_u64_e32 vcc, v[6:7], v[8:9]
; GFX1164-DPP-NEXT: v_mov_b32_e32 v9, v7
; GFX1164-DPP-NEXT: v_mov_b32_e32 v8, v6
+; GFX1164-DPP-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1164-DPP-NEXT: s_or_b64 s[2:3], vcc, s[2:3]
-; GFX1164-DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164-DPP-NEXT: s_and_not1_b64 exec, exec, s[2:3]
+; GFX1164-DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164-DPP-NEXT: s_cbranch_execnz .LBB10_2
; GFX1164-DPP-NEXT: .LBB10_3:
; GFX1164-DPP-NEXT: s_endpgm
@@ -7429,6 +7471,7 @@ define amdgpu_kernel void @global_atomic_fsub_double_uni_address_div_value_agent
; GFX1132-DPP-NEXT: s_mov_b32 s2, 0
; GFX1132-DPP-NEXT: s_mov_b32 s0, exec_lo
; GFX1132-DPP-NEXT: v_cmpx_eq_u32_e32 0, v6
+; GFX1132-DPP-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1132-DPP-NEXT: s_cbranch_execz .LBB10_3
; GFX1132-DPP-NEXT: ; %bb.1:
; GFX1132-DPP-NEXT: s_load_b64 s[0:1], s[34:35], 0x24
@@ -7443,7 +7486,7 @@ define amdgpu_kernel void @global_atomic_fsub_double_uni_address_div_value_agent
; GFX1132-DPP-NEXT: v_cmp_eq_u64_e32 vcc_lo, v[6:7], v[8:9]
; GFX1132-DPP-NEXT: v_dual_mov_b32 v9, v7 :: v_dual_mov_b32 v8, v6
; GFX1132-DPP-NEXT: s_or_b32 s2, vcc_lo, s2
-; GFX1132-DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1132-DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1132-DPP-NEXT: s_and_not1_b32 exec_lo, exec_lo, s2
; GFX1132-DPP-NEXT: s_cbranch_execnz .LBB10_2
; GFX1132-DPP-NEXT: .LBB10_3:
@@ -7639,7 +7682,7 @@ define amdgpu_kernel void @global_atomic_fsub_double_uni_address_uni_value_one_a
; GFX1164-NEXT: scratch_store_b32 off, v1, off
; GFX1164-NEXT: scratch_load_b64 v[0:1], off, off
; GFX1164-NEXT: v_mbcnt_hi_u32_b32 v2, exec_hi, v2
-; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164-NEXT: v_cmpx_eq_u32_e32 0, v2
; GFX1164-NEXT: s_cbranch_execz .LBB11_3
; GFX1164-NEXT: ; %bb.1:
@@ -7664,9 +7707,10 @@ define amdgpu_kernel void @global_atomic_fsub_double_uni_address_uni_value_one_a
; GFX1164-NEXT: v_cmp_eq_u64_e32 vcc, v[0:1], v[2:3]
; GFX1164-NEXT: v_mov_b32_e32 v3, v1
; GFX1164-NEXT: v_mov_b32_e32 v2, v0
+; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1164-NEXT: s_or_b64 s[2:3], vcc, s[2:3]
-; GFX1164-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164-NEXT: s_and_not1_b64 exec, exec, s[2:3]
+; GFX1164-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164-NEXT: s_cbranch_execnz .LBB11_2
; GFX1164-NEXT: .LBB11_3:
; GFX1164-NEXT: s_endpgm
@@ -7684,6 +7728,7 @@ define amdgpu_kernel void @global_atomic_fsub_double_uni_address_uni_value_one_a
; GFX1132-NEXT: scratch_store_b32 off, v1, off
; GFX1132-NEXT: scratch_load_b64 v[0:1], off, off
; GFX1132-NEXT: v_cmpx_eq_u32_e32 0, v2
+; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1132-NEXT: s_cbranch_execz .LBB11_3
; GFX1132-NEXT: ; %bb.1:
; GFX1132-NEXT: s_waitcnt vmcnt(0)
@@ -7705,7 +7750,7 @@ define amdgpu_kernel void @global_atomic_fsub_double_uni_address_uni_value_one_a
; GFX1132-NEXT: v_cmp_eq_u64_e32 vcc_lo, v[0:1], v[2:3]
; GFX1132-NEXT: v_dual_mov_b32 v3, v1 :: v_dual_mov_b32 v2, v0
; GFX1132-NEXT: s_or_b32 s2, vcc_lo, s2
-; GFX1132-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1132-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1132-NEXT: s_and_not1_b32 exec_lo, exec_lo, s2
; GFX1132-NEXT: s_cbranch_execnz .LBB11_2
; GFX1132-NEXT: .LBB11_3:
@@ -7896,7 +7941,7 @@ define amdgpu_kernel void @global_atomic_fsub_double_uni_address_uni_value_one_a
; GFX1164-DPP-NEXT: scratch_store_b32 off, v1, off
; GFX1164-DPP-NEXT: scratch_load_b64 v[0:1], off, off
; GFX1164-DPP-NEXT: v_mbcnt_hi_u32_b32 v2, exec_hi, v2
-; GFX1164-DPP-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1164-DPP-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164-DPP-NEXT: v_cmpx_eq_u32_e32 0, v2
; GFX1164-DPP-NEXT: s_cbranch_execz .LBB11_3
; GFX1164-DPP-NEXT: ; %bb.1:
@@ -7921,9 +7966,10 @@ define amdgpu_kernel void @global_atomic_fsub_double_uni_address_uni_value_one_a
; GFX1164-DPP-NEXT: v_cmp_eq_u64_e32 vcc, v[0:1], v[2:3]
; GFX1164-DPP-NEXT: v_mov_b32_e32 v3, v1
; GFX1164-DPP-NEXT: v_mov_b32_e32 v2, v0
+; GFX1164-DPP-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1164-DPP-NEXT: s_or_b64 s[2:3], vcc, s[2:3]
-; GFX1164-DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164-DPP-NEXT: s_and_not1_b64 exec, exec, s[2:3]
+; GFX1164-DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164-DPP-NEXT: s_cbranch_execnz .LBB11_2
; GFX1164-DPP-NEXT: .LBB11_3:
; GFX1164-DPP-NEXT: s_endpgm
@@ -7941,6 +7987,7 @@ define amdgpu_kernel void @global_atomic_fsub_double_uni_address_uni_value_one_a
; GFX1132-DPP-NEXT: scratch_store_b32 off, v1, off
; GFX1132-DPP-NEXT: scratch_load_b64 v[0:1], off, off
; GFX1132-DPP-NEXT: v_cmpx_eq_u32_e32 0, v2
+; GFX1132-DPP-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1132-DPP-NEXT: s_cbranch_execz .LBB11_3
; GFX1132-DPP-NEXT: ; %bb.1:
; GFX1132-DPP-NEXT: s_waitcnt vmcnt(0)
@@ -7962,7 +8009,7 @@ define amdgpu_kernel void @global_atomic_fsub_double_uni_address_uni_value_one_a
; GFX1132-DPP-NEXT: v_cmp_eq_u64_e32 vcc_lo, v[0:1], v[2:3]
; GFX1132-DPP-NEXT: v_dual_mov_b32 v3, v1 :: v_dual_mov_b32 v2, v0
; GFX1132-DPP-NEXT: s_or_b32 s2, vcc_lo, s2
-; GFX1132-DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1132-DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1132-DPP-NEXT: s_and_not1_b32 exec_lo, exec_lo, s2
; GFX1132-DPP-NEXT: s_cbranch_execnz .LBB11_2
; GFX1132-DPP-NEXT: .LBB11_3:
@@ -8299,6 +8346,7 @@ define amdgpu_kernel void @global_atomic_fsub_double_uni_address_div_value_one_a
; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
; GFX1164-NEXT: v_cmpx_eq_u32_e32 0, v0
+; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1164-NEXT: s_xor_b64 s[0:1], exec, s[0:1]
; GFX1164-NEXT: s_cbranch_execz .LBB12_5
; GFX1164-NEXT: ; %bb.3:
@@ -8316,9 +8364,10 @@ define amdgpu_kernel void @global_atomic_fsub_double_uni_address_div_value_one_a
; GFX1164-NEXT: v_cmp_eq_u64_e32 vcc, v[0:1], v[2:3]
; GFX1164-NEXT: v_mov_b32_e32 v3, v1
; GFX1164-NEXT: v_mov_b32_e32 v2, v0
+; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1164-NEXT: s_or_b64 s[2:3], vcc, s[2:3]
-; GFX1164-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164-NEXT: s_and_not1_b64 exec, exec, s[2:3]
+; GFX1164-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164-NEXT: s_cbranch_execnz .LBB12_4
; GFX1164-NEXT: .LBB12_5:
; GFX1164-NEXT: s_endpgm
@@ -8362,9 +8411,10 @@ define amdgpu_kernel void @global_atomic_fsub_double_uni_address_div_value_one_a
; GFX1132-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
; GFX1132-NEXT: s_mov_b32 s2, 0
; GFX1132-NEXT: s_mov_b32 s0, exec_lo
-; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX1132-NEXT: s_xor_b32 s0, exec_lo, s0
+; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1132-NEXT: s_cbranch_execz .LBB12_5
; GFX1132-NEXT: ; %bb.3:
; GFX1132-NEXT: s_load_b64 s[0:1], s[34:35], 0x24
@@ -8380,7 +8430,7 @@ define amdgpu_kernel void @global_atomic_fsub_double_uni_address_div_value_one_a
; GFX1132-NEXT: v_cmp_eq_u64_e32 vcc_lo, v[0:1], v[2:3]
; GFX1132-NEXT: v_dual_mov_b32 v3, v1 :: v_dual_mov_b32 v2, v0
; GFX1132-NEXT: s_or_b32 s2, vcc_lo, s2
-; GFX1132-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1132-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1132-NEXT: s_and_not1_b32 exec_lo, exec_lo, s2
; GFX1132-NEXT: s_cbranch_execnz .LBB12_4
; GFX1132-NEXT: .LBB12_5:
@@ -8784,6 +8834,7 @@ define amdgpu_kernel void @global_atomic_fsub_double_uni_address_div_value_one_a
; GFX1164-DPP-NEXT: v_mov_b32_e32 v0, v2
; GFX1164-DPP-NEXT: v_cmpx_eq_u32_e32 0, v6
; GFX1164-DPP-NEXT: s_waitcnt_depctr depctr_sa_sdst(0)
+; GFX1164-DPP-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1164-DPP-NEXT: s_cbranch_execz .LBB12_3
; GFX1164-DPP-NEXT: ; %bb.1:
; GFX1164-DPP-NEXT: s_load_b64 s[0:1], s[34:35], 0x24
@@ -8799,9 +8850,10 @@ define amdgpu_kernel void @global_atomic_fsub_double_uni_address_div_value_one_a
; GFX1164-DPP-NEXT: v_cmp_eq_u64_e32 vcc, v[6:7], v[8:9]
; GFX1164-DPP-NEXT: v_mov_b32_e32 v9, v7
; GFX1164-DPP-NEXT: v_mov_b32_e32 v8, v6
+; GFX1164-DPP-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1164-DPP-NEXT: s_or_b64 s[2:3], vcc, s[2:3]
-; GFX1164-DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164-DPP-NEXT: s_and_not1_b64 exec, exec, s[2:3]
+; GFX1164-DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164-DPP-NEXT: s_cbranch_execnz .LBB12_2
; GFX1164-DPP-NEXT: .LBB12_3:
; GFX1164-DPP-NEXT: s_endpgm
@@ -8869,6 +8921,7 @@ define amdgpu_kernel void @global_atomic_fsub_double_uni_address_div_value_one_a
; GFX1132-DPP-NEXT: s_mov_b32 s2, 0
; GFX1132-DPP-NEXT: s_mov_b32 s0, exec_lo
; GFX1132-DPP-NEXT: v_cmpx_eq_u32_e32 0, v6
+; GFX1132-DPP-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1132-DPP-NEXT: s_cbranch_execz .LBB12_3
; GFX1132-DPP-NEXT: ; %bb.1:
; GFX1132-DPP-NEXT: s_load_b64 s[0:1], s[34:35], 0x24
@@ -8883,7 +8936,7 @@ define amdgpu_kernel void @global_atomic_fsub_double_uni_address_div_value_one_a
; GFX1132-DPP-NEXT: v_cmp_eq_u64_e32 vcc_lo, v[6:7], v[8:9]
; GFX1132-DPP-NEXT: v_dual_mov_b32 v9, v7 :: v_dual_mov_b32 v8, v6
; GFX1132-DPP-NEXT: s_or_b32 s2, vcc_lo, s2
-; GFX1132-DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1132-DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1132-DPP-NEXT: s_and_not1_b32 exec_lo, exec_lo, s2
; GFX1132-DPP-NEXT: s_cbranch_execnz .LBB12_2
; GFX1132-DPP-NEXT: .LBB12_3:
@@ -9079,7 +9132,7 @@ define amdgpu_kernel void @global_atomic_fsub_double_uni_address_uni_value_agent
; GFX1164-NEXT: scratch_store_b32 off, v1, off
; GFX1164-NEXT: scratch_load_b64 v[0:1], off, off
; GFX1164-NEXT: v_mbcnt_hi_u32_b32 v2, exec_hi, v2
-; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164-NEXT: v_cmpx_eq_u32_e32 0, v2
; GFX1164-NEXT: s_cbranch_execz .LBB13_3
; GFX1164-NEXT: ; %bb.1:
@@ -9104,9 +9157,10 @@ define amdgpu_kernel void @global_atomic_fsub_double_uni_address_uni_value_agent
; GFX1164-NEXT: v_cmp_eq_u64_e32 vcc, v[0:1], v[2:3]
; GFX1164-NEXT: v_mov_b32_e32 v3, v1
; GFX1164-NEXT: v_mov_b32_e32 v2, v0
+; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1164-NEXT: s_or_b64 s[2:3], vcc, s[2:3]
-; GFX1164-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164-NEXT: s_and_not1_b64 exec, exec, s[2:3]
+; GFX1164-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164-NEXT: s_cbranch_execnz .LBB13_2
; GFX1164-NEXT: .LBB13_3:
; GFX1164-NEXT: s_endpgm
@@ -9124,6 +9178,7 @@ define amdgpu_kernel void @global_atomic_fsub_double_uni_address_uni_value_agent
; GFX1132-NEXT: scratch_store_b32 off, v1, off
; GFX1132-NEXT: scratch_load_b64 v[0:1], off, off
; GFX1132-NEXT: v_cmpx_eq_u32_e32 0, v2
+; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1132-NEXT: s_cbranch_execz .LBB13_3
; GFX1132-NEXT: ; %bb.1:
; GFX1132-NEXT: s_waitcnt vmcnt(0)
@@ -9145,7 +9200,7 @@ define amdgpu_kernel void @global_atomic_fsub_double_uni_address_uni_value_agent
; GFX1132-NEXT: v_cmp_eq_u64_e32 vcc_lo, v[0:1], v[2:3]
; GFX1132-NEXT: v_dual_mov_b32 v3, v1 :: v_dual_mov_b32 v2, v0
; GFX1132-NEXT: s_or_b32 s2, vcc_lo, s2
-; GFX1132-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1132-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1132-NEXT: s_and_not1_b32 exec_lo, exec_lo, s2
; GFX1132-NEXT: s_cbranch_execnz .LBB13_2
; GFX1132-NEXT: .LBB13_3:
@@ -9336,7 +9391,7 @@ define amdgpu_kernel void @global_atomic_fsub_double_uni_address_uni_value_agent
; GFX1164-DPP-NEXT: scratch_store_b32 off, v1, off
; GFX1164-DPP-NEXT: scratch_load_b64 v[0:1], off, off
; GFX1164-DPP-NEXT: v_mbcnt_hi_u32_b32 v2, exec_hi, v2
-; GFX1164-DPP-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1164-DPP-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164-DPP-NEXT: v_cmpx_eq_u32_e32 0, v2
; GFX1164-DPP-NEXT: s_cbranch_execz .LBB13_3
; GFX1164-DPP-NEXT: ; %bb.1:
@@ -9361,9 +9416,10 @@ define amdgpu_kernel void @global_atomic_fsub_double_uni_address_uni_value_agent
; GFX1164-DPP-NEXT: v_cmp_eq_u64_e32 vcc, v[0:1], v[2:3]
; GFX1164-DPP-NEXT: v_mov_b32_e32 v3, v1
; GFX1164-DPP-NEXT: v_mov_b32_e32 v2, v0
+; GFX1164-DPP-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1164-DPP-NEXT: s_or_b64 s[2:3], vcc, s[2:3]
-; GFX1164-DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164-DPP-NEXT: s_and_not1_b64 exec, exec, s[2:3]
+; GFX1164-DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164-DPP-NEXT: s_cbranch_execnz .LBB13_2
; GFX1164-DPP-NEXT: .LBB13_3:
; GFX1164-DPP-NEXT: s_endpgm
@@ -9381,6 +9437,7 @@ define amdgpu_kernel void @global_atomic_fsub_double_uni_address_uni_value_agent
; GFX1132-DPP-NEXT: scratch_store_b32 off, v1, off
; GFX1132-DPP-NEXT: scratch_load_b64 v[0:1], off, off
; GFX1132-DPP-NEXT: v_cmpx_eq_u32_e32 0, v2
+; GFX1132-DPP-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1132-DPP-NEXT: s_cbranch_execz .LBB13_3
; GFX1132-DPP-NEXT: ; %bb.1:
; GFX1132-DPP-NEXT: s_waitcnt vmcnt(0)
@@ -9402,7 +9459,7 @@ define amdgpu_kernel void @global_atomic_fsub_double_uni_address_uni_value_agent
; GFX1132-DPP-NEXT: v_cmp_eq_u64_e32 vcc_lo, v[0:1], v[2:3]
; GFX1132-DPP-NEXT: v_dual_mov_b32 v3, v1 :: v_dual_mov_b32 v2, v0
; GFX1132-DPP-NEXT: s_or_b32 s2, vcc_lo, s2
-; GFX1132-DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1132-DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1132-DPP-NEXT: s_and_not1_b32 exec_lo, exec_lo, s2
; GFX1132-DPP-NEXT: s_cbranch_execnz .LBB13_2
; GFX1132-DPP-NEXT: .LBB13_3:
@@ -9740,6 +9797,7 @@ define amdgpu_kernel void @global_atomic_fsub_double_uni_address_div_value_agent
; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
; GFX1164-NEXT: v_cmpx_eq_u32_e32 0, v0
+; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1164-NEXT: s_xor_b64 s[0:1], exec, s[0:1]
; GFX1164-NEXT: s_cbranch_execz .LBB14_5
; GFX1164-NEXT: ; %bb.3:
@@ -9757,9 +9815,10 @@ define amdgpu_kernel void @global_atomic_fsub_double_uni_address_div_value_agent
; GFX1164-NEXT: v_cmp_eq_u64_e32 vcc, v[0:1], v[2:3]
; GFX1164-NEXT: v_mov_b32_e32 v3, v1
; GFX1164-NEXT: v_mov_b32_e32 v2, v0
+; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1164-NEXT: s_or_b64 s[2:3], vcc, s[2:3]
-; GFX1164-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164-NEXT: s_and_not1_b64 exec, exec, s[2:3]
+; GFX1164-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164-NEXT: s_cbranch_execnz .LBB14_4
; GFX1164-NEXT: .LBB14_5:
; GFX1164-NEXT: s_endpgm
@@ -9803,9 +9862,10 @@ define amdgpu_kernel void @global_atomic_fsub_double_uni_address_div_value_agent
; GFX1132-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
; GFX1132-NEXT: s_mov_b32 s2, 0
; GFX1132-NEXT: s_mov_b32 s0, exec_lo
-; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX1132-NEXT: s_xor_b32 s0, exec_lo, s0
+; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1132-NEXT: s_cbranch_execz .LBB14_5
; GFX1132-NEXT: ; %bb.3:
; GFX1132-NEXT: s_load_b64 s[0:1], s[34:35], 0x24
@@ -9821,7 +9881,7 @@ define amdgpu_kernel void @global_atomic_fsub_double_uni_address_div_value_agent
; GFX1132-NEXT: v_cmp_eq_u64_e32 vcc_lo, v[0:1], v[2:3]
; GFX1132-NEXT: v_dual_mov_b32 v3, v1 :: v_dual_mov_b32 v2, v0
; GFX1132-NEXT: s_or_b32 s2, vcc_lo, s2
-; GFX1132-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1132-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1132-NEXT: s_and_not1_b32 exec_lo, exec_lo, s2
; GFX1132-NEXT: s_cbranch_execnz .LBB14_4
; GFX1132-NEXT: .LBB14_5:
@@ -10225,6 +10285,7 @@ define amdgpu_kernel void @global_atomic_fsub_double_uni_address_div_value_agent
; GFX1164-DPP-NEXT: v_mov_b32_e32 v0, v2
; GFX1164-DPP-NEXT: v_cmpx_eq_u32_e32 0, v6
; GFX1164-DPP-NEXT: s_waitcnt_depctr depctr_sa_sdst(0)
+; GFX1164-DPP-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1164-DPP-NEXT: s_cbranch_execz .LBB14_3
; GFX1164-DPP-NEXT: ; %bb.1:
; GFX1164-DPP-NEXT: s_load_b64 s[0:1], s[34:35], 0x24
@@ -10240,9 +10301,10 @@ define amdgpu_kernel void @global_atomic_fsub_double_uni_address_div_value_agent
; GFX1164-DPP-NEXT: v_cmp_eq_u64_e32 vcc, v[6:7], v[8:9]
; GFX1164-DPP-NEXT: v_mov_b32_e32 v9, v7
; GFX1164-DPP-NEXT: v_mov_b32_e32 v8, v6
+; GFX1164-DPP-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1164-DPP-NEXT: s_or_b64 s[2:3], vcc, s[2:3]
-; GFX1164-DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164-DPP-NEXT: s_and_not1_b64 exec, exec, s[2:3]
+; GFX1164-DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164-DPP-NEXT: s_cbranch_execnz .LBB14_2
; GFX1164-DPP-NEXT: .LBB14_3:
; GFX1164-DPP-NEXT: s_endpgm
@@ -10310,6 +10372,7 @@ define amdgpu_kernel void @global_atomic_fsub_double_uni_address_div_value_agent
; GFX1132-DPP-NEXT: s_mov_b32 s2, 0
; GFX1132-DPP-NEXT: s_mov_b32 s0, exec_lo
; GFX1132-DPP-NEXT: v_cmpx_eq_u32_e32 0, v6
+; GFX1132-DPP-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1132-DPP-NEXT: s_cbranch_execz .LBB14_3
; GFX1132-DPP-NEXT: ; %bb.1:
; GFX1132-DPP-NEXT: s_load_b64 s[0:1], s[34:35], 0x24
@@ -10324,7 +10387,7 @@ define amdgpu_kernel void @global_atomic_fsub_double_uni_address_div_value_agent
; GFX1132-DPP-NEXT: v_cmp_eq_u64_e32 vcc_lo, v[6:7], v[8:9]
; GFX1132-DPP-NEXT: v_dual_mov_b32 v9, v7 :: v_dual_mov_b32 v8, v6
; GFX1132-DPP-NEXT: s_or_b32 s2, vcc_lo, s2
-; GFX1132-DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1132-DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1132-DPP-NEXT: s_and_not1_b32 exec_lo, exec_lo, s2
; GFX1132-DPP-NEXT: s_cbranch_execnz .LBB14_2
; GFX1132-DPP-NEXT: .LBB14_3:
@@ -10663,6 +10726,7 @@ define amdgpu_kernel void @global_atomic_fsub_double_uni_address_div_value_agent
; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
; GFX1164-NEXT: v_cmpx_eq_u32_e32 0, v0
+; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1164-NEXT: s_xor_b64 s[0:1], exec, s[0:1]
; GFX1164-NEXT: s_cbranch_execz .LBB15_5
; GFX1164-NEXT: ; %bb.3:
@@ -10680,9 +10744,10 @@ define amdgpu_kernel void @global_atomic_fsub_double_uni_address_div_value_agent
; GFX1164-NEXT: v_cmp_eq_u64_e32 vcc, v[0:1], v[2:3]
; GFX1164-NEXT: v_mov_b32_e32 v3, v1
; GFX1164-NEXT: v_mov_b32_e32 v2, v0
+; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1164-NEXT: s_or_b64 s[2:3], vcc, s[2:3]
-; GFX1164-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164-NEXT: s_and_not1_b64 exec, exec, s[2:3]
+; GFX1164-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164-NEXT: s_cbranch_execnz .LBB15_4
; GFX1164-NEXT: .LBB15_5:
; GFX1164-NEXT: s_endpgm
@@ -10726,9 +10791,10 @@ define amdgpu_kernel void @global_atomic_fsub_double_uni_address_div_value_agent
; GFX1132-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
; GFX1132-NEXT: s_mov_b32 s2, 0
; GFX1132-NEXT: s_mov_b32 s0, exec_lo
-; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX1132-NEXT: s_xor_b32 s0, exec_lo, s0
+; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1132-NEXT: s_cbranch_execz .LBB15_5
; GFX1132-NEXT: ; %bb.3:
; GFX1132-NEXT: s_load_b64 s[0:1], s[34:35], 0x24
@@ -10744,7 +10810,7 @@ define amdgpu_kernel void @global_atomic_fsub_double_uni_address_div_value_agent
; GFX1132-NEXT: v_cmp_eq_u64_e32 vcc_lo, v[0:1], v[2:3]
; GFX1132-NEXT: v_dual_mov_b32 v3, v1 :: v_dual_mov_b32 v2, v0
; GFX1132-NEXT: s_or_b32 s2, vcc_lo, s2
-; GFX1132-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1132-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1132-NEXT: s_and_not1_b32 exec_lo, exec_lo, s2
; GFX1132-NEXT: s_cbranch_execnz .LBB15_4
; GFX1132-NEXT: .LBB15_5:
@@ -11148,6 +11214,7 @@ define amdgpu_kernel void @global_atomic_fsub_double_uni_address_div_value_agent
; GFX1164-DPP-NEXT: v_mov_b32_e32 v0, v2
; GFX1164-DPP-NEXT: v_cmpx_eq_u32_e32 0, v6
; GFX1164-DPP-NEXT: s_waitcnt_depctr depctr_sa_sdst(0)
+; GFX1164-DPP-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1164-DPP-NEXT: s_cbranch_execz .LBB15_3
; GFX1164-DPP-NEXT: ; %bb.1:
; GFX1164-DPP-NEXT: s_load_b64 s[0:1], s[34:35], 0x24
@@ -11163,9 +11230,10 @@ define amdgpu_kernel void @global_atomic_fsub_double_uni_address_div_value_agent
; GFX1164-DPP-NEXT: v_cmp_eq_u64_e32 vcc, v[6:7], v[8:9]
; GFX1164-DPP-NEXT: v_mov_b32_e32 v9, v7
; GFX1164-DPP-NEXT: v_mov_b32_e32 v8, v6
+; GFX1164-DPP-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1164-DPP-NEXT: s_or_b64 s[2:3], vcc, s[2:3]
-; GFX1164-DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164-DPP-NEXT: s_and_not1_b64 exec, exec, s[2:3]
+; GFX1164-DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164-DPP-NEXT: s_cbranch_execnz .LBB15_2
; GFX1164-DPP-NEXT: .LBB15_3:
; GFX1164-DPP-NEXT: s_endpgm
@@ -11233,6 +11301,7 @@ define amdgpu_kernel void @global_atomic_fsub_double_uni_address_div_value_agent
; GFX1132-DPP-NEXT: s_mov_b32 s2, 0
; GFX1132-DPP-NEXT: s_mov_b32 s0, exec_lo
; GFX1132-DPP-NEXT: v_cmpx_eq_u32_e32 0, v6
+; GFX1132-DPP-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1132-DPP-NEXT: s_cbranch_execz .LBB15_3
; GFX1132-DPP-NEXT: ; %bb.1:
; GFX1132-DPP-NEXT: s_load_b64 s[0:1], s[34:35], 0x24
@@ -11247,7 +11316,7 @@ define amdgpu_kernel void @global_atomic_fsub_double_uni_address_div_value_agent
; GFX1132-DPP-NEXT: v_cmp_eq_u64_e32 vcc_lo, v[6:7], v[8:9]
; GFX1132-DPP-NEXT: v_dual_mov_b32 v9, v7 :: v_dual_mov_b32 v8, v6
; GFX1132-DPP-NEXT: s_or_b32 s2, vcc_lo, s2
-; GFX1132-DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1132-DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1132-DPP-NEXT: s_and_not1_b32 exec_lo, exec_lo, s2
; GFX1132-DPP-NEXT: s_cbranch_execnz .LBB15_2
; GFX1132-DPP-NEXT: .LBB15_3:
@@ -11442,7 +11511,7 @@ define amdgpu_kernel void @global_atomic_fsub_double_uni_address_uni_value_defau
; GFX1164-NEXT: scratch_store_b32 off, v1, off
; GFX1164-NEXT: scratch_load_b64 v[0:1], off, off
; GFX1164-NEXT: v_mbcnt_hi_u32_b32 v2, exec_hi, v2
-; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164-NEXT: v_cmpx_eq_u32_e32 0, v2
; GFX1164-NEXT: s_cbranch_execz .LBB16_3
; GFX1164-NEXT: ; %bb.1:
@@ -11467,9 +11536,10 @@ define amdgpu_kernel void @global_atomic_fsub_double_uni_address_uni_value_defau
; GFX1164-NEXT: v_cmp_eq_u64_e32 vcc, v[0:1], v[2:3]
; GFX1164-NEXT: v_mov_b32_e32 v3, v1
; GFX1164-NEXT: v_mov_b32_e32 v2, v0
+; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1164-NEXT: s_or_b64 s[2:3], vcc, s[2:3]
-; GFX1164-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164-NEXT: s_and_not1_b64 exec, exec, s[2:3]
+; GFX1164-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164-NEXT: s_cbranch_execnz .LBB16_2
; GFX1164-NEXT: .LBB16_3:
; GFX1164-NEXT: s_endpgm
@@ -11487,6 +11557,7 @@ define amdgpu_kernel void @global_atomic_fsub_double_uni_address_uni_value_defau
; GFX1132-NEXT: scratch_store_b32 off, v1, off
; GFX1132-NEXT: scratch_load_b64 v[0:1], off, off
; GFX1132-NEXT: v_cmpx_eq_u32_e32 0, v2
+; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1132-NEXT: s_cbranch_execz .LBB16_3
; GFX1132-NEXT: ; %bb.1:
; GFX1132-NEXT: s_waitcnt vmcnt(0)
@@ -11508,7 +11579,7 @@ define amdgpu_kernel void @global_atomic_fsub_double_uni_address_uni_value_defau
; GFX1132-NEXT: v_cmp_eq_u64_e32 vcc_lo, v[0:1], v[2:3]
; GFX1132-NEXT: v_dual_mov_b32 v3, v1 :: v_dual_mov_b32 v2, v0
; GFX1132-NEXT: s_or_b32 s2, vcc_lo, s2
-; GFX1132-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1132-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1132-NEXT: s_and_not1_b32 exec_lo, exec_lo, s2
; GFX1132-NEXT: s_cbranch_execnz .LBB16_2
; GFX1132-NEXT: .LBB16_3:
@@ -11699,7 +11770,7 @@ define amdgpu_kernel void @global_atomic_fsub_double_uni_address_uni_value_defau
; GFX1164-DPP-NEXT: scratch_store_b32 off, v1, off
; GFX1164-DPP-NEXT: scratch_load_b64 v[0:1], off, off
; GFX1164-DPP-NEXT: v_mbcnt_hi_u32_b32 v2, exec_hi, v2
-; GFX1164-DPP-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1164-DPP-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164-DPP-NEXT: v_cmpx_eq_u32_e32 0, v2
; GFX1164-DPP-NEXT: s_cbranch_execz .LBB16_3
; GFX1164-DPP-NEXT: ; %bb.1:
@@ -11724,9 +11795,10 @@ define amdgpu_kernel void @global_atomic_fsub_double_uni_address_uni_value_defau
; GFX1164-DPP-NEXT: v_cmp_eq_u64_e32 vcc, v[0:1], v[2:3]
; GFX1164-DPP-NEXT: v_mov_b32_e32 v3, v1
; GFX1164-DPP-NEXT: v_mov_b32_e32 v2, v0
+; GFX1164-DPP-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1164-DPP-NEXT: s_or_b64 s[2:3], vcc, s[2:3]
-; GFX1164-DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164-DPP-NEXT: s_and_not1_b64 exec, exec, s[2:3]
+; GFX1164-DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164-DPP-NEXT: s_cbranch_execnz .LBB16_2
; GFX1164-DPP-NEXT: .LBB16_3:
; GFX1164-DPP-NEXT: s_endpgm
@@ -11744,6 +11816,7 @@ define amdgpu_kernel void @global_atomic_fsub_double_uni_address_uni_value_defau
; GFX1132-DPP-NEXT: scratch_store_b32 off, v1, off
; GFX1132-DPP-NEXT: scratch_load_b64 v[0:1], off, off
; GFX1132-DPP-NEXT: v_cmpx_eq_u32_e32 0, v2
+; GFX1132-DPP-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1132-DPP-NEXT: s_cbranch_execz .LBB16_3
; GFX1132-DPP-NEXT: ; %bb.1:
; GFX1132-DPP-NEXT: s_waitcnt vmcnt(0)
@@ -11765,7 +11838,7 @@ define amdgpu_kernel void @global_atomic_fsub_double_uni_address_uni_value_defau
; GFX1132-DPP-NEXT: v_cmp_eq_u64_e32 vcc_lo, v[0:1], v[2:3]
; GFX1132-DPP-NEXT: v_dual_mov_b32 v3, v1 :: v_dual_mov_b32 v2, v0
; GFX1132-DPP-NEXT: s_or_b32 s2, vcc_lo, s2
-; GFX1132-DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1132-DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1132-DPP-NEXT: s_and_not1_b32 exec_lo, exec_lo, s2
; GFX1132-DPP-NEXT: s_cbranch_execnz .LBB16_2
; GFX1132-DPP-NEXT: .LBB16_3:
@@ -12103,6 +12176,7 @@ define amdgpu_kernel void @global_atomic_fsub_double_uni_address_div_value_defau
; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
; GFX1164-NEXT: v_cmpx_eq_u32_e32 0, v0
+; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1164-NEXT: s_xor_b64 s[0:1], exec, s[0:1]
; GFX1164-NEXT: s_cbranch_execz .LBB17_5
; GFX1164-NEXT: ; %bb.3:
@@ -12120,9 +12194,10 @@ define amdgpu_kernel void @global_atomic_fsub_double_uni_address_div_value_defau
; GFX1164-NEXT: v_cmp_eq_u64_e32 vcc, v[0:1], v[2:3]
; GFX1164-NEXT: v_mov_b32_e32 v3, v1
; GFX1164-NEXT: v_mov_b32_e32 v2, v0
+; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1164-NEXT: s_or_b64 s[2:3], vcc, s[2:3]
-; GFX1164-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164-NEXT: s_and_not1_b64 exec, exec, s[2:3]
+; GFX1164-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164-NEXT: s_cbranch_execnz .LBB17_4
; GFX1164-NEXT: .LBB17_5:
; GFX1164-NEXT: s_endpgm
@@ -12166,9 +12241,10 @@ define amdgpu_kernel void @global_atomic_fsub_double_uni_address_div_value_defau
; GFX1132-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
; GFX1132-NEXT: s_mov_b32 s2, 0
; GFX1132-NEXT: s_mov_b32 s0, exec_lo
-; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX1132-NEXT: s_xor_b32 s0, exec_lo, s0
+; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1132-NEXT: s_cbranch_execz .LBB17_5
; GFX1132-NEXT: ; %bb.3:
; GFX1132-NEXT: s_load_b64 s[0:1], s[34:35], 0x24
@@ -12184,7 +12260,7 @@ define amdgpu_kernel void @global_atomic_fsub_double_uni_address_div_value_defau
; GFX1132-NEXT: v_cmp_eq_u64_e32 vcc_lo, v[0:1], v[2:3]
; GFX1132-NEXT: v_dual_mov_b32 v3, v1 :: v_dual_mov_b32 v2, v0
; GFX1132-NEXT: s_or_b32 s2, vcc_lo, s2
-; GFX1132-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1132-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1132-NEXT: s_and_not1_b32 exec_lo, exec_lo, s2
; GFX1132-NEXT: s_cbranch_execnz .LBB17_4
; GFX1132-NEXT: .LBB17_5:
@@ -12588,6 +12664,7 @@ define amdgpu_kernel void @global_atomic_fsub_double_uni_address_div_value_defau
; GFX1164-DPP-NEXT: v_mov_b32_e32 v0, v2
; GFX1164-DPP-NEXT: v_cmpx_eq_u32_e32 0, v6
; GFX1164-DPP-NEXT: s_waitcnt_depctr depctr_sa_sdst(0)
+; GFX1164-DPP-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1164-DPP-NEXT: s_cbranch_execz .LBB17_3
; GFX1164-DPP-NEXT: ; %bb.1:
; GFX1164-DPP-NEXT: s_load_b64 s[0:1], s[34:35], 0x24
@@ -12603,9 +12680,10 @@ define amdgpu_kernel void @global_atomic_fsub_double_uni_address_div_value_defau
; GFX1164-DPP-NEXT: v_cmp_eq_u64_e32 vcc, v[6:7], v[8:9]
; GFX1164-DPP-NEXT: v_mov_b32_e32 v9, v7
; GFX1164-DPP-NEXT: v_mov_b32_e32 v8, v6
+; GFX1164-DPP-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1164-DPP-NEXT: s_or_b64 s[2:3], vcc, s[2:3]
-; GFX1164-DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164-DPP-NEXT: s_and_not1_b64 exec, exec, s[2:3]
+; GFX1164-DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164-DPP-NEXT: s_cbranch_execnz .LBB17_2
; GFX1164-DPP-NEXT: .LBB17_3:
; GFX1164-DPP-NEXT: s_endpgm
@@ -12673,6 +12751,7 @@ define amdgpu_kernel void @global_atomic_fsub_double_uni_address_div_value_defau
; GFX1132-DPP-NEXT: s_mov_b32 s2, 0
; GFX1132-DPP-NEXT: s_mov_b32 s0, exec_lo
; GFX1132-DPP-NEXT: v_cmpx_eq_u32_e32 0, v6
+; GFX1132-DPP-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1132-DPP-NEXT: s_cbranch_execz .LBB17_3
; GFX1132-DPP-NEXT: ; %bb.1:
; GFX1132-DPP-NEXT: s_load_b64 s[0:1], s[34:35], 0x24
@@ -12687,7 +12766,7 @@ define amdgpu_kernel void @global_atomic_fsub_double_uni_address_div_value_defau
; GFX1132-DPP-NEXT: v_cmp_eq_u64_e32 vcc_lo, v[6:7], v[8:9]
; GFX1132-DPP-NEXT: v_dual_mov_b32 v9, v7 :: v_dual_mov_b32 v8, v6
; GFX1132-DPP-NEXT: s_or_b32 s2, vcc_lo, s2
-; GFX1132-DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1132-DPP-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1132-DPP-NEXT: s_and_not1_b32 exec_lo, exec_lo, s2
; GFX1132-DPP-NEXT: s_cbranch_execnz .LBB17_2
; GFX1132-DPP-NEXT: .LBB17_3:
diff --git a/llvm/test/CodeGen/AMDGPU/group-image-instructions.ll b/llvm/test/CodeGen/AMDGPU/group-image-instructions.ll
index 08d7087a3e9468..7e146ba64bba08 100644
--- a/llvm/test/CodeGen/AMDGPU/group-image-instructions.ll
+++ b/llvm/test/CodeGen/AMDGPU/group-image-instructions.ll
@@ -21,6 +21,7 @@ define amdgpu_ps void @group_image_sample(i32 inreg noundef %globalTable, i32 in
; GFX11-NEXT: lds_param_load v2, attr0.y wait_vdst:15
; GFX11-NEXT: lds_param_load v3, attr0.x wait_vdst:15
; GFX11-NEXT: s_mov_b32 exec_lo, s16
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: v_interp_p10_f32 v4, v2, v0, v2 wait_exp:1
; GFX11-NEXT: v_interp_p10_f32 v0, v3, v0, v3 wait_exp:0
; GFX11-NEXT: s_waitcnt lgkmcnt(0)
diff --git a/llvm/test/CodeGen/AMDGPU/i1-to-bf16.ll b/llvm/test/CodeGen/AMDGPU/i1-to-bf16.ll
index c3807581d13e17..05b64ef99b4e74 100644
--- a/llvm/test/CodeGen/AMDGPU/i1-to-bf16.ll
+++ b/llvm/test/CodeGen/AMDGPU/i1-to-bf16.ll
@@ -106,38 +106,38 @@ define amdgpu_ps i32 @s_uitofp_i1_to_bf16(i1 inreg %num) {
; GFX11-LABEL: s_uitofp_i1_to_bf16:
; GFX11: ; %bb.0:
; GFX11-NEXT: s_bitcmp1_b32 s0, 0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_cselect_b32 s0, -1, 0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_cndmask_b32_e64 v0, 0, 1.0, s0
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_2) | instid1(VALU_DEP_3)
; GFX11-NEXT: v_bfe_u32 v1, v0, 16, 1
; GFX11-NEXT: v_or_b32_e32 v2, 0x400000, v0
; GFX11-NEXT: v_cmp_u_f32_e32 vcc_lo, v0, v0
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_nc_u32_e32 v1, v1, v0
-; GFX11-NEXT: v_add_nc_u32_e32 v1, 0x7fff, v1
; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX11-NEXT: v_add_nc_u32_e32 v1, 0x7fff, v1
; GFX11-NEXT: v_cndmask_b32_e32 v0, v1, v2, vcc_lo
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_lshrrev_b32_e32 v0, 16, v0
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_readfirstlane_b32 s0, v0
; GFX11-NEXT: ; return to shader part epilog
;
; GFX12-LABEL: s_uitofp_i1_to_bf16:
; GFX12: ; %bb.0:
; GFX12-NEXT: s_bitcmp1_b32 s0, 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s0, -1, 0
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX12-NEXT: v_cndmask_b32_e64 v0, 0, 1.0, s0
+; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_2) | instid1(VALU_DEP_3)
; GFX12-NEXT: v_bfe_u32 v1, v0, 16, 1
; GFX12-NEXT: v_or_b32_e32 v2, 0x400000, v0
; GFX12-NEXT: v_cmp_u_f32_e32 vcc_lo, v0, v0
-; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX12-NEXT: v_add_nc_u32_e32 v1, v1, v0
-; GFX12-NEXT: v_add_nc_u32_e32 v1, 0x7fff, v1
; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX12-NEXT: v_add_nc_u32_e32 v1, 0x7fff, v1
; GFX12-NEXT: v_cndmask_b32_e32 v0, v1, v2, vcc_lo
+; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX12-NEXT: v_lshrrev_b32_e32 v0, 16, v0
-; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-NEXT: v_readfirstlane_b32 s0, v0
; GFX12-NEXT: s_wait_alu depctr_va_sdst(0)
; GFX12-NEXT: ; return to shader part epilog
@@ -367,32 +367,32 @@ define amdgpu_ps <2 x i32> @s_uitofp_v2i1_to_v2bf16(<2 x i1> inreg %num) {
; GFX11: ; %bb.0:
; GFX11-NEXT: s_and_b32 s1, 1, s1
; GFX11-NEXT: s_bitcmp1_b32 s0, 0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_3) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_cselect_b32 s0, -1, 0
; GFX11-NEXT: s_cmp_eq_u32 s1, 1
; GFX11-NEXT: v_cndmask_b32_e64 v0, 0, 1.0, s0
; GFX11-NEXT: s_cselect_b32 s0, -1, 0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-NEXT: v_cndmask_b32_e64 v1, 0, 1.0, s0
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_2) | instid1(VALU_DEP_4)
; GFX11-NEXT: v_bfe_u32 v2, v0, 16, 1
; GFX11-NEXT: v_or_b32_e32 v4, 0x400000, v0
; GFX11-NEXT: v_cmp_u_f32_e32 vcc_lo, v0, v0
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX11-NEXT: v_bfe_u32 v3, v1, 16, 1
; GFX11-NEXT: v_or_b32_e32 v5, 0x400000, v1
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_nc_u32_e32 v3, v3, v1
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_nc_u32_e32 v3, 0x7fff, v3
; GFX11-NEXT: v_add_nc_u32_e32 v2, v2, v0
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_nc_u32_e32 v2, 0x7fff, v2
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX11-NEXT: v_cndmask_b32_e32 v0, v2, v4, vcc_lo
; GFX11-NEXT: v_cmp_u_f32_e32 vcc_lo, v1, v1
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX11-NEXT: v_lshrrev_b32_e32 v0, 16, v0
; GFX11-NEXT: v_cndmask_b32_e32 v1, v3, v5, vcc_lo
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-NEXT: v_readfirstlane_b32 s0, v0
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_lshrrev_b32_e32 v1, 16, v1
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_readfirstlane_b32 s1, v1
; GFX11-NEXT: ; return to shader part epilog
;
@@ -400,6 +400,7 @@ define amdgpu_ps <2 x i32> @s_uitofp_v2i1_to_v2bf16(<2 x i1> inreg %num) {
; GFX12: ; %bb.0:
; GFX12-NEXT: s_and_b32 s1, 1, s1
; GFX12-NEXT: s_bitcmp1_b32 s0, 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s0, -1, 0
; GFX12-NEXT: s_cmp_eq_u32 s1, 1
; GFX12-NEXT: v_cndmask_b32_e64 v0, 0, 1.0, s0
@@ -734,6 +735,7 @@ define amdgpu_ps <3 x i32> @s_uitofp_v3i1_to_v3bf16(<3 x i1> inreg %num) {
; GFX11-NEXT: s_and_b32 s2, 1, s2
; GFX11-NEXT: s_and_b32 s1, 1, s1
; GFX11-NEXT: s_bitcmp1_b32 s0, 0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cselect_b32 s0, -1, 0
; GFX11-NEXT: s_cmp_eq_u32 s1, 1
; GFX11-NEXT: v_cndmask_b32_e64 v1, 0, 1.0, s0
@@ -777,6 +779,7 @@ define amdgpu_ps <3 x i32> @s_uitofp_v3i1_to_v3bf16(<3 x i1> inreg %num) {
; GFX12-NEXT: s_and_b32 s2, 1, s2
; GFX12-NEXT: s_and_b32 s1, 1, s1
; GFX12-NEXT: s_bitcmp1_b32 s0, 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s0, -1, 0
; GFX12-NEXT: s_cmp_eq_u32 s1, 1
; GFX12-NEXT: v_cndmask_b32_e64 v1, 0, 1.0, s0
@@ -1197,6 +1200,7 @@ define amdgpu_ps <4 x i32> @s_uitofp_v4i1_to_v4bf16(<4 x i1> inreg %num) {
; GFX11-NEXT: s_and_b32 s2, 1, s2
; GFX11-NEXT: s_and_b32 s1, 1, s1
; GFX11-NEXT: s_bitcmp1_b32 s0, 0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cselect_b32 s0, -1, 0
; GFX11-NEXT: s_cmp_eq_u32 s1, 1
; GFX11-NEXT: v_cndmask_b32_e64 v3, 0, 1.0, s0
@@ -1252,6 +1256,7 @@ define amdgpu_ps <4 x i32> @s_uitofp_v4i1_to_v4bf16(<4 x i1> inreg %num) {
; GFX12-NEXT: s_and_b32 s2, 1, s2
; GFX12-NEXT: s_and_b32 s1, 1, s1
; GFX12-NEXT: s_bitcmp1_b32 s0, 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s0, -1, 0
; GFX12-NEXT: s_cmp_eq_u32 s1, 1
; GFX12-NEXT: v_cndmask_b32_e64 v3, 0, 1.0, s0
@@ -1412,38 +1417,38 @@ define amdgpu_ps i32 @s_sitofp_i1_to_bf16(i1 inreg %num) {
; GFX11-LABEL: s_sitofp_i1_to_bf16:
; GFX11: ; %bb.0:
; GFX11-NEXT: s_bitcmp1_b32 s0, 0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_cselect_b32 s0, -1, 0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_cndmask_b32_e64 v0, 0, -1.0, s0
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_2) | instid1(VALU_DEP_3)
; GFX11-NEXT: v_bfe_u32 v1, v0, 16, 1
; GFX11-NEXT: v_or_b32_e32 v2, 0x400000, v0
; GFX11-NEXT: v_cmp_u_f32_e32 vcc_lo, v0, v0
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_nc_u32_e32 v1, v1, v0
-; GFX11-NEXT: v_add_nc_u32_e32 v1, 0x7fff, v1
; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX11-NEXT: v_add_nc_u32_e32 v1, 0x7fff, v1
; GFX11-NEXT: v_cndmask_b32_e32 v0, v1, v2, vcc_lo
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_ashrrev_i32_e32 v0, 16, v0
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_readfirstlane_b32 s0, v0
; GFX11-NEXT: ; return to shader part epilog
;
; GFX12-LABEL: s_sitofp_i1_to_bf16:
; GFX12: ; %bb.0:
; GFX12-NEXT: s_bitcmp1_b32 s0, 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s0, -1, 0
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX12-NEXT: v_cndmask_b32_e64 v0, 0, -1.0, s0
+; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_2) | instid1(VALU_DEP_3)
; GFX12-NEXT: v_bfe_u32 v1, v0, 16, 1
; GFX12-NEXT: v_or_b32_e32 v2, 0x400000, v0
; GFX12-NEXT: v_cmp_u_f32_e32 vcc_lo, v0, v0
-; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX12-NEXT: v_add_nc_u32_e32 v1, v1, v0
-; GFX12-NEXT: v_add_nc_u32_e32 v1, 0x7fff, v1
; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX12-NEXT: v_add_nc_u32_e32 v1, 0x7fff, v1
; GFX12-NEXT: v_cndmask_b32_e32 v0, v1, v2, vcc_lo
+; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX12-NEXT: v_ashrrev_i32_e32 v0, 16, v0
-; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-NEXT: v_readfirstlane_b32 s0, v0
; GFX12-NEXT: s_wait_alu depctr_va_sdst(0)
; GFX12-NEXT: ; return to shader part epilog
@@ -1673,33 +1678,33 @@ define amdgpu_ps <2 x i32> @s_sitofp_v2i1_to_v2bf16(<2 x i1> inreg %num) {
; GFX11: ; %bb.0:
; GFX11-NEXT: s_and_b32 s0, 1, s0
; GFX11-NEXT: s_bitcmp1_b32 s1, 0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_3) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_cselect_b32 s1, -1, 0
; GFX11-NEXT: s_cmp_eq_u32 s0, 1
; GFX11-NEXT: v_cndmask_b32_e64 v0, 0, -1.0, s1
; GFX11-NEXT: s_cselect_b32 s0, -1, 0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-NEXT: v_cndmask_b32_e64 v1, 0, -1.0, s0
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_3)
; GFX11-NEXT: v_bfe_u32 v3, v0, 16, 1
; GFX11-NEXT: v_or_b32_e32 v5, 0x400000, v0
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_1) | instid1(VALU_DEP_4)
; GFX11-NEXT: v_bfe_u32 v2, v1, 16, 1
; GFX11-NEXT: v_or_b32_e32 v4, 0x400000, v1
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(SKIP_1) | instid1(VALU_DEP_4)
; GFX11-NEXT: v_add_nc_u32_e32 v3, v3, v0
; GFX11-NEXT: v_cmp_u_f32_e32 vcc_lo, v1, v1
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_3)
; GFX11-NEXT: v_add_nc_u32_e32 v2, v2, v1
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-NEXT: v_add_nc_u32_e32 v3, 0x7fff, v3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_nc_u32_e32 v2, 0x7fff, v2
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_4)
; GFX11-NEXT: v_cndmask_b32_e32 v1, v2, v4, vcc_lo
; GFX11-NEXT: v_cmp_u_f32_e32 vcc_lo, v0, v0
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_3)
; GFX11-NEXT: v_cndmask_b32_e32 v0, v3, v5, vcc_lo
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-NEXT: v_ashrrev_i32_e32 v1, 16, v1
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-NEXT: v_ashrrev_i32_e32 v0, 16, v0
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-NEXT: v_readfirstlane_b32 s0, v1
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX11-NEXT: v_readfirstlane_b32 s1, v0
; GFX11-NEXT: ; return to shader part epilog
;
@@ -1707,6 +1712,7 @@ define amdgpu_ps <2 x i32> @s_sitofp_v2i1_to_v2bf16(<2 x i1> inreg %num) {
; GFX12: ; %bb.0:
; GFX12-NEXT: s_and_b32 s0, 1, s0
; GFX12-NEXT: s_bitcmp1_b32 s1, 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s1, -1, 0
; GFX12-NEXT: s_cmp_eq_u32 s0, 1
; GFX12-NEXT: v_cndmask_b32_e64 v0, 0, -1.0, s1
@@ -2042,6 +2048,7 @@ define amdgpu_ps <3 x i32> @s_sitofp_v3i1_to_v3bf16(<3 x i1> inreg %num) {
; GFX11-NEXT: s_and_b32 s0, 1, s0
; GFX11-NEXT: s_and_b32 s1, 1, s1
; GFX11-NEXT: s_bitcmp1_b32 s2, 0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cselect_b32 s2, -1, 0
; GFX11-NEXT: s_cmp_eq_u32 s1, 1
; GFX11-NEXT: v_cndmask_b32_e64 v2, 0, -1.0, s2
@@ -2087,6 +2094,7 @@ define amdgpu_ps <3 x i32> @s_sitofp_v3i1_to_v3bf16(<3 x i1> inreg %num) {
; GFX12-NEXT: s_and_b32 s0, 1, s0
; GFX12-NEXT: s_and_b32 s1, 1, s1
; GFX12-NEXT: s_bitcmp1_b32 s2, 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s2, -1, 0
; GFX12-NEXT: s_cmp_eq_u32 s1, 1
; GFX12-NEXT: v_cndmask_b32_e64 v2, 0, -1.0, s2
@@ -2509,6 +2517,7 @@ define amdgpu_ps <4 x i32> @s_sitofp_v4i1_to_v4bf16(<4 x i1> inreg %num) {
; GFX11-NEXT: s_and_b32 s1, 1, s1
; GFX11-NEXT: s_and_b32 s2, 1, s2
; GFX11-NEXT: s_bitcmp1_b32 s3, 0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cselect_b32 s3, -1, 0
; GFX11-NEXT: s_cmp_eq_u32 s2, 1
; GFX11-NEXT: v_cndmask_b32_e64 v1, 0, -1.0, s3
@@ -2567,6 +2576,7 @@ define amdgpu_ps <4 x i32> @s_sitofp_v4i1_to_v4bf16(<4 x i1> inreg %num) {
; GFX12-NEXT: s_and_b32 s1, 1, s1
; GFX12-NEXT: s_and_b32 s2, 1, s2
; GFX12-NEXT: s_bitcmp1_b32 s3, 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s3, -1, 0
; GFX12-NEXT: s_cmp_eq_u32 s2, 1
; GFX12-NEXT: v_cndmask_b32_e64 v1, 0, -1.0, s3
diff --git a/llvm/test/CodeGen/AMDGPU/idiv-licm.ll b/llvm/test/CodeGen/AMDGPU/idiv-licm.ll
index 655b6e811933c2..9d8ee6a94b1598 100644
--- a/llvm/test/CodeGen/AMDGPU/idiv-licm.ll
+++ b/llvm/test/CodeGen/AMDGPU/idiv-licm.ll
@@ -118,13 +118,14 @@ define amdgpu_kernel void @udiv32_invariant_denom(ptr addrspace(1) nocapture %ar
; GFX11-NEXT: s_mul_i32 s7, s5, s6
; GFX11-NEXT: s_add_i32 s8, s5, 1
; GFX11-NEXT: s_sub_i32 s7, s2, s7
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_sub_i32 s9, s7, s6
; GFX11-NEXT: s_cmp_ge_u32 s7, s6
; GFX11-NEXT: s_cselect_b32 s5, s8, s5
; GFX11-NEXT: s_cselect_b32 s7, s9, s7
; GFX11-NEXT: s_add_i32 s8, s5, 1
; GFX11-NEXT: s_cmp_ge_u32 s7, s6
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cselect_b32 s5, s8, s5
; GFX11-NEXT: s_lshl_b64 s[8:9], s[2:3], 2
; GFX11-NEXT: v_mov_b32_e32 v1, s5
@@ -267,10 +268,11 @@ define amdgpu_kernel void @urem32_invariant_denom(ptr addrspace(1) nocapture %ar
; GFX11-NEXT: s_sub_i32 s5, s2, s5
; GFX11-NEXT: s_sub_i32 s7, s5, s6
; GFX11-NEXT: s_cmp_ge_u32 s5, s6
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_cselect_b32 s5, s7, s5
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_sub_i32 s7, s5, s6
; GFX11-NEXT: s_cmp_ge_u32 s5, s6
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cselect_b32 s5, s7, s5
; GFX11-NEXT: s_lshl_b64 s[8:9], s[2:3], 2
; GFX11-NEXT: v_mov_b32_e32 v1, s5
@@ -419,18 +421,19 @@ define amdgpu_kernel void @sdiv32_invariant_denom(ptr addrspace(1) nocapture %ar
; GFX11-NEXT: s_mul_i32 s7, s6, s2
; GFX11-NEXT: s_add_i32 s8, s6, 1
; GFX11-NEXT: s_sub_i32 s7, s4, s7
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_sub_i32 s9, s7, s2
; GFX11-NEXT: s_cmp_ge_u32 s7, s2
; GFX11-NEXT: s_cselect_b32 s6, s8, s6
; GFX11-NEXT: s_cselect_b32 s7, s9, s7
; GFX11-NEXT: s_add_i32 s8, s6, 1
; GFX11-NEXT: s_cmp_ge_u32 s7, s2
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_cselect_b32 s6, s8, s6
; GFX11-NEXT: s_add_i32 s4, s4, 1
; GFX11-NEXT: s_xor_b32 s6, s6, s3
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_sub_i32 s6, s6, s3
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_4) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: v_mov_b32_e32 v1, s6
; GFX11-NEXT: global_store_b32 v0, v1, s[0:1]
; GFX11-NEXT: s_add_u32 s0, s0, 4
@@ -566,10 +569,11 @@ define amdgpu_kernel void @srem32_invariant_denom(ptr addrspace(1) nocapture %ar
; GFX11-NEXT: s_sub_i32 s5, s3, s5
; GFX11-NEXT: s_sub_i32 s6, s5, s2
; GFX11-NEXT: s_cmp_ge_u32 s5, s2
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_cselect_b32 s5, s6, s5
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_sub_i32 s6, s5, s2
; GFX11-NEXT: s_cmp_ge_u32 s5, s2
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cselect_b32 s5, s6, s5
; GFX11-NEXT: s_add_i32 s3, s3, 1
; GFX11-NEXT: v_mov_b32_e32 v1, s5
@@ -578,6 +582,7 @@ define amdgpu_kernel void @srem32_invariant_denom(ptr addrspace(1) nocapture %ar
; GFX11-NEXT: s_add_u32 s0, s0, 4
; GFX11-NEXT: s_addc_u32 s1, s1, 0
; GFX11-NEXT: s_cmpk_eq_i32 s3, 0x400
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc0 .LBB3_1
; GFX11-NEXT: ; %bb.2: ; %bb2
; GFX11-NEXT: s_endpgm
diff --git a/llvm/test/CodeGen/AMDGPU/indirect-call-known-callees.ll b/llvm/test/CodeGen/AMDGPU/indirect-call-known-callees.ll
index 9f67a48cf3f533..fe0d3633b2cc4b 100644
--- a/llvm/test/CodeGen/AMDGPU/indirect-call-known-callees.ll
+++ b/llvm/test/CodeGen/AMDGPU/indirect-call-known-callees.ll
@@ -66,7 +66,7 @@ define amdgpu_kernel void @indirect_call_known_no_special_inputs() {
; GFX12-NEXT: s_mov_b32 s32, 0
; GFX12-NEXT: s_wait_kmcnt 0x0
; GFX12-NEXT: s_and_b32 s12, 1, s14
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_eq_u32 s12, 1
; GFX12-NEXT: s_cselect_b32 s13, s7, s5
; GFX12-NEXT: s_cselect_b32 s12, s6, s4
diff --git a/llvm/test/CodeGen/AMDGPU/insert-delay-alu-bug.ll b/llvm/test/CodeGen/AMDGPU/insert-delay-alu-bug.ll
index 1b1ef8b76da181..be3414a42c8079 100644
--- a/llvm/test/CodeGen/AMDGPU/insert-delay-alu-bug.ll
+++ b/llvm/test/CodeGen/AMDGPU/insert-delay-alu-bug.ll
@@ -69,13 +69,14 @@ define amdgpu_kernel void @f2(i32 %arg, i32 %arg1, i32 %arg2, i1 %arg3, i32 %arg
; GFX11-NEXT: s_mov_b32 s32, 0
; GFX11-NEXT: s_waitcnt lgkmcnt(0)
; GFX11-NEXT: v_mul_lo_u32 v0, s26, v0
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX11-NEXT: s_cbranch_execz .LBB2_14
; GFX11-NEXT: ; %bb.1: ; %bb14
; GFX11-NEXT: s_load_b128 s[20:23], s[18:19], 0x2c
; GFX11-NEXT: s_waitcnt lgkmcnt(0)
; GFX11-NEXT: s_bitcmp1_b32 s21, 0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cselect_b32 s25, -1, 0
; GFX11-NEXT: s_bitcmp0_b32 s21, 0
; GFX11-NEXT: s_mov_b32 s21, 0
@@ -102,12 +103,13 @@ define amdgpu_kernel void @f2(i32 %arg, i32 %arg1, i32 %arg2, i1 %arg3, i32 %arg
; GFX11-NEXT: .LBB2_4: ; %Flow10
; GFX11-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-NEXT: s_cselect_b32 s0, 1, 0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_cmp_lg_u32 s0, 1
; GFX11-NEXT: s_cbranch_scc1 .LBB2_13
; GFX11-NEXT: ; %bb.5: ; %bb16
; GFX11-NEXT: s_load_b32 s0, s[18:19], 0x54
; GFX11-NEXT: s_bitcmp1_b32 s23, 0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cselect_b32 s8, -1, 0
; GFX11-NEXT: s_and_b32 s1, s23, 1
; GFX11-NEXT: s_waitcnt lgkmcnt(0)
@@ -115,6 +117,7 @@ define amdgpu_kernel void @f2(i32 %arg, i32 %arg1, i32 %arg2, i1 %arg3, i32 %arg
; GFX11-NEXT: s_mov_b32 s0, -1
; GFX11-NEXT: s_cselect_b32 s3, -1, 0
; GFX11-NEXT: s_cmp_eq_u32 s1, 0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc0 .LBB2_9
; GFX11-NEXT: ; %bb.6: ; %bb18.preheader
; GFX11-NEXT: s_load_b128 s[28:31], s[18:19], 0x44
@@ -133,7 +136,7 @@ define amdgpu_kernel void @f2(i32 %arg, i32 %arg1, i32 %arg2, i1 %arg3, i32 %arg
; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_mul_i32 s0, s0, s20
; GFX11-NEXT: s_or_b32 s0, s26, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_4) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_lshl_b64 s[22:23], s[0:1], 1
; GFX11-NEXT: global_load_u16 v0, v0, s[22:23]
; GFX11-NEXT: s_waitcnt vmcnt(0)
@@ -161,22 +164,23 @@ define amdgpu_kernel void @f2(i32 %arg, i32 %arg1, i32 %arg2, i1 %arg3, i32 %arg
; GFX11-NEXT: s_cselect_b32 s15, 1, 0
; GFX11-NEXT: s_and_b32 s16, s8, exec_lo
; GFX11-NEXT: s_cselect_b32 s13, s13, s15
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_bitcmp1_b32 s13, 0
; GFX11-NEXT: s_cselect_b32 s13, 0x100, 0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_or_b32 s9, s13, s9
; GFX11-NEXT: s_cbranch_vccz .LBB2_7
; GFX11-NEXT: ; %bb.8: ; %Flow
; GFX11-NEXT: s_mov_b32 s0, 0
; GFX11-NEXT: .LBB2_9: ; %Flow12
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_b32 vcc_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_vccz .LBB2_13
; GFX11-NEXT: ; %bb.10:
; GFX11-NEXT: s_xor_b32 s0, s3, -1
; GFX11-NEXT: .LBB2_11: ; %bb17
; GFX11-NEXT: ; =>This Inner Loop Header: Depth=1
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_b32 vcc_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_vccz .LBB2_11
; GFX11-NEXT: ; %bb.12: ; %Flow6
@@ -186,6 +190,7 @@ define amdgpu_kernel void @f2(i32 %arg, i32 %arg1, i32 %arg2, i1 %arg3, i32 %arg
; GFX11-NEXT: s_or_not1_b32 s0, s21, exec_lo
; GFX11-NEXT: .LBB2_14: ; %Flow9
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s24
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_and_saveexec_b32 s21, s0
; GFX11-NEXT: s_cbranch_execz .LBB2_16
; GFX11-NEXT: ; %bb.15: ; %bb43
@@ -203,6 +208,7 @@ define amdgpu_kernel void @f2(i32 %arg, i32 %arg1, i32 %arg2, i1 %arg3, i32 %arg
; GFX11-NEXT: s_or_b32 s20, s20, exec_lo
; GFX11-NEXT: .LBB2_16: ; %Flow14
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s21
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_and_saveexec_b32 s0, s20
; GFX11-NEXT: ; %bb.17: ; %UnifiedUnreachableBlock
; GFX11-NEXT: ; divergent unreachable
diff --git a/llvm/test/CodeGen/AMDGPU/insert-delay-alu.mir b/llvm/test/CodeGen/AMDGPU/insert-delay-alu.mir
index c427309353ca07..62bf3bb61a8d92 100644
--- a/llvm/test/CodeGen/AMDGPU/insert-delay-alu.mir
+++ b/llvm/test/CodeGen/AMDGPU/insert-delay-alu.mir
@@ -201,7 +201,6 @@
; CHECK-LABEL: explicit_cmp_cndmask:
; CHECK: ; %bb.0:
; CHECK-NEXT: v_cmp_eq_i32_e64 s[0:1], v0, v1
- ; CHECK-NEXT: s_delay_alu instid0(VALU_DEP_1)
; CHECK-NEXT: v_cndmask_b32_e64 v2, v3, v4, s[0:1]
ret void
}
diff --git a/llvm/test/CodeGen/AMDGPU/insert_vector_elt.v2bf16.ll b/llvm/test/CodeGen/AMDGPU/insert_vector_elt.v2bf16.ll
index 1526d59749a1f3..a2969a14ae6888 100644
--- a/llvm/test/CodeGen/AMDGPU/insert_vector_elt.v2bf16.ll
+++ b/llvm/test/CodeGen/AMDGPU/insert_vector_elt.v2bf16.ll
@@ -1695,18 +1695,22 @@ define amdgpu_kernel void @v_insertelement_v8bf16_dynamic(ptr addrspace(1) %out,
; GFX1250-REAL16-NEXT: s_wait_xcnt 0x0
; GFX1250-REAL16-NEXT: s_cselect_b32 s2, -1, 0
; GFX1250-REAL16-NEXT: s_cmp_eq_u32 s5, 7
+; GFX1250-REAL16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX1250-REAL16-NEXT: s_cselect_b32 s3, -1, 0
; GFX1250-REAL16-NEXT: s_cmp_eq_u32 s5, 4
; GFX1250-REAL16-NEXT: s_cselect_b32 s6, -1, 0
; GFX1250-REAL16-NEXT: s_cmp_eq_u32 s5, 5
+; GFX1250-REAL16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX1250-REAL16-NEXT: s_cselect_b32 s7, -1, 0
; GFX1250-REAL16-NEXT: s_cmp_eq_u32 s5, 2
; GFX1250-REAL16-NEXT: s_cselect_b32 s8, -1, 0
; GFX1250-REAL16-NEXT: s_cmp_eq_u32 s5, 3
+; GFX1250-REAL16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX1250-REAL16-NEXT: s_cselect_b32 s9, -1, 0
; GFX1250-REAL16-NEXT: s_cmp_eq_u32 s5, 0
; GFX1250-REAL16-NEXT: s_cselect_b32 s10, -1, 0
; GFX1250-REAL16-NEXT: s_cmp_eq_u32 s5, 1
+; GFX1250-REAL16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1250-REAL16-NEXT: s_cselect_b32 s5, -1, 0
; GFX1250-REAL16-NEXT: s_wait_loadcnt 0x0
; GFX1250-REAL16-NEXT: v_cndmask_b16 v5.l, v3.l, s4, s2
@@ -2344,34 +2348,42 @@ define amdgpu_kernel void @v_insertelement_v16bf16_dynamic(ptr addrspace(1) %out
; GFX1250-REAL16-NEXT: s_wait_xcnt 0x0
; GFX1250-REAL16-NEXT: s_cselect_b32 s2, -1, 0
; GFX1250-REAL16-NEXT: s_cmp_eq_u32 s5, 7
+; GFX1250-REAL16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX1250-REAL16-NEXT: s_cselect_b32 s3, -1, 0
; GFX1250-REAL16-NEXT: s_cmp_eq_u32 s5, 4
; GFX1250-REAL16-NEXT: s_cselect_b32 s6, -1, 0
; GFX1250-REAL16-NEXT: s_cmp_eq_u32 s5, 5
+; GFX1250-REAL16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX1250-REAL16-NEXT: s_cselect_b32 s7, -1, 0
; GFX1250-REAL16-NEXT: s_cmp_eq_u32 s5, 2
; GFX1250-REAL16-NEXT: s_cselect_b32 s8, -1, 0
; GFX1250-REAL16-NEXT: s_cmp_eq_u32 s5, 3
+; GFX1250-REAL16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX1250-REAL16-NEXT: s_cselect_b32 s9, -1, 0
; GFX1250-REAL16-NEXT: s_cmp_eq_u32 s5, 0
; GFX1250-REAL16-NEXT: s_cselect_b32 s10, -1, 0
; GFX1250-REAL16-NEXT: s_cmp_eq_u32 s5, 1
+; GFX1250-REAL16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX1250-REAL16-NEXT: s_cselect_b32 s11, -1, 0
; GFX1250-REAL16-NEXT: s_cmp_eq_u32 s5, 14
; GFX1250-REAL16-NEXT: s_cselect_b32 s12, -1, 0
; GFX1250-REAL16-NEXT: s_cmp_eq_u32 s5, 15
+; GFX1250-REAL16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX1250-REAL16-NEXT: s_cselect_b32 s13, -1, 0
; GFX1250-REAL16-NEXT: s_cmp_eq_u32 s5, 12
; GFX1250-REAL16-NEXT: s_cselect_b32 s14, -1, 0
; GFX1250-REAL16-NEXT: s_cmp_eq_u32 s5, 13
+; GFX1250-REAL16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX1250-REAL16-NEXT: s_cselect_b32 s15, -1, 0
; GFX1250-REAL16-NEXT: s_cmp_eq_u32 s5, 10
; GFX1250-REAL16-NEXT: s_cselect_b32 s16, -1, 0
; GFX1250-REAL16-NEXT: s_cmp_eq_u32 s5, 11
+; GFX1250-REAL16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX1250-REAL16-NEXT: s_cselect_b32 s17, -1, 0
; GFX1250-REAL16-NEXT: s_cmp_eq_u32 s5, 8
; GFX1250-REAL16-NEXT: s_cselect_b32 s18, -1, 0
; GFX1250-REAL16-NEXT: s_cmp_eq_u32 s5, 9
+; GFX1250-REAL16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1250-REAL16-NEXT: s_cselect_b32 s5, -1, 0
; GFX1250-REAL16-NEXT: s_wait_loadcnt 0x1
; GFX1250-REAL16-NEXT: v_cndmask_b16 v11.l, v3.l, s4, s2
diff --git a/llvm/test/CodeGen/AMDGPU/insert_vector_elt.v2i16.ll b/llvm/test/CodeGen/AMDGPU/insert_vector_elt.v2i16.ll
index 7bccc761fc5bbd..c46430c34f5e01 100644
--- a/llvm/test/CodeGen/AMDGPU/insert_vector_elt.v2i16.ll
+++ b/llvm/test/CodeGen/AMDGPU/insert_vector_elt.v2i16.ll
@@ -2911,25 +2911,29 @@ define amdgpu_kernel void @v_insertelement_v8f16_dynamic(ptr addrspace(1) %out,
; GFX11-TRUE16-NEXT: s_load_b128 s[0:3], s[4:5], 0x0
; GFX11-TRUE16-NEXT: v_and_b32_e32 v0, 0x3ff, v0
; GFX11-TRUE16-NEXT: s_load_b64 s[4:5], s[4:5], 0x10
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_3) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_lshlrev_b32_e32 v5, 4, v0
; GFX11-TRUE16-NEXT: s_waitcnt lgkmcnt(0)
; GFX11-TRUE16-NEXT: global_load_b128 v[0:3], v5, s[2:3]
; GFX11-TRUE16-NEXT: s_cmp_eq_u32 s5, 6
; GFX11-TRUE16-NEXT: s_cselect_b32 s2, -1, 0
; GFX11-TRUE16-NEXT: s_cmp_eq_u32 s5, 7
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_cselect_b32 s3, -1, 0
; GFX11-TRUE16-NEXT: s_cmp_eq_u32 s5, 4
; GFX11-TRUE16-NEXT: s_cselect_b32 s6, -1, 0
; GFX11-TRUE16-NEXT: s_cmp_eq_u32 s5, 5
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_cselect_b32 s7, -1, 0
; GFX11-TRUE16-NEXT: s_cmp_eq_u32 s5, 2
; GFX11-TRUE16-NEXT: s_cselect_b32 s8, -1, 0
; GFX11-TRUE16-NEXT: s_cmp_eq_u32 s5, 3
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_cselect_b32 s9, -1, 0
; GFX11-TRUE16-NEXT: s_cmp_eq_u32 s5, 0
; GFX11-TRUE16-NEXT: s_cselect_b32 s10, -1, 0
; GFX11-TRUE16-NEXT: s_cmp_eq_u32 s5, 1
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_cselect_b32 s5, -1, 0
; GFX11-TRUE16-NEXT: s_waitcnt vmcnt(0)
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v6, 16, v3
@@ -2952,7 +2956,7 @@ define amdgpu_kernel void @v_insertelement_v8f16_dynamic(ptr addrspace(1) %out,
; GFX11-FAKE16-NEXT: s_load_b128 s[0:3], s[4:5], 0x0
; GFX11-FAKE16-NEXT: v_and_b32_e32 v0, 0x3ff, v0
; GFX11-FAKE16-NEXT: s_load_b64 s[4:5], s[4:5], 0x10
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_3) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v4, 4, v0
; GFX11-FAKE16-NEXT: s_waitcnt lgkmcnt(0)
; GFX11-FAKE16-NEXT: global_load_b128 v[0:3], v4, s[2:3]
@@ -3559,34 +3563,42 @@ define amdgpu_kernel void @v_insertelement_v16f16_dynamic(ptr addrspace(1) %out,
; GFX11-TRUE16-NEXT: global_load_b128 v[0:3], v12, s[2:3]
; GFX11-TRUE16-NEXT: global_load_b128 v[4:7], v12, s[2:3] offset:16
; GFX11-TRUE16-NEXT: s_cmp_eq_u32 s5, 6
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_cselect_b32 s2, -1, 0
; GFX11-TRUE16-NEXT: s_cmp_eq_u32 s5, 7
; GFX11-TRUE16-NEXT: s_cselect_b32 s3, -1, 0
; GFX11-TRUE16-NEXT: s_cmp_eq_u32 s5, 4
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_cselect_b32 s6, -1, 0
; GFX11-TRUE16-NEXT: s_cmp_eq_u32 s5, 5
; GFX11-TRUE16-NEXT: s_cselect_b32 s7, -1, 0
; GFX11-TRUE16-NEXT: s_cmp_eq_u32 s5, 2
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_cselect_b32 s8, -1, 0
; GFX11-TRUE16-NEXT: s_cmp_eq_u32 s5, 3
; GFX11-TRUE16-NEXT: s_cselect_b32 s9, -1, 0
; GFX11-TRUE16-NEXT: s_cmp_eq_u32 s5, 0
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_cselect_b32 s10, -1, 0
; GFX11-TRUE16-NEXT: s_cmp_eq_u32 s5, 1
; GFX11-TRUE16-NEXT: s_cselect_b32 s11, -1, 0
; GFX11-TRUE16-NEXT: s_cmp_eq_u32 s5, 14
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_cselect_b32 s12, -1, 0
; GFX11-TRUE16-NEXT: s_cmp_eq_u32 s5, 15
; GFX11-TRUE16-NEXT: s_cselect_b32 s13, -1, 0
; GFX11-TRUE16-NEXT: s_cmp_eq_u32 s5, 12
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_cselect_b32 s14, -1, 0
; GFX11-TRUE16-NEXT: s_cmp_eq_u32 s5, 13
; GFX11-TRUE16-NEXT: s_cselect_b32 s15, -1, 0
; GFX11-TRUE16-NEXT: s_cmp_eq_u32 s5, 10
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_cselect_b32 s16, -1, 0
; GFX11-TRUE16-NEXT: s_cmp_eq_u32 s5, 11
; GFX11-TRUE16-NEXT: s_cselect_b32 s17, -1, 0
; GFX11-TRUE16-NEXT: s_cmp_eq_u32 s5, 8
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_cselect_b32 s18, -1, 0
; GFX11-TRUE16-NEXT: s_cmp_eq_u32 s5, 9
; GFX11-TRUE16-NEXT: s_cselect_b32 s5, -1, 0
@@ -3633,6 +3645,7 @@ define amdgpu_kernel void @v_insertelement_v16f16_dynamic(ptr addrspace(1) %out,
; GFX11-FAKE16-NEXT: global_load_b128 v[0:3], v8, s[2:3]
; GFX11-FAKE16-NEXT: global_load_b128 v[4:7], v8, s[2:3] offset:16
; GFX11-FAKE16-NEXT: s_cmp_eq_u32 s5, 6
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_cselect_b32 s2, -1, 0
; GFX11-FAKE16-NEXT: s_cmp_eq_u32 s5, 7
; GFX11-FAKE16-NEXT: s_waitcnt vmcnt(1)
diff --git a/llvm/test/CodeGen/AMDGPU/insert_waitcnt_for_precise_memory.ll b/llvm/test/CodeGen/AMDGPU/insert_waitcnt_for_precise_memory.ll
index 5eccd311a970b9..cf281f16f9b069 100644
--- a/llvm/test/CodeGen/AMDGPU/insert_waitcnt_for_precise_memory.ll
+++ b/llvm/test/CodeGen/AMDGPU/insert_waitcnt_for_precise_memory.ll
@@ -109,7 +109,7 @@ define void @syncscope_workgroup_nortn(ptr %addr, float %val) {
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-NEXT: v_mov_b32_e32 v4, v3
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB0_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -247,11 +247,12 @@ define i32 @atomic_nand_i32_global(ptr addrspace(1) %ptr) nounwind {
; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v3
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB1_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: v_mov_b32_e32 v0, v2
; GFX11-NEXT: s_setpc_b64 s[30:31]
;
@@ -281,9 +282,11 @@ define i32 @atomic_nand_i32_global(ptr addrspace(1) %ptr) nounwind {
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB1_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: v_mov_b32_e32 v0, v2
; GFX12-NEXT: s_setpc_b64 s[30:31]
%result = atomicrmw nand ptr addrspace(1) %ptr, i32 4 seq_cst
@@ -565,6 +568,7 @@ define amdgpu_kernel void @udiv_i32(ptr addrspace(1) %out, i32 %x, i32 %y) {
; GFX11-NEXT: s_add_i32 s5, s4, 1
; GFX11-NEXT: s_sub_i32 s6, s2, s3
; GFX11-NEXT: s_cmp_ge_u32 s2, s3
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_3) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_cselect_b32 s4, s5, s4
; GFX11-NEXT: s_cselect_b32 s2, s6, s2
; GFX11-NEXT: s_add_i32 s5, s4, 1
@@ -766,7 +770,7 @@ define amdgpu_kernel void @atomic_add_local(ptr addrspace(3) %local) {
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: s_mov_b32 s1, exec_lo
; GFX11-NEXT: v_mbcnt_lo_u32_b32 v0, s0, 0
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX11-NEXT: s_cbranch_execz .LBB5_2
; GFX11-NEXT: ; %bb.1:
@@ -787,7 +791,7 @@ define amdgpu_kernel void @atomic_add_local(ptr addrspace(3) %local) {
; GFX12-NEXT: s_mov_b32 s0, exec_lo
; GFX12-NEXT: s_mov_b32 s1, exec_lo
; GFX12-NEXT: v_mbcnt_lo_u32_b32 v0, s0, 0
-; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX12-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX12-NEXT: s_cbranch_execz .LBB5_2
; GFX12-NEXT: ; %bb.1:
@@ -1004,7 +1008,7 @@ define amdgpu_kernel void @atomic_add_ret_local(ptr addrspace(1) %out, ptr addrs
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_mbcnt_lo_u32_b32 v0, s1, 0
; GFX11-NEXT: ; implicit-def: $vgpr1
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX11-NEXT: s_cbranch_execz .LBB7_2
; GFX11-NEXT: ; %bb.1:
@@ -1035,7 +1039,7 @@ define amdgpu_kernel void @atomic_add_ret_local(ptr addrspace(1) %out, ptr addrs
; GFX12-NEXT: s_mov_b32 s0, exec_lo
; GFX12-NEXT: v_mbcnt_lo_u32_b32 v0, s1, 0
; GFX12-NEXT: ; implicit-def: $vgpr1
-; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX12-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX12-NEXT: s_cbranch_execz .LBB7_2
; GFX12-NEXT: ; %bb.1:
@@ -1191,7 +1195,7 @@ define amdgpu_kernel void @add_i32_constant(ptr addrspace(1) %out, ptr addrspace
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_mbcnt_lo_u32_b32 v0, s1, 0
; GFX11-NEXT: ; implicit-def: $vgpr1
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX11-NEXT: s_cbranch_execz .LBB8_2
; GFX11-NEXT: ; %bb.1:
@@ -1221,7 +1225,7 @@ define amdgpu_kernel void @add_i32_constant(ptr addrspace(1) %out, ptr addrspace
; GFX12-NEXT: s_mov_b32 s0, exec_lo
; GFX12-NEXT: v_mbcnt_lo_u32_b32 v0, s1, 0
; GFX12-NEXT: ; implicit-def: $vgpr1
-; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX12-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX12-NEXT: s_cbranch_execz .LBB8_2
; GFX12-NEXT: ; %bb.1:
diff --git a/llvm/test/CodeGen/AMDGPU/integer-mad-patterns.ll b/llvm/test/CodeGen/AMDGPU/integer-mad-patterns.ll
index 0a3874e6f70c0e..c75225c37e3613 100644
--- a/llvm/test/CodeGen/AMDGPU/integer-mad-patterns.ll
+++ b/llvm/test/CodeGen/AMDGPU/integer-mad-patterns.ll
@@ -1010,14 +1010,14 @@ define <3 x i16> @clpeak_imad_pat_v3i16(<3 x i16> %x, <3 x i16> %y) {
; GFX9-SDAG: ; %bb.0: ; %entry
; GFX9-SDAG-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX9-SDAG-NEXT: v_pk_add_u16 v0, v0, 1 op_sel_hi:[1,0]
-; GFX9-SDAG-NEXT: v_pk_add_u16 v1, v1, 1
+; GFX9-SDAG-NEXT: v_pk_add_u16 v1, v1, 1 op_sel_hi:[1,0]
; GFX9-SDAG-NEXT: v_pk_mad_u16 v4, v1, v3, v1
; GFX9-SDAG-NEXT: v_pk_mad_u16 v5, v0, v2, v0
; GFX9-SDAG-NEXT: v_pk_mul_lo_u16 v6, v5, v2
; GFX9-SDAG-NEXT: v_pk_mul_lo_u16 v7, v4, v3
; GFX9-SDAG-NEXT: v_pk_mad_u16 v0, v0, v2, 1 op_sel_hi:[1,1,0]
-; GFX9-SDAG-NEXT: v_pk_mad_u16 v1, v1, v3, 1
-; GFX9-SDAG-NEXT: v_pk_mad_u16 v3, v4, v3, 1
+; GFX9-SDAG-NEXT: v_pk_mad_u16 v1, v1, v3, 1 op_sel_hi:[1,1,0]
+; GFX9-SDAG-NEXT: v_pk_mad_u16 v3, v4, v3, 1 op_sel_hi:[1,1,0]
; GFX9-SDAG-NEXT: v_pk_mad_u16 v2, v5, v2, 1 op_sel_hi:[1,1,0]
; GFX9-SDAG-NEXT: v_pk_mul_lo_u16 v1, v7, v1
; GFX9-SDAG-NEXT: v_pk_mul_lo_u16 v0, v6, v0
@@ -1048,14 +1048,14 @@ define <3 x i16> @clpeak_imad_pat_v3i16(<3 x i16> %x, <3 x i16> %y) {
; GFX10-SDAG: ; %bb.0: ; %entry
; GFX10-SDAG-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX10-SDAG-NEXT: v_pk_add_u16 v0, v0, 1 op_sel_hi:[1,0]
-; GFX10-SDAG-NEXT: v_pk_add_u16 v1, v1, 1
+; GFX10-SDAG-NEXT: v_pk_add_u16 v1, v1, 1 op_sel_hi:[1,0]
; GFX10-SDAG-NEXT: v_pk_mad_u16 v4, v0, v2, v0
; GFX10-SDAG-NEXT: v_pk_mad_u16 v5, v1, v3, v1
; GFX10-SDAG-NEXT: v_pk_mad_u16 v0, v0, v2, 1 op_sel_hi:[1,1,0]
-; GFX10-SDAG-NEXT: v_pk_mad_u16 v1, v1, v3, 1
+; GFX10-SDAG-NEXT: v_pk_mad_u16 v1, v1, v3, 1 op_sel_hi:[1,1,0]
; GFX10-SDAG-NEXT: v_pk_mul_lo_u16 v6, v4, v2
; GFX10-SDAG-NEXT: v_pk_mul_lo_u16 v7, v5, v3
-; GFX10-SDAG-NEXT: v_pk_mad_u16 v3, v5, v3, 1
+; GFX10-SDAG-NEXT: v_pk_mad_u16 v3, v5, v3, 1 op_sel_hi:[1,1,0]
; GFX10-SDAG-NEXT: v_pk_mad_u16 v2, v4, v2, 1 op_sel_hi:[1,1,0]
; GFX10-SDAG-NEXT: v_pk_mul_lo_u16 v0, v6, v0
; GFX10-SDAG-NEXT: v_pk_mul_lo_u16 v1, v7, v1
@@ -1086,16 +1086,16 @@ define <3 x i16> @clpeak_imad_pat_v3i16(<3 x i16> %x, <3 x i16> %y) {
; GFX11-SDAG: ; %bb.0: ; %entry
; GFX11-SDAG-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-SDAG-NEXT: v_pk_add_u16 v0, v0, 1 op_sel_hi:[1,0]
-; GFX11-SDAG-NEXT: v_pk_add_u16 v1, v1, 1
+; GFX11-SDAG-NEXT: v_pk_add_u16 v1, v1, 1 op_sel_hi:[1,0]
; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-SDAG-NEXT: v_pk_mad_u16 v4, v0, v2, v0
; GFX11-SDAG-NEXT: v_pk_mad_u16 v5, v1, v3, v1
; GFX11-SDAG-NEXT: v_pk_mad_u16 v0, v0, v2, 1 op_sel_hi:[1,1,0]
-; GFX11-SDAG-NEXT: v_pk_mad_u16 v1, v1, v3, 1
+; GFX11-SDAG-NEXT: v_pk_mad_u16 v1, v1, v3, 1 op_sel_hi:[1,1,0]
; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
; GFX11-SDAG-NEXT: v_pk_mul_lo_u16 v6, v4, v2
; GFX11-SDAG-NEXT: v_pk_mul_lo_u16 v7, v5, v3
-; GFX11-SDAG-NEXT: v_pk_mad_u16 v3, v5, v3, 1
+; GFX11-SDAG-NEXT: v_pk_mad_u16 v3, v5, v3, 1 op_sel_hi:[1,1,0]
; GFX11-SDAG-NEXT: v_pk_mad_u16 v2, v4, v2, 1 op_sel_hi:[1,1,0]
; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
; GFX11-SDAG-NEXT: v_pk_mul_lo_u16 v0, v6, v0
@@ -1136,16 +1136,16 @@ define <3 x i16> @clpeak_imad_pat_v3i16(<3 x i16> %x, <3 x i16> %y) {
; GFX1200-SDAG-NEXT: s_wait_bvhcnt 0x0
; GFX1200-SDAG-NEXT: s_wait_kmcnt 0x0
; GFX1200-SDAG-NEXT: v_pk_add_u16 v0, v0, 1 op_sel_hi:[1,0]
-; GFX1200-SDAG-NEXT: v_pk_add_u16 v1, v1, 1
+; GFX1200-SDAG-NEXT: v_pk_add_u16 v1, v1, 1 op_sel_hi:[1,0]
; GFX1200-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1200-SDAG-NEXT: v_pk_mad_u16 v4, v0, v2, v0
; GFX1200-SDAG-NEXT: v_pk_mad_u16 v5, v1, v3, v1
; GFX1200-SDAG-NEXT: v_pk_mad_u16 v0, v0, v2, 1 op_sel_hi:[1,1,0]
-; GFX1200-SDAG-NEXT: v_pk_mad_u16 v1, v1, v3, 1
+; GFX1200-SDAG-NEXT: v_pk_mad_u16 v1, v1, v3, 1 op_sel_hi:[1,1,0]
; GFX1200-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
; GFX1200-SDAG-NEXT: v_pk_mul_lo_u16 v6, v4, v2
; GFX1200-SDAG-NEXT: v_pk_mul_lo_u16 v7, v5, v3
-; GFX1200-SDAG-NEXT: v_pk_mad_u16 v3, v5, v3, 1
+; GFX1200-SDAG-NEXT: v_pk_mad_u16 v3, v5, v3, 1 op_sel_hi:[1,1,0]
; GFX1200-SDAG-NEXT: v_pk_mad_u16 v2, v4, v2, 1 op_sel_hi:[1,1,0]
; GFX1200-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
; GFX1200-SDAG-NEXT: v_pk_mul_lo_u16 v0, v6, v0
@@ -1187,16 +1187,16 @@ define <3 x i16> @clpeak_imad_pat_v3i16(<3 x i16> %x, <3 x i16> %y) {
; GFX1250-SDAG-NEXT: s_wait_loadcnt_dscnt 0x0
; GFX1250-SDAG-NEXT: s_wait_kmcnt 0x0
; GFX1250-SDAG-NEXT: v_pk_add_u16 v0, v0, 1 op_sel_hi:[1,0]
-; GFX1250-SDAG-NEXT: v_pk_add_u16 v1, v1, 1
+; GFX1250-SDAG-NEXT: v_pk_add_u16 v1, v1, 1 op_sel_hi:[1,0]
; GFX1250-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-SDAG-NEXT: v_pk_mad_u16 v4, v0, v2, v0
; GFX1250-SDAG-NEXT: v_pk_mad_u16 v5, v1, v3, v1
; GFX1250-SDAG-NEXT: v_pk_mad_u16 v0, v0, v2, 1 op_sel_hi:[1,1,0]
-; GFX1250-SDAG-NEXT: v_pk_mad_u16 v1, v1, v3, 1
+; GFX1250-SDAG-NEXT: v_pk_mad_u16 v1, v1, v3, 1 op_sel_hi:[1,1,0]
; GFX1250-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
; GFX1250-SDAG-NEXT: v_pk_mul_lo_u16 v6, v4, v2
; GFX1250-SDAG-NEXT: v_pk_mul_lo_u16 v7, v5, v3
-; GFX1250-SDAG-NEXT: v_pk_mad_u16 v3, v5, v3, 1
+; GFX1250-SDAG-NEXT: v_pk_mad_u16 v3, v5, v3, 1 op_sel_hi:[1,1,0]
; GFX1250-SDAG-NEXT: v_pk_mad_u16 v2, v4, v2, 1 op_sel_hi:[1,1,0]
; GFX1250-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
; GFX1250-SDAG-NEXT: v_pk_mul_lo_u16 v0, v6, v0
@@ -2534,14 +2534,14 @@ define <3 x i16> @clpeak_umad_pat_v3i16(<3 x i16> %x, <3 x i16> %y) {
; GFX9-SDAG: ; %bb.0: ; %entry
; GFX9-SDAG-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX9-SDAG-NEXT: v_pk_add_u16 v0, v0, 1 op_sel_hi:[1,0]
-; GFX9-SDAG-NEXT: v_pk_add_u16 v1, v1, 1
+; GFX9-SDAG-NEXT: v_pk_add_u16 v1, v1, 1 op_sel_hi:[1,0]
; GFX9-SDAG-NEXT: v_pk_mad_u16 v4, v1, v3, v1
; GFX9-SDAG-NEXT: v_pk_mad_u16 v5, v0, v2, v0
; GFX9-SDAG-NEXT: v_pk_mul_lo_u16 v6, v5, v2
; GFX9-SDAG-NEXT: v_pk_mul_lo_u16 v7, v4, v3
; GFX9-SDAG-NEXT: v_pk_mad_u16 v0, v0, v2, 1 op_sel_hi:[1,1,0]
-; GFX9-SDAG-NEXT: v_pk_mad_u16 v1, v1, v3, 1
-; GFX9-SDAG-NEXT: v_pk_mad_u16 v3, v4, v3, 1
+; GFX9-SDAG-NEXT: v_pk_mad_u16 v1, v1, v3, 1 op_sel_hi:[1,1,0]
+; GFX9-SDAG-NEXT: v_pk_mad_u16 v3, v4, v3, 1 op_sel_hi:[1,1,0]
; GFX9-SDAG-NEXT: v_pk_mad_u16 v2, v5, v2, 1 op_sel_hi:[1,1,0]
; GFX9-SDAG-NEXT: v_pk_mul_lo_u16 v1, v7, v1
; GFX9-SDAG-NEXT: v_pk_mul_lo_u16 v0, v6, v0
@@ -2572,14 +2572,14 @@ define <3 x i16> @clpeak_umad_pat_v3i16(<3 x i16> %x, <3 x i16> %y) {
; GFX10-SDAG: ; %bb.0: ; %entry
; GFX10-SDAG-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX10-SDAG-NEXT: v_pk_add_u16 v0, v0, 1 op_sel_hi:[1,0]
-; GFX10-SDAG-NEXT: v_pk_add_u16 v1, v1, 1
+; GFX10-SDAG-NEXT: v_pk_add_u16 v1, v1, 1 op_sel_hi:[1,0]
; GFX10-SDAG-NEXT: v_pk_mad_u16 v4, v0, v2, v0
; GFX10-SDAG-NEXT: v_pk_mad_u16 v5, v1, v3, v1
; GFX10-SDAG-NEXT: v_pk_mad_u16 v0, v0, v2, 1 op_sel_hi:[1,1,0]
-; GFX10-SDAG-NEXT: v_pk_mad_u16 v1, v1, v3, 1
+; GFX10-SDAG-NEXT: v_pk_mad_u16 v1, v1, v3, 1 op_sel_hi:[1,1,0]
; GFX10-SDAG-NEXT: v_pk_mul_lo_u16 v6, v4, v2
; GFX10-SDAG-NEXT: v_pk_mul_lo_u16 v7, v5, v3
-; GFX10-SDAG-NEXT: v_pk_mad_u16 v3, v5, v3, 1
+; GFX10-SDAG-NEXT: v_pk_mad_u16 v3, v5, v3, 1 op_sel_hi:[1,1,0]
; GFX10-SDAG-NEXT: v_pk_mad_u16 v2, v4, v2, 1 op_sel_hi:[1,1,0]
; GFX10-SDAG-NEXT: v_pk_mul_lo_u16 v0, v6, v0
; GFX10-SDAG-NEXT: v_pk_mul_lo_u16 v1, v7, v1
@@ -2610,16 +2610,16 @@ define <3 x i16> @clpeak_umad_pat_v3i16(<3 x i16> %x, <3 x i16> %y) {
; GFX11-SDAG: ; %bb.0: ; %entry
; GFX11-SDAG-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-SDAG-NEXT: v_pk_add_u16 v0, v0, 1 op_sel_hi:[1,0]
-; GFX11-SDAG-NEXT: v_pk_add_u16 v1, v1, 1
+; GFX11-SDAG-NEXT: v_pk_add_u16 v1, v1, 1 op_sel_hi:[1,0]
; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-SDAG-NEXT: v_pk_mad_u16 v4, v0, v2, v0
; GFX11-SDAG-NEXT: v_pk_mad_u16 v5, v1, v3, v1
; GFX11-SDAG-NEXT: v_pk_mad_u16 v0, v0, v2, 1 op_sel_hi:[1,1,0]
-; GFX11-SDAG-NEXT: v_pk_mad_u16 v1, v1, v3, 1
+; GFX11-SDAG-NEXT: v_pk_mad_u16 v1, v1, v3, 1 op_sel_hi:[1,1,0]
; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
; GFX11-SDAG-NEXT: v_pk_mul_lo_u16 v6, v4, v2
; GFX11-SDAG-NEXT: v_pk_mul_lo_u16 v7, v5, v3
-; GFX11-SDAG-NEXT: v_pk_mad_u16 v3, v5, v3, 1
+; GFX11-SDAG-NEXT: v_pk_mad_u16 v3, v5, v3, 1 op_sel_hi:[1,1,0]
; GFX11-SDAG-NEXT: v_pk_mad_u16 v2, v4, v2, 1 op_sel_hi:[1,1,0]
; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
; GFX11-SDAG-NEXT: v_pk_mul_lo_u16 v0, v6, v0
@@ -2660,16 +2660,16 @@ define <3 x i16> @clpeak_umad_pat_v3i16(<3 x i16> %x, <3 x i16> %y) {
; GFX1200-SDAG-NEXT: s_wait_bvhcnt 0x0
; GFX1200-SDAG-NEXT: s_wait_kmcnt 0x0
; GFX1200-SDAG-NEXT: v_pk_add_u16 v0, v0, 1 op_sel_hi:[1,0]
-; GFX1200-SDAG-NEXT: v_pk_add_u16 v1, v1, 1
+; GFX1200-SDAG-NEXT: v_pk_add_u16 v1, v1, 1 op_sel_hi:[1,0]
; GFX1200-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1200-SDAG-NEXT: v_pk_mad_u16 v4, v0, v2, v0
; GFX1200-SDAG-NEXT: v_pk_mad_u16 v5, v1, v3, v1
; GFX1200-SDAG-NEXT: v_pk_mad_u16 v0, v0, v2, 1 op_sel_hi:[1,1,0]
-; GFX1200-SDAG-NEXT: v_pk_mad_u16 v1, v1, v3, 1
+; GFX1200-SDAG-NEXT: v_pk_mad_u16 v1, v1, v3, 1 op_sel_hi:[1,1,0]
; GFX1200-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
; GFX1200-SDAG-NEXT: v_pk_mul_lo_u16 v6, v4, v2
; GFX1200-SDAG-NEXT: v_pk_mul_lo_u16 v7, v5, v3
-; GFX1200-SDAG-NEXT: v_pk_mad_u16 v3, v5, v3, 1
+; GFX1200-SDAG-NEXT: v_pk_mad_u16 v3, v5, v3, 1 op_sel_hi:[1,1,0]
; GFX1200-SDAG-NEXT: v_pk_mad_u16 v2, v4, v2, 1 op_sel_hi:[1,1,0]
; GFX1200-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
; GFX1200-SDAG-NEXT: v_pk_mul_lo_u16 v0, v6, v0
@@ -2711,16 +2711,16 @@ define <3 x i16> @clpeak_umad_pat_v3i16(<3 x i16> %x, <3 x i16> %y) {
; GFX1250-SDAG-NEXT: s_wait_loadcnt_dscnt 0x0
; GFX1250-SDAG-NEXT: s_wait_kmcnt 0x0
; GFX1250-SDAG-NEXT: v_pk_add_u16 v0, v0, 1 op_sel_hi:[1,0]
-; GFX1250-SDAG-NEXT: v_pk_add_u16 v1, v1, 1
+; GFX1250-SDAG-NEXT: v_pk_add_u16 v1, v1, 1 op_sel_hi:[1,0]
; GFX1250-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-SDAG-NEXT: v_pk_mad_u16 v4, v0, v2, v0
; GFX1250-SDAG-NEXT: v_pk_mad_u16 v5, v1, v3, v1
; GFX1250-SDAG-NEXT: v_pk_mad_u16 v0, v0, v2, 1 op_sel_hi:[1,1,0]
-; GFX1250-SDAG-NEXT: v_pk_mad_u16 v1, v1, v3, 1
+; GFX1250-SDAG-NEXT: v_pk_mad_u16 v1, v1, v3, 1 op_sel_hi:[1,1,0]
; GFX1250-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
; GFX1250-SDAG-NEXT: v_pk_mul_lo_u16 v6, v4, v2
; GFX1250-SDAG-NEXT: v_pk_mul_lo_u16 v7, v5, v3
-; GFX1250-SDAG-NEXT: v_pk_mad_u16 v3, v5, v3, 1
+; GFX1250-SDAG-NEXT: v_pk_mad_u16 v3, v5, v3, 1 op_sel_hi:[1,1,0]
; GFX1250-SDAG-NEXT: v_pk_mad_u16 v2, v4, v2, 1 op_sel_hi:[1,1,0]
; GFX1250-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
; GFX1250-SDAG-NEXT: v_pk_mul_lo_u16 v0, v6, v0
@@ -6793,32 +6793,31 @@ define i64 @clpeak_imad_pat_i64(i64 %x, i64 %y) {
; GFX11-SDAG: ; %bb.0: ; %entry
; GFX11-SDAG-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-SDAG-NEXT: v_add_co_u32 v4, vcc_lo, v0, 1
-; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-SDAG-NEXT: v_add_co_ci_u32_e64 v5, null, 0, v1, vcc_lo
+; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_3)
; GFX11-SDAG-NEXT: v_mul_lo_u32 v7, v4, v3
; GFX11-SDAG-NEXT: v_mad_u64_u32 v[0:1], null, v4, v2, 0
-; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-SDAG-NEXT: v_mul_lo_u32 v6, v5, v2
+; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_3)
; GFX11-SDAG-NEXT: v_add3_u32 v1, v1, v7, v6
-; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-SDAG-NEXT: v_add_co_u32 v6, vcc_lo, v0, v4
+; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-SDAG-NEXT: v_add_co_ci_u32_e64 v5, null, v1, v5, vcc_lo
-; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_3)
; GFX11-SDAG-NEXT: v_mul_lo_u32 v7, v6, v3
; GFX11-SDAG-NEXT: v_mad_u64_u32 v[3:4], null, v6, v2, 0
+; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-SDAG-NEXT: v_mul_lo_u32 v2, v5, v2
-; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-SDAG-NEXT: v_mul_lo_u32 v1, v3, v1
+; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-SDAG-NEXT: v_add3_u32 v4, v4, v7, v2
-; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-SDAG-NEXT: v_mul_lo_u32 v2, v4, v0
; GFX11-SDAG-NEXT: v_mad_u64_u32 v[5:6], null, v3, v0, v[3:4]
+; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-SDAG-NEXT: v_add3_u32 v6, v2, v6, v1
-; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-SDAG-NEXT: v_mul_lo_u32 v2, v5, v4
+; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-SDAG-NEXT: v_mul_lo_u32 v4, v6, v3
; GFX11-SDAG-NEXT: v_mad_u64_u32 v[0:1], null, v5, v3, v[5:6]
-; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-SDAG-NEXT: v_add3_u32 v1, v4, v1, v2
; GFX11-SDAG-NEXT: s_setpc_b64 s[30:31]
;
@@ -6826,20 +6825,19 @@ define i64 @clpeak_imad_pat_i64(i64 %x, i64 %y) {
; GFX11-GISEL: ; %bb.0: ; %entry
; GFX11-GISEL-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-GISEL-NEXT: v_add_co_u32 v8, vcc_lo, v0, 1
-; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-GISEL-NEXT: v_add_co_ci_u32_e64 v9, null, 0, v1, vcc_lo
+; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-GISEL-NEXT: v_mad_u64_u32 v[0:1], null, v8, v2, 0
-; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-GISEL-NEXT: v_mad_u64_u32 v[4:5], null, v8, v3, v[1:2]
-; GFX11-GISEL-NEXT: v_add_co_u32 v1, vcc_lo, v0, v8
; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX11-GISEL-NEXT: v_add_co_u32 v1, vcc_lo, v0, v8
; GFX11-GISEL-NEXT: v_mad_u64_u32 v[6:7], null, v9, v2, v[4:5]
-; GFX11-GISEL-NEXT: v_mad_u64_u32 v[4:5], null, v1, v2, 0
; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX11-GISEL-NEXT: v_mad_u64_u32 v[4:5], null, v1, v2, 0
; GFX11-GISEL-NEXT: v_add_co_ci_u32_e64 v10, null, v6, v9, vcc_lo
+; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_2) | instid1(VALU_DEP_2)
; GFX11-GISEL-NEXT: v_mad_u64_u32 v[7:8], null, v1, v3, v[5:6]
; GFX11-GISEL-NEXT: v_add_co_u32 v11, vcc_lo, v0, 1
-; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-GISEL-NEXT: v_add_co_ci_u32_e64 v0, null, 0, v6, vcc_lo
; GFX11-GISEL-NEXT: v_mad_u64_u32 v[5:6], null, v4, v11, 0
; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(SKIP_1) | instid1(VALU_DEP_3)
@@ -7646,7 +7644,6 @@ define <2 x i64> @clpeak_imad_pat_v2i64(<2 x i64> %x, <2 x i64> %y) {
; GFX11-SDAG: ; %bb.0: ; %entry
; GFX11-SDAG-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-SDAG-NEXT: v_add_co_u32 v8, vcc_lo, v0, 1
-; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-SDAG-NEXT: v_add_co_ci_u32_e64 v9, null, 0, v1, vcc_lo
; GFX11-SDAG-NEXT: v_add_co_u32 v10, vcc_lo, v2, 1
; GFX11-SDAG-NEXT: v_add_co_ci_u32_e64 v11, null, 0, v3, vcc_lo
@@ -7661,7 +7658,7 @@ define <2 x i64> @clpeak_imad_pat_v2i64(<2 x i64> %x, <2 x i64> %y) {
; GFX11-SDAG-NEXT: v_add3_u32 v1, v1, v13, v12
; GFX11-SDAG-NEXT: v_add3_u32 v12, v3, v15, v14
; GFX11-SDAG-NEXT: v_add_co_u32 v3, vcc_lo, v0, v8
-; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
+; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_1) | instid1(VALU_DEP_4)
; GFX11-SDAG-NEXT: v_add_co_ci_u32_e64 v8, null, v1, v9, vcc_lo
; GFX11-SDAG-NEXT: v_add_co_u32 v9, vcc_lo, v2, v10
; GFX11-SDAG-NEXT: v_add_co_ci_u32_e64 v10, null, v12, v11, vcc_lo
@@ -12374,7 +12371,7 @@ define i64 @mul_u24_add64(i32 %x, i32 %y, i64 %z) {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: v_mul_u32_u24_e32 v4, v0, v1
; GFX11-NEXT: v_mul_hi_u32_u24_e32 v1, v0, v1
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-NEXT: v_add_co_u32 v0, vcc_lo, v4, v2
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, v1, v3, vcc_lo
; GFX11-NEXT: s_setpc_b64 s[30:31]
@@ -12445,7 +12442,7 @@ define i64 @mul_u24_zext_add64(i32 %x, i32 %y, i64 %z) {
; GFX11: ; %bb.0:
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: v_mul_u32_u24_e32 v0, v0, v1
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v3, vcc_lo
; GFX11-NEXT: s_setpc_b64 s[30:31]
@@ -12561,7 +12558,7 @@ define i64 @mul_u24_known_24bit_add64(i16 %x.s, i16 %y.s, i64 %z) {
; GFX11-GISEL-NEXT: v_mul_u32_u24_e32 v4, v0, v1
; GFX11-GISEL-NEXT: v_mul_hi_u32_u24_e32 v1, v0, v1
; GFX11-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, v4, v2
-; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX11-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, v1, v3, vcc_lo
; GFX11-GISEL-NEXT: s_setpc_b64 s[30:31]
;
diff --git a/llvm/test/CodeGen/AMDGPU/integer-select-src-modifiers.ll b/llvm/test/CodeGen/AMDGPU/integer-select-src-modifiers.ll
index 0298b288c8e659..a6a41370683915 100644
--- a/llvm/test/CodeGen/AMDGPU/integer-select-src-modifiers.ll
+++ b/llvm/test/CodeGen/AMDGPU/integer-select-src-modifiers.ll
@@ -101,8 +101,8 @@ define i32 @s_fneg_select_i32_1(i32 inreg %cond, i32 inreg %a, i32 inreg %b) {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_xor_b32 s1, s1, 0x80000000
; GFX11-NEXT: s_cmp_eq_u32 s0, 0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_cselect_b32 s0, s1, s2
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: v_mov_b32_e32 v0, s0
; GFX11-NEXT: s_setpc_b64 s[30:31]
%neg.a = xor i32 %a, u0x80000000
@@ -124,8 +124,8 @@ define i32 @s_fneg_1_fabs_2_select_i32(i32 inreg %cond, i32 %a, i32 %b) {
; GFX11: ; %bb.0:
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_cmp_eq_u32 s0, 0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_cselect_b32 s0, -1, 0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: v_cndmask_b32_e64 v0, |v0|, -v0, s0
; GFX11-NEXT: s_setpc_b64 s[30:31]
%neg.a = xor i32 %a, u0x80000000
@@ -360,6 +360,7 @@ define <2 x i32> @s_fneg_select_v2i32_1(<2 x i32> inreg %cond, <2 x i32> inreg %
; GFX11-NEXT: s_mov_b32 s5, s4
; GFX11-NEXT: s_xor_b64 s[2:3], s[2:3], s[4:5]
; GFX11-NEXT: s_cmp_eq_u32 s0, 0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_cselect_b32 s0, s2, s16
; GFX11-NEXT: s_cmp_eq_u32 s1, 0
; GFX11-NEXT: s_cselect_b32 s1, s3, s17
@@ -395,6 +396,7 @@ define <2 x i32> @s_fneg_fabs_select_v2i32_2(<2 x i32> inreg %cond, <2 x i32> in
; GFX11-NEXT: s_mov_b32 s5, s4
; GFX11-NEXT: s_or_b64 s[2:3], s[2:3], s[4:5]
; GFX11-NEXT: s_cmp_eq_u32 s0, 0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_cselect_b32 s0, s16, s2
; GFX11-NEXT: s_cmp_eq_u32 s1, 0
; GFX11-NEXT: s_cselect_b32 s1, s17, s3
@@ -591,9 +593,9 @@ define i64 @s_fneg_select_i64_1(i64 inreg %cond, i64 inreg %a, i64 inreg %b) {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_xor_b32 s3, s3, 0x80000000
; GFX11-NEXT: s_cmp_eq_u64 s[0:1], 0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_cselect_b32 s0, s2, s16
; GFX11-NEXT: s_cselect_b32 s1, s3, s17
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: v_dual_mov_b32 v0, s0 :: v_dual_mov_b32 v1, s1
; GFX11-NEXT: s_setpc_b64 s[30:31]
%neg.a = xor i64 %a, u0x8000000000000000
@@ -631,9 +633,9 @@ define i64 @s_fneg_select_i64_2(i64 inreg %cond, i64 inreg %a, i64 inreg %b) {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_xor_b32 s3, s3, 0x80000000
; GFX11-NEXT: s_cmp_eq_u64 s[0:1], 0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_cselect_b32 s0, s16, s2
; GFX11-NEXT: s_cselect_b32 s1, s17, s3
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: v_dual_mov_b32 v0, s0 :: v_dual_mov_b32 v1, s1
; GFX11-NEXT: s_setpc_b64 s[30:31]
%neg.a = xor i64 %a, u0x8000000000000000
@@ -674,9 +676,9 @@ define i64 @s_fneg_1_fabs_2_select_i64(i64 inreg %cond, i64 inreg %a, i64 inreg
; GFX11-NEXT: s_xor_b32 s3, s3, 0x80000000
; GFX11-NEXT: s_bitset0_b32 s17, 31
; GFX11-NEXT: s_cmp_eq_u64 s[0:1], 0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_cselect_b32 s0, s2, s16
; GFX11-NEXT: s_cselect_b32 s1, s3, s17
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: v_dual_mov_b32 v0, s0 :: v_dual_mov_b32 v1, s1
; GFX11-NEXT: s_setpc_b64 s[30:31]
%neg.a = xor i64 %a, u0x8000000000000000
@@ -715,9 +717,9 @@ define i64 @s_fabs_select_i64_1(i64 inreg %cond, i64 inreg %a, i64 inreg %b) {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_bitset0_b32 s3, 31
; GFX11-NEXT: s_cmp_eq_u64 s[0:1], 0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_cselect_b32 s0, s2, s16
; GFX11-NEXT: s_cselect_b32 s1, s3, s17
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: v_dual_mov_b32 v0, s0 :: v_dual_mov_b32 v1, s1
; GFX11-NEXT: s_setpc_b64 s[30:31]
%neg.a = and i64 %a, u0x7fffffffffffffff
@@ -755,9 +757,9 @@ define i64 @s_fabs_select_i64_2(i64 inreg %cond, i64 inreg %a, i64 inreg %b) {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_bitset0_b32 s3, 31
; GFX11-NEXT: s_cmp_eq_u64 s[0:1], 0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_cselect_b32 s0, s16, s2
; GFX11-NEXT: s_cselect_b32 s1, s17, s3
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: v_dual_mov_b32 v0, s0 :: v_dual_mov_b32 v1, s1
; GFX11-NEXT: s_setpc_b64 s[30:31]
%neg.a = and i64 %a, u0x7fffffffffffffff
@@ -795,9 +797,9 @@ define i64 @s_fneg_fabs_select_i64_1(i64 inreg %cond, i64 inreg %a, i64 inreg %b
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_bitset1_b32 s3, 31
; GFX11-NEXT: s_cmp_eq_u64 s[0:1], 0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_cselect_b32 s0, s2, s16
; GFX11-NEXT: s_cselect_b32 s1, s3, s17
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: v_dual_mov_b32 v0, s0 :: v_dual_mov_b32 v1, s1
; GFX11-NEXT: s_setpc_b64 s[30:31]
%neg.a = or i64 %a, u0x8000000000000000
@@ -835,9 +837,9 @@ define i64 @s_fneg_fabs_select_i64_2(i64 inreg %cond, i64 inreg %a, i64 inreg %b
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_bitset1_b32 s3, 31
; GFX11-NEXT: s_cmp_eq_u64 s[0:1], 0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_cselect_b32 s0, s16, s2
; GFX11-NEXT: s_cselect_b32 s1, s17, s3
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: v_dual_mov_b32 v0, s0 :: v_dual_mov_b32 v1, s1
; GFX11-NEXT: s_setpc_b64 s[30:31]
%neg.a = or i64 %a, u0x8000000000000000
diff --git a/llvm/test/CodeGen/AMDGPU/intrinsic-amdgcn-s-alloc-vgpr.ll b/llvm/test/CodeGen/AMDGPU/intrinsic-amdgcn-s-alloc-vgpr.ll
index 8c896403db33c2..ab64b19c4ec98b 100644
--- a/llvm/test/CodeGen/AMDGPU/intrinsic-amdgcn-s-alloc-vgpr.ll
+++ b/llvm/test/CodeGen/AMDGPU/intrinsic-amdgcn-s-alloc-vgpr.ll
@@ -13,15 +13,17 @@ define amdgpu_cs void @test_alloc_vreg_const(ptr addrspace(1) %out) #0 {
; GISEL-NEXT: v_nop
; GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GISEL-NEXT: s_getreg_b32 s33, hwreg(HW_REG_WAVE_HW_ID2, 8, 2)
-; GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_3) | instid1(SALU_CYCLE_1)
+; GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GISEL-NEXT: s_cmp_lg_u32 0, s33
; GISEL-NEXT: s_cmovk_i32 s33, 0x1c0
; GISEL-NEXT: s_alloc_vgpr 45
+; GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GISEL-NEXT: s_cselect_b32 s0, 1, 0
; GISEL-NEXT: s_and_b32 s0, s0, 1
-; GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
+; GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GISEL-NEXT: s_cmp_lg_u32 s0, 0
; GISEL-NEXT: s_cselect_b32 s0, 1, 0
+; GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GISEL-NEXT: v_mov_b32_e32 v2, s0
; GISEL-NEXT: global_store_b32 v[0:1], v2, off
; GISEL-NEXT: s_alloc_vgpr 0
@@ -34,10 +36,11 @@ define amdgpu_cs void @test_alloc_vreg_const(ptr addrspace(1) %out) #0 {
; DAGISEL-NEXT: v_nop
; DAGISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; DAGISEL-NEXT: s_getreg_b32 s33, hwreg(HW_REG_WAVE_HW_ID2, 8, 2)
-; DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_3) | instid1(SALU_CYCLE_1)
+; DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; DAGISEL-NEXT: s_cmp_lg_u32 0, s33
; DAGISEL-NEXT: s_cmovk_i32 s33, 0x1c0
; DAGISEL-NEXT: s_alloc_vgpr 45
+; DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; DAGISEL-NEXT: s_cselect_b32 s0, 1, 0
; DAGISEL-NEXT: v_mov_b32_e32 v2, s0
; DAGISEL-NEXT: global_store_b32 v[0:1], v2, off
@@ -51,15 +54,17 @@ define amdgpu_cs void @test_alloc_vreg_const(ptr addrspace(1) %out) #0 {
; NRBS-NEXT: v_nop
; NRBS-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; NRBS-NEXT: s_getreg_b32 s33, hwreg(HW_REG_WAVE_HW_ID2, 8, 2)
-; NRBS-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_3) | instid1(SALU_CYCLE_1)
+; NRBS-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; NRBS-NEXT: s_cmp_lg_u32 0, s33
; NRBS-NEXT: s_cmovk_i32 s33, 0x1c0
; NRBS-NEXT: s_alloc_vgpr 45
+; NRBS-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; NRBS-NEXT: s_cselect_b32 s0, 1, 0
; NRBS-NEXT: s_and_b32 s0, s0, 1
-; NRBS-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
+; NRBS-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; NRBS-NEXT: s_cmp_lg_u32 s0, 0
; NRBS-NEXT: s_cselect_b32 s0, 1, 0
+; NRBS-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; NRBS-NEXT: v_mov_b32_e32 v2, s0
; NRBS-NEXT: global_store_b32 v[0:1], v2, off
; NRBS-NEXT: s_alloc_vgpr 0
@@ -79,15 +84,17 @@ define amdgpu_cs void @test_alloc_vreg_var(i32 inreg %n, ptr addrspace(1) %out)
; GISEL-NEXT: v_nop
; GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GISEL-NEXT: s_getreg_b32 s33, hwreg(HW_REG_WAVE_HW_ID2, 8, 2)
-; GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_3) | instid1(SALU_CYCLE_1)
+; GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GISEL-NEXT: s_cmp_lg_u32 0, s33
; GISEL-NEXT: s_cmovk_i32 s33, 0x1c0
; GISEL-NEXT: s_alloc_vgpr s0
+; GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GISEL-NEXT: s_cselect_b32 s0, 1, 0
; GISEL-NEXT: s_and_b32 s0, s0, 1
-; GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
+; GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GISEL-NEXT: s_cmp_lg_u32 s0, 0
; GISEL-NEXT: s_cselect_b32 s0, 1, 0
+; GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GISEL-NEXT: v_mov_b32_e32 v2, s0
; GISEL-NEXT: global_store_b32 v[0:1], v2, off
; GISEL-NEXT: s_alloc_vgpr 0
@@ -100,10 +107,11 @@ define amdgpu_cs void @test_alloc_vreg_var(i32 inreg %n, ptr addrspace(1) %out)
; DAGISEL-NEXT: v_nop
; DAGISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; DAGISEL-NEXT: s_getreg_b32 s33, hwreg(HW_REG_WAVE_HW_ID2, 8, 2)
-; DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_3) | instid1(SALU_CYCLE_1)
+; DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; DAGISEL-NEXT: s_cmp_lg_u32 0, s33
; DAGISEL-NEXT: s_cmovk_i32 s33, 0x1c0
; DAGISEL-NEXT: s_alloc_vgpr s0
+; DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; DAGISEL-NEXT: s_cselect_b32 s0, 1, 0
; DAGISEL-NEXT: v_mov_b32_e32 v2, s0
; DAGISEL-NEXT: global_store_b32 v[0:1], v2, off
@@ -117,15 +125,17 @@ define amdgpu_cs void @test_alloc_vreg_var(i32 inreg %n, ptr addrspace(1) %out)
; NRBS-NEXT: v_nop
; NRBS-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; NRBS-NEXT: s_getreg_b32 s33, hwreg(HW_REG_WAVE_HW_ID2, 8, 2)
-; NRBS-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_3) | instid1(SALU_CYCLE_1)
+; NRBS-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; NRBS-NEXT: s_cmp_lg_u32 0, s33
; NRBS-NEXT: s_cmovk_i32 s33, 0x1c0
; NRBS-NEXT: s_alloc_vgpr s0
+; NRBS-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; NRBS-NEXT: s_cselect_b32 s0, 1, 0
; NRBS-NEXT: s_and_b32 s0, s0, 1
-; NRBS-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
+; NRBS-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; NRBS-NEXT: s_cmp_lg_u32 s0, 0
; NRBS-NEXT: s_cselect_b32 s0, 1, 0
+; NRBS-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; NRBS-NEXT: v_mov_b32_e32 v2, s0
; NRBS-NEXT: global_store_b32 v[0:1], v2, off
; NRBS-NEXT: s_alloc_vgpr 0
@@ -148,14 +158,15 @@ define amdgpu_cs void @test_alloc_vreg_vgpr(i32 %n, ptr addrspace(1) %out) #0 {
; GISEL-NEXT: s_getreg_b32 s33, hwreg(HW_REG_WAVE_HW_ID2, 8, 2)
; GISEL-NEXT: v_dual_mov_b32 v4, v1 :: v_dual_mov_b32 v5, v2
; GISEL-NEXT: s_cmp_lg_u32 0, s33
+; GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GISEL-NEXT: s_cmovk_i32 s33, 0x1c0
; GISEL-NEXT: s_alloc_vgpr s0
; GISEL-NEXT: s_cselect_b32 s0, 1, 0
; GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GISEL-NEXT: s_and_b32 s0, s0, 1
; GISEL-NEXT: s_cmp_lg_u32 s0, 0
+; GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GISEL-NEXT: s_cselect_b32 s0, 1, 0
-; GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GISEL-NEXT: v_mov_b32_e32 v0, s0
; GISEL-NEXT: global_store_b32 v[4:5], v0, off
; GISEL-NEXT: s_alloc_vgpr 0
@@ -171,6 +182,7 @@ define amdgpu_cs void @test_alloc_vreg_vgpr(i32 %n, ptr addrspace(1) %out) #0 {
; DAGISEL-NEXT: s_getreg_b32 s33, hwreg(HW_REG_WAVE_HW_ID2, 8, 2)
; DAGISEL-NEXT: v_dual_mov_b32 v3, v2 :: v_dual_mov_b32 v2, v1
; DAGISEL-NEXT: s_cmp_lg_u32 0, s33
+; DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; DAGISEL-NEXT: s_cmovk_i32 s33, 0x1c0
; DAGISEL-NEXT: s_alloc_vgpr s0
; DAGISEL-NEXT: s_cselect_b32 s0, 1, 0
@@ -190,14 +202,15 @@ define amdgpu_cs void @test_alloc_vreg_vgpr(i32 %n, ptr addrspace(1) %out) #0 {
; NRBS-NEXT: s_getreg_b32 s33, hwreg(HW_REG_WAVE_HW_ID2, 8, 2)
; NRBS-NEXT: v_dual_mov_b32 v4, v1 :: v_dual_mov_b32 v5, v2
; NRBS-NEXT: s_cmp_lg_u32 0, s33
+; NRBS-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; NRBS-NEXT: s_cmovk_i32 s33, 0x1c0
; NRBS-NEXT: s_alloc_vgpr s0
; NRBS-NEXT: s_cselect_b32 s0, 1, 0
; NRBS-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; NRBS-NEXT: s_and_b32 s0, s0, 1
; NRBS-NEXT: s_cmp_lg_u32 s0, 0
+; NRBS-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; NRBS-NEXT: s_cselect_b32 s0, 1, 0
-; NRBS-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; NRBS-NEXT: v_mov_b32_e32 v0, s0
; NRBS-NEXT: global_store_b32 v[4:5], v0, off
; NRBS-NEXT: s_alloc_vgpr 0
diff --git a/llvm/test/CodeGen/AMDGPU/issue130120-eliminate-frame-index.ll b/llvm/test/CodeGen/AMDGPU/issue130120-eliminate-frame-index.ll
index 52ec0a62463733..084e7c6052ee82 100644
--- a/llvm/test/CodeGen/AMDGPU/issue130120-eliminate-frame-index.ll
+++ b/llvm/test/CodeGen/AMDGPU/issue130120-eliminate-frame-index.ll
@@ -74,6 +74,7 @@ define amdgpu_gfx [13 x i32] @issue130120() {
; CHECK-NEXT: scratch_store_b32 off, v0, vcc_lo
; CHECK-NEXT: scratch_store_b32 off, v0, vcc_hi
; CHECK-NEXT: s_mov_b32 vcc_lo, exec_lo
+; CHECK-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; CHECK-NEXT: s_cbranch_vccnz .LBB0_1
; CHECK-NEXT: ; %bb.2: ; %DummyReturnBlock
; CHECK-NEXT: s_setpc_b64 s[30:31]
diff --git a/llvm/test/CodeGen/AMDGPU/issue92561-restore-undef-scc-verifier-error.ll b/llvm/test/CodeGen/AMDGPU/issue92561-restore-undef-scc-verifier-error.ll
index 124ba142a56f68..504922631e3c95 100644
--- a/llvm/test/CodeGen/AMDGPU/issue92561-restore-undef-scc-verifier-error.ll
+++ b/llvm/test/CodeGen/AMDGPU/issue92561-restore-undef-scc-verifier-error.ll
@@ -42,6 +42,7 @@ define void @issue92561(ptr addrspace(1) %arg) {
; SDAG-NEXT: s_cbranch_execnz .LBB0_1
; SDAG-NEXT: ; %bb.2:
; SDAG-NEXT: s_mov_b32 exec_lo, s12
+; SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; SDAG-NEXT: v_dual_mov_b32 v0, 0x7fc00000 :: v_dual_mov_b32 v1, 1.0
; SDAG-NEXT: s_mov_b32 s0, s8
; SDAG-NEXT: s_mov_b32 s1, s8
@@ -111,9 +112,11 @@ define void @issue92561(ptr addrspace(1) %arg) {
; GISEL-NEXT: ; implicit-def: $vgpr0_vgpr1_vgpr2_vgpr3
; GISEL-NEXT: ; implicit-def: $vgpr8
; GISEL-NEXT: s_xor_b32 exec_lo, exec_lo, s0
+; GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GISEL-NEXT: s_cbranch_execnz .LBB0_1
; GISEL-NEXT: ; %bb.2:
; GISEL-NEXT: s_mov_b32 exec_lo, s3
+; GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GISEL-NEXT: v_dual_mov_b32 v1, 0 :: v_dual_mov_b32 v0, 0x7fc00000
; GISEL-NEXT: v_mov_b32_e32 v2, 1.0
; GISEL-NEXT: s_clause 0x2
diff --git a/llvm/test/CodeGen/AMDGPU/lds-misaligned-bug.ll b/llvm/test/CodeGen/AMDGPU/lds-misaligned-bug.ll
index 2375e67b7069b7..02d96ea2378a14 100644
--- a/llvm/test/CodeGen/AMDGPU/lds-misaligned-bug.ll
+++ b/llvm/test/CodeGen/AMDGPU/lds-misaligned-bug.ll
@@ -291,7 +291,6 @@ define amdgpu_kernel void @test_flat_misaligned_v2(ptr %arg) {
; ALIGNED-GFX11-NEXT: v_lshlrev_b32_e32 v0, 2, v0
; ALIGNED-GFX11-NEXT: s_waitcnt lgkmcnt(0)
; ALIGNED-GFX11-NEXT: v_add_co_u32 v3, s0, s0, v0
-; ALIGNED-GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; ALIGNED-GFX11-NEXT: v_add_co_ci_u32_e64 v4, null, s1, 0, s0
; ALIGNED-GFX11-NEXT: flat_load_b64 v[0:1], v[3:4]
; ALIGNED-GFX11-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
@@ -307,7 +306,6 @@ define amdgpu_kernel void @test_flat_misaligned_v2(ptr %arg) {
; UNALIGNED-GFX11-NEXT: v_lshlrev_b32_e32 v0, 2, v0
; UNALIGNED-GFX11-NEXT: s_waitcnt lgkmcnt(0)
; UNALIGNED-GFX11-NEXT: v_add_co_u32 v3, s0, s0, v0
-; UNALIGNED-GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; UNALIGNED-GFX11-NEXT: v_add_co_ci_u32_e64 v4, null, s1, 0, s0
; UNALIGNED-GFX11-NEXT: flat_load_b64 v[0:1], v[3:4]
; UNALIGNED-GFX11-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
@@ -392,7 +390,6 @@ define amdgpu_kernel void @test_flat_misaligned_v4(ptr %arg) {
; ALIGNED-GFX11-NEXT: v_lshlrev_b32_e32 v0, 2, v0
; ALIGNED-GFX11-NEXT: s_waitcnt lgkmcnt(0)
; ALIGNED-GFX11-NEXT: v_add_co_u32 v7, s0, s0, v0
-; ALIGNED-GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; ALIGNED-GFX11-NEXT: v_add_co_ci_u32_e64 v8, null, s1, 0, s0
; ALIGNED-GFX11-NEXT: flat_load_b128 v[0:3], v[7:8]
; ALIGNED-GFX11-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
@@ -409,7 +406,6 @@ define amdgpu_kernel void @test_flat_misaligned_v4(ptr %arg) {
; UNALIGNED-GFX11-NEXT: v_lshlrev_b32_e32 v0, 2, v0
; UNALIGNED-GFX11-NEXT: s_waitcnt lgkmcnt(0)
; UNALIGNED-GFX11-NEXT: v_add_co_u32 v7, s0, s0, v0
-; UNALIGNED-GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; UNALIGNED-GFX11-NEXT: v_add_co_ci_u32_e64 v8, null, s1, 0, s0
; UNALIGNED-GFX11-NEXT: flat_load_b128 v[0:3], v[7:8]
; UNALIGNED-GFX11-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
@@ -493,7 +489,6 @@ define amdgpu_kernel void @test_flat_misaligned_v3(ptr %arg) {
; ALIGNED-GFX11-NEXT: v_lshlrev_b32_e32 v0, 2, v0
; ALIGNED-GFX11-NEXT: s_waitcnt lgkmcnt(0)
; ALIGNED-GFX11-NEXT: v_add_co_u32 v5, s0, s0, v0
-; ALIGNED-GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; ALIGNED-GFX11-NEXT: v_add_co_ci_u32_e64 v6, null, s1, 0, s0
; ALIGNED-GFX11-NEXT: flat_load_b96 v[0:2], v[5:6]
; ALIGNED-GFX11-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
@@ -509,7 +504,6 @@ define amdgpu_kernel void @test_flat_misaligned_v3(ptr %arg) {
; UNALIGNED-GFX11-NEXT: v_lshlrev_b32_e32 v0, 2, v0
; UNALIGNED-GFX11-NEXT: s_waitcnt lgkmcnt(0)
; UNALIGNED-GFX11-NEXT: v_add_co_u32 v5, s0, s0, v0
-; UNALIGNED-GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; UNALIGNED-GFX11-NEXT: v_add_co_ci_u32_e64 v6, null, s1, 0, s0
; UNALIGNED-GFX11-NEXT: flat_load_b96 v[0:2], v[5:6]
; UNALIGNED-GFX11-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
@@ -725,7 +719,6 @@ define amdgpu_kernel void @test_flat_aligned_v2(ptr %arg) {
; ALIGNED-GFX11-NEXT: v_lshlrev_b32_e32 v0, 2, v0
; ALIGNED-GFX11-NEXT: s_waitcnt lgkmcnt(0)
; ALIGNED-GFX11-NEXT: v_add_co_u32 v3, s0, s0, v0
-; ALIGNED-GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; ALIGNED-GFX11-NEXT: v_add_co_ci_u32_e64 v4, null, s1, 0, s0
; ALIGNED-GFX11-NEXT: flat_load_b64 v[0:1], v[3:4]
; ALIGNED-GFX11-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
@@ -741,7 +734,6 @@ define amdgpu_kernel void @test_flat_aligned_v2(ptr %arg) {
; UNALIGNED-GFX11-NEXT: v_lshlrev_b32_e32 v0, 2, v0
; UNALIGNED-GFX11-NEXT: s_waitcnt lgkmcnt(0)
; UNALIGNED-GFX11-NEXT: v_add_co_u32 v3, s0, s0, v0
-; UNALIGNED-GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; UNALIGNED-GFX11-NEXT: v_add_co_ci_u32_e64 v4, null, s1, 0, s0
; UNALIGNED-GFX11-NEXT: flat_load_b64 v[0:1], v[3:4]
; UNALIGNED-GFX11-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
@@ -814,7 +806,6 @@ define amdgpu_kernel void @test_flat_aligned_v4(ptr %arg) {
; ALIGNED-GFX11-NEXT: v_lshlrev_b32_e32 v0, 2, v0
; ALIGNED-GFX11-NEXT: s_waitcnt lgkmcnt(0)
; ALIGNED-GFX11-NEXT: v_add_co_u32 v7, s0, s0, v0
-; ALIGNED-GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; ALIGNED-GFX11-NEXT: v_add_co_ci_u32_e64 v8, null, s1, 0, s0
; ALIGNED-GFX11-NEXT: flat_load_b128 v[0:3], v[7:8]
; ALIGNED-GFX11-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
@@ -831,7 +822,6 @@ define amdgpu_kernel void @test_flat_aligned_v4(ptr %arg) {
; UNALIGNED-GFX11-NEXT: v_lshlrev_b32_e32 v0, 2, v0
; UNALIGNED-GFX11-NEXT: s_waitcnt lgkmcnt(0)
; UNALIGNED-GFX11-NEXT: v_add_co_u32 v7, s0, s0, v0
-; UNALIGNED-GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; UNALIGNED-GFX11-NEXT: v_add_co_ci_u32_e64 v8, null, s1, 0, s0
; UNALIGNED-GFX11-NEXT: flat_load_b128 v[0:3], v[7:8]
; UNALIGNED-GFX11-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
@@ -994,7 +984,6 @@ define amdgpu_kernel void @test_flat_v4_aligned8(ptr %arg) {
; ALIGNED-GFX11-NEXT: v_lshlrev_b32_e32 v0, 2, v0
; ALIGNED-GFX11-NEXT: s_waitcnt lgkmcnt(0)
; ALIGNED-GFX11-NEXT: v_add_co_u32 v7, s0, s0, v0
-; ALIGNED-GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; ALIGNED-GFX11-NEXT: v_add_co_ci_u32_e64 v8, null, s1, 0, s0
; ALIGNED-GFX11-NEXT: flat_load_b128 v[0:3], v[7:8]
; ALIGNED-GFX11-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
@@ -1011,7 +1000,6 @@ define amdgpu_kernel void @test_flat_v4_aligned8(ptr %arg) {
; UNALIGNED-GFX11-NEXT: v_lshlrev_b32_e32 v0, 2, v0
; UNALIGNED-GFX11-NEXT: s_waitcnt lgkmcnt(0)
; UNALIGNED-GFX11-NEXT: v_add_co_u32 v7, s0, s0, v0
-; UNALIGNED-GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; UNALIGNED-GFX11-NEXT: v_add_co_ci_u32_e64 v8, null, s1, 0, s0
; UNALIGNED-GFX11-NEXT: flat_load_b128 v[0:3], v[7:8]
; UNALIGNED-GFX11-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
diff --git a/llvm/test/CodeGen/AMDGPU/literal64.ll b/llvm/test/CodeGen/AMDGPU/literal64.ll
index 2415fdd2c76828..be6ac2a242e428 100644
--- a/llvm/test/CodeGen/AMDGPU/literal64.ll
+++ b/llvm/test/CodeGen/AMDGPU/literal64.ll
@@ -36,7 +36,6 @@ define amdgpu_ps void @v_add_u64(i64 %a, ptr addrspace(1) %out) {
; GFX13-LABEL: v_add_u64:
; GFX13: ; %bb.0:
; GFX13-NEXT: v_add_co_u32 v0, vcc_lo, 0x12345678, v0
-; GFX13-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX13-NEXT: v_add_co_ci_u32_e64 v1, null, 15, v1, vcc_lo
; GFX13-NEXT: global_store_b64 v[2:3], v[0:1], off
; GFX13-NEXT: s_endpgm
@@ -77,7 +76,6 @@ define amdgpu_ps void @v_add_neg_u64(i64 %a, ptr addrspace(1) %out) {
; GFX13-LABEL: v_add_neg_u64:
; GFX13: ; %bb.0:
; GFX13-NEXT: v_add_co_u32 v0, vcc_lo, 0xedcba988, v0
-; GFX13-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX13-NEXT: v_add_co_ci_u32_e64 v1, null, -16, v1, vcc_lo
; GFX13-NEXT: global_store_b64 v[2:3], v[0:1], off
; GFX13-NEXT: s_endpgm
@@ -118,7 +116,6 @@ define amdgpu_ps void @v_sub_u64(i64 %a, ptr addrspace(1) %out) {
; GFX13-LABEL: v_sub_u64:
; GFX13: ; %bb.0:
; GFX13-NEXT: v_sub_co_u32 v0, vcc_lo, 0x12345678, v0
-; GFX13-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX13-NEXT: v_sub_co_ci_u32_e64 v1, null, 15, v1, vcc_lo
; GFX13-NEXT: global_store_b64 v[2:3], v[0:1], off
; GFX13-NEXT: s_endpgm
@@ -154,7 +151,7 @@ define void @v_mov_b64_double(ptr addrspace(1) %ptr) {
; GFX13-NEXT: v_cmp_eq_u64_e32 vcc_lo, v[2:3], v[4:5]
; GFX13-NEXT: v_dual_mov_b32 v5, v3 :: v_dual_mov_b32 v4, v2
; GFX13-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX13-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX13-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX13-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX13-NEXT: s_cbranch_execnz .LBB6_1
; GFX13-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -364,6 +361,7 @@ define amdgpu_ps <2 x float> @v_add_f64_200.1(double %a) {
; GFX1250-NEXT: v_nop
; GFX1250-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 2, 2), 0 ; msbs: dst=0 src0=0 src1=0 src2=0
+; GFX1250-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1250-NEXT: v_add_f64_e32 v[0:1], 0x4069033333333333, v[0:1]
; GFX1250-NEXT: ; return to shader part epilog
;
@@ -386,6 +384,7 @@ define amdgpu_ps <2 x float> @v_add_f64_200.0(double %a) {
; GFX1250-NEXT: v_nop
; GFX1250-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 2, 2), 0 ; msbs: dst=0 src0=0 src1=0 src2=0
+; GFX1250-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1250-NEXT: v_add_f64_e32 v[0:1], 0x40690000, v[0:1]
; GFX1250-NEXT: ; return to shader part epilog
;
@@ -426,7 +425,7 @@ define amdgpu_ps <2 x float> @v_lshl_add_u64(i64 %a) {
; GFX13-LABEL: v_lshl_add_u64:
; GFX13: ; %bb.0:
; GFX13-NEXT: v_lshlrev_b64_e32 v[0:1], 1, v[0:1]
-; GFX13-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX13-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX13-NEXT: v_add_co_u32 v0, vcc_lo, 0x12345678, v0
; GFX13-NEXT: v_add_co_ci_u32_e64 v1, null, 15, v1, vcc_lo
; GFX13-NEXT: ; return to shader part epilog
@@ -511,6 +510,7 @@ define amdgpu_ps <2 x float> @v_add_neg_f64(double %a) {
; GFX1250-SDAG-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-SDAG-NEXT: s_mov_b64 s[0:1], 0x4069033333333333
; GFX1250-SDAG-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 2, 2), 0 ; msbs: dst=0 src0=0 src1=0 src2=0
+; GFX1250-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1250-SDAG-NEXT: v_add_f64_e64 v[0:1], -v[0:1], s[0:1]
; GFX1250-SDAG-NEXT: ; return to shader part epilog
;
@@ -523,7 +523,7 @@ define amdgpu_ps <2 x float> @v_add_neg_f64(double %a) {
; GFX1250-GISEL-NEXT: v_max_num_f64_e32 v[0:1], v[0:1], v[0:1]
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], 0x4069033333333333
; GFX1250-GISEL-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 2, 2), 0 ; msbs: dst=0 src0=0 src1=0 src2=0
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX1250-GISEL-NEXT: v_add_f64_e64 v[0:1], -v[0:1], v[2:3]
; GFX1250-GISEL-NEXT: ; return to shader part epilog
;
diff --git a/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.av.load.b128.ll b/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.av.load.b128.ll
index ec1c395f306a63..398cc841d1161f 100644
--- a/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.av.load.b128.ll
+++ b/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.av.load.b128.ll
@@ -1112,7 +1112,6 @@ define <4 x float> @global_load_i8_offset_4096(ptr addrspace(1) %sbase) {
; GFX1100-SDAG: ; %bb.0:
; GFX1100-SDAG-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX1100-SDAG-NEXT: v_add_co_u32 v0, vcc_lo, 0x1000, v0
-; GFX1100-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1100-SDAG-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1100-SDAG-NEXT: global_load_b128 v[0:3], v[0:1], off glc
; GFX1100-SDAG-NEXT: s_waitcnt vmcnt(0)
@@ -1169,7 +1168,6 @@ define <4 x float> @global_load_i8_offset_4096(ptr addrspace(1) %sbase) {
; GFX1100-ISEL: ; %bb.0:
; GFX1100-ISEL-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX1100-ISEL-NEXT: v_add_co_u32 v0, vcc_lo, 0x1000, v0
-; GFX1100-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1100-ISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1100-ISEL-NEXT: global_load_b128 v[0:3], v[0:1], off glc
; GFX1100-ISEL-NEXT: s_waitcnt vmcnt(0)
@@ -1233,7 +1231,6 @@ define <4 x float> @global_load_i8_offset_4097(ptr addrspace(1) %sbase) {
; GFX1100-SDAG: ; %bb.0:
; GFX1100-SDAG-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX1100-SDAG-NEXT: v_add_co_u32 v0, vcc_lo, 0x1000, v0
-; GFX1100-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1100-SDAG-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1100-SDAG-NEXT: global_load_b128 v[0:3], v[0:1], off offset:1 glc
; GFX1100-SDAG-NEXT: s_waitcnt vmcnt(0)
@@ -1290,7 +1287,6 @@ define <4 x float> @global_load_i8_offset_4097(ptr addrspace(1) %sbase) {
; GFX1100-ISEL: ; %bb.0:
; GFX1100-ISEL-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX1100-ISEL-NEXT: v_add_co_u32 v0, vcc_lo, 0x1001, v0
-; GFX1100-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1100-ISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1100-ISEL-NEXT: global_load_b128 v[0:3], v[0:1], off glc
; GFX1100-ISEL-NEXT: s_waitcnt vmcnt(0)
@@ -1459,7 +1455,6 @@ define <4 x float> @global_load_i8_offset_neg4097(ptr addrspace(1) %sbase) {
; GFX1100-SDAG: ; %bb.0:
; GFX1100-SDAG-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX1100-SDAG-NEXT: v_add_co_u32 v0, vcc_lo, 0xfffff000, v0
-; GFX1100-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1100-SDAG-NEXT: v_add_co_ci_u32_e64 v1, null, -1, v1, vcc_lo
; GFX1100-SDAG-NEXT: global_load_b128 v[0:3], v[0:1], off offset:-1 glc
; GFX1100-SDAG-NEXT: s_waitcnt vmcnt(0)
@@ -1516,7 +1511,6 @@ define <4 x float> @global_load_i8_offset_neg4097(ptr addrspace(1) %sbase) {
; GFX1100-ISEL: ; %bb.0:
; GFX1100-ISEL-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX1100-ISEL-NEXT: v_add_co_u32 v0, vcc_lo, 0xffffefff, v0
-; GFX1100-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1100-ISEL-NEXT: v_add_co_ci_u32_e64 v1, null, -1, v1, vcc_lo
; GFX1100-ISEL-NEXT: global_load_b128 v[0:3], v[0:1], off glc
; GFX1100-ISEL-NEXT: s_waitcnt vmcnt(0)
@@ -1580,7 +1574,6 @@ define <4 x float> @global_load_i8_offset_neg4098(ptr addrspace(1) %sbase) {
; GFX1100-SDAG: ; %bb.0:
; GFX1100-SDAG-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX1100-SDAG-NEXT: v_add_co_u32 v0, vcc_lo, 0xfffff000, v0
-; GFX1100-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1100-SDAG-NEXT: v_add_co_ci_u32_e64 v1, null, -1, v1, vcc_lo
; GFX1100-SDAG-NEXT: global_load_b128 v[0:3], v[0:1], off offset:-2 glc
; GFX1100-SDAG-NEXT: s_waitcnt vmcnt(0)
@@ -1637,7 +1630,6 @@ define <4 x float> @global_load_i8_offset_neg4098(ptr addrspace(1) %sbase) {
; GFX1100-ISEL: ; %bb.0:
; GFX1100-ISEL-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX1100-ISEL-NEXT: v_add_co_u32 v0, vcc_lo, 0xffffeffe, v0
-; GFX1100-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1100-ISEL-NEXT: v_add_co_ci_u32_e64 v1, null, -1, v1, vcc_lo
; GFX1100-ISEL-NEXT: global_load_b128 v[0:3], v[0:1], off glc
; GFX1100-ISEL-NEXT: s_waitcnt vmcnt(0)
@@ -2327,7 +2319,6 @@ define <4 x float> @global_load_i8_offset_0x7FFFFF(ptr addrspace(1) %sbase) {
; GFX1100-SDAG: ; %bb.0:
; GFX1100-SDAG-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX1100-SDAG-NEXT: v_add_co_u32 v0, vcc_lo, 0x7ff000, v0
-; GFX1100-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1100-SDAG-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1100-SDAG-NEXT: global_load_b128 v[0:3], v[0:1], off offset:4095 glc
; GFX1100-SDAG-NEXT: s_waitcnt vmcnt(0)
@@ -2384,7 +2375,6 @@ define <4 x float> @global_load_i8_offset_0x7FFFFF(ptr addrspace(1) %sbase) {
; GFX1100-ISEL: ; %bb.0:
; GFX1100-ISEL-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX1100-ISEL-NEXT: v_add_co_u32 v0, vcc_lo, 0x7fffff, v0
-; GFX1100-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1100-ISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1100-ISEL-NEXT: global_load_b128 v[0:3], v[0:1], off glc
; GFX1100-ISEL-NEXT: s_waitcnt vmcnt(0)
@@ -2448,7 +2438,6 @@ define <4 x float> @global_load_i8_offset_0xFFFFFF(ptr addrspace(1) %sbase) {
; GFX1100-SDAG: ; %bb.0:
; GFX1100-SDAG-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX1100-SDAG-NEXT: v_add_co_u32 v0, vcc_lo, 0xff800000, v0
-; GFX1100-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1100-SDAG-NEXT: v_add_co_ci_u32_e64 v1, null, -1, v1, vcc_lo
; GFX1100-SDAG-NEXT: global_load_b128 v[0:3], v[0:1], off glc
; GFX1100-SDAG-NEXT: s_waitcnt vmcnt(0)
@@ -2505,7 +2494,6 @@ define <4 x float> @global_load_i8_offset_0xFFFFFF(ptr addrspace(1) %sbase) {
; GFX1100-ISEL: ; %bb.0:
; GFX1100-ISEL-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX1100-ISEL-NEXT: v_add_co_u32 v0, vcc_lo, 0xff800000, v0
-; GFX1100-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1100-ISEL-NEXT: v_add_co_ci_u32_e64 v1, null, -1, v1, vcc_lo
; GFX1100-ISEL-NEXT: global_load_b128 v[0:3], v[0:1], off glc
; GFX1100-ISEL-NEXT: s_waitcnt vmcnt(0)
@@ -2569,7 +2557,6 @@ define <4 x float> @global_load_i8_offset_0xFFFFFFFF(ptr addrspace(1) %sbase) {
; GFX1100-SDAG: ; %bb.0:
; GFX1100-SDAG-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX1100-SDAG-NEXT: v_add_co_u32 v0, vcc_lo, 0xfffff000, v0
-; GFX1100-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1100-SDAG-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1100-SDAG-NEXT: global_load_b128 v[0:3], v[0:1], off offset:4095 glc
; GFX1100-SDAG-NEXT: s_waitcnt vmcnt(0)
@@ -2580,7 +2567,6 @@ define <4 x float> @global_load_i8_offset_0xFFFFFFFF(ptr addrspace(1) %sbase) {
; GFX1250-SDAG-NEXT: s_wait_loadcnt_dscnt 0x0
; GFX1250-SDAG-NEXT: s_wait_kmcnt 0x0
; GFX1250-SDAG-NEXT: v_add_co_u32 v0, vcc_lo, 0xff800000, v0
-; GFX1250-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1250-SDAG-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1250-SDAG-NEXT: global_load_b128 v[0:3], v[0:1], off offset:8388607 scope:SCOPE_SYS
; GFX1250-SDAG-NEXT: s_wait_loadcnt 0x0
@@ -2594,7 +2580,6 @@ define <4 x float> @global_load_i8_offset_0xFFFFFFFF(ptr addrspace(1) %sbase) {
; GFX1310-SDAG-NEXT: s_wait_bvhcnt 0x0
; GFX1310-SDAG-NEXT: s_wait_kmcnt 0x0
; GFX1310-SDAG-NEXT: v_add_co_u32 v0, vcc_lo, 0xff800000, v0
-; GFX1310-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1310-SDAG-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1310-SDAG-NEXT: global_load_b128 v[0:3], v[0:1], off offset:8388607 scope:SCOPE_SYS
; GFX1310-SDAG-NEXT: s_wait_loadcnt 0x0
@@ -2632,7 +2617,6 @@ define <4 x float> @global_load_i8_offset_0xFFFFFFFF(ptr addrspace(1) %sbase) {
; GFX1100-ISEL: ; %bb.0:
; GFX1100-ISEL-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX1100-ISEL-NEXT: v_add_co_u32 v0, vcc_lo, v0, -1
-; GFX1100-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1100-ISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1100-ISEL-NEXT: global_load_b128 v[0:3], v[0:1], off glc
; GFX1100-ISEL-NEXT: s_waitcnt vmcnt(0)
@@ -2643,7 +2627,6 @@ define <4 x float> @global_load_i8_offset_0xFFFFFFFF(ptr addrspace(1) %sbase) {
; GFX1250-ISEL-NEXT: s_wait_loadcnt_dscnt 0x0
; GFX1250-ISEL-NEXT: s_wait_kmcnt 0x0
; GFX1250-ISEL-NEXT: v_add_co_u32 v0, vcc_lo, v0, -1
-; GFX1250-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1250-ISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1250-ISEL-NEXT: global_load_b128 v[0:3], v[0:1], off scope:SCOPE_SYS
; GFX1250-ISEL-NEXT: s_wait_loadcnt 0x0
@@ -2657,7 +2640,6 @@ define <4 x float> @global_load_i8_offset_0xFFFFFFFF(ptr addrspace(1) %sbase) {
; GFX1310-ISEL-NEXT: s_wait_bvhcnt 0x0
; GFX1310-ISEL-NEXT: s_wait_kmcnt 0x0
; GFX1310-ISEL-NEXT: v_add_co_u32 v0, vcc_lo, v0, -1
-; GFX1310-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1310-ISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1310-ISEL-NEXT: global_load_b128 v[0:3], v[0:1], off scope:SCOPE_SYS
; GFX1310-ISEL-NEXT: s_wait_loadcnt 0x0
@@ -2755,7 +2737,6 @@ define <4 x float> @global_load_i8_offset_0x100000000(ptr addrspace(1) %sbase) {
; GFX1100-ISEL: ; %bb.0:
; GFX1100-ISEL-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX1100-ISEL-NEXT: v_add_co_u32 v0, vcc_lo, v0, 0
-; GFX1100-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1100-ISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 1, v1, vcc_lo
; GFX1100-ISEL-NEXT: global_load_b128 v[0:3], v[0:1], off
; GFX1100-ISEL-NEXT: s_waitcnt vmcnt(0)
@@ -2766,7 +2747,6 @@ define <4 x float> @global_load_i8_offset_0x100000000(ptr addrspace(1) %sbase) {
; GFX1250-ISEL-NEXT: s_wait_loadcnt_dscnt 0x0
; GFX1250-ISEL-NEXT: s_wait_kmcnt 0x0
; GFX1250-ISEL-NEXT: v_add_co_u32 v0, vcc_lo, v0, 0
-; GFX1250-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1250-ISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 1, v1, vcc_lo
; GFX1250-ISEL-NEXT: global_load_b128 v[0:3], v[0:1], off
; GFX1250-ISEL-NEXT: s_wait_loadcnt 0x0
@@ -2780,7 +2760,6 @@ define <4 x float> @global_load_i8_offset_0x100000000(ptr addrspace(1) %sbase) {
; GFX1310-ISEL-NEXT: s_wait_bvhcnt 0x0
; GFX1310-ISEL-NEXT: s_wait_kmcnt 0x0
; GFX1310-ISEL-NEXT: v_add_co_u32 v0, vcc_lo, v0, 0
-; GFX1310-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1310-ISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 1, v1, vcc_lo
; GFX1310-ISEL-NEXT: global_load_b128 v[0:3], v[0:1], off
; GFX1310-ISEL-NEXT: s_wait_loadcnt 0x0
@@ -2825,7 +2804,6 @@ define <4 x float> @global_load_i8_offset_0x100000001(ptr addrspace(1) %sbase) {
; GFX1100-SDAG: ; %bb.0:
; GFX1100-SDAG-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX1100-SDAG-NEXT: v_add_co_u32 v0, vcc_lo, 0, v0
-; GFX1100-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1100-SDAG-NEXT: v_add_co_ci_u32_e64 v1, null, 1, v1, vcc_lo
; GFX1100-SDAG-NEXT: global_load_b128 v[0:3], v[0:1], off offset:1 glc
; GFX1100-SDAG-NEXT: s_waitcnt vmcnt(0)
@@ -2836,7 +2814,6 @@ define <4 x float> @global_load_i8_offset_0x100000001(ptr addrspace(1) %sbase) {
; GFX1250-SDAG-NEXT: s_wait_loadcnt_dscnt 0x0
; GFX1250-SDAG-NEXT: s_wait_kmcnt 0x0
; GFX1250-SDAG-NEXT: v_add_co_u32 v0, vcc_lo, 0, v0
-; GFX1250-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1250-SDAG-NEXT: v_add_co_ci_u32_e64 v1, null, 1, v1, vcc_lo
; GFX1250-SDAG-NEXT: global_load_b128 v[0:3], v[0:1], off offset:1
; GFX1250-SDAG-NEXT: s_wait_loadcnt 0x0
@@ -2850,7 +2827,6 @@ define <4 x float> @global_load_i8_offset_0x100000001(ptr addrspace(1) %sbase) {
; GFX1310-SDAG-NEXT: s_wait_bvhcnt 0x0
; GFX1310-SDAG-NEXT: s_wait_kmcnt 0x0
; GFX1310-SDAG-NEXT: v_add_co_u32 v0, vcc_lo, 0, v0
-; GFX1310-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1310-SDAG-NEXT: v_add_co_ci_u32_e64 v1, null, 1, v1, vcc_lo
; GFX1310-SDAG-NEXT: global_load_b128 v[0:3], v[0:1], off offset:1 scope:SCOPE_SE
; GFX1310-SDAG-NEXT: s_wait_loadcnt 0x0
@@ -2888,7 +2864,6 @@ define <4 x float> @global_load_i8_offset_0x100000001(ptr addrspace(1) %sbase) {
; GFX1100-ISEL: ; %bb.0:
; GFX1100-ISEL-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX1100-ISEL-NEXT: v_add_co_u32 v0, vcc_lo, v0, 1
-; GFX1100-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1100-ISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 1, v1, vcc_lo
; GFX1100-ISEL-NEXT: global_load_b128 v[0:3], v[0:1], off glc
; GFX1100-ISEL-NEXT: s_waitcnt vmcnt(0)
@@ -2899,7 +2874,6 @@ define <4 x float> @global_load_i8_offset_0x100000001(ptr addrspace(1) %sbase) {
; GFX1250-ISEL-NEXT: s_wait_loadcnt_dscnt 0x0
; GFX1250-ISEL-NEXT: s_wait_kmcnt 0x0
; GFX1250-ISEL-NEXT: v_add_co_u32 v0, vcc_lo, v0, 1
-; GFX1250-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1250-ISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 1, v1, vcc_lo
; GFX1250-ISEL-NEXT: global_load_b128 v[0:3], v[0:1], off
; GFX1250-ISEL-NEXT: s_wait_loadcnt 0x0
@@ -2913,7 +2887,6 @@ define <4 x float> @global_load_i8_offset_0x100000001(ptr addrspace(1) %sbase) {
; GFX1310-ISEL-NEXT: s_wait_bvhcnt 0x0
; GFX1310-ISEL-NEXT: s_wait_kmcnt 0x0
; GFX1310-ISEL-NEXT: v_add_co_u32 v0, vcc_lo, v0, 1
-; GFX1310-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1310-ISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 1, v1, vcc_lo
; GFX1310-ISEL-NEXT: global_load_b128 v[0:3], v[0:1], off scope:SCOPE_SE
; GFX1310-ISEL-NEXT: s_wait_loadcnt 0x0
@@ -2958,7 +2931,6 @@ define <4 x float> @global_load_i8_offset_0x100000FFF(ptr addrspace(1) %sbase) {
; GFX1100-SDAG: ; %bb.0:
; GFX1100-SDAG-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX1100-SDAG-NEXT: v_add_co_u32 v0, vcc_lo, 0, v0
-; GFX1100-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1100-SDAG-NEXT: v_add_co_ci_u32_e64 v1, null, 1, v1, vcc_lo
; GFX1100-SDAG-NEXT: global_load_b128 v[0:3], v[0:1], off offset:4095 glc
; GFX1100-SDAG-NEXT: s_waitcnt vmcnt(0)
@@ -2969,7 +2941,6 @@ define <4 x float> @global_load_i8_offset_0x100000FFF(ptr addrspace(1) %sbase) {
; GFX1250-SDAG-NEXT: s_wait_loadcnt_dscnt 0x0
; GFX1250-SDAG-NEXT: s_wait_kmcnt 0x0
; GFX1250-SDAG-NEXT: v_add_co_u32 v0, vcc_lo, 0, v0
-; GFX1250-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1250-SDAG-NEXT: v_add_co_ci_u32_e64 v1, null, 1, v1, vcc_lo
; GFX1250-SDAG-NEXT: global_load_b128 v[0:3], v[0:1], off offset:4095 scope:SCOPE_DEV
; GFX1250-SDAG-NEXT: s_wait_loadcnt 0x0
@@ -2983,7 +2954,6 @@ define <4 x float> @global_load_i8_offset_0x100000FFF(ptr addrspace(1) %sbase) {
; GFX1310-SDAG-NEXT: s_wait_bvhcnt 0x0
; GFX1310-SDAG-NEXT: s_wait_kmcnt 0x0
; GFX1310-SDAG-NEXT: v_add_co_u32 v0, vcc_lo, 0, v0
-; GFX1310-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1310-SDAG-NEXT: v_add_co_ci_u32_e64 v1, null, 1, v1, vcc_lo
; GFX1310-SDAG-NEXT: global_load_b128 v[0:3], v[0:1], off offset:4095 scope:SCOPE_DEV
; GFX1310-SDAG-NEXT: s_wait_loadcnt 0x0
@@ -3021,7 +2991,6 @@ define <4 x float> @global_load_i8_offset_0x100000FFF(ptr addrspace(1) %sbase) {
; GFX1100-ISEL: ; %bb.0:
; GFX1100-ISEL-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX1100-ISEL-NEXT: v_add_co_u32 v0, vcc_lo, 0xfff, v0
-; GFX1100-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1100-ISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 1, v1, vcc_lo
; GFX1100-ISEL-NEXT: global_load_b128 v[0:3], v[0:1], off glc
; GFX1100-ISEL-NEXT: s_waitcnt vmcnt(0)
@@ -3032,7 +3001,6 @@ define <4 x float> @global_load_i8_offset_0x100000FFF(ptr addrspace(1) %sbase) {
; GFX1250-ISEL-NEXT: s_wait_loadcnt_dscnt 0x0
; GFX1250-ISEL-NEXT: s_wait_kmcnt 0x0
; GFX1250-ISEL-NEXT: v_add_co_u32 v0, vcc_lo, 0xfff, v0
-; GFX1250-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1250-ISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 1, v1, vcc_lo
; GFX1250-ISEL-NEXT: global_load_b128 v[0:3], v[0:1], off scope:SCOPE_DEV
; GFX1250-ISEL-NEXT: s_wait_loadcnt 0x0
@@ -3046,7 +3014,6 @@ define <4 x float> @global_load_i8_offset_0x100000FFF(ptr addrspace(1) %sbase) {
; GFX1310-ISEL-NEXT: s_wait_bvhcnt 0x0
; GFX1310-ISEL-NEXT: s_wait_kmcnt 0x0
; GFX1310-ISEL-NEXT: v_add_co_u32 v0, vcc_lo, 0xfff, v0
-; GFX1310-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1310-ISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 1, v1, vcc_lo
; GFX1310-ISEL-NEXT: global_load_b128 v[0:3], v[0:1], off scope:SCOPE_DEV
; GFX1310-ISEL-NEXT: s_wait_loadcnt 0x0
@@ -3091,7 +3058,6 @@ define <4 x float> @global_load_i8_offset_0x100001000(ptr addrspace(1) %sbase) {
; GFX1100-SDAG: ; %bb.0:
; GFX1100-SDAG-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX1100-SDAG-NEXT: v_add_co_u32 v0, vcc_lo, 0x1000, v0
-; GFX1100-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1100-SDAG-NEXT: v_add_co_ci_u32_e64 v1, null, 1, v1, vcc_lo
; GFX1100-SDAG-NEXT: global_load_b128 v[0:3], v[0:1], off glc
; GFX1100-SDAG-NEXT: s_waitcnt vmcnt(0)
@@ -3102,7 +3068,6 @@ define <4 x float> @global_load_i8_offset_0x100001000(ptr addrspace(1) %sbase) {
; GFX1250-SDAG-NEXT: s_wait_loadcnt_dscnt 0x0
; GFX1250-SDAG-NEXT: s_wait_kmcnt 0x0
; GFX1250-SDAG-NEXT: v_add_co_u32 v0, vcc_lo, 0, v0
-; GFX1250-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1250-SDAG-NEXT: v_add_co_ci_u32_e64 v1, null, 1, v1, vcc_lo
; GFX1250-SDAG-NEXT: global_load_b128 v[0:3], v[0:1], off offset:4096 scope:SCOPE_SYS
; GFX1250-SDAG-NEXT: s_wait_loadcnt 0x0
@@ -3116,7 +3081,6 @@ define <4 x float> @global_load_i8_offset_0x100001000(ptr addrspace(1) %sbase) {
; GFX1310-SDAG-NEXT: s_wait_bvhcnt 0x0
; GFX1310-SDAG-NEXT: s_wait_kmcnt 0x0
; GFX1310-SDAG-NEXT: v_add_co_u32 v0, vcc_lo, 0, v0
-; GFX1310-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1310-SDAG-NEXT: v_add_co_ci_u32_e64 v1, null, 1, v1, vcc_lo
; GFX1310-SDAG-NEXT: global_load_b128 v[0:3], v[0:1], off offset:4096 scope:SCOPE_SYS
; GFX1310-SDAG-NEXT: s_wait_loadcnt 0x0
@@ -3154,7 +3118,6 @@ define <4 x float> @global_load_i8_offset_0x100001000(ptr addrspace(1) %sbase) {
; GFX1100-ISEL: ; %bb.0:
; GFX1100-ISEL-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX1100-ISEL-NEXT: v_add_co_u32 v0, vcc_lo, 0x1000, v0
-; GFX1100-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1100-ISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 1, v1, vcc_lo
; GFX1100-ISEL-NEXT: global_load_b128 v[0:3], v[0:1], off glc
; GFX1100-ISEL-NEXT: s_waitcnt vmcnt(0)
@@ -3165,7 +3128,6 @@ define <4 x float> @global_load_i8_offset_0x100001000(ptr addrspace(1) %sbase) {
; GFX1250-ISEL-NEXT: s_wait_loadcnt_dscnt 0x0
; GFX1250-ISEL-NEXT: s_wait_kmcnt 0x0
; GFX1250-ISEL-NEXT: v_add_co_u32 v0, vcc_lo, 0x1000, v0
-; GFX1250-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1250-ISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 1, v1, vcc_lo
; GFX1250-ISEL-NEXT: global_load_b128 v[0:3], v[0:1], off scope:SCOPE_SYS
; GFX1250-ISEL-NEXT: s_wait_loadcnt 0x0
@@ -3179,7 +3141,6 @@ define <4 x float> @global_load_i8_offset_0x100001000(ptr addrspace(1) %sbase) {
; GFX1310-ISEL-NEXT: s_wait_bvhcnt 0x0
; GFX1310-ISEL-NEXT: s_wait_kmcnt 0x0
; GFX1310-ISEL-NEXT: v_add_co_u32 v0, vcc_lo, 0x1000, v0
-; GFX1310-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1310-ISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 1, v1, vcc_lo
; GFX1310-ISEL-NEXT: global_load_b128 v[0:3], v[0:1], off scope:SCOPE_SYS
; GFX1310-ISEL-NEXT: s_wait_loadcnt 0x0
@@ -3224,7 +3185,6 @@ define <4 x float> @global_load_i8_offset_neg0xFFFFFFFF(ptr addrspace(1) %sbase)
; GFX1100-SDAG: ; %bb.0:
; GFX1100-SDAG-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX1100-SDAG-NEXT: v_add_co_u32 v0, vcc_lo, 0x1000, v0
-; GFX1100-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1100-SDAG-NEXT: v_add_co_ci_u32_e64 v1, null, -1, v1, vcc_lo
; GFX1100-SDAG-NEXT: global_load_b128 v[0:3], v[0:1], off offset:-4095
; GFX1100-SDAG-NEXT: s_waitcnt vmcnt(0)
@@ -3235,7 +3195,6 @@ define <4 x float> @global_load_i8_offset_neg0xFFFFFFFF(ptr addrspace(1) %sbase)
; GFX1250-SDAG-NEXT: s_wait_loadcnt_dscnt 0x0
; GFX1250-SDAG-NEXT: s_wait_kmcnt 0x0
; GFX1250-SDAG-NEXT: v_add_co_u32 v0, vcc_lo, 0x800000, v0
-; GFX1250-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1250-SDAG-NEXT: v_add_co_ci_u32_e64 v1, null, -1, v1, vcc_lo
; GFX1250-SDAG-NEXT: global_load_b128 v[0:3], v[0:1], off offset:-8388607
; GFX1250-SDAG-NEXT: s_wait_loadcnt 0x0
@@ -3249,7 +3208,6 @@ define <4 x float> @global_load_i8_offset_neg0xFFFFFFFF(ptr addrspace(1) %sbase)
; GFX1310-SDAG-NEXT: s_wait_bvhcnt 0x0
; GFX1310-SDAG-NEXT: s_wait_kmcnt 0x0
; GFX1310-SDAG-NEXT: v_add_co_u32 v0, vcc_lo, 0x800000, v0
-; GFX1310-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1310-SDAG-NEXT: v_add_co_ci_u32_e64 v1, null, -1, v1, vcc_lo
; GFX1310-SDAG-NEXT: global_load_b128 v[0:3], v[0:1], off offset:-8388607
; GFX1310-SDAG-NEXT: s_wait_loadcnt 0x0
@@ -3287,7 +3245,6 @@ define <4 x float> @global_load_i8_offset_neg0xFFFFFFFF(ptr addrspace(1) %sbase)
; GFX1100-ISEL: ; %bb.0:
; GFX1100-ISEL-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX1100-ISEL-NEXT: v_add_co_u32 v0, vcc_lo, v0, 1
-; GFX1100-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1100-ISEL-NEXT: v_add_co_ci_u32_e64 v1, null, -1, v1, vcc_lo
; GFX1100-ISEL-NEXT: global_load_b128 v[0:3], v[0:1], off
; GFX1100-ISEL-NEXT: s_waitcnt vmcnt(0)
@@ -3298,7 +3255,6 @@ define <4 x float> @global_load_i8_offset_neg0xFFFFFFFF(ptr addrspace(1) %sbase)
; GFX1250-ISEL-NEXT: s_wait_loadcnt_dscnt 0x0
; GFX1250-ISEL-NEXT: s_wait_kmcnt 0x0
; GFX1250-ISEL-NEXT: v_add_co_u32 v0, vcc_lo, v0, 1
-; GFX1250-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1250-ISEL-NEXT: v_add_co_ci_u32_e64 v1, null, -1, v1, vcc_lo
; GFX1250-ISEL-NEXT: global_load_b128 v[0:3], v[0:1], off
; GFX1250-ISEL-NEXT: s_wait_loadcnt 0x0
@@ -3312,7 +3268,6 @@ define <4 x float> @global_load_i8_offset_neg0xFFFFFFFF(ptr addrspace(1) %sbase)
; GFX1310-ISEL-NEXT: s_wait_bvhcnt 0x0
; GFX1310-ISEL-NEXT: s_wait_kmcnt 0x0
; GFX1310-ISEL-NEXT: v_add_co_u32 v0, vcc_lo, v0, 1
-; GFX1310-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1310-ISEL-NEXT: v_add_co_ci_u32_e64 v1, null, -1, v1, vcc_lo
; GFX1310-ISEL-NEXT: global_load_b128 v[0:3], v[0:1], off
; GFX1310-ISEL-NEXT: s_wait_loadcnt 0x0
@@ -3410,7 +3365,6 @@ define <4 x float> @global_load_i8_offset_neg0x100000000(ptr addrspace(1) %sbase
; GFX1100-ISEL: ; %bb.0:
; GFX1100-ISEL-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX1100-ISEL-NEXT: v_add_co_u32 v0, vcc_lo, v0, 0
-; GFX1100-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1100-ISEL-NEXT: v_add_co_ci_u32_e64 v1, null, -1, v1, vcc_lo
; GFX1100-ISEL-NEXT: global_load_b128 v[0:3], v[0:1], off glc
; GFX1100-ISEL-NEXT: s_waitcnt vmcnt(0)
@@ -3421,7 +3375,6 @@ define <4 x float> @global_load_i8_offset_neg0x100000000(ptr addrspace(1) %sbase
; GFX1250-ISEL-NEXT: s_wait_loadcnt_dscnt 0x0
; GFX1250-ISEL-NEXT: s_wait_kmcnt 0x0
; GFX1250-ISEL-NEXT: v_add_co_u32 v0, vcc_lo, v0, 0
-; GFX1250-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1250-ISEL-NEXT: v_add_co_ci_u32_e64 v1, null, -1, v1, vcc_lo
; GFX1250-ISEL-NEXT: global_load_b128 v[0:3], v[0:1], off
; GFX1250-ISEL-NEXT: s_wait_loadcnt 0x0
@@ -3435,7 +3388,6 @@ define <4 x float> @global_load_i8_offset_neg0x100000000(ptr addrspace(1) %sbase
; GFX1310-ISEL-NEXT: s_wait_bvhcnt 0x0
; GFX1310-ISEL-NEXT: s_wait_kmcnt 0x0
; GFX1310-ISEL-NEXT: v_add_co_u32 v0, vcc_lo, v0, 0
-; GFX1310-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1310-ISEL-NEXT: v_add_co_ci_u32_e64 v1, null, -1, v1, vcc_lo
; GFX1310-ISEL-NEXT: global_load_b128 v[0:3], v[0:1], off scope:SCOPE_SE
; GFX1310-ISEL-NEXT: s_wait_loadcnt 0x0
@@ -3480,7 +3432,6 @@ define <4 x float> @global_load_i8_offset_neg0x100000001(ptr addrspace(1) %sbase
; GFX1100-SDAG: ; %bb.0:
; GFX1100-SDAG-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX1100-SDAG-NEXT: v_add_co_u32 v0, vcc_lo, 0, v0
-; GFX1100-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1100-SDAG-NEXT: v_add_co_ci_u32_e64 v1, null, -1, v1, vcc_lo
; GFX1100-SDAG-NEXT: global_load_b128 v[0:3], v[0:1], off offset:-1 glc
; GFX1100-SDAG-NEXT: s_waitcnt vmcnt(0)
@@ -3491,7 +3442,6 @@ define <4 x float> @global_load_i8_offset_neg0x100000001(ptr addrspace(1) %sbase
; GFX1250-SDAG-NEXT: s_wait_loadcnt_dscnt 0x0
; GFX1250-SDAG-NEXT: s_wait_kmcnt 0x0
; GFX1250-SDAG-NEXT: v_add_co_u32 v0, vcc_lo, 0, v0
-; GFX1250-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1250-SDAG-NEXT: v_add_co_ci_u32_e64 v1, null, -1, v1, vcc_lo
; GFX1250-SDAG-NEXT: global_load_b128 v[0:3], v[0:1], off offset:-1 scope:SCOPE_DEV
; GFX1250-SDAG-NEXT: s_wait_loadcnt 0x0
@@ -3505,7 +3455,6 @@ define <4 x float> @global_load_i8_offset_neg0x100000001(ptr addrspace(1) %sbase
; GFX1310-SDAG-NEXT: s_wait_bvhcnt 0x0
; GFX1310-SDAG-NEXT: s_wait_kmcnt 0x0
; GFX1310-SDAG-NEXT: v_add_co_u32 v0, vcc_lo, 0, v0
-; GFX1310-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1310-SDAG-NEXT: v_add_co_ci_u32_e64 v1, null, -1, v1, vcc_lo
; GFX1310-SDAG-NEXT: global_load_b128 v[0:3], v[0:1], off offset:-1 scope:SCOPE_DEV
; GFX1310-SDAG-NEXT: s_wait_loadcnt 0x0
@@ -3543,7 +3492,6 @@ define <4 x float> @global_load_i8_offset_neg0x100000001(ptr addrspace(1) %sbase
; GFX1100-ISEL: ; %bb.0:
; GFX1100-ISEL-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX1100-ISEL-NEXT: v_add_co_u32 v0, vcc_lo, v0, -1
-; GFX1100-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1100-ISEL-NEXT: v_add_co_ci_u32_e64 v1, null, -2, v1, vcc_lo
; GFX1100-ISEL-NEXT: global_load_b128 v[0:3], v[0:1], off glc
; GFX1100-ISEL-NEXT: s_waitcnt vmcnt(0)
@@ -3554,7 +3502,6 @@ define <4 x float> @global_load_i8_offset_neg0x100000001(ptr addrspace(1) %sbase
; GFX1250-ISEL-NEXT: s_wait_loadcnt_dscnt 0x0
; GFX1250-ISEL-NEXT: s_wait_kmcnt 0x0
; GFX1250-ISEL-NEXT: v_add_co_u32 v0, vcc_lo, v0, -1
-; GFX1250-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1250-ISEL-NEXT: v_add_co_ci_u32_e64 v1, null, -2, v1, vcc_lo
; GFX1250-ISEL-NEXT: global_load_b128 v[0:3], v[0:1], off scope:SCOPE_DEV
; GFX1250-ISEL-NEXT: s_wait_loadcnt 0x0
@@ -3568,7 +3515,6 @@ define <4 x float> @global_load_i8_offset_neg0x100000001(ptr addrspace(1) %sbase
; GFX1310-ISEL-NEXT: s_wait_bvhcnt 0x0
; GFX1310-ISEL-NEXT: s_wait_kmcnt 0x0
; GFX1310-ISEL-NEXT: v_add_co_u32 v0, vcc_lo, v0, -1
-; GFX1310-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1310-ISEL-NEXT: v_add_co_ci_u32_e64 v1, null, -2, v1, vcc_lo
; GFX1310-ISEL-NEXT: global_load_b128 v[0:3], v[0:1], off scope:SCOPE_DEV
; GFX1310-ISEL-NEXT: s_wait_loadcnt 0x0
@@ -3616,7 +3562,6 @@ define <4 x float> @global_load_i8_zext_vgpr(ptr addrspace(1) %sbase, i32 %voffs
; GFX1100-SDAG: ; %bb.0:
; GFX1100-SDAG-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX1100-SDAG-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
-; GFX1100-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1100-SDAG-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1100-SDAG-NEXT: global_load_b128 v[0:3], v[0:1], off glc
; GFX1100-SDAG-NEXT: s_waitcnt vmcnt(0)
@@ -3641,7 +3586,6 @@ define <4 x float> @global_load_i8_zext_vgpr(ptr addrspace(1) %sbase, i32 %voffs
; GFX1310-SDAG-NEXT: s_wait_bvhcnt 0x0
; GFX1310-SDAG-NEXT: s_wait_kmcnt 0x0
; GFX1310-SDAG-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
-; GFX1310-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1310-SDAG-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1310-SDAG-NEXT: global_load_b128 v[0:3], v[0:1], off scope:SCOPE_SYS
; GFX1310-SDAG-NEXT: s_wait_loadcnt 0x0
@@ -3679,7 +3623,6 @@ define <4 x float> @global_load_i8_zext_vgpr(ptr addrspace(1) %sbase, i32 %voffs
; GFX1100-ISEL: ; %bb.0:
; GFX1100-ISEL-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX1100-ISEL-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
-; GFX1100-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1100-ISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1100-ISEL-NEXT: global_load_b128 v[0:3], v[0:1], off glc
; GFX1100-ISEL-NEXT: s_waitcnt vmcnt(0)
@@ -3690,7 +3633,6 @@ define <4 x float> @global_load_i8_zext_vgpr(ptr addrspace(1) %sbase, i32 %voffs
; GFX1250-ISEL-NEXT: s_wait_loadcnt_dscnt 0x0
; GFX1250-ISEL-NEXT: s_wait_kmcnt 0x0
; GFX1250-ISEL-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
-; GFX1250-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1250-ISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1250-ISEL-NEXT: global_load_b128 v[0:3], v[0:1], off scope:SCOPE_SYS
; GFX1250-ISEL-NEXT: s_wait_loadcnt 0x0
@@ -3704,7 +3646,6 @@ define <4 x float> @global_load_i8_zext_vgpr(ptr addrspace(1) %sbase, i32 %voffs
; GFX1310-ISEL-NEXT: s_wait_bvhcnt 0x0
; GFX1310-ISEL-NEXT: s_wait_kmcnt 0x0
; GFX1310-ISEL-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
-; GFX1310-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1310-ISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1310-ISEL-NEXT: global_load_b128 v[0:3], v[0:1], off scope:SCOPE_SYS
; GFX1310-ISEL-NEXT: s_wait_loadcnt 0x0
@@ -3751,7 +3692,6 @@ define <4 x float> @global_load_i8_zext_vgpr_offset_4095(ptr addrspace(1) %sbase
; GFX1100-SDAG: ; %bb.0:
; GFX1100-SDAG-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX1100-SDAG-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
-; GFX1100-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1100-SDAG-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1100-SDAG-NEXT: global_load_b128 v[0:3], v[0:1], off offset:4095
; GFX1100-SDAG-NEXT: s_waitcnt vmcnt(0)
@@ -3776,7 +3716,6 @@ define <4 x float> @global_load_i8_zext_vgpr_offset_4095(ptr addrspace(1) %sbase
; GFX1310-SDAG-NEXT: s_wait_bvhcnt 0x0
; GFX1310-SDAG-NEXT: s_wait_kmcnt 0x0
; GFX1310-SDAG-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
-; GFX1310-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1310-SDAG-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1310-SDAG-NEXT: global_load_b128 v[0:3], v[0:1], off offset:4095
; GFX1310-SDAG-NEXT: s_wait_loadcnt 0x0
@@ -3816,7 +3755,6 @@ define <4 x float> @global_load_i8_zext_vgpr_offset_4095(ptr addrspace(1) %sbase
; GFX1100-ISEL: ; %bb.0:
; GFX1100-ISEL-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX1100-ISEL-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
-; GFX1100-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1100-ISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1100-ISEL-NEXT: global_load_b128 v[0:3], v[0:1], off offset:4095
; GFX1100-ISEL-NEXT: s_waitcnt vmcnt(0)
@@ -3827,7 +3765,6 @@ define <4 x float> @global_load_i8_zext_vgpr_offset_4095(ptr addrspace(1) %sbase
; GFX1250-ISEL-NEXT: s_wait_loadcnt_dscnt 0x0
; GFX1250-ISEL-NEXT: s_wait_kmcnt 0x0
; GFX1250-ISEL-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
-; GFX1250-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1250-ISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1250-ISEL-NEXT: global_load_b128 v[0:3], v[0:1], off offset:4095
; GFX1250-ISEL-NEXT: s_wait_loadcnt 0x0
@@ -3841,7 +3778,6 @@ define <4 x float> @global_load_i8_zext_vgpr_offset_4095(ptr addrspace(1) %sbase
; GFX1310-ISEL-NEXT: s_wait_bvhcnt 0x0
; GFX1310-ISEL-NEXT: s_wait_kmcnt 0x0
; GFX1310-ISEL-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
-; GFX1310-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1310-ISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1310-ISEL-NEXT: global_load_b128 v[0:3], v[0:1], off offset:4095
; GFX1310-ISEL-NEXT: s_wait_loadcnt 0x0
@@ -3894,10 +3830,9 @@ define <4 x float> @global_load_i8_zext_vgpr_offset_4096(ptr addrspace(1) %sbase
; GFX1100-SDAG: ; %bb.0:
; GFX1100-SDAG-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX1100-SDAG-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
-; GFX1100-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1100-SDAG-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
+; GFX1100-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1100-SDAG-NEXT: v_add_co_u32 v0, vcc_lo, 0x1000, v0
-; GFX1100-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1100-SDAG-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1100-SDAG-NEXT: global_load_b128 v[0:3], v[0:1], off glc
; GFX1100-SDAG-NEXT: s_waitcnt vmcnt(0)
@@ -3922,7 +3857,6 @@ define <4 x float> @global_load_i8_zext_vgpr_offset_4096(ptr addrspace(1) %sbase
; GFX1310-SDAG-NEXT: s_wait_bvhcnt 0x0
; GFX1310-SDAG-NEXT: s_wait_kmcnt 0x0
; GFX1310-SDAG-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
-; GFX1310-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1310-SDAG-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1310-SDAG-NEXT: global_load_b128 v[0:3], v[0:1], off offset:4096 scope:SCOPE_SE
; GFX1310-SDAG-NEXT: s_wait_loadcnt 0x0
@@ -3967,10 +3901,9 @@ define <4 x float> @global_load_i8_zext_vgpr_offset_4096(ptr addrspace(1) %sbase
; GFX1100-ISEL: ; %bb.0:
; GFX1100-ISEL-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX1100-ISEL-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
-; GFX1100-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1100-ISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
+; GFX1100-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1100-ISEL-NEXT: v_add_co_u32 v0, vcc_lo, 0x1000, v0
-; GFX1100-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1100-ISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1100-ISEL-NEXT: global_load_b128 v[0:3], v[0:1], off glc
; GFX1100-ISEL-NEXT: s_waitcnt vmcnt(0)
@@ -3981,7 +3914,6 @@ define <4 x float> @global_load_i8_zext_vgpr_offset_4096(ptr addrspace(1) %sbase
; GFX1250-ISEL-NEXT: s_wait_loadcnt_dscnt 0x0
; GFX1250-ISEL-NEXT: s_wait_kmcnt 0x0
; GFX1250-ISEL-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
-; GFX1250-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1250-ISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1250-ISEL-NEXT: global_load_b128 v[0:3], v[0:1], off offset:4096
; GFX1250-ISEL-NEXT: s_wait_loadcnt 0x0
@@ -3995,7 +3927,6 @@ define <4 x float> @global_load_i8_zext_vgpr_offset_4096(ptr addrspace(1) %sbase
; GFX1310-ISEL-NEXT: s_wait_bvhcnt 0x0
; GFX1310-ISEL-NEXT: s_wait_kmcnt 0x0
; GFX1310-ISEL-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
-; GFX1310-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1310-ISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1310-ISEL-NEXT: global_load_b128 v[0:3], v[0:1], off offset:4096 scope:SCOPE_SE
; GFX1310-ISEL-NEXT: s_wait_loadcnt 0x0
@@ -4043,7 +3974,6 @@ define <4 x float> @global_load_i8_zext_vgpr_offset_neg4096(ptr addrspace(1) %sb
; GFX1100-SDAG: ; %bb.0:
; GFX1100-SDAG-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX1100-SDAG-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
-; GFX1100-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1100-SDAG-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1100-SDAG-NEXT: global_load_b128 v[0:3], v[0:1], off offset:-4096 glc
; GFX1100-SDAG-NEXT: s_waitcnt vmcnt(0)
@@ -4068,7 +3998,6 @@ define <4 x float> @global_load_i8_zext_vgpr_offset_neg4096(ptr addrspace(1) %sb
; GFX1310-SDAG-NEXT: s_wait_bvhcnt 0x0
; GFX1310-SDAG-NEXT: s_wait_kmcnt 0x0
; GFX1310-SDAG-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
-; GFX1310-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1310-SDAG-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1310-SDAG-NEXT: global_load_b128 v[0:3], v[0:1], off offset:-4096 scope:SCOPE_DEV
; GFX1310-SDAG-NEXT: s_wait_loadcnt 0x0
@@ -4108,7 +4037,6 @@ define <4 x float> @global_load_i8_zext_vgpr_offset_neg4096(ptr addrspace(1) %sb
; GFX1100-ISEL: ; %bb.0:
; GFX1100-ISEL-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX1100-ISEL-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
-; GFX1100-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1100-ISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1100-ISEL-NEXT: global_load_b128 v[0:3], v[0:1], off offset:-4096 glc
; GFX1100-ISEL-NEXT: s_waitcnt vmcnt(0)
@@ -4119,7 +4047,6 @@ define <4 x float> @global_load_i8_zext_vgpr_offset_neg4096(ptr addrspace(1) %sb
; GFX1250-ISEL-NEXT: s_wait_loadcnt_dscnt 0x0
; GFX1250-ISEL-NEXT: s_wait_kmcnt 0x0
; GFX1250-ISEL-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
-; GFX1250-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1250-ISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1250-ISEL-NEXT: global_load_b128 v[0:3], v[0:1], off offset:-4096 scope:SCOPE_DEV
; GFX1250-ISEL-NEXT: s_wait_loadcnt 0x0
@@ -4133,7 +4060,6 @@ define <4 x float> @global_load_i8_zext_vgpr_offset_neg4096(ptr addrspace(1) %sb
; GFX1310-ISEL-NEXT: s_wait_bvhcnt 0x0
; GFX1310-ISEL-NEXT: s_wait_kmcnt 0x0
; GFX1310-ISEL-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
-; GFX1310-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1310-ISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1310-ISEL-NEXT: global_load_b128 v[0:3], v[0:1], off offset:-4096 scope:SCOPE_DEV
; GFX1310-ISEL-NEXT: s_wait_loadcnt 0x0
@@ -4186,10 +4112,9 @@ define <4 x float> @global_load_i8_zext_vgpr_offset_neg4097(ptr addrspace(1) %sb
; GFX1100-SDAG: ; %bb.0:
; GFX1100-SDAG-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX1100-SDAG-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
-; GFX1100-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1100-SDAG-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
+; GFX1100-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1100-SDAG-NEXT: v_add_co_u32 v0, vcc_lo, 0xfffff000, v0
-; GFX1100-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1100-SDAG-NEXT: v_add_co_ci_u32_e64 v1, null, -1, v1, vcc_lo
; GFX1100-SDAG-NEXT: global_load_b128 v[0:3], v[0:1], off offset:-1 glc
; GFX1100-SDAG-NEXT: s_waitcnt vmcnt(0)
@@ -4214,7 +4139,6 @@ define <4 x float> @global_load_i8_zext_vgpr_offset_neg4097(ptr addrspace(1) %sb
; GFX1310-SDAG-NEXT: s_wait_bvhcnt 0x0
; GFX1310-SDAG-NEXT: s_wait_kmcnt 0x0
; GFX1310-SDAG-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
-; GFX1310-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1310-SDAG-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1310-SDAG-NEXT: global_load_b128 v[0:3], v[0:1], off offset:-4097 scope:SCOPE_SYS
; GFX1310-SDAG-NEXT: s_wait_loadcnt 0x0
@@ -4259,10 +4183,9 @@ define <4 x float> @global_load_i8_zext_vgpr_offset_neg4097(ptr addrspace(1) %sb
; GFX1100-ISEL: ; %bb.0:
; GFX1100-ISEL-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX1100-ISEL-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
-; GFX1100-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1100-ISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
+; GFX1100-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1100-ISEL-NEXT: v_add_co_u32 v0, vcc_lo, 0xffffefff, v0
-; GFX1100-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1100-ISEL-NEXT: v_add_co_ci_u32_e64 v1, null, -1, v1, vcc_lo
; GFX1100-ISEL-NEXT: global_load_b128 v[0:3], v[0:1], off glc
; GFX1100-ISEL-NEXT: s_waitcnt vmcnt(0)
@@ -4273,7 +4196,6 @@ define <4 x float> @global_load_i8_zext_vgpr_offset_neg4097(ptr addrspace(1) %sb
; GFX1250-ISEL-NEXT: s_wait_loadcnt_dscnt 0x0
; GFX1250-ISEL-NEXT: s_wait_kmcnt 0x0
; GFX1250-ISEL-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
-; GFX1250-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1250-ISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1250-ISEL-NEXT: global_load_b128 v[0:3], v[0:1], off offset:-4097 scope:SCOPE_SYS
; GFX1250-ISEL-NEXT: s_wait_loadcnt 0x0
@@ -4287,7 +4209,6 @@ define <4 x float> @global_load_i8_zext_vgpr_offset_neg4097(ptr addrspace(1) %sb
; GFX1310-ISEL-NEXT: s_wait_bvhcnt 0x0
; GFX1310-ISEL-NEXT: s_wait_kmcnt 0x0
; GFX1310-ISEL-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
-; GFX1310-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1310-ISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1310-ISEL-NEXT: global_load_b128 v[0:3], v[0:1], off offset:-4097 scope:SCOPE_SYS
; GFX1310-ISEL-NEXT: s_wait_loadcnt 0x0
@@ -4333,7 +4254,6 @@ define <4 x float> @global_load_i8_zext_vgpr_offset_2047(ptr addrspace(1) %sbase
; GFX1100-SDAG: ; %bb.0:
; GFX1100-SDAG-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX1100-SDAG-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
-; GFX1100-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1100-SDAG-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1100-SDAG-NEXT: global_load_b128 v[0:3], v[0:1], off offset:2047
; GFX1100-SDAG-NEXT: s_waitcnt vmcnt(0)
@@ -4358,7 +4278,6 @@ define <4 x float> @global_load_i8_zext_vgpr_offset_2047(ptr addrspace(1) %sbase
; GFX1310-SDAG-NEXT: s_wait_bvhcnt 0x0
; GFX1310-SDAG-NEXT: s_wait_kmcnt 0x0
; GFX1310-SDAG-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
-; GFX1310-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1310-SDAG-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1310-SDAG-NEXT: global_load_b128 v[0:3], v[0:1], off offset:2047
; GFX1310-SDAG-NEXT: s_wait_loadcnt 0x0
@@ -4396,7 +4315,6 @@ define <4 x float> @global_load_i8_zext_vgpr_offset_2047(ptr addrspace(1) %sbase
; GFX1100-ISEL: ; %bb.0:
; GFX1100-ISEL-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX1100-ISEL-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
-; GFX1100-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1100-ISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1100-ISEL-NEXT: global_load_b128 v[0:3], v[0:1], off offset:2047
; GFX1100-ISEL-NEXT: s_waitcnt vmcnt(0)
@@ -4407,7 +4325,6 @@ define <4 x float> @global_load_i8_zext_vgpr_offset_2047(ptr addrspace(1) %sbase
; GFX1250-ISEL-NEXT: s_wait_loadcnt_dscnt 0x0
; GFX1250-ISEL-NEXT: s_wait_kmcnt 0x0
; GFX1250-ISEL-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
-; GFX1250-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1250-ISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1250-ISEL-NEXT: global_load_b128 v[0:3], v[0:1], off offset:2047
; GFX1250-ISEL-NEXT: s_wait_loadcnt 0x0
@@ -4421,7 +4338,6 @@ define <4 x float> @global_load_i8_zext_vgpr_offset_2047(ptr addrspace(1) %sbase
; GFX1310-ISEL-NEXT: s_wait_bvhcnt 0x0
; GFX1310-ISEL-NEXT: s_wait_kmcnt 0x0
; GFX1310-ISEL-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
-; GFX1310-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1310-ISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1310-ISEL-NEXT: global_load_b128 v[0:3], v[0:1], off offset:2047
; GFX1310-ISEL-NEXT: s_wait_loadcnt 0x0
@@ -4469,7 +4385,6 @@ define <4 x float> @global_load_i8_zext_vgpr_offset_2048(ptr addrspace(1) %sbase
; GFX1100-SDAG: ; %bb.0:
; GFX1100-SDAG-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX1100-SDAG-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
-; GFX1100-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1100-SDAG-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1100-SDAG-NEXT: global_load_b128 v[0:3], v[0:1], off offset:2048 glc
; GFX1100-SDAG-NEXT: s_waitcnt vmcnt(0)
@@ -4494,7 +4409,6 @@ define <4 x float> @global_load_i8_zext_vgpr_offset_2048(ptr addrspace(1) %sbase
; GFX1310-SDAG-NEXT: s_wait_bvhcnt 0x0
; GFX1310-SDAG-NEXT: s_wait_kmcnt 0x0
; GFX1310-SDAG-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
-; GFX1310-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1310-SDAG-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1310-SDAG-NEXT: global_load_b128 v[0:3], v[0:1], off offset:2048 scope:SCOPE_SE
; GFX1310-SDAG-NEXT: s_wait_loadcnt 0x0
@@ -4534,7 +4448,6 @@ define <4 x float> @global_load_i8_zext_vgpr_offset_2048(ptr addrspace(1) %sbase
; GFX1100-ISEL: ; %bb.0:
; GFX1100-ISEL-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX1100-ISEL-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
-; GFX1100-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1100-ISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1100-ISEL-NEXT: global_load_b128 v[0:3], v[0:1], off offset:2048 glc
; GFX1100-ISEL-NEXT: s_waitcnt vmcnt(0)
@@ -4545,7 +4458,6 @@ define <4 x float> @global_load_i8_zext_vgpr_offset_2048(ptr addrspace(1) %sbase
; GFX1250-ISEL-NEXT: s_wait_loadcnt_dscnt 0x0
; GFX1250-ISEL-NEXT: s_wait_kmcnt 0x0
; GFX1250-ISEL-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
-; GFX1250-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1250-ISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1250-ISEL-NEXT: global_load_b128 v[0:3], v[0:1], off offset:2048
; GFX1250-ISEL-NEXT: s_wait_loadcnt 0x0
@@ -4559,7 +4471,6 @@ define <4 x float> @global_load_i8_zext_vgpr_offset_2048(ptr addrspace(1) %sbase
; GFX1310-ISEL-NEXT: s_wait_bvhcnt 0x0
; GFX1310-ISEL-NEXT: s_wait_kmcnt 0x0
; GFX1310-ISEL-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
-; GFX1310-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1310-ISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1310-ISEL-NEXT: global_load_b128 v[0:3], v[0:1], off offset:2048 scope:SCOPE_SE
; GFX1310-ISEL-NEXT: s_wait_loadcnt 0x0
@@ -4605,7 +4516,6 @@ define <4 x float> @global_load_i8_zext_vgpr_offset_neg2048(ptr addrspace(1) %sb
; GFX1100-SDAG: ; %bb.0:
; GFX1100-SDAG-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX1100-SDAG-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
-; GFX1100-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1100-SDAG-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1100-SDAG-NEXT: global_load_b128 v[0:3], v[0:1], off offset:-2048 glc
; GFX1100-SDAG-NEXT: s_waitcnt vmcnt(0)
@@ -4630,7 +4540,6 @@ define <4 x float> @global_load_i8_zext_vgpr_offset_neg2048(ptr addrspace(1) %sb
; GFX1310-SDAG-NEXT: s_wait_bvhcnt 0x0
; GFX1310-SDAG-NEXT: s_wait_kmcnt 0x0
; GFX1310-SDAG-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
-; GFX1310-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1310-SDAG-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1310-SDAG-NEXT: global_load_b128 v[0:3], v[0:1], off offset:-2048 scope:SCOPE_DEV
; GFX1310-SDAG-NEXT: s_wait_loadcnt 0x0
@@ -4668,7 +4577,6 @@ define <4 x float> @global_load_i8_zext_vgpr_offset_neg2048(ptr addrspace(1) %sb
; GFX1100-ISEL: ; %bb.0:
; GFX1100-ISEL-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX1100-ISEL-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
-; GFX1100-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1100-ISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1100-ISEL-NEXT: global_load_b128 v[0:3], v[0:1], off offset:-2048 glc
; GFX1100-ISEL-NEXT: s_waitcnt vmcnt(0)
@@ -4679,7 +4587,6 @@ define <4 x float> @global_load_i8_zext_vgpr_offset_neg2048(ptr addrspace(1) %sb
; GFX1250-ISEL-NEXT: s_wait_loadcnt_dscnt 0x0
; GFX1250-ISEL-NEXT: s_wait_kmcnt 0x0
; GFX1250-ISEL-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
-; GFX1250-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1250-ISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1250-ISEL-NEXT: global_load_b128 v[0:3], v[0:1], off offset:-2048 scope:SCOPE_DEV
; GFX1250-ISEL-NEXT: s_wait_loadcnt 0x0
@@ -4693,7 +4600,6 @@ define <4 x float> @global_load_i8_zext_vgpr_offset_neg2048(ptr addrspace(1) %sb
; GFX1310-ISEL-NEXT: s_wait_bvhcnt 0x0
; GFX1310-ISEL-NEXT: s_wait_kmcnt 0x0
; GFX1310-ISEL-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
-; GFX1310-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1310-ISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1310-ISEL-NEXT: global_load_b128 v[0:3], v[0:1], off offset:-2048 scope:SCOPE_DEV
; GFX1310-ISEL-NEXT: s_wait_loadcnt 0x0
@@ -4741,7 +4647,6 @@ define <4 x float> @global_load_i8_zext_vgpr_offset_neg2049(ptr addrspace(1) %sb
; GFX1100-SDAG: ; %bb.0:
; GFX1100-SDAG-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX1100-SDAG-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
-; GFX1100-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1100-SDAG-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1100-SDAG-NEXT: global_load_b128 v[0:3], v[0:1], off offset:-2049 glc
; GFX1100-SDAG-NEXT: s_waitcnt vmcnt(0)
@@ -4766,7 +4671,6 @@ define <4 x float> @global_load_i8_zext_vgpr_offset_neg2049(ptr addrspace(1) %sb
; GFX1310-SDAG-NEXT: s_wait_bvhcnt 0x0
; GFX1310-SDAG-NEXT: s_wait_kmcnt 0x0
; GFX1310-SDAG-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
-; GFX1310-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1310-SDAG-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1310-SDAG-NEXT: global_load_b128 v[0:3], v[0:1], off offset:-2049 scope:SCOPE_SYS
; GFX1310-SDAG-NEXT: s_wait_loadcnt 0x0
@@ -4806,7 +4710,6 @@ define <4 x float> @global_load_i8_zext_vgpr_offset_neg2049(ptr addrspace(1) %sb
; GFX1100-ISEL: ; %bb.0:
; GFX1100-ISEL-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX1100-ISEL-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
-; GFX1100-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1100-ISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1100-ISEL-NEXT: global_load_b128 v[0:3], v[0:1], off offset:-2049 glc
; GFX1100-ISEL-NEXT: s_waitcnt vmcnt(0)
@@ -4817,7 +4720,6 @@ define <4 x float> @global_load_i8_zext_vgpr_offset_neg2049(ptr addrspace(1) %sb
; GFX1250-ISEL-NEXT: s_wait_loadcnt_dscnt 0x0
; GFX1250-ISEL-NEXT: s_wait_kmcnt 0x0
; GFX1250-ISEL-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
-; GFX1250-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1250-ISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1250-ISEL-NEXT: global_load_b128 v[0:3], v[0:1], off offset:-2049 scope:SCOPE_SYS
; GFX1250-ISEL-NEXT: s_wait_loadcnt 0x0
@@ -4831,7 +4733,6 @@ define <4 x float> @global_load_i8_zext_vgpr_offset_neg2049(ptr addrspace(1) %sb
; GFX1310-ISEL-NEXT: s_wait_bvhcnt 0x0
; GFX1310-ISEL-NEXT: s_wait_kmcnt 0x0
; GFX1310-ISEL-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
-; GFX1310-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1310-ISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1310-ISEL-NEXT: global_load_b128 v[0:3], v[0:1], off offset:-2049 scope:SCOPE_SYS
; GFX1310-ISEL-NEXT: s_wait_loadcnt 0x0
@@ -4884,10 +4785,9 @@ define <4 x float> @global_load_i8_zext_vgpr_offset_0x7FFFFF(ptr addrspace(1) %s
; GFX1100-SDAG: ; %bb.0:
; GFX1100-SDAG-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX1100-SDAG-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
-; GFX1100-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1100-SDAG-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
+; GFX1100-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1100-SDAG-NEXT: v_add_co_u32 v0, vcc_lo, 0x7ff000, v0
-; GFX1100-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1100-SDAG-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1100-SDAG-NEXT: global_load_b128 v[0:3], v[0:1], off offset:4095
; GFX1100-SDAG-NEXT: s_waitcnt vmcnt(0)
@@ -4912,7 +4812,6 @@ define <4 x float> @global_load_i8_zext_vgpr_offset_0x7FFFFF(ptr addrspace(1) %s
; GFX1310-SDAG-NEXT: s_wait_bvhcnt 0x0
; GFX1310-SDAG-NEXT: s_wait_kmcnt 0x0
; GFX1310-SDAG-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
-; GFX1310-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1310-SDAG-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1310-SDAG-NEXT: global_load_b128 v[0:3], v[0:1], off offset:8388607
; GFX1310-SDAG-NEXT: s_wait_loadcnt 0x0
@@ -4957,10 +4856,9 @@ define <4 x float> @global_load_i8_zext_vgpr_offset_0x7FFFFF(ptr addrspace(1) %s
; GFX1100-ISEL: ; %bb.0:
; GFX1100-ISEL-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX1100-ISEL-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
-; GFX1100-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1100-ISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
+; GFX1100-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1100-ISEL-NEXT: v_add_co_u32 v0, vcc_lo, 0x7fffff, v0
-; GFX1100-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1100-ISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1100-ISEL-NEXT: global_load_b128 v[0:3], v[0:1], off
; GFX1100-ISEL-NEXT: s_waitcnt vmcnt(0)
@@ -4971,7 +4869,6 @@ define <4 x float> @global_load_i8_zext_vgpr_offset_0x7FFFFF(ptr addrspace(1) %s
; GFX1250-ISEL-NEXT: s_wait_loadcnt_dscnt 0x0
; GFX1250-ISEL-NEXT: s_wait_kmcnt 0x0
; GFX1250-ISEL-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
-; GFX1250-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1250-ISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1250-ISEL-NEXT: global_load_b128 v[0:3], v[0:1], off offset:8388607
; GFX1250-ISEL-NEXT: s_wait_loadcnt 0x0
@@ -4985,7 +4882,6 @@ define <4 x float> @global_load_i8_zext_vgpr_offset_0x7FFFFF(ptr addrspace(1) %s
; GFX1310-ISEL-NEXT: s_wait_bvhcnt 0x0
; GFX1310-ISEL-NEXT: s_wait_kmcnt 0x0
; GFX1310-ISEL-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
-; GFX1310-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1310-ISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1310-ISEL-NEXT: global_load_b128 v[0:3], v[0:1], off offset:8388607
; GFX1310-ISEL-NEXT: s_wait_loadcnt 0x0
@@ -5037,10 +4933,9 @@ define <4 x float> @global_load_i8_zext_vgpr_offset_0xFFFFFF(ptr addrspace(1) %s
; GFX1100-SDAG: ; %bb.0:
; GFX1100-SDAG-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX1100-SDAG-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
-; GFX1100-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1100-SDAG-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
+; GFX1100-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1100-SDAG-NEXT: v_add_co_u32 v0, vcc_lo, 0xff800000, v0
-; GFX1100-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1100-SDAG-NEXT: v_add_co_ci_u32_e64 v1, null, -1, v1, vcc_lo
; GFX1100-SDAG-NEXT: global_load_b128 v[0:3], v[0:1], off glc
; GFX1100-SDAG-NEXT: s_waitcnt vmcnt(0)
@@ -5065,7 +4960,6 @@ define <4 x float> @global_load_i8_zext_vgpr_offset_0xFFFFFF(ptr addrspace(1) %s
; GFX1310-SDAG-NEXT: s_wait_bvhcnt 0x0
; GFX1310-SDAG-NEXT: s_wait_kmcnt 0x0
; GFX1310-SDAG-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
-; GFX1310-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1310-SDAG-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1310-SDAG-NEXT: global_load_b128 v[0:3], v[0:1], off offset:-8388608 scope:SCOPE_SE
; GFX1310-SDAG-NEXT: s_wait_loadcnt 0x0
@@ -5110,10 +5004,9 @@ define <4 x float> @global_load_i8_zext_vgpr_offset_0xFFFFFF(ptr addrspace(1) %s
; GFX1100-ISEL: ; %bb.0:
; GFX1100-ISEL-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX1100-ISEL-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
-; GFX1100-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1100-ISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
+; GFX1100-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1100-ISEL-NEXT: v_add_co_u32 v0, vcc_lo, 0xff800000, v0
-; GFX1100-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1100-ISEL-NEXT: v_add_co_ci_u32_e64 v1, null, -1, v1, vcc_lo
; GFX1100-ISEL-NEXT: global_load_b128 v[0:3], v[0:1], off glc
; GFX1100-ISEL-NEXT: s_waitcnt vmcnt(0)
@@ -5124,7 +5017,6 @@ define <4 x float> @global_load_i8_zext_vgpr_offset_0xFFFFFF(ptr addrspace(1) %s
; GFX1250-ISEL-NEXT: s_wait_loadcnt_dscnt 0x0
; GFX1250-ISEL-NEXT: s_wait_kmcnt 0x0
; GFX1250-ISEL-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
-; GFX1250-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1250-ISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1250-ISEL-NEXT: global_load_b128 v[0:3], v[0:1], off offset:-8388608
; GFX1250-ISEL-NEXT: s_wait_loadcnt 0x0
@@ -5138,7 +5030,6 @@ define <4 x float> @global_load_i8_zext_vgpr_offset_0xFFFFFF(ptr addrspace(1) %s
; GFX1310-ISEL-NEXT: s_wait_bvhcnt 0x0
; GFX1310-ISEL-NEXT: s_wait_kmcnt 0x0
; GFX1310-ISEL-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
-; GFX1310-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1310-ISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1310-ISEL-NEXT: global_load_b128 v[0:3], v[0:1], off offset:-8388608 scope:SCOPE_SE
; GFX1310-ISEL-NEXT: s_wait_loadcnt 0x0
@@ -5186,7 +5077,6 @@ define <4 x float> @global_load_i8_zext_vgpr_offset_4095_gep_order(ptr addrspace
; GFX1100-SDAG: ; %bb.0:
; GFX1100-SDAG-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX1100-SDAG-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
-; GFX1100-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1100-SDAG-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1100-SDAG-NEXT: global_load_b128 v[0:3], v[0:1], off offset:4095 glc
; GFX1100-SDAG-NEXT: s_waitcnt vmcnt(0)
@@ -5211,7 +5101,6 @@ define <4 x float> @global_load_i8_zext_vgpr_offset_4095_gep_order(ptr addrspace
; GFX1310-SDAG-NEXT: s_wait_bvhcnt 0x0
; GFX1310-SDAG-NEXT: s_wait_kmcnt 0x0
; GFX1310-SDAG-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
-; GFX1310-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1310-SDAG-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1310-SDAG-NEXT: global_load_b128 v[0:3], v[0:1], off offset:4095 scope:SCOPE_DEV
; GFX1310-SDAG-NEXT: s_wait_loadcnt 0x0
@@ -5251,7 +5140,6 @@ define <4 x float> @global_load_i8_zext_vgpr_offset_4095_gep_order(ptr addrspace
; GFX1100-ISEL: ; %bb.0:
; GFX1100-ISEL-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX1100-ISEL-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
-; GFX1100-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1100-ISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1100-ISEL-NEXT: global_load_b128 v[0:3], v[0:1], off offset:4095 glc
; GFX1100-ISEL-NEXT: s_waitcnt vmcnt(0)
@@ -5262,7 +5150,6 @@ define <4 x float> @global_load_i8_zext_vgpr_offset_4095_gep_order(ptr addrspace
; GFX1250-ISEL-NEXT: s_wait_loadcnt_dscnt 0x0
; GFX1250-ISEL-NEXT: s_wait_kmcnt 0x0
; GFX1250-ISEL-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
-; GFX1250-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1250-ISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1250-ISEL-NEXT: global_load_b128 v[0:3], v[0:1], off offset:4095 scope:SCOPE_DEV
; GFX1250-ISEL-NEXT: s_wait_loadcnt 0x0
@@ -5276,7 +5163,6 @@ define <4 x float> @global_load_i8_zext_vgpr_offset_4095_gep_order(ptr addrspace
; GFX1310-ISEL-NEXT: s_wait_bvhcnt 0x0
; GFX1310-ISEL-NEXT: s_wait_kmcnt 0x0
; GFX1310-ISEL-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
-; GFX1310-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1310-ISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1310-ISEL-NEXT: global_load_b128 v[0:3], v[0:1], off offset:4095 scope:SCOPE_DEV
; GFX1310-ISEL-NEXT: s_wait_loadcnt 0x0
@@ -5322,7 +5208,6 @@ define <4 x float> @global_load_i8_zext_vgpr_ptrtoint(ptr addrspace(1) %sbase, i
; GFX1100-SDAG: ; %bb.0:
; GFX1100-SDAG-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX1100-SDAG-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
-; GFX1100-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1100-SDAG-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1100-SDAG-NEXT: global_load_b128 v[0:3], v[0:1], off glc
; GFX1100-SDAG-NEXT: s_waitcnt vmcnt(0)
@@ -5347,7 +5232,6 @@ define <4 x float> @global_load_i8_zext_vgpr_ptrtoint(ptr addrspace(1) %sbase, i
; GFX1310-SDAG-NEXT: s_wait_bvhcnt 0x0
; GFX1310-SDAG-NEXT: s_wait_kmcnt 0x0
; GFX1310-SDAG-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
-; GFX1310-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1310-SDAG-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1310-SDAG-NEXT: global_load_b128 v[0:3], v[0:1], off scope:SCOPE_SYS
; GFX1310-SDAG-NEXT: s_wait_loadcnt 0x0
@@ -5385,7 +5269,6 @@ define <4 x float> @global_load_i8_zext_vgpr_ptrtoint(ptr addrspace(1) %sbase, i
; GFX1100-ISEL: ; %bb.0:
; GFX1100-ISEL-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX1100-ISEL-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
-; GFX1100-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1100-ISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1100-ISEL-NEXT: global_load_b128 v[0:3], v[0:1], off glc
; GFX1100-ISEL-NEXT: s_waitcnt vmcnt(0)
@@ -5396,7 +5279,6 @@ define <4 x float> @global_load_i8_zext_vgpr_ptrtoint(ptr addrspace(1) %sbase, i
; GFX1250-ISEL-NEXT: s_wait_loadcnt_dscnt 0x0
; GFX1250-ISEL-NEXT: s_wait_kmcnt 0x0
; GFX1250-ISEL-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
-; GFX1250-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1250-ISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1250-ISEL-NEXT: global_load_b128 v[0:3], v[0:1], off scope:SCOPE_SYS
; GFX1250-ISEL-NEXT: s_wait_loadcnt 0x0
@@ -5410,7 +5292,6 @@ define <4 x float> @global_load_i8_zext_vgpr_ptrtoint(ptr addrspace(1) %sbase, i
; GFX1310-ISEL-NEXT: s_wait_bvhcnt 0x0
; GFX1310-ISEL-NEXT: s_wait_kmcnt 0x0
; GFX1310-ISEL-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
-; GFX1310-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1310-ISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1310-ISEL-NEXT: global_load_b128 v[0:3], v[0:1], off scope:SCOPE_SYS
; GFX1310-ISEL-NEXT: s_wait_loadcnt 0x0
@@ -5457,7 +5338,6 @@ define <4 x float> @global_load_i8_zext_vgpr_ptrtoint_commute_add(ptr addrspace(
; GFX1100-SDAG: ; %bb.0:
; GFX1100-SDAG-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX1100-SDAG-NEXT: v_add_co_u32 v0, vcc_lo, v2, v0
-; GFX1100-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1100-SDAG-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1100-SDAG-NEXT: global_load_b128 v[0:3], v[0:1], off
; GFX1100-SDAG-NEXT: s_waitcnt vmcnt(0)
@@ -5482,7 +5362,6 @@ define <4 x float> @global_load_i8_zext_vgpr_ptrtoint_commute_add(ptr addrspace(
; GFX1310-SDAG-NEXT: s_wait_bvhcnt 0x0
; GFX1310-SDAG-NEXT: s_wait_kmcnt 0x0
; GFX1310-SDAG-NEXT: v_add_co_u32 v0, vcc_lo, v2, v0
-; GFX1310-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1310-SDAG-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1310-SDAG-NEXT: global_load_b128 v[0:3], v[0:1], off
; GFX1310-SDAG-NEXT: s_wait_loadcnt 0x0
@@ -5520,7 +5399,6 @@ define <4 x float> @global_load_i8_zext_vgpr_ptrtoint_commute_add(ptr addrspace(
; GFX1100-ISEL: ; %bb.0:
; GFX1100-ISEL-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX1100-ISEL-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
-; GFX1100-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1100-ISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1100-ISEL-NEXT: global_load_b128 v[0:3], v[0:1], off
; GFX1100-ISEL-NEXT: s_waitcnt vmcnt(0)
@@ -5531,7 +5409,6 @@ define <4 x float> @global_load_i8_zext_vgpr_ptrtoint_commute_add(ptr addrspace(
; GFX1250-ISEL-NEXT: s_wait_loadcnt_dscnt 0x0
; GFX1250-ISEL-NEXT: s_wait_kmcnt 0x0
; GFX1250-ISEL-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
-; GFX1250-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1250-ISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1250-ISEL-NEXT: global_load_b128 v[0:3], v[0:1], off
; GFX1250-ISEL-NEXT: s_wait_loadcnt 0x0
@@ -5545,7 +5422,6 @@ define <4 x float> @global_load_i8_zext_vgpr_ptrtoint_commute_add(ptr addrspace(
; GFX1310-ISEL-NEXT: s_wait_bvhcnt 0x0
; GFX1310-ISEL-NEXT: s_wait_kmcnt 0x0
; GFX1310-ISEL-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
-; GFX1310-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1310-ISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1310-ISEL-NEXT: global_load_b128 v[0:3], v[0:1], off
; GFX1310-ISEL-NEXT: s_wait_loadcnt 0x0
@@ -5592,7 +5468,6 @@ define <4 x float> @global_load_i8_zext_vgpr_ptrtoint_commute_add_imm_offset0(pt
; GFX1100-SDAG: ; %bb.0:
; GFX1100-SDAG-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX1100-SDAG-NEXT: v_add_co_u32 v0, vcc_lo, v2, v0
-; GFX1100-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1100-SDAG-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1100-SDAG-NEXT: global_load_b128 v[0:3], v[0:1], off offset:128 glc
; GFX1100-SDAG-NEXT: s_waitcnt vmcnt(0)
@@ -5617,7 +5492,6 @@ define <4 x float> @global_load_i8_zext_vgpr_ptrtoint_commute_add_imm_offset0(pt
; GFX1310-SDAG-NEXT: s_wait_bvhcnt 0x0
; GFX1310-SDAG-NEXT: s_wait_kmcnt 0x0
; GFX1310-SDAG-NEXT: v_add_co_u32 v0, vcc_lo, v2, v0
-; GFX1310-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1310-SDAG-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1310-SDAG-NEXT: global_load_b128 v[0:3], v[0:1], off offset:128 scope:SCOPE_SE
; GFX1310-SDAG-NEXT: s_wait_loadcnt 0x0
@@ -5655,7 +5529,6 @@ define <4 x float> @global_load_i8_zext_vgpr_ptrtoint_commute_add_imm_offset0(pt
; GFX1100-ISEL: ; %bb.0:
; GFX1100-ISEL-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX1100-ISEL-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
-; GFX1100-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1100-ISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1100-ISEL-NEXT: global_load_b128 v[0:3], v[0:1], off offset:128 glc
; GFX1100-ISEL-NEXT: s_waitcnt vmcnt(0)
@@ -5666,7 +5539,6 @@ define <4 x float> @global_load_i8_zext_vgpr_ptrtoint_commute_add_imm_offset0(pt
; GFX1250-ISEL-NEXT: s_wait_loadcnt_dscnt 0x0
; GFX1250-ISEL-NEXT: s_wait_kmcnt 0x0
; GFX1250-ISEL-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
-; GFX1250-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1250-ISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1250-ISEL-NEXT: global_load_b128 v[0:3], v[0:1], off offset:128
; GFX1250-ISEL-NEXT: s_wait_loadcnt 0x0
@@ -5680,7 +5552,6 @@ define <4 x float> @global_load_i8_zext_vgpr_ptrtoint_commute_add_imm_offset0(pt
; GFX1310-ISEL-NEXT: s_wait_bvhcnt 0x0
; GFX1310-ISEL-NEXT: s_wait_kmcnt 0x0
; GFX1310-ISEL-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
-; GFX1310-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1310-ISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1310-ISEL-NEXT: global_load_b128 v[0:3], v[0:1], off offset:128 scope:SCOPE_SE
; GFX1310-ISEL-NEXT: s_wait_loadcnt 0x0
@@ -5728,7 +5599,6 @@ define <4 x float> @global_load_i8_zext_vgpr_ptrtoint_commute_add_imm_offset1(pt
; GFX1100-SDAG: ; %bb.0:
; GFX1100-SDAG-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX1100-SDAG-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
-; GFX1100-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1100-SDAG-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1100-SDAG-NEXT: global_load_b128 v[0:3], v[0:1], off offset:128 glc
; GFX1100-SDAG-NEXT: s_waitcnt vmcnt(0)
@@ -5753,7 +5623,6 @@ define <4 x float> @global_load_i8_zext_vgpr_ptrtoint_commute_add_imm_offset1(pt
; GFX1310-SDAG-NEXT: s_wait_bvhcnt 0x0
; GFX1310-SDAG-NEXT: s_wait_kmcnt 0x0
; GFX1310-SDAG-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
-; GFX1310-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1310-SDAG-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1310-SDAG-NEXT: global_load_b128 v[0:3], v[0:1], off offset:128 scope:SCOPE_DEV
; GFX1310-SDAG-NEXT: s_wait_loadcnt 0x0
@@ -5791,7 +5660,6 @@ define <4 x float> @global_load_i8_zext_vgpr_ptrtoint_commute_add_imm_offset1(pt
; GFX1100-ISEL: ; %bb.0:
; GFX1100-ISEL-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX1100-ISEL-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
-; GFX1100-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1100-ISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1100-ISEL-NEXT: global_load_b128 v[0:3], v[0:1], off offset:128 glc
; GFX1100-ISEL-NEXT: s_waitcnt vmcnt(0)
@@ -5802,7 +5670,6 @@ define <4 x float> @global_load_i8_zext_vgpr_ptrtoint_commute_add_imm_offset1(pt
; GFX1250-ISEL-NEXT: s_wait_loadcnt_dscnt 0x0
; GFX1250-ISEL-NEXT: s_wait_kmcnt 0x0
; GFX1250-ISEL-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
-; GFX1250-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1250-ISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1250-ISEL-NEXT: global_load_b128 v[0:3], v[0:1], off offset:128 scope:SCOPE_DEV
; GFX1250-ISEL-NEXT: s_wait_loadcnt 0x0
@@ -5816,7 +5683,6 @@ define <4 x float> @global_load_i8_zext_vgpr_ptrtoint_commute_add_imm_offset1(pt
; GFX1310-ISEL-NEXT: s_wait_bvhcnt 0x0
; GFX1310-ISEL-NEXT: s_wait_kmcnt 0x0
; GFX1310-ISEL-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
-; GFX1310-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1310-ISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1310-ISEL-NEXT: global_load_b128 v[0:3], v[0:1], off offset:128 scope:SCOPE_DEV
; GFX1310-ISEL-NEXT: s_wait_loadcnt 0x0
@@ -5868,7 +5734,6 @@ define <4 x float> @global_load_i8_zext_uniform_offset(ptr addrspace(1) %sbase,
; GFX1100-SDAG: ; %bb.0:
; GFX1100-SDAG-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX1100-SDAG-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
-; GFX1100-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1100-SDAG-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1100-SDAG-NEXT: global_load_b128 v[0:3], v[0:1], off glc
; GFX1100-SDAG-NEXT: s_waitcnt vmcnt(0)
@@ -5893,7 +5758,6 @@ define <4 x float> @global_load_i8_zext_uniform_offset(ptr addrspace(1) %sbase,
; GFX1310-SDAG-NEXT: s_wait_bvhcnt 0x0
; GFX1310-SDAG-NEXT: s_wait_kmcnt 0x0
; GFX1310-SDAG-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
-; GFX1310-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1310-SDAG-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1310-SDAG-NEXT: global_load_b128 v[0:3], v[0:1], off scope:SCOPE_SYS
; GFX1310-SDAG-NEXT: s_wait_loadcnt 0x0
@@ -5931,7 +5795,6 @@ define <4 x float> @global_load_i8_zext_uniform_offset(ptr addrspace(1) %sbase,
; GFX1100-ISEL: ; %bb.0:
; GFX1100-ISEL-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX1100-ISEL-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
-; GFX1100-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1100-ISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1100-ISEL-NEXT: global_load_b128 v[0:3], v[0:1], off glc
; GFX1100-ISEL-NEXT: s_waitcnt vmcnt(0)
@@ -5942,7 +5805,6 @@ define <4 x float> @global_load_i8_zext_uniform_offset(ptr addrspace(1) %sbase,
; GFX1250-ISEL-NEXT: s_wait_loadcnt_dscnt 0x0
; GFX1250-ISEL-NEXT: s_wait_kmcnt 0x0
; GFX1250-ISEL-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
-; GFX1250-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1250-ISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1250-ISEL-NEXT: global_load_b128 v[0:3], v[0:1], off scope:SCOPE_SYS
; GFX1250-ISEL-NEXT: s_wait_loadcnt 0x0
@@ -5956,7 +5818,6 @@ define <4 x float> @global_load_i8_zext_uniform_offset(ptr addrspace(1) %sbase,
; GFX1310-ISEL-NEXT: s_wait_bvhcnt 0x0
; GFX1310-ISEL-NEXT: s_wait_kmcnt 0x0
; GFX1310-ISEL-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
-; GFX1310-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1310-ISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1310-ISEL-NEXT: global_load_b128 v[0:3], v[0:1], off scope:SCOPE_SYS
; GFX1310-ISEL-NEXT: s_wait_loadcnt 0x0
@@ -6001,7 +5862,6 @@ define <4 x float> @global_load_i8_zext_uniform_offset_immoffset(ptr addrspace(1
; GFX1100-SDAG: ; %bb.0:
; GFX1100-SDAG-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX1100-SDAG-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
-; GFX1100-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1100-SDAG-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1100-SDAG-NEXT: global_load_b128 v[0:3], v[0:1], off offset:-24
; GFX1100-SDAG-NEXT: s_waitcnt vmcnt(0)
@@ -6026,7 +5886,6 @@ define <4 x float> @global_load_i8_zext_uniform_offset_immoffset(ptr addrspace(1
; GFX1310-SDAG-NEXT: s_wait_bvhcnt 0x0
; GFX1310-SDAG-NEXT: s_wait_kmcnt 0x0
; GFX1310-SDAG-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
-; GFX1310-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1310-SDAG-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1310-SDAG-NEXT: global_load_b128 v[0:3], v[0:1], off offset:-24
; GFX1310-SDAG-NEXT: s_wait_loadcnt 0x0
@@ -6064,7 +5923,6 @@ define <4 x float> @global_load_i8_zext_uniform_offset_immoffset(ptr addrspace(1
; GFX1100-ISEL: ; %bb.0:
; GFX1100-ISEL-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX1100-ISEL-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
-; GFX1100-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1100-ISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1100-ISEL-NEXT: global_load_b128 v[0:3], v[0:1], off offset:-24
; GFX1100-ISEL-NEXT: s_waitcnt vmcnt(0)
@@ -6075,7 +5933,6 @@ define <4 x float> @global_load_i8_zext_uniform_offset_immoffset(ptr addrspace(1
; GFX1250-ISEL-NEXT: s_wait_loadcnt_dscnt 0x0
; GFX1250-ISEL-NEXT: s_wait_kmcnt 0x0
; GFX1250-ISEL-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
-; GFX1250-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1250-ISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1250-ISEL-NEXT: global_load_b128 v[0:3], v[0:1], off offset:-24
; GFX1250-ISEL-NEXT: s_wait_loadcnt 0x0
@@ -6089,7 +5946,6 @@ define <4 x float> @global_load_i8_zext_uniform_offset_immoffset(ptr addrspace(1
; GFX1310-ISEL-NEXT: s_wait_bvhcnt 0x0
; GFX1310-ISEL-NEXT: s_wait_kmcnt 0x0
; GFX1310-ISEL-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
-; GFX1310-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1310-ISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1310-ISEL-NEXT: global_load_b128 v[0:3], v[0:1], off offset:-24
; GFX1310-ISEL-NEXT: s_wait_loadcnt 0x0
@@ -6135,7 +5991,6 @@ define <4 x float> @global_load_i8_zext_sgpr_ptrtoint_commute_add(ptr addrspace(
; GFX1100-SDAG: ; %bb.0:
; GFX1100-SDAG-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX1100-SDAG-NEXT: v_add_co_u32 v0, vcc_lo, v2, v0
-; GFX1100-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1100-SDAG-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1100-SDAG-NEXT: global_load_b128 v[0:3], v[0:1], off glc
; GFX1100-SDAG-NEXT: s_waitcnt vmcnt(0)
@@ -6160,7 +6015,6 @@ define <4 x float> @global_load_i8_zext_sgpr_ptrtoint_commute_add(ptr addrspace(
; GFX1310-SDAG-NEXT: s_wait_bvhcnt 0x0
; GFX1310-SDAG-NEXT: s_wait_kmcnt 0x0
; GFX1310-SDAG-NEXT: v_add_co_u32 v0, vcc_lo, v2, v0
-; GFX1310-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1310-SDAG-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1310-SDAG-NEXT: global_load_b128 v[0:3], v[0:1], off scope:SCOPE_SE
; GFX1310-SDAG-NEXT: s_wait_loadcnt 0x0
@@ -6198,7 +6052,6 @@ define <4 x float> @global_load_i8_zext_sgpr_ptrtoint_commute_add(ptr addrspace(
; GFX1100-ISEL: ; %bb.0:
; GFX1100-ISEL-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX1100-ISEL-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
-; GFX1100-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1100-ISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1100-ISEL-NEXT: global_load_b128 v[0:3], v[0:1], off glc
; GFX1100-ISEL-NEXT: s_waitcnt vmcnt(0)
@@ -6209,7 +6062,6 @@ define <4 x float> @global_load_i8_zext_sgpr_ptrtoint_commute_add(ptr addrspace(
; GFX1250-ISEL-NEXT: s_wait_loadcnt_dscnt 0x0
; GFX1250-ISEL-NEXT: s_wait_kmcnt 0x0
; GFX1250-ISEL-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
-; GFX1250-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1250-ISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1250-ISEL-NEXT: global_load_b128 v[0:3], v[0:1], off
; GFX1250-ISEL-NEXT: s_wait_loadcnt 0x0
@@ -6223,7 +6075,6 @@ define <4 x float> @global_load_i8_zext_sgpr_ptrtoint_commute_add(ptr addrspace(
; GFX1310-ISEL-NEXT: s_wait_bvhcnt 0x0
; GFX1310-ISEL-NEXT: s_wait_kmcnt 0x0
; GFX1310-ISEL-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
-; GFX1310-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1310-ISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1310-ISEL-NEXT: global_load_b128 v[0:3], v[0:1], off scope:SCOPE_SE
; GFX1310-ISEL-NEXT: s_wait_loadcnt 0x0
@@ -6270,7 +6121,6 @@ define <4 x float> @global_load_i8_zext_sgpr_ptrtoint_commute_add_imm_offset0(pt
; GFX1100-SDAG: ; %bb.0:
; GFX1100-SDAG-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX1100-SDAG-NEXT: v_add_co_u32 v0, vcc_lo, v2, v0
-; GFX1100-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1100-SDAG-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1100-SDAG-NEXT: global_load_b128 v[0:3], v[0:1], off offset:128 glc
; GFX1100-SDAG-NEXT: s_waitcnt vmcnt(0)
@@ -6295,7 +6145,6 @@ define <4 x float> @global_load_i8_zext_sgpr_ptrtoint_commute_add_imm_offset0(pt
; GFX1310-SDAG-NEXT: s_wait_bvhcnt 0x0
; GFX1310-SDAG-NEXT: s_wait_kmcnt 0x0
; GFX1310-SDAG-NEXT: v_add_co_u32 v0, vcc_lo, v2, v0
-; GFX1310-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1310-SDAG-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1310-SDAG-NEXT: global_load_b128 v[0:3], v[0:1], off offset:128 scope:SCOPE_DEV
; GFX1310-SDAG-NEXT: s_wait_loadcnt 0x0
@@ -6333,7 +6182,6 @@ define <4 x float> @global_load_i8_zext_sgpr_ptrtoint_commute_add_imm_offset0(pt
; GFX1100-ISEL: ; %bb.0:
; GFX1100-ISEL-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX1100-ISEL-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
-; GFX1100-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1100-ISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1100-ISEL-NEXT: global_load_b128 v[0:3], v[0:1], off offset:128 glc
; GFX1100-ISEL-NEXT: s_waitcnt vmcnt(0)
@@ -6344,7 +6192,6 @@ define <4 x float> @global_load_i8_zext_sgpr_ptrtoint_commute_add_imm_offset0(pt
; GFX1250-ISEL-NEXT: s_wait_loadcnt_dscnt 0x0
; GFX1250-ISEL-NEXT: s_wait_kmcnt 0x0
; GFX1250-ISEL-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
-; GFX1250-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1250-ISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1250-ISEL-NEXT: global_load_b128 v[0:3], v[0:1], off offset:128 scope:SCOPE_DEV
; GFX1250-ISEL-NEXT: s_wait_loadcnt 0x0
@@ -6358,7 +6205,6 @@ define <4 x float> @global_load_i8_zext_sgpr_ptrtoint_commute_add_imm_offset0(pt
; GFX1310-ISEL-NEXT: s_wait_bvhcnt 0x0
; GFX1310-ISEL-NEXT: s_wait_kmcnt 0x0
; GFX1310-ISEL-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
-; GFX1310-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1310-ISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1310-ISEL-NEXT: global_load_b128 v[0:3], v[0:1], off offset:128 scope:SCOPE_DEV
; GFX1310-ISEL-NEXT: s_wait_loadcnt 0x0
@@ -6406,7 +6252,6 @@ define <4 x float> @global_load_i8_vgpr64_sgpr32(ptr addrspace(1) %vbase, i32 %s
; GFX1100-SDAG: ; %bb.0:
; GFX1100-SDAG-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX1100-SDAG-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
-; GFX1100-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1100-SDAG-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1100-SDAG-NEXT: global_load_b128 v[0:3], v[0:1], off glc
; GFX1100-SDAG-NEXT: s_waitcnt vmcnt(0)
@@ -6431,7 +6276,6 @@ define <4 x float> @global_load_i8_vgpr64_sgpr32(ptr addrspace(1) %vbase, i32 %s
; GFX1310-SDAG-NEXT: s_wait_bvhcnt 0x0
; GFX1310-SDAG-NEXT: s_wait_kmcnt 0x0
; GFX1310-SDAG-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
-; GFX1310-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1310-SDAG-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1310-SDAG-NEXT: global_load_b128 v[0:3], v[0:1], off scope:SCOPE_SYS
; GFX1310-SDAG-NEXT: s_wait_loadcnt 0x0
@@ -6469,7 +6313,6 @@ define <4 x float> @global_load_i8_vgpr64_sgpr32(ptr addrspace(1) %vbase, i32 %s
; GFX1100-ISEL: ; %bb.0:
; GFX1100-ISEL-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX1100-ISEL-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
-; GFX1100-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1100-ISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1100-ISEL-NEXT: global_load_b128 v[0:3], v[0:1], off glc
; GFX1100-ISEL-NEXT: s_waitcnt vmcnt(0)
@@ -6480,7 +6323,6 @@ define <4 x float> @global_load_i8_vgpr64_sgpr32(ptr addrspace(1) %vbase, i32 %s
; GFX1250-ISEL-NEXT: s_wait_loadcnt_dscnt 0x0
; GFX1250-ISEL-NEXT: s_wait_kmcnt 0x0
; GFX1250-ISEL-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
-; GFX1250-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1250-ISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1250-ISEL-NEXT: global_load_b128 v[0:3], v[0:1], off scope:SCOPE_SYS
; GFX1250-ISEL-NEXT: s_wait_loadcnt 0x0
@@ -6494,7 +6336,6 @@ define <4 x float> @global_load_i8_vgpr64_sgpr32(ptr addrspace(1) %vbase, i32 %s
; GFX1310-ISEL-NEXT: s_wait_bvhcnt 0x0
; GFX1310-ISEL-NEXT: s_wait_kmcnt 0x0
; GFX1310-ISEL-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
-; GFX1310-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1310-ISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1310-ISEL-NEXT: global_load_b128 v[0:3], v[0:1], off scope:SCOPE_SYS
; GFX1310-ISEL-NEXT: s_wait_loadcnt 0x0
@@ -6541,7 +6382,6 @@ define <4 x float> @global_load_i8_vgpr64_sgpr32_offset_4095(ptr addrspace(1) %v
; GFX1100-SDAG: ; %bb.0:
; GFX1100-SDAG-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX1100-SDAG-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
-; GFX1100-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1100-SDAG-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1100-SDAG-NEXT: global_load_b128 v[0:3], v[0:1], off offset:4095
; GFX1100-SDAG-NEXT: s_waitcnt vmcnt(0)
@@ -6566,7 +6406,6 @@ define <4 x float> @global_load_i8_vgpr64_sgpr32_offset_4095(ptr addrspace(1) %v
; GFX1310-SDAG-NEXT: s_wait_bvhcnt 0x0
; GFX1310-SDAG-NEXT: s_wait_kmcnt 0x0
; GFX1310-SDAG-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
-; GFX1310-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1310-SDAG-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1310-SDAG-NEXT: global_load_b128 v[0:3], v[0:1], off offset:4095
; GFX1310-SDAG-NEXT: s_wait_loadcnt 0x0
@@ -6606,7 +6445,6 @@ define <4 x float> @global_load_i8_vgpr64_sgpr32_offset_4095(ptr addrspace(1) %v
; GFX1100-ISEL: ; %bb.0:
; GFX1100-ISEL-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX1100-ISEL-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
-; GFX1100-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1100-ISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1100-ISEL-NEXT: global_load_b128 v[0:3], v[0:1], off offset:4095
; GFX1100-ISEL-NEXT: s_waitcnt vmcnt(0)
@@ -6617,7 +6455,6 @@ define <4 x float> @global_load_i8_vgpr64_sgpr32_offset_4095(ptr addrspace(1) %v
; GFX1250-ISEL-NEXT: s_wait_loadcnt_dscnt 0x0
; GFX1250-ISEL-NEXT: s_wait_kmcnt 0x0
; GFX1250-ISEL-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
-; GFX1250-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1250-ISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1250-ISEL-NEXT: global_load_b128 v[0:3], v[0:1], off offset:4095
; GFX1250-ISEL-NEXT: s_wait_loadcnt 0x0
@@ -6631,7 +6468,6 @@ define <4 x float> @global_load_i8_vgpr64_sgpr32_offset_4095(ptr addrspace(1) %v
; GFX1310-ISEL-NEXT: s_wait_bvhcnt 0x0
; GFX1310-ISEL-NEXT: s_wait_kmcnt 0x0
; GFX1310-ISEL-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
-; GFX1310-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1310-ISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1310-ISEL-NEXT: global_load_b128 v[0:3], v[0:1], off offset:4095
; GFX1310-ISEL-NEXT: s_wait_loadcnt 0x0
@@ -6696,7 +6532,7 @@ define <4 x float> @global_load_f32_natural_addressing(ptr addrspace(1) %sbase,
; GFX1100-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1100-SDAG-NEXT: v_lshlrev_b64 v[2:3], 2, v[2:3]
; GFX1100-SDAG-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
-; GFX1100-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1100-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX1100-SDAG-NEXT: v_add_co_ci_u32_e64 v1, null, v1, v3, vcc_lo
; GFX1100-SDAG-NEXT: global_load_b128 v[0:3], v[0:1], off glc
; GFX1100-SDAG-NEXT: s_waitcnt vmcnt(0)
@@ -6729,7 +6565,7 @@ define <4 x float> @global_load_f32_natural_addressing(ptr addrspace(1) %sbase,
; GFX1310-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1310-SDAG-NEXT: v_lshlrev_b64_e32 v[2:3], 2, v[2:3]
; GFX1310-SDAG-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
-; GFX1310-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1310-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX1310-SDAG-NEXT: v_add_co_ci_u32_e64 v1, null, v1, v3, vcc_lo
; GFX1310-SDAG-NEXT: global_load_b128 v[0:3], v[0:1], off scope:SCOPE_SE
; GFX1310-SDAG-NEXT: s_wait_loadcnt 0x0
@@ -6781,7 +6617,7 @@ define <4 x float> @global_load_f32_natural_addressing(ptr addrspace(1) %sbase,
; GFX1100-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1100-ISEL-NEXT: v_lshlrev_b64 v[2:3], 2, v[2:3]
; GFX1100-ISEL-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
-; GFX1100-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1100-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX1100-ISEL-NEXT: v_add_co_ci_u32_e64 v1, null, v1, v3, vcc_lo
; GFX1100-ISEL-NEXT: global_load_b128 v[0:3], v[0:1], off glc
; GFX1100-ISEL-NEXT: s_waitcnt vmcnt(0)
@@ -6814,7 +6650,7 @@ define <4 x float> @global_load_f32_natural_addressing(ptr addrspace(1) %sbase,
; GFX1310-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1310-ISEL-NEXT: v_lshlrev_b64_e32 v[2:3], 2, v[2:3]
; GFX1310-ISEL-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
-; GFX1310-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1310-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX1310-ISEL-NEXT: v_add_co_ci_u32_e64 v1, null, v1, v3, vcc_lo
; GFX1310-ISEL-NEXT: global_load_b128 v[0:3], v[0:1], off scope:SCOPE_SE
; GFX1310-ISEL-NEXT: s_wait_loadcnt 0x0
@@ -6868,7 +6704,6 @@ define <4 x float> @global_load_f32_natural_addressing_immoffset(ptr addrspace(1
; GFX1100-SDAG-NEXT: global_load_b32 v2, v[2:3], off
; GFX1100-SDAG-NEXT: s_waitcnt vmcnt(0)
; GFX1100-SDAG-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
-; GFX1100-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1100-SDAG-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1100-SDAG-NEXT: global_load_b128 v[0:3], v[0:1], off offset:128 glc
; GFX1100-SDAG-NEXT: s_waitcnt vmcnt(0)
@@ -6898,7 +6733,6 @@ define <4 x float> @global_load_f32_natural_addressing_immoffset(ptr addrspace(1
; GFX1310-SDAG-NEXT: global_load_b32 v2, v[2:3], off
; GFX1310-SDAG-NEXT: s_wait_loadcnt 0x0
; GFX1310-SDAG-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
-; GFX1310-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1310-SDAG-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1310-SDAG-NEXT: global_load_b128 v[0:3], v[0:1], off offset:128 scope:SCOPE_DEV
; GFX1310-SDAG-NEXT: s_wait_loadcnt 0x0
@@ -6944,7 +6778,6 @@ define <4 x float> @global_load_f32_natural_addressing_immoffset(ptr addrspace(1
; GFX1100-ISEL-NEXT: global_load_b32 v2, v[2:3], off
; GFX1100-ISEL-NEXT: s_waitcnt vmcnt(0)
; GFX1100-ISEL-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
-; GFX1100-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1100-ISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1100-ISEL-NEXT: global_load_b128 v[0:3], v[0:1], off offset:128 glc
; GFX1100-ISEL-NEXT: s_waitcnt vmcnt(0)
@@ -6957,7 +6790,6 @@ define <4 x float> @global_load_f32_natural_addressing_immoffset(ptr addrspace(1
; GFX1250-ISEL-NEXT: global_load_b32 v2, v[2:3], off
; GFX1250-ISEL-NEXT: s_wait_loadcnt 0x0
; GFX1250-ISEL-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
-; GFX1250-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1250-ISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1250-ISEL-NEXT: global_load_b128 v[0:3], v[0:1], off offset:128 scope:SCOPE_DEV
; GFX1250-ISEL-NEXT: s_wait_loadcnt 0x0
@@ -6973,7 +6805,6 @@ define <4 x float> @global_load_f32_natural_addressing_immoffset(ptr addrspace(1
; GFX1310-ISEL-NEXT: global_load_b32 v2, v[2:3], off
; GFX1310-ISEL-NEXT: s_wait_loadcnt 0x0
; GFX1310-ISEL-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
-; GFX1310-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1310-ISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1310-ISEL-NEXT: global_load_b128 v[0:3], v[0:1], off offset:128 scope:SCOPE_DEV
; GFX1310-ISEL-NEXT: s_wait_loadcnt 0x0
@@ -7031,7 +6862,7 @@ define <4 x float> @global_load_f32_zext_vgpr_range(ptr addrspace(1) %sbase, ptr
; GFX1100-SDAG-NEXT: global_load_b32 v2, v[2:3], off
; GFX1100-SDAG-NEXT: s_waitcnt vmcnt(0)
; GFX1100-SDAG-NEXT: v_lshlrev_b32_e32 v2, 2, v2
-; GFX1100-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1100-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1100-SDAG-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
; GFX1100-SDAG-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1100-SDAG-NEXT: global_load_b128 v[0:3], v[0:1], off glc
@@ -7062,7 +6893,7 @@ define <4 x float> @global_load_f32_zext_vgpr_range(ptr addrspace(1) %sbase, ptr
; GFX1310-SDAG-NEXT: global_load_b32 v2, v[2:3], off
; GFX1310-SDAG-NEXT: s_wait_loadcnt 0x0
; GFX1310-SDAG-NEXT: v_lshlrev_b32_e32 v2, 2, v2
-; GFX1310-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1310-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1310-SDAG-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
; GFX1310-SDAG-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1310-SDAG-NEXT: global_load_b128 v[0:3], v[0:1], off scope:SCOPE_SYS
@@ -7112,7 +6943,7 @@ define <4 x float> @global_load_f32_zext_vgpr_range(ptr addrspace(1) %sbase, ptr
; GFX1100-ISEL-NEXT: global_load_b32 v2, v[2:3], off
; GFX1100-ISEL-NEXT: s_waitcnt vmcnt(0)
; GFX1100-ISEL-NEXT: v_lshlrev_b32_e32 v2, 2, v2
-; GFX1100-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1100-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1100-ISEL-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
; GFX1100-ISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1100-ISEL-NEXT: global_load_b128 v[0:3], v[0:1], off glc
@@ -7127,7 +6958,7 @@ define <4 x float> @global_load_f32_zext_vgpr_range(ptr addrspace(1) %sbase, ptr
; GFX1250-ISEL-NEXT: s_wait_loadcnt 0x0
; GFX1250-ISEL-NEXT: s_wait_xcnt 0x0
; GFX1250-ISEL-NEXT: v_lshlrev_b32_e32 v2, 2, v2
-; GFX1250-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1250-ISEL-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
; GFX1250-ISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1250-ISEL-NEXT: global_load_b128 v[0:3], v[0:1], off scope:SCOPE_SYS
@@ -7144,7 +6975,7 @@ define <4 x float> @global_load_f32_zext_vgpr_range(ptr addrspace(1) %sbase, ptr
; GFX1310-ISEL-NEXT: global_load_b32 v2, v[2:3], off
; GFX1310-ISEL-NEXT: s_wait_loadcnt 0x0
; GFX1310-ISEL-NEXT: v_lshlrev_b32_e32 v2, 2, v2
-; GFX1310-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1310-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1310-ISEL-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
; GFX1310-ISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1310-ISEL-NEXT: global_load_b128 v[0:3], v[0:1], off scope:SCOPE_SYS
@@ -7202,7 +7033,7 @@ define <4 x float> @global_load_f32_zext_vgpr_range_imm_offset(ptr addrspace(1)
; GFX1100-SDAG-NEXT: global_load_b32 v2, v[2:3], off
; GFX1100-SDAG-NEXT: s_waitcnt vmcnt(0)
; GFX1100-SDAG-NEXT: v_lshlrev_b32_e32 v2, 2, v2
-; GFX1100-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1100-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1100-SDAG-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
; GFX1100-SDAG-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1100-SDAG-NEXT: global_load_b128 v[0:3], v[0:1], off offset:400
@@ -7233,7 +7064,7 @@ define <4 x float> @global_load_f32_zext_vgpr_range_imm_offset(ptr addrspace(1)
; GFX1310-SDAG-NEXT: global_load_b32 v2, v[2:3], off
; GFX1310-SDAG-NEXT: s_wait_loadcnt 0x0
; GFX1310-SDAG-NEXT: v_lshlrev_b32_e32 v2, 2, v2
-; GFX1310-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1310-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1310-SDAG-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
; GFX1310-SDAG-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1310-SDAG-NEXT: global_load_b128 v[0:3], v[0:1], off offset:400
@@ -7283,7 +7114,7 @@ define <4 x float> @global_load_f32_zext_vgpr_range_imm_offset(ptr addrspace(1)
; GFX1100-ISEL-NEXT: global_load_b32 v2, v[2:3], off
; GFX1100-ISEL-NEXT: s_waitcnt vmcnt(0)
; GFX1100-ISEL-NEXT: v_lshlrev_b32_e32 v2, 2, v2
-; GFX1100-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1100-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1100-ISEL-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
; GFX1100-ISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1100-ISEL-NEXT: global_load_b128 v[0:3], v[0:1], off offset:400
@@ -7298,7 +7129,7 @@ define <4 x float> @global_load_f32_zext_vgpr_range_imm_offset(ptr addrspace(1)
; GFX1250-ISEL-NEXT: s_wait_loadcnt 0x0
; GFX1250-ISEL-NEXT: s_wait_xcnt 0x0
; GFX1250-ISEL-NEXT: v_lshlrev_b32_e32 v2, 2, v2
-; GFX1250-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1250-ISEL-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
; GFX1250-ISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1250-ISEL-NEXT: global_load_b128 v[0:3], v[0:1], off offset:400
@@ -7315,7 +7146,7 @@ define <4 x float> @global_load_f32_zext_vgpr_range_imm_offset(ptr addrspace(1)
; GFX1310-ISEL-NEXT: global_load_b32 v2, v[2:3], off
; GFX1310-ISEL-NEXT: s_wait_loadcnt 0x0
; GFX1310-ISEL-NEXT: v_lshlrev_b32_e32 v2, 2, v2
-; GFX1310-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1310-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1310-ISEL-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
; GFX1310-ISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1310-ISEL-NEXT: global_load_b128 v[0:3], v[0:1], off offset:400
@@ -7378,7 +7209,7 @@ define <4 x float> @global_load_f32_zext_vgpr_range_too_large(ptr addrspace(1) %
; GFX1100-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1100-SDAG-NEXT: v_lshlrev_b64 v[2:3], 2, v[2:3]
; GFX1100-SDAG-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
-; GFX1100-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1100-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX1100-SDAG-NEXT: v_add_co_ci_u32_e64 v1, null, v1, v3, vcc_lo
; GFX1100-SDAG-NEXT: global_load_b128 v[0:3], v[0:1], off glc
; GFX1100-SDAG-NEXT: s_waitcnt vmcnt(0)
@@ -7411,7 +7242,7 @@ define <4 x float> @global_load_f32_zext_vgpr_range_too_large(ptr addrspace(1) %
; GFX1310-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1310-SDAG-NEXT: v_lshlrev_b64_e32 v[2:3], 2, v[2:3]
; GFX1310-SDAG-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
-; GFX1310-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1310-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX1310-SDAG-NEXT: v_add_co_ci_u32_e64 v1, null, v1, v3, vcc_lo
; GFX1310-SDAG-NEXT: global_load_b128 v[0:3], v[0:1], off scope:SCOPE_SE
; GFX1310-SDAG-NEXT: s_wait_loadcnt 0x0
@@ -7463,7 +7294,7 @@ define <4 x float> @global_load_f32_zext_vgpr_range_too_large(ptr addrspace(1) %
; GFX1100-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1100-ISEL-NEXT: v_lshlrev_b64 v[2:3], 2, v[2:3]
; GFX1100-ISEL-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
-; GFX1100-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1100-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX1100-ISEL-NEXT: v_add_co_ci_u32_e64 v1, null, v1, v3, vcc_lo
; GFX1100-ISEL-NEXT: global_load_b128 v[0:3], v[0:1], off glc
; GFX1100-ISEL-NEXT: s_waitcnt vmcnt(0)
@@ -7496,7 +7327,7 @@ define <4 x float> @global_load_f32_zext_vgpr_range_too_large(ptr addrspace(1) %
; GFX1310-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1310-ISEL-NEXT: v_lshlrev_b64_e32 v[2:3], 2, v[2:3]
; GFX1310-ISEL-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
-; GFX1310-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1310-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX1310-ISEL-NEXT: v_add_co_ci_u32_e64 v1, null, v1, v3, vcc_lo
; GFX1310-ISEL-NEXT: global_load_b128 v[0:3], v[0:1], off scope:SCOPE_SE
; GFX1310-ISEL-NEXT: s_wait_loadcnt 0x0
@@ -7829,13 +7660,13 @@ define <4 x float> @global_addr_64bit_lsr_iv(ptr addrspace(1) %arg) {
; GFX1100-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1100-SDAG-NEXT: s_add_i32 s0, s0, 1
; GFX1100-SDAG-NEXT: s_cmpk_eq_i32 s0, 0xff
+; GFX1100-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1100-SDAG-NEXT: s_cbranch_scc0 .LBB60_1
; GFX1100-SDAG-NEXT: ; %bb.2: ; %bb2
; GFX1100-SDAG-NEXT: s_mov_b32 s1, 0
; GFX1100-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1100-SDAG-NEXT: s_lshl_b64 s[0:1], s[0:1], 2
; GFX1100-SDAG-NEXT: v_add_co_u32 v0, vcc_lo, v0, s0
-; GFX1100-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1100-SDAG-NEXT: v_add_co_ci_u32_e64 v1, null, s1, v1, vcc_lo
; GFX1100-SDAG-NEXT: global_load_b128 v[0:3], v[0:1], off
; GFX1100-SDAG-NEXT: s_waitcnt vmcnt(0)
@@ -7851,6 +7682,7 @@ define <4 x float> @global_addr_64bit_lsr_iv(ptr addrspace(1) %arg) {
; GFX1250-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1250-SDAG-NEXT: s_add_co_i32 s0, s0, 1
; GFX1250-SDAG-NEXT: s_cmp_eq_u32 s0, 0xff
+; GFX1250-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1250-SDAG-NEXT: s_cbranch_scc0 .LBB60_1
; GFX1250-SDAG-NEXT: ; %bb.2: ; %bb2
; GFX1250-SDAG-NEXT: s_mov_b32 s1, 0
@@ -7873,13 +7705,13 @@ define <4 x float> @global_addr_64bit_lsr_iv(ptr addrspace(1) %arg) {
; GFX1310-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1310-SDAG-NEXT: s_add_co_i32 s0, s0, 1
; GFX1310-SDAG-NEXT: s_cmp_eq_u32 s0, 0xff
+; GFX1310-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1310-SDAG-NEXT: s_cbranch_scc0 .LBB60_1
; GFX1310-SDAG-NEXT: ; %bb.2: ; %bb2
; GFX1310-SDAG-NEXT: s_mov_b32 s1, 0
; GFX1310-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1310-SDAG-NEXT: s_lshl_b64 s[0:1], s[0:1], 2
; GFX1310-SDAG-NEXT: v_add_co_u32 v0, vcc_lo, v0, s0
-; GFX1310-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1310-SDAG-NEXT: v_add_co_ci_u32_e64 v1, null, s1, v1, vcc_lo
; GFX1310-SDAG-NEXT: global_load_b128 v[0:3], v[0:1], off
; GFX1310-SDAG-NEXT: s_wait_loadcnt 0x0
@@ -7954,13 +7786,14 @@ define <4 x float> @global_addr_64bit_lsr_iv(ptr addrspace(1) %arg) {
; GFX1100-ISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1100-ISEL-NEXT: s_add_i32 s0, s0, 1
; GFX1100-ISEL-NEXT: s_cmpk_eq_i32 s0, 0xff
+; GFX1100-ISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1100-ISEL-NEXT: s_cbranch_scc0 .LBB60_1
; GFX1100-ISEL-NEXT: ; %bb.2: ; %bb2
; GFX1100-ISEL-NEXT: s_mov_b32 s1, 0
; GFX1100-ISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1100-ISEL-NEXT: s_lshl_b64 s[0:1], s[0:1], 2
; GFX1100-ISEL-NEXT: v_dual_mov_b32 v3, s1 :: v_dual_mov_b32 v2, s0
-; GFX1100-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1100-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1100-ISEL-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
; GFX1100-ISEL-NEXT: v_add_co_ci_u32_e64 v1, null, v1, v3, vcc_lo
; GFX1100-ISEL-NEXT: global_load_b128 v[0:3], v[0:1], off
@@ -7977,13 +7810,14 @@ define <4 x float> @global_addr_64bit_lsr_iv(ptr addrspace(1) %arg) {
; GFX1250-ISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1250-ISEL-NEXT: s_add_co_i32 s0, s0, 1
; GFX1250-ISEL-NEXT: s_cmp_eq_u32 s0, 0xff
+; GFX1250-ISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1250-ISEL-NEXT: s_cbranch_scc0 .LBB60_1
; GFX1250-ISEL-NEXT: ; %bb.2: ; %bb2
; GFX1250-ISEL-NEXT: s_mov_b32 s1, 0
; GFX1250-ISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1250-ISEL-NEXT: s_lshl_b64 s[0:1], s[0:1], 2
; GFX1250-ISEL-NEXT: v_mov_b64_e32 v[2:3], s[0:1]
-; GFX1250-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-ISEL-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
; GFX1250-ISEL-NEXT: v_add_co_ci_u32_e64 v1, null, v1, v3, vcc_lo
; GFX1250-ISEL-NEXT: global_load_b128 v[0:3], v[0:1], off
@@ -8003,13 +7837,14 @@ define <4 x float> @global_addr_64bit_lsr_iv(ptr addrspace(1) %arg) {
; GFX1310-ISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1310-ISEL-NEXT: s_add_co_i32 s0, s0, 1
; GFX1310-ISEL-NEXT: s_cmp_eq_u32 s0, 0xff
+; GFX1310-ISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1310-ISEL-NEXT: s_cbranch_scc0 .LBB60_1
; GFX1310-ISEL-NEXT: ; %bb.2: ; %bb2
; GFX1310-ISEL-NEXT: s_mov_b32 s1, 0
; GFX1310-ISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1310-ISEL-NEXT: s_lshl_b64 s[0:1], s[0:1], 2
; GFX1310-ISEL-NEXT: v_dual_mov_b32 v3, s1 :: v_dual_mov_b32 v2, s0
-; GFX1310-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1310-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1310-ISEL-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
; GFX1310-ISEL-NEXT: v_add_co_ci_u32_e64 v1, null, v1, v3, vcc_lo
; GFX1310-ISEL-NEXT: global_load_b128 v[0:3], v[0:1], off
@@ -8098,13 +7933,13 @@ define <4 x float> @global_addr_64bit_lsr_iv_multiload(ptr addrspace(1) %arg, pt
; GFX1100-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1100-SDAG-NEXT: s_add_i32 s0, s0, 1
; GFX1100-SDAG-NEXT: s_cmpk_eq_i32 s0, 0xff
+; GFX1100-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1100-SDAG-NEXT: s_cbranch_scc0 .LBB61_1
; GFX1100-SDAG-NEXT: ; %bb.2: ; %bb2
; GFX1100-SDAG-NEXT: s_mov_b32 s1, 0
; GFX1100-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1100-SDAG-NEXT: s_lshl_b64 s[0:1], s[0:1], 2
; GFX1100-SDAG-NEXT: v_add_co_u32 v0, vcc_lo, v0, s0
-; GFX1100-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1100-SDAG-NEXT: v_add_co_ci_u32_e64 v1, null, s1, v1, vcc_lo
; GFX1100-SDAG-NEXT: global_load_b128 v[0:3], v[0:1], off glc
; GFX1100-SDAG-NEXT: s_waitcnt vmcnt(0)
@@ -8120,6 +7955,7 @@ define <4 x float> @global_addr_64bit_lsr_iv_multiload(ptr addrspace(1) %arg, pt
; GFX1250-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1250-SDAG-NEXT: s_add_co_i32 s0, s0, 1
; GFX1250-SDAG-NEXT: s_cmp_eq_u32 s0, 0xff
+; GFX1250-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1250-SDAG-NEXT: s_cbranch_scc0 .LBB61_1
; GFX1250-SDAG-NEXT: ; %bb.2: ; %bb2
; GFX1250-SDAG-NEXT: s_mov_b32 s1, 0
@@ -8142,13 +7978,13 @@ define <4 x float> @global_addr_64bit_lsr_iv_multiload(ptr addrspace(1) %arg, pt
; GFX1310-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1310-SDAG-NEXT: s_add_co_i32 s0, s0, 1
; GFX1310-SDAG-NEXT: s_cmp_eq_u32 s0, 0xff
+; GFX1310-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1310-SDAG-NEXT: s_cbranch_scc0 .LBB61_1
; GFX1310-SDAG-NEXT: ; %bb.2: ; %bb2
; GFX1310-SDAG-NEXT: s_mov_b32 s1, 0
; GFX1310-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1310-SDAG-NEXT: s_lshl_b64 s[0:1], s[0:1], 2
; GFX1310-SDAG-NEXT: v_add_co_u32 v0, vcc_lo, v0, s0
-; GFX1310-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1310-SDAG-NEXT: v_add_co_ci_u32_e64 v1, null, s1, v1, vcc_lo
; GFX1310-SDAG-NEXT: global_load_b128 v[0:3], v[0:1], off scope:SCOPE_SE
; GFX1310-SDAG-NEXT: s_wait_loadcnt 0x0
@@ -8223,13 +8059,14 @@ define <4 x float> @global_addr_64bit_lsr_iv_multiload(ptr addrspace(1) %arg, pt
; GFX1100-ISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1100-ISEL-NEXT: s_add_i32 s0, s0, 1
; GFX1100-ISEL-NEXT: s_cmpk_eq_i32 s0, 0xff
+; GFX1100-ISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1100-ISEL-NEXT: s_cbranch_scc0 .LBB61_1
; GFX1100-ISEL-NEXT: ; %bb.2: ; %bb2
; GFX1100-ISEL-NEXT: s_mov_b32 s1, 0
; GFX1100-ISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1100-ISEL-NEXT: s_lshl_b64 s[0:1], s[0:1], 2
; GFX1100-ISEL-NEXT: v_dual_mov_b32 v3, s1 :: v_dual_mov_b32 v2, s0
-; GFX1100-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1100-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1100-ISEL-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
; GFX1100-ISEL-NEXT: v_add_co_ci_u32_e64 v1, null, v1, v3, vcc_lo
; GFX1100-ISEL-NEXT: global_load_b128 v[0:3], v[0:1], off glc
@@ -8246,13 +8083,14 @@ define <4 x float> @global_addr_64bit_lsr_iv_multiload(ptr addrspace(1) %arg, pt
; GFX1250-ISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1250-ISEL-NEXT: s_add_co_i32 s0, s0, 1
; GFX1250-ISEL-NEXT: s_cmp_eq_u32 s0, 0xff
+; GFX1250-ISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1250-ISEL-NEXT: s_cbranch_scc0 .LBB61_1
; GFX1250-ISEL-NEXT: ; %bb.2: ; %bb2
; GFX1250-ISEL-NEXT: s_mov_b32 s1, 0
; GFX1250-ISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1250-ISEL-NEXT: s_lshl_b64 s[0:1], s[0:1], 2
; GFX1250-ISEL-NEXT: v_mov_b64_e32 v[2:3], s[0:1]
-; GFX1250-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-ISEL-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
; GFX1250-ISEL-NEXT: v_add_co_ci_u32_e64 v1, null, v1, v3, vcc_lo
; GFX1250-ISEL-NEXT: global_load_b128 v[0:3], v[0:1], off
@@ -8272,13 +8110,14 @@ define <4 x float> @global_addr_64bit_lsr_iv_multiload(ptr addrspace(1) %arg, pt
; GFX1310-ISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1310-ISEL-NEXT: s_add_co_i32 s0, s0, 1
; GFX1310-ISEL-NEXT: s_cmp_eq_u32 s0, 0xff
+; GFX1310-ISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1310-ISEL-NEXT: s_cbranch_scc0 .LBB61_1
; GFX1310-ISEL-NEXT: ; %bb.2: ; %bb2
; GFX1310-ISEL-NEXT: s_mov_b32 s1, 0
; GFX1310-ISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1310-ISEL-NEXT: s_lshl_b64 s[0:1], s[0:1], 2
; GFX1310-ISEL-NEXT: v_dual_mov_b32 v3, s1 :: v_dual_mov_b32 v2, s0
-; GFX1310-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1310-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1310-ISEL-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
; GFX1310-ISEL-NEXT: v_add_co_ci_u32_e64 v1, null, v1, v3, vcc_lo
; GFX1310-ISEL-NEXT: global_load_b128 v[0:3], v[0:1], off scope:SCOPE_SE
@@ -9152,7 +8991,6 @@ define <4 x float> @global_load_saddr_i8_offset_neg4097(ptr addrspace(1) inreg %
; GFX1100-SDAG: ; %bb.0:
; GFX1100-SDAG-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX1100-SDAG-NEXT: v_add_co_u32 v0, s0, 0xfffff000, s0
-; GFX1100-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1100-SDAG-NEXT: v_add_co_ci_u32_e64 v1, null, -1, s1, s0
; GFX1100-SDAG-NEXT: global_load_b128 v[0:3], v[0:1], off offset:-1
; GFX1100-SDAG-NEXT: s_waitcnt vmcnt(0)
@@ -9328,7 +9166,6 @@ define <4 x float> @global_load_saddr_i8_offset_neg4098(ptr addrspace(1) inreg %
; GFX1100-SDAG: ; %bb.0:
; GFX1100-SDAG-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX1100-SDAG-NEXT: v_add_co_u32 v0, s0, 0xfffff000, s0
-; GFX1100-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1100-SDAG-NEXT: v_add_co_ci_u32_e64 v1, null, -1, s1, s0
; GFX1100-SDAG-NEXT: global_load_b128 v[0:3], v[0:1], off offset:-2 glc
; GFX1100-SDAG-NEXT: s_waitcnt vmcnt(0)
@@ -10625,7 +10462,6 @@ define <4 x float> @global_load_saddr_i8_offset_0xFFFFFF(ptr addrspace(1) inreg
; GFX1100-SDAG: ; %bb.0:
; GFX1100-SDAG-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX1100-SDAG-NEXT: v_add_co_u32 v0, s0, 0xff800000, s0
-; GFX1100-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1100-SDAG-NEXT: v_add_co_ci_u32_e64 v1, null, -1, s1, s0
; GFX1100-SDAG-NEXT: global_load_b128 v[0:3], v[0:1], off glc
; GFX1100-SDAG-NEXT: s_waitcnt vmcnt(0)
@@ -10805,7 +10641,6 @@ define <4 x float> @global_load_saddr_i8_offset_0xFFFFFFFF(ptr addrspace(1) inre
; GFX1250-SDAG-NEXT: s_wait_loadcnt_dscnt 0x0
; GFX1250-SDAG-NEXT: s_wait_kmcnt 0x0
; GFX1250-SDAG-NEXT: v_add_co_u32 v0, s0, 0xff800000, s0
-; GFX1250-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1250-SDAG-NEXT: v_add_co_ci_u32_e64 v1, null, 0, s1, s0
; GFX1250-SDAG-NEXT: global_load_b128 v[0:3], v[0:1], off offset:8388607 scope:SCOPE_DEV
; GFX1250-SDAG-NEXT: s_wait_loadcnt 0x0
@@ -10819,7 +10654,6 @@ define <4 x float> @global_load_saddr_i8_offset_0xFFFFFFFF(ptr addrspace(1) inre
; GFX1310-SDAG-NEXT: s_wait_bvhcnt 0x0
; GFX1310-SDAG-NEXT: s_wait_kmcnt 0x0
; GFX1310-SDAG-NEXT: v_add_co_u32 v0, s0, 0xff800000, s0
-; GFX1310-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1310-SDAG-NEXT: v_add_co_ci_u32_e64 v1, null, 0, s1, s0
; GFX1310-SDAG-NEXT: global_load_b128 v[0:3], v[0:1], off offset:8388607 scope:SCOPE_DEV
; GFX1310-SDAG-NEXT: s_wait_loadcnt 0x0
@@ -11153,7 +10987,6 @@ define <4 x float> @global_load_saddr_i8_offset_0x100000001(ptr addrspace(1) inr
; GFX1100-SDAG: ; %bb.0:
; GFX1100-SDAG-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX1100-SDAG-NEXT: v_add_co_u32 v0, s0, 0, s0
-; GFX1100-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1100-SDAG-NEXT: v_add_co_ci_u32_e64 v1, null, 1, s1, s0
; GFX1100-SDAG-NEXT: global_load_b128 v[0:3], v[0:1], off offset:1
; GFX1100-SDAG-NEXT: s_waitcnt vmcnt(0)
@@ -11164,7 +10997,6 @@ define <4 x float> @global_load_saddr_i8_offset_0x100000001(ptr addrspace(1) inr
; GFX1250-SDAG-NEXT: s_wait_loadcnt_dscnt 0x0
; GFX1250-SDAG-NEXT: s_wait_kmcnt 0x0
; GFX1250-SDAG-NEXT: v_add_co_u32 v0, s0, 0, s0
-; GFX1250-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1250-SDAG-NEXT: v_add_co_ci_u32_e64 v1, null, 1, s1, s0
; GFX1250-SDAG-NEXT: global_load_b128 v[0:3], v[0:1], off offset:1
; GFX1250-SDAG-NEXT: s_wait_loadcnt 0x0
@@ -11178,7 +11010,6 @@ define <4 x float> @global_load_saddr_i8_offset_0x100000001(ptr addrspace(1) inr
; GFX1310-SDAG-NEXT: s_wait_bvhcnt 0x0
; GFX1310-SDAG-NEXT: s_wait_kmcnt 0x0
; GFX1310-SDAG-NEXT: v_add_co_u32 v0, s0, 0, s0
-; GFX1310-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1310-SDAG-NEXT: v_add_co_ci_u32_e64 v1, null, 1, s1, s0
; GFX1310-SDAG-NEXT: global_load_b128 v[0:3], v[0:1], off offset:1
; GFX1310-SDAG-NEXT: s_wait_loadcnt 0x0
@@ -11340,7 +11171,6 @@ define <4 x float> @global_load_saddr_i8_offset_0x100000FFF(ptr addrspace(1) inr
; GFX1100-SDAG: ; %bb.0:
; GFX1100-SDAG-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX1100-SDAG-NEXT: v_add_co_u32 v0, s0, 0, s0
-; GFX1100-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1100-SDAG-NEXT: v_add_co_ci_u32_e64 v1, null, 1, s1, s0
; GFX1100-SDAG-NEXT: global_load_b128 v[0:3], v[0:1], off offset:4095 glc
; GFX1100-SDAG-NEXT: s_waitcnt vmcnt(0)
@@ -11351,7 +11181,6 @@ define <4 x float> @global_load_saddr_i8_offset_0x100000FFF(ptr addrspace(1) inr
; GFX1250-SDAG-NEXT: s_wait_loadcnt_dscnt 0x0
; GFX1250-SDAG-NEXT: s_wait_kmcnt 0x0
; GFX1250-SDAG-NEXT: v_add_co_u32 v0, s0, 0, s0
-; GFX1250-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1250-SDAG-NEXT: v_add_co_ci_u32_e64 v1, null, 1, s1, s0
; GFX1250-SDAG-NEXT: global_load_b128 v[0:3], v[0:1], off offset:4095
; GFX1250-SDAG-NEXT: s_wait_loadcnt 0x0
@@ -11365,7 +11194,6 @@ define <4 x float> @global_load_saddr_i8_offset_0x100000FFF(ptr addrspace(1) inr
; GFX1310-SDAG-NEXT: s_wait_bvhcnt 0x0
; GFX1310-SDAG-NEXT: s_wait_kmcnt 0x0
; GFX1310-SDAG-NEXT: v_add_co_u32 v0, s0, 0, s0
-; GFX1310-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1310-SDAG-NEXT: v_add_co_ci_u32_e64 v1, null, 1, s1, s0
; GFX1310-SDAG-NEXT: global_load_b128 v[0:3], v[0:1], off offset:4095 scope:SCOPE_SE
; GFX1310-SDAG-NEXT: s_wait_loadcnt 0x0
@@ -11526,7 +11354,6 @@ define <4 x float> @global_load_saddr_i8_offset_0x100001000(ptr addrspace(1) inr
; GFX1100-SDAG: ; %bb.0:
; GFX1100-SDAG-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX1100-SDAG-NEXT: v_add_co_u32 v0, s0, 0x1000, s0
-; GFX1100-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1100-SDAG-NEXT: v_add_co_ci_u32_e64 v1, null, 1, s1, s0
; GFX1100-SDAG-NEXT: global_load_b128 v[0:3], v[0:1], off glc
; GFX1100-SDAG-NEXT: s_waitcnt vmcnt(0)
@@ -11537,7 +11364,6 @@ define <4 x float> @global_load_saddr_i8_offset_0x100001000(ptr addrspace(1) inr
; GFX1250-SDAG-NEXT: s_wait_loadcnt_dscnt 0x0
; GFX1250-SDAG-NEXT: s_wait_kmcnt 0x0
; GFX1250-SDAG-NEXT: v_add_co_u32 v0, s0, 0, s0
-; GFX1250-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1250-SDAG-NEXT: v_add_co_ci_u32_e64 v1, null, 1, s1, s0
; GFX1250-SDAG-NEXT: global_load_b128 v[0:3], v[0:1], off offset:4096 scope:SCOPE_DEV
; GFX1250-SDAG-NEXT: s_wait_loadcnt 0x0
@@ -11551,7 +11377,6 @@ define <4 x float> @global_load_saddr_i8_offset_0x100001000(ptr addrspace(1) inr
; GFX1310-SDAG-NEXT: s_wait_bvhcnt 0x0
; GFX1310-SDAG-NEXT: s_wait_kmcnt 0x0
; GFX1310-SDAG-NEXT: v_add_co_u32 v0, s0, 0, s0
-; GFX1310-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1310-SDAG-NEXT: v_add_co_ci_u32_e64 v1, null, 1, s1, s0
; GFX1310-SDAG-NEXT: global_load_b128 v[0:3], v[0:1], off offset:4096 scope:SCOPE_DEV
; GFX1310-SDAG-NEXT: s_wait_loadcnt 0x0
@@ -11715,7 +11540,6 @@ define <4 x float> @global_load_saddr_i8_offset_neg0xFFFFFFFF(ptr addrspace(1) i
; GFX1100-SDAG: ; %bb.0:
; GFX1100-SDAG-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX1100-SDAG-NEXT: v_add_co_u32 v0, s0, 0x1000, s0
-; GFX1100-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1100-SDAG-NEXT: v_add_co_ci_u32_e64 v1, null, -1, s1, s0
; GFX1100-SDAG-NEXT: global_load_b128 v[0:3], v[0:1], off offset:-4095 glc
; GFX1100-SDAG-NEXT: s_waitcnt vmcnt(0)
@@ -11726,7 +11550,6 @@ define <4 x float> @global_load_saddr_i8_offset_neg0xFFFFFFFF(ptr addrspace(1) i
; GFX1250-SDAG-NEXT: s_wait_loadcnt_dscnt 0x0
; GFX1250-SDAG-NEXT: s_wait_kmcnt 0x0
; GFX1250-SDAG-NEXT: v_add_co_u32 v0, s0, 0x800000, s0
-; GFX1250-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1250-SDAG-NEXT: v_add_co_ci_u32_e64 v1, null, -1, s1, s0
; GFX1250-SDAG-NEXT: global_load_b128 v[0:3], v[0:1], off offset:-8388607 scope:SCOPE_SYS
; GFX1250-SDAG-NEXT: s_wait_loadcnt 0x0
@@ -11740,7 +11563,6 @@ define <4 x float> @global_load_saddr_i8_offset_neg0xFFFFFFFF(ptr addrspace(1) i
; GFX1310-SDAG-NEXT: s_wait_bvhcnt 0x0
; GFX1310-SDAG-NEXT: s_wait_kmcnt 0x0
; GFX1310-SDAG-NEXT: v_add_co_u32 v0, s0, 0x800000, s0
-; GFX1310-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1310-SDAG-NEXT: v_add_co_ci_u32_e64 v1, null, -1, s1, s0
; GFX1310-SDAG-NEXT: global_load_b128 v[0:3], v[0:1], off offset:-8388607 scope:SCOPE_SYS
; GFX1310-SDAG-NEXT: s_wait_loadcnt 0x0
@@ -12085,7 +11907,6 @@ define <4 x float> @global_load_saddr_i8_offset_neg0x100000001(ptr addrspace(1)
; GFX1100-SDAG: ; %bb.0:
; GFX1100-SDAG-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX1100-SDAG-NEXT: v_add_co_u32 v0, s0, 0, s0
-; GFX1100-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1100-SDAG-NEXT: v_add_co_ci_u32_e64 v1, null, -1, s1, s0
; GFX1100-SDAG-NEXT: global_load_b128 v[0:3], v[0:1], off offset:-1 glc
; GFX1100-SDAG-NEXT: s_waitcnt vmcnt(0)
@@ -12096,7 +11917,6 @@ define <4 x float> @global_load_saddr_i8_offset_neg0x100000001(ptr addrspace(1)
; GFX1250-SDAG-NEXT: s_wait_loadcnt_dscnt 0x0
; GFX1250-SDAG-NEXT: s_wait_kmcnt 0x0
; GFX1250-SDAG-NEXT: v_add_co_u32 v0, s0, 0, s0
-; GFX1250-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1250-SDAG-NEXT: v_add_co_ci_u32_e64 v1, null, -1, s1, s0
; GFX1250-SDAG-NEXT: global_load_b128 v[0:3], v[0:1], off offset:-1
; GFX1250-SDAG-NEXT: s_wait_loadcnt 0x0
@@ -12110,7 +11930,6 @@ define <4 x float> @global_load_saddr_i8_offset_neg0x100000001(ptr addrspace(1)
; GFX1310-SDAG-NEXT: s_wait_bvhcnt 0x0
; GFX1310-SDAG-NEXT: s_wait_kmcnt 0x0
; GFX1310-SDAG-NEXT: v_add_co_u32 v0, s0, 0, s0
-; GFX1310-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1310-SDAG-NEXT: v_add_co_ci_u32_e64 v1, null, -1, s1, s0
; GFX1310-SDAG-NEXT: global_load_b128 v[0:3], v[0:1], off offset:-1 scope:SCOPE_SE
; GFX1310-SDAG-NEXT: s_wait_loadcnt 0x0
@@ -12290,7 +12109,6 @@ define <4 x float> @global_load_saddr_i8_zext_vgpr(ptr addrspace(1) inreg %sbase
; GFX1310-SDAG-NEXT: s_wait_bvhcnt 0x0
; GFX1310-SDAG-NEXT: s_wait_kmcnt 0x0
; GFX1310-SDAG-NEXT: v_add_co_u32 v0, s0, s0, v0
-; GFX1310-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1310-SDAG-NEXT: v_add_co_ci_u32_e64 v1, null, s1, 0, s0
; GFX1310-SDAG-NEXT: global_load_b128 v[0:3], v[0:1], off scope:SCOPE_DEV
; GFX1310-SDAG-NEXT: s_wait_loadcnt 0x0
@@ -12329,7 +12147,7 @@ define <4 x float> @global_load_saddr_i8_zext_vgpr(ptr addrspace(1) inreg %sbase
; GFX1250-ISEL-NEXT: s_wait_loadcnt_dscnt 0x0
; GFX1250-ISEL-NEXT: s_wait_kmcnt 0x0
; GFX1250-ISEL-NEXT: v_mov_b64_e32 v[2:3], s[0:1]
-; GFX1250-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-ISEL-NEXT: v_add_co_u32 v0, vcc_lo, v2, v0
; GFX1250-ISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v3, vcc_lo
; GFX1250-ISEL-NEXT: global_load_b128 v[0:3], v[0:1], off scope:SCOPE_DEV
@@ -12344,7 +12162,7 @@ define <4 x float> @global_load_saddr_i8_zext_vgpr(ptr addrspace(1) inreg %sbase
; GFX1310-ISEL-NEXT: s_wait_bvhcnt 0x0
; GFX1310-ISEL-NEXT: s_wait_kmcnt 0x0
; GFX1310-ISEL-NEXT: v_dual_mov_b32 v2, s1 :: v_dual_mov_b32 v1, s0
-; GFX1310-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1310-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1310-ISEL-NEXT: v_add_co_u32 v0, vcc_lo, v1, v0
; GFX1310-ISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v2, vcc_lo
; GFX1310-ISEL-NEXT: global_load_b128 v[0:3], v[0:1], off scope:SCOPE_DEV
@@ -12410,7 +12228,6 @@ define <4 x float> @global_load_saddr_i8_zext_vgpr_offset_4095(ptr addrspace(1)
; GFX1310-SDAG-NEXT: s_wait_bvhcnt 0x0
; GFX1310-SDAG-NEXT: s_wait_kmcnt 0x0
; GFX1310-SDAG-NEXT: v_add_co_u32 v0, s0, s0, v0
-; GFX1310-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1310-SDAG-NEXT: v_add_co_ci_u32_e64 v1, null, s1, 0, s0
; GFX1310-SDAG-NEXT: global_load_b128 v[0:3], v[0:1], off offset:4095 scope:SCOPE_SYS
; GFX1310-SDAG-NEXT: s_wait_loadcnt 0x0
@@ -12455,7 +12272,7 @@ define <4 x float> @global_load_saddr_i8_zext_vgpr_offset_4095(ptr addrspace(1)
; GFX1250-ISEL-NEXT: s_wait_loadcnt_dscnt 0x0
; GFX1250-ISEL-NEXT: s_wait_kmcnt 0x0
; GFX1250-ISEL-NEXT: v_mov_b64_e32 v[2:3], s[0:1]
-; GFX1250-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-ISEL-NEXT: v_add_co_u32 v0, vcc_lo, v2, v0
; GFX1250-ISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v3, vcc_lo
; GFX1250-ISEL-NEXT: global_load_b128 v[0:3], v[0:1], off offset:4095 scope:SCOPE_SYS
@@ -12470,7 +12287,7 @@ define <4 x float> @global_load_saddr_i8_zext_vgpr_offset_4095(ptr addrspace(1)
; GFX1310-ISEL-NEXT: s_wait_bvhcnt 0x0
; GFX1310-ISEL-NEXT: s_wait_kmcnt 0x0
; GFX1310-ISEL-NEXT: v_dual_mov_b32 v2, s1 :: v_dual_mov_b32 v1, s0
-; GFX1310-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1310-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1310-ISEL-NEXT: v_add_co_u32 v0, vcc_lo, v1, v0
; GFX1310-ISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v2, vcc_lo
; GFX1310-ISEL-NEXT: global_load_b128 v[0:3], v[0:1], off offset:4095 scope:SCOPE_SYS
@@ -12525,10 +12342,9 @@ define <4 x float> @global_load_saddr_i8_zext_vgpr_offset_4096(ptr addrspace(1)
; GFX1100-SDAG: ; %bb.0:
; GFX1100-SDAG-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX1100-SDAG-NEXT: v_add_co_u32 v0, s0, s0, v0
-; GFX1100-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1100-SDAG-NEXT: v_add_co_ci_u32_e64 v1, null, s1, 0, s0
+; GFX1100-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1100-SDAG-NEXT: v_add_co_u32 v0, vcc_lo, 0x1000, v0
-; GFX1100-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1100-SDAG-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1100-SDAG-NEXT: global_load_b128 v[0:3], v[0:1], off
; GFX1100-SDAG-NEXT: s_waitcnt vmcnt(0)
@@ -12553,7 +12369,6 @@ define <4 x float> @global_load_saddr_i8_zext_vgpr_offset_4096(ptr addrspace(1)
; GFX1310-SDAG-NEXT: s_wait_bvhcnt 0x0
; GFX1310-SDAG-NEXT: s_wait_kmcnt 0x0
; GFX1310-SDAG-NEXT: v_add_co_u32 v0, s0, s0, v0
-; GFX1310-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1310-SDAG-NEXT: v_add_co_ci_u32_e64 v1, null, s1, 0, s0
; GFX1310-SDAG-NEXT: global_load_b128 v[0:3], v[0:1], off offset:4096
; GFX1310-SDAG-NEXT: s_wait_loadcnt 0x0
@@ -12603,10 +12418,10 @@ define <4 x float> @global_load_saddr_i8_zext_vgpr_offset_4096(ptr addrspace(1)
; GFX1100-ISEL: ; %bb.0:
; GFX1100-ISEL-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX1100-ISEL-NEXT: v_dual_mov_b32 v2, s1 :: v_dual_mov_b32 v1, s0
-; GFX1100-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1100-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1100-ISEL-NEXT: v_add_co_u32 v0, vcc_lo, v1, v0
; GFX1100-ISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v2, vcc_lo
-; GFX1100-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1100-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1100-ISEL-NEXT: v_add_co_u32 v0, vcc_lo, 0x1000, v0
; GFX1100-ISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1100-ISEL-NEXT: global_load_b128 v[0:3], v[0:1], off
@@ -12618,7 +12433,7 @@ define <4 x float> @global_load_saddr_i8_zext_vgpr_offset_4096(ptr addrspace(1)
; GFX1250-ISEL-NEXT: s_wait_loadcnt_dscnt 0x0
; GFX1250-ISEL-NEXT: s_wait_kmcnt 0x0
; GFX1250-ISEL-NEXT: v_mov_b64_e32 v[2:3], s[0:1]
-; GFX1250-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-ISEL-NEXT: v_add_co_u32 v0, vcc_lo, v2, v0
; GFX1250-ISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v3, vcc_lo
; GFX1250-ISEL-NEXT: global_load_b128 v[0:3], v[0:1], off offset:4096
@@ -12633,7 +12448,7 @@ define <4 x float> @global_load_saddr_i8_zext_vgpr_offset_4096(ptr addrspace(1)
; GFX1310-ISEL-NEXT: s_wait_bvhcnt 0x0
; GFX1310-ISEL-NEXT: s_wait_kmcnt 0x0
; GFX1310-ISEL-NEXT: v_dual_mov_b32 v2, s1 :: v_dual_mov_b32 v1, s0
-; GFX1310-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1310-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1310-ISEL-NEXT: v_add_co_u32 v0, vcc_lo, v1, v0
; GFX1310-ISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v2, vcc_lo
; GFX1310-ISEL-NEXT: global_load_b128 v[0:3], v[0:1], off offset:4096
@@ -12700,7 +12515,6 @@ define <4 x float> @global_load_saddr_i8_zext_vgpr_offset_neg4096(ptr addrspace(
; GFX1310-SDAG-NEXT: s_wait_bvhcnt 0x0
; GFX1310-SDAG-NEXT: s_wait_kmcnt 0x0
; GFX1310-SDAG-NEXT: v_add_co_u32 v0, s0, s0, v0
-; GFX1310-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1310-SDAG-NEXT: v_add_co_ci_u32_e64 v1, null, s1, 0, s0
; GFX1310-SDAG-NEXT: global_load_b128 v[0:3], v[0:1], off offset:-4096 scope:SCOPE_SE
; GFX1310-SDAG-NEXT: s_wait_loadcnt 0x0
@@ -12745,7 +12559,7 @@ define <4 x float> @global_load_saddr_i8_zext_vgpr_offset_neg4096(ptr addrspace(
; GFX1250-ISEL-NEXT: s_wait_loadcnt_dscnt 0x0
; GFX1250-ISEL-NEXT: s_wait_kmcnt 0x0
; GFX1250-ISEL-NEXT: v_mov_b64_e32 v[2:3], s[0:1]
-; GFX1250-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-ISEL-NEXT: v_add_co_u32 v0, vcc_lo, v2, v0
; GFX1250-ISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v3, vcc_lo
; GFX1250-ISEL-NEXT: global_load_b128 v[0:3], v[0:1], off offset:-4096
@@ -12760,7 +12574,7 @@ define <4 x float> @global_load_saddr_i8_zext_vgpr_offset_neg4096(ptr addrspace(
; GFX1310-ISEL-NEXT: s_wait_bvhcnt 0x0
; GFX1310-ISEL-NEXT: s_wait_kmcnt 0x0
; GFX1310-ISEL-NEXT: v_dual_mov_b32 v2, s1 :: v_dual_mov_b32 v1, s0
-; GFX1310-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1310-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1310-ISEL-NEXT: v_add_co_u32 v0, vcc_lo, v1, v0
; GFX1310-ISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v2, vcc_lo
; GFX1310-ISEL-NEXT: global_load_b128 v[0:3], v[0:1], off offset:-4096 scope:SCOPE_SE
@@ -12815,10 +12629,9 @@ define <4 x float> @global_load_saddr_i8_zext_vgpr_offset_neg4097(ptr addrspace(
; GFX1100-SDAG: ; %bb.0:
; GFX1100-SDAG-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX1100-SDAG-NEXT: v_add_co_u32 v0, s0, s0, v0
-; GFX1100-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1100-SDAG-NEXT: v_add_co_ci_u32_e64 v1, null, s1, 0, s0
+; GFX1100-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1100-SDAG-NEXT: v_add_co_u32 v0, vcc_lo, 0xfffff000, v0
-; GFX1100-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1100-SDAG-NEXT: v_add_co_ci_u32_e64 v1, null, -1, v1, vcc_lo
; GFX1100-SDAG-NEXT: global_load_b128 v[0:3], v[0:1], off offset:-1 glc
; GFX1100-SDAG-NEXT: s_waitcnt vmcnt(0)
@@ -12843,7 +12656,6 @@ define <4 x float> @global_load_saddr_i8_zext_vgpr_offset_neg4097(ptr addrspace(
; GFX1310-SDAG-NEXT: s_wait_bvhcnt 0x0
; GFX1310-SDAG-NEXT: s_wait_kmcnt 0x0
; GFX1310-SDAG-NEXT: v_add_co_u32 v0, s0, s0, v0
-; GFX1310-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1310-SDAG-NEXT: v_add_co_ci_u32_e64 v1, null, s1, 0, s0
; GFX1310-SDAG-NEXT: global_load_b128 v[0:3], v[0:1], off offset:-4097 scope:SCOPE_DEV
; GFX1310-SDAG-NEXT: s_wait_loadcnt 0x0
@@ -12893,10 +12705,10 @@ define <4 x float> @global_load_saddr_i8_zext_vgpr_offset_neg4097(ptr addrspace(
; GFX1100-ISEL: ; %bb.0:
; GFX1100-ISEL-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX1100-ISEL-NEXT: v_dual_mov_b32 v2, s1 :: v_dual_mov_b32 v1, s0
-; GFX1100-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1100-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1100-ISEL-NEXT: v_add_co_u32 v0, vcc_lo, v1, v0
; GFX1100-ISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v2, vcc_lo
-; GFX1100-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1100-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1100-ISEL-NEXT: v_add_co_u32 v0, vcc_lo, 0xffffefff, v0
; GFX1100-ISEL-NEXT: v_add_co_ci_u32_e64 v1, null, -1, v1, vcc_lo
; GFX1100-ISEL-NEXT: global_load_b128 v[0:3], v[0:1], off glc
@@ -12908,7 +12720,7 @@ define <4 x float> @global_load_saddr_i8_zext_vgpr_offset_neg4097(ptr addrspace(
; GFX1250-ISEL-NEXT: s_wait_loadcnt_dscnt 0x0
; GFX1250-ISEL-NEXT: s_wait_kmcnt 0x0
; GFX1250-ISEL-NEXT: v_mov_b64_e32 v[2:3], s[0:1]
-; GFX1250-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-ISEL-NEXT: v_add_co_u32 v0, vcc_lo, v2, v0
; GFX1250-ISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v3, vcc_lo
; GFX1250-ISEL-NEXT: global_load_b128 v[0:3], v[0:1], off offset:-4097 scope:SCOPE_DEV
@@ -12923,7 +12735,7 @@ define <4 x float> @global_load_saddr_i8_zext_vgpr_offset_neg4097(ptr addrspace(
; GFX1310-ISEL-NEXT: s_wait_bvhcnt 0x0
; GFX1310-ISEL-NEXT: s_wait_kmcnt 0x0
; GFX1310-ISEL-NEXT: v_dual_mov_b32 v2, s1 :: v_dual_mov_b32 v1, s0
-; GFX1310-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1310-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1310-ISEL-NEXT: v_add_co_u32 v0, vcc_lo, v1, v0
; GFX1310-ISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v2, vcc_lo
; GFX1310-ISEL-NEXT: global_load_b128 v[0:3], v[0:1], off offset:-4097 scope:SCOPE_DEV
@@ -12986,7 +12798,6 @@ define <4 x float> @global_load_saddr_i8_zext_vgpr_offset_2047(ptr addrspace(1)
; GFX1310-SDAG-NEXT: s_wait_bvhcnt 0x0
; GFX1310-SDAG-NEXT: s_wait_kmcnt 0x0
; GFX1310-SDAG-NEXT: v_add_co_u32 v0, s0, s0, v0
-; GFX1310-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1310-SDAG-NEXT: v_add_co_ci_u32_e64 v1, null, s1, 0, s0
; GFX1310-SDAG-NEXT: global_load_b128 v[0:3], v[0:1], off offset:2047 scope:SCOPE_SYS
; GFX1310-SDAG-NEXT: s_wait_loadcnt 0x0
@@ -13025,7 +12836,7 @@ define <4 x float> @global_load_saddr_i8_zext_vgpr_offset_2047(ptr addrspace(1)
; GFX1250-ISEL-NEXT: s_wait_loadcnt_dscnt 0x0
; GFX1250-ISEL-NEXT: s_wait_kmcnt 0x0
; GFX1250-ISEL-NEXT: v_mov_b64_e32 v[2:3], s[0:1]
-; GFX1250-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-ISEL-NEXT: v_add_co_u32 v0, vcc_lo, v2, v0
; GFX1250-ISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v3, vcc_lo
; GFX1250-ISEL-NEXT: global_load_b128 v[0:3], v[0:1], off offset:2047 scope:SCOPE_SYS
@@ -13040,7 +12851,7 @@ define <4 x float> @global_load_saddr_i8_zext_vgpr_offset_2047(ptr addrspace(1)
; GFX1310-ISEL-NEXT: s_wait_bvhcnt 0x0
; GFX1310-ISEL-NEXT: s_wait_kmcnt 0x0
; GFX1310-ISEL-NEXT: v_dual_mov_b32 v2, s1 :: v_dual_mov_b32 v1, s0
-; GFX1310-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1310-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1310-ISEL-NEXT: v_add_co_u32 v0, vcc_lo, v1, v0
; GFX1310-ISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v2, vcc_lo
; GFX1310-ISEL-NEXT: global_load_b128 v[0:3], v[0:1], off offset:2047 scope:SCOPE_SYS
@@ -13107,7 +12918,6 @@ define <4 x float> @global_load_saddr_i8_zext_vgpr_offset_2048(ptr addrspace(1)
; GFX1310-SDAG-NEXT: s_wait_bvhcnt 0x0
; GFX1310-SDAG-NEXT: s_wait_kmcnt 0x0
; GFX1310-SDAG-NEXT: v_add_co_u32 v0, s0, s0, v0
-; GFX1310-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1310-SDAG-NEXT: v_add_co_ci_u32_e64 v1, null, s1, 0, s0
; GFX1310-SDAG-NEXT: global_load_b128 v[0:3], v[0:1], off offset:2048
; GFX1310-SDAG-NEXT: s_wait_loadcnt 0x0
@@ -13152,7 +12962,7 @@ define <4 x float> @global_load_saddr_i8_zext_vgpr_offset_2048(ptr addrspace(1)
; GFX1250-ISEL-NEXT: s_wait_loadcnt_dscnt 0x0
; GFX1250-ISEL-NEXT: s_wait_kmcnt 0x0
; GFX1250-ISEL-NEXT: v_mov_b64_e32 v[2:3], s[0:1]
-; GFX1250-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-ISEL-NEXT: v_add_co_u32 v0, vcc_lo, v2, v0
; GFX1250-ISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v3, vcc_lo
; GFX1250-ISEL-NEXT: global_load_b128 v[0:3], v[0:1], off offset:2048
@@ -13167,7 +12977,7 @@ define <4 x float> @global_load_saddr_i8_zext_vgpr_offset_2048(ptr addrspace(1)
; GFX1310-ISEL-NEXT: s_wait_bvhcnt 0x0
; GFX1310-ISEL-NEXT: s_wait_kmcnt 0x0
; GFX1310-ISEL-NEXT: v_dual_mov_b32 v2, s1 :: v_dual_mov_b32 v1, s0
-; GFX1310-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1310-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1310-ISEL-NEXT: v_add_co_u32 v0, vcc_lo, v1, v0
; GFX1310-ISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v2, vcc_lo
; GFX1310-ISEL-NEXT: global_load_b128 v[0:3], v[0:1], off offset:2048
@@ -13230,7 +13040,6 @@ define <4 x float> @global_load_saddr_i8_zext_vgpr_offset_neg2048(ptr addrspace(
; GFX1310-SDAG-NEXT: s_wait_bvhcnt 0x0
; GFX1310-SDAG-NEXT: s_wait_kmcnt 0x0
; GFX1310-SDAG-NEXT: v_add_co_u32 v0, s0, s0, v0
-; GFX1310-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1310-SDAG-NEXT: v_add_co_ci_u32_e64 v1, null, s1, 0, s0
; GFX1310-SDAG-NEXT: global_load_b128 v[0:3], v[0:1], off offset:-2048 scope:SCOPE_SE
; GFX1310-SDAG-NEXT: s_wait_loadcnt 0x0
@@ -13269,7 +13078,7 @@ define <4 x float> @global_load_saddr_i8_zext_vgpr_offset_neg2048(ptr addrspace(
; GFX1250-ISEL-NEXT: s_wait_loadcnt_dscnt 0x0
; GFX1250-ISEL-NEXT: s_wait_kmcnt 0x0
; GFX1250-ISEL-NEXT: v_mov_b64_e32 v[2:3], s[0:1]
-; GFX1250-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-ISEL-NEXT: v_add_co_u32 v0, vcc_lo, v2, v0
; GFX1250-ISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v3, vcc_lo
; GFX1250-ISEL-NEXT: global_load_b128 v[0:3], v[0:1], off offset:-2048
@@ -13284,7 +13093,7 @@ define <4 x float> @global_load_saddr_i8_zext_vgpr_offset_neg2048(ptr addrspace(
; GFX1310-ISEL-NEXT: s_wait_bvhcnt 0x0
; GFX1310-ISEL-NEXT: s_wait_kmcnt 0x0
; GFX1310-ISEL-NEXT: v_dual_mov_b32 v2, s1 :: v_dual_mov_b32 v1, s0
-; GFX1310-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1310-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1310-ISEL-NEXT: v_add_co_u32 v0, vcc_lo, v1, v0
; GFX1310-ISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v2, vcc_lo
; GFX1310-ISEL-NEXT: global_load_b128 v[0:3], v[0:1], off offset:-2048 scope:SCOPE_SE
@@ -13351,7 +13160,6 @@ define <4 x float> @global_load_saddr_i8_zext_vgpr_offset_neg2049(ptr addrspace(
; GFX1310-SDAG-NEXT: s_wait_bvhcnt 0x0
; GFX1310-SDAG-NEXT: s_wait_kmcnt 0x0
; GFX1310-SDAG-NEXT: v_add_co_u32 v0, s0, s0, v0
-; GFX1310-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1310-SDAG-NEXT: v_add_co_ci_u32_e64 v1, null, s1, 0, s0
; GFX1310-SDAG-NEXT: global_load_b128 v[0:3], v[0:1], off offset:-2049 scope:SCOPE_DEV
; GFX1310-SDAG-NEXT: s_wait_loadcnt 0x0
@@ -13396,7 +13204,7 @@ define <4 x float> @global_load_saddr_i8_zext_vgpr_offset_neg2049(ptr addrspace(
; GFX1250-ISEL-NEXT: s_wait_loadcnt_dscnt 0x0
; GFX1250-ISEL-NEXT: s_wait_kmcnt 0x0
; GFX1250-ISEL-NEXT: v_mov_b64_e32 v[2:3], s[0:1]
-; GFX1250-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-ISEL-NEXT: v_add_co_u32 v0, vcc_lo, v2, v0
; GFX1250-ISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v3, vcc_lo
; GFX1250-ISEL-NEXT: global_load_b128 v[0:3], v[0:1], off offset:-2049 scope:SCOPE_DEV
@@ -13411,7 +13219,7 @@ define <4 x float> @global_load_saddr_i8_zext_vgpr_offset_neg2049(ptr addrspace(
; GFX1310-ISEL-NEXT: s_wait_bvhcnt 0x0
; GFX1310-ISEL-NEXT: s_wait_kmcnt 0x0
; GFX1310-ISEL-NEXT: v_dual_mov_b32 v2, s1 :: v_dual_mov_b32 v1, s0
-; GFX1310-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1310-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1310-ISEL-NEXT: v_add_co_u32 v0, vcc_lo, v1, v0
; GFX1310-ISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v2, vcc_lo
; GFX1310-ISEL-NEXT: global_load_b128 v[0:3], v[0:1], off offset:-2049 scope:SCOPE_DEV
@@ -13466,10 +13274,9 @@ define <4 x float> @global_load_saddr_i8_zext_vgpr_offset_0x7FFFFF(ptr addrspace
; GFX1100-SDAG: ; %bb.0:
; GFX1100-SDAG-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX1100-SDAG-NEXT: v_add_co_u32 v0, s0, s0, v0
-; GFX1100-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1100-SDAG-NEXT: v_add_co_ci_u32_e64 v1, null, s1, 0, s0
+; GFX1100-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1100-SDAG-NEXT: v_add_co_u32 v0, vcc_lo, 0x7ff000, v0
-; GFX1100-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1100-SDAG-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1100-SDAG-NEXT: global_load_b128 v[0:3], v[0:1], off offset:4095 glc
; GFX1100-SDAG-NEXT: s_waitcnt vmcnt(0)
@@ -13494,7 +13301,6 @@ define <4 x float> @global_load_saddr_i8_zext_vgpr_offset_0x7FFFFF(ptr addrspace
; GFX1310-SDAG-NEXT: s_wait_bvhcnt 0x0
; GFX1310-SDAG-NEXT: s_wait_kmcnt 0x0
; GFX1310-SDAG-NEXT: v_add_co_u32 v0, s0, s0, v0
-; GFX1310-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1310-SDAG-NEXT: v_add_co_ci_u32_e64 v1, null, s1, 0, s0
; GFX1310-SDAG-NEXT: global_load_b128 v[0:3], v[0:1], off offset:8388607 scope:SCOPE_SYS
; GFX1310-SDAG-NEXT: s_wait_loadcnt 0x0
@@ -13544,10 +13350,10 @@ define <4 x float> @global_load_saddr_i8_zext_vgpr_offset_0x7FFFFF(ptr addrspace
; GFX1100-ISEL: ; %bb.0:
; GFX1100-ISEL-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX1100-ISEL-NEXT: v_dual_mov_b32 v2, s1 :: v_dual_mov_b32 v1, s0
-; GFX1100-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1100-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1100-ISEL-NEXT: v_add_co_u32 v0, vcc_lo, v1, v0
; GFX1100-ISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v2, vcc_lo
-; GFX1100-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1100-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1100-ISEL-NEXT: v_add_co_u32 v0, vcc_lo, 0x7fffff, v0
; GFX1100-ISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1100-ISEL-NEXT: global_load_b128 v[0:3], v[0:1], off glc
@@ -13559,7 +13365,7 @@ define <4 x float> @global_load_saddr_i8_zext_vgpr_offset_0x7FFFFF(ptr addrspace
; GFX1250-ISEL-NEXT: s_wait_loadcnt_dscnt 0x0
; GFX1250-ISEL-NEXT: s_wait_kmcnt 0x0
; GFX1250-ISEL-NEXT: v_mov_b64_e32 v[2:3], s[0:1]
-; GFX1250-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-ISEL-NEXT: v_add_co_u32 v0, vcc_lo, v2, v0
; GFX1250-ISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v3, vcc_lo
; GFX1250-ISEL-NEXT: global_load_b128 v[0:3], v[0:1], off offset:8388607 scope:SCOPE_SYS
@@ -13574,7 +13380,7 @@ define <4 x float> @global_load_saddr_i8_zext_vgpr_offset_0x7FFFFF(ptr addrspace
; GFX1310-ISEL-NEXT: s_wait_bvhcnt 0x0
; GFX1310-ISEL-NEXT: s_wait_kmcnt 0x0
; GFX1310-ISEL-NEXT: v_dual_mov_b32 v2, s1 :: v_dual_mov_b32 v1, s0
-; GFX1310-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1310-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1310-ISEL-NEXT: v_add_co_u32 v0, vcc_lo, v1, v0
; GFX1310-ISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v2, vcc_lo
; GFX1310-ISEL-NEXT: global_load_b128 v[0:3], v[0:1], off offset:8388607 scope:SCOPE_SYS
@@ -13628,10 +13434,9 @@ define <4 x float> @global_load_saddr_i8_zext_vgpr_offset_0xFFFFFF(ptr addrspace
; GFX1100-SDAG: ; %bb.0:
; GFX1100-SDAG-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX1100-SDAG-NEXT: v_add_co_u32 v0, s0, s0, v0
-; GFX1100-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1100-SDAG-NEXT: v_add_co_ci_u32_e64 v1, null, s1, 0, s0
+; GFX1100-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1100-SDAG-NEXT: v_add_co_u32 v0, vcc_lo, 0xff800000, v0
-; GFX1100-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1100-SDAG-NEXT: v_add_co_ci_u32_e64 v1, null, -1, v1, vcc_lo
; GFX1100-SDAG-NEXT: global_load_b128 v[0:3], v[0:1], off
; GFX1100-SDAG-NEXT: s_waitcnt vmcnt(0)
@@ -13656,7 +13461,6 @@ define <4 x float> @global_load_saddr_i8_zext_vgpr_offset_0xFFFFFF(ptr addrspace
; GFX1310-SDAG-NEXT: s_wait_bvhcnt 0x0
; GFX1310-SDAG-NEXT: s_wait_kmcnt 0x0
; GFX1310-SDAG-NEXT: v_add_co_u32 v0, s0, s0, v0
-; GFX1310-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1310-SDAG-NEXT: v_add_co_ci_u32_e64 v1, null, s1, 0, s0
; GFX1310-SDAG-NEXT: global_load_b128 v[0:3], v[0:1], off offset:-8388608
; GFX1310-SDAG-NEXT: s_wait_loadcnt 0x0
@@ -13706,10 +13510,10 @@ define <4 x float> @global_load_saddr_i8_zext_vgpr_offset_0xFFFFFF(ptr addrspace
; GFX1100-ISEL: ; %bb.0:
; GFX1100-ISEL-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX1100-ISEL-NEXT: v_dual_mov_b32 v2, s1 :: v_dual_mov_b32 v1, s0
-; GFX1100-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1100-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1100-ISEL-NEXT: v_add_co_u32 v0, vcc_lo, v1, v0
; GFX1100-ISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v2, vcc_lo
-; GFX1100-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1100-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1100-ISEL-NEXT: v_add_co_u32 v0, vcc_lo, 0xff800000, v0
; GFX1100-ISEL-NEXT: v_add_co_ci_u32_e64 v1, null, -1, v1, vcc_lo
; GFX1100-ISEL-NEXT: global_load_b128 v[0:3], v[0:1], off
@@ -13721,7 +13525,7 @@ define <4 x float> @global_load_saddr_i8_zext_vgpr_offset_0xFFFFFF(ptr addrspace
; GFX1250-ISEL-NEXT: s_wait_loadcnt_dscnt 0x0
; GFX1250-ISEL-NEXT: s_wait_kmcnt 0x0
; GFX1250-ISEL-NEXT: v_mov_b64_e32 v[2:3], s[0:1]
-; GFX1250-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-ISEL-NEXT: v_add_co_u32 v0, vcc_lo, v2, v0
; GFX1250-ISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v3, vcc_lo
; GFX1250-ISEL-NEXT: global_load_b128 v[0:3], v[0:1], off offset:-8388608
@@ -13736,7 +13540,7 @@ define <4 x float> @global_load_saddr_i8_zext_vgpr_offset_0xFFFFFF(ptr addrspace
; GFX1310-ISEL-NEXT: s_wait_bvhcnt 0x0
; GFX1310-ISEL-NEXT: s_wait_kmcnt 0x0
; GFX1310-ISEL-NEXT: v_dual_mov_b32 v2, s1 :: v_dual_mov_b32 v1, s0
-; GFX1310-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1310-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1310-ISEL-NEXT: v_add_co_u32 v0, vcc_lo, v1, v0
; GFX1310-ISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v2, vcc_lo
; GFX1310-ISEL-NEXT: global_load_b128 v[0:3], v[0:1], off offset:-8388608
@@ -13803,7 +13607,6 @@ define <4 x float> @global_load_saddr_i8_zext_vgpr_offset_4095_gep_order(ptr add
; GFX1310-SDAG-NEXT: s_wait_bvhcnt 0x0
; GFX1310-SDAG-NEXT: s_wait_kmcnt 0x0
; GFX1310-SDAG-NEXT: v_add_co_u32 v0, s0, s0, v0
-; GFX1310-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1310-SDAG-NEXT: v_add_co_ci_u32_e64 v1, null, s1, 0, s0
; GFX1310-SDAG-NEXT: global_load_b128 v[0:3], v[0:1], off offset:4095 scope:SCOPE_SE
; GFX1310-SDAG-NEXT: s_wait_loadcnt 0x0
@@ -13848,7 +13651,7 @@ define <4 x float> @global_load_saddr_i8_zext_vgpr_offset_4095_gep_order(ptr add
; GFX1250-ISEL-NEXT: s_wait_loadcnt_dscnt 0x0
; GFX1250-ISEL-NEXT: s_wait_kmcnt 0x0
; GFX1250-ISEL-NEXT: v_mov_b64_e32 v[2:3], s[0:1]
-; GFX1250-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-ISEL-NEXT: v_add_co_u32 v0, vcc_lo, v2, v0
; GFX1250-ISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v3, vcc_lo
; GFX1250-ISEL-NEXT: global_load_b128 v[0:3], v[0:1], off offset:4095
@@ -13863,7 +13666,7 @@ define <4 x float> @global_load_saddr_i8_zext_vgpr_offset_4095_gep_order(ptr add
; GFX1310-ISEL-NEXT: s_wait_bvhcnt 0x0
; GFX1310-ISEL-NEXT: s_wait_kmcnt 0x0
; GFX1310-ISEL-NEXT: v_dual_mov_b32 v2, s1 :: v_dual_mov_b32 v1, s0
-; GFX1310-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1310-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1310-ISEL-NEXT: v_add_co_u32 v0, vcc_lo, v1, v0
; GFX1310-ISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v2, vcc_lo
; GFX1310-ISEL-NEXT: global_load_b128 v[0:3], v[0:1], off offset:4095 scope:SCOPE_SE
@@ -13926,7 +13729,6 @@ define <4 x float> @global_load_saddr_i8_zext_vgpr_ptrtoint(ptr addrspace(1) inr
; GFX1310-SDAG-NEXT: s_wait_bvhcnt 0x0
; GFX1310-SDAG-NEXT: s_wait_kmcnt 0x0
; GFX1310-SDAG-NEXT: v_add_co_u32 v0, s0, s0, v0
-; GFX1310-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1310-SDAG-NEXT: v_add_co_ci_u32_e64 v1, null, s1, 0, s0
; GFX1310-SDAG-NEXT: global_load_b128 v[0:3], v[0:1], off scope:SCOPE_DEV
; GFX1310-SDAG-NEXT: s_wait_loadcnt 0x0
@@ -13965,7 +13767,7 @@ define <4 x float> @global_load_saddr_i8_zext_vgpr_ptrtoint(ptr addrspace(1) inr
; GFX1250-ISEL-NEXT: s_wait_loadcnt_dscnt 0x0
; GFX1250-ISEL-NEXT: s_wait_kmcnt 0x0
; GFX1250-ISEL-NEXT: v_mov_b64_e32 v[2:3], s[0:1]
-; GFX1250-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-ISEL-NEXT: v_add_co_u32 v0, vcc_lo, v2, v0
; GFX1250-ISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v3, vcc_lo
; GFX1250-ISEL-NEXT: global_load_b128 v[0:3], v[0:1], off scope:SCOPE_DEV
@@ -13980,7 +13782,7 @@ define <4 x float> @global_load_saddr_i8_zext_vgpr_ptrtoint(ptr addrspace(1) inr
; GFX1310-ISEL-NEXT: s_wait_bvhcnt 0x0
; GFX1310-ISEL-NEXT: s_wait_kmcnt 0x0
; GFX1310-ISEL-NEXT: v_dual_mov_b32 v2, s1 :: v_dual_mov_b32 v1, s0
-; GFX1310-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1310-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1310-ISEL-NEXT: v_add_co_u32 v0, vcc_lo, v1, v0
; GFX1310-ISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v2, vcc_lo
; GFX1310-ISEL-NEXT: global_load_b128 v[0:3], v[0:1], off scope:SCOPE_DEV
@@ -14044,7 +13846,6 @@ define <4 x float> @global_load_saddr_i8_zext_vgpr_ptrtoint_commute_add(ptr addr
; GFX1310-SDAG-NEXT: s_wait_bvhcnt 0x0
; GFX1310-SDAG-NEXT: s_wait_kmcnt 0x0
; GFX1310-SDAG-NEXT: v_add_co_u32 v0, s0, v0, s0
-; GFX1310-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1310-SDAG-NEXT: v_add_co_ci_u32_e64 v1, null, 0, s1, s0
; GFX1310-SDAG-NEXT: global_load_b128 v[0:3], v[0:1], off scope:SCOPE_SYS
; GFX1310-SDAG-NEXT: s_wait_loadcnt 0x0
@@ -14083,7 +13884,7 @@ define <4 x float> @global_load_saddr_i8_zext_vgpr_ptrtoint_commute_add(ptr addr
; GFX1250-ISEL-NEXT: s_wait_loadcnt_dscnt 0x0
; GFX1250-ISEL-NEXT: s_wait_kmcnt 0x0
; GFX1250-ISEL-NEXT: v_mov_b64_e32 v[2:3], s[0:1]
-; GFX1250-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-ISEL-NEXT: v_add_co_u32 v0, vcc_lo, v2, v0
; GFX1250-ISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v3, vcc_lo
; GFX1250-ISEL-NEXT: global_load_b128 v[0:3], v[0:1], off scope:SCOPE_SYS
@@ -14098,7 +13899,7 @@ define <4 x float> @global_load_saddr_i8_zext_vgpr_ptrtoint_commute_add(ptr addr
; GFX1310-ISEL-NEXT: s_wait_bvhcnt 0x0
; GFX1310-ISEL-NEXT: s_wait_kmcnt 0x0
; GFX1310-ISEL-NEXT: v_dual_mov_b32 v2, s1 :: v_dual_mov_b32 v1, s0
-; GFX1310-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1310-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1310-ISEL-NEXT: v_add_co_u32 v0, vcc_lo, v1, v0
; GFX1310-ISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v2, vcc_lo
; GFX1310-ISEL-NEXT: global_load_b128 v[0:3], v[0:1], off scope:SCOPE_SYS
@@ -14162,7 +13963,6 @@ define <4 x float> @global_load_saddr_i8_zext_vgpr_ptrtoint_commute_add_imm_offs
; GFX1310-SDAG-NEXT: s_wait_bvhcnt 0x0
; GFX1310-SDAG-NEXT: s_wait_kmcnt 0x0
; GFX1310-SDAG-NEXT: v_add_co_u32 v0, s0, v0, s0
-; GFX1310-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1310-SDAG-NEXT: v_add_co_ci_u32_e64 v1, null, 0, s1, s0
; GFX1310-SDAG-NEXT: global_load_b128 v[0:3], v[0:1], off offset:128
; GFX1310-SDAG-NEXT: s_wait_loadcnt 0x0
@@ -14201,7 +14001,7 @@ define <4 x float> @global_load_saddr_i8_zext_vgpr_ptrtoint_commute_add_imm_offs
; GFX1250-ISEL-NEXT: s_wait_loadcnt_dscnt 0x0
; GFX1250-ISEL-NEXT: s_wait_kmcnt 0x0
; GFX1250-ISEL-NEXT: v_mov_b64_e32 v[2:3], s[0:1]
-; GFX1250-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-ISEL-NEXT: v_add_co_u32 v0, vcc_lo, v2, v0
; GFX1250-ISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v3, vcc_lo
; GFX1250-ISEL-NEXT: global_load_b128 v[0:3], v[0:1], off offset:128
@@ -14216,7 +14016,7 @@ define <4 x float> @global_load_saddr_i8_zext_vgpr_ptrtoint_commute_add_imm_offs
; GFX1310-ISEL-NEXT: s_wait_bvhcnt 0x0
; GFX1310-ISEL-NEXT: s_wait_kmcnt 0x0
; GFX1310-ISEL-NEXT: v_dual_mov_b32 v2, s1 :: v_dual_mov_b32 v1, s0
-; GFX1310-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1310-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1310-ISEL-NEXT: v_add_co_u32 v0, vcc_lo, v1, v0
; GFX1310-ISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v2, vcc_lo
; GFX1310-ISEL-NEXT: global_load_b128 v[0:3], v[0:1], off offset:128
@@ -14281,7 +14081,6 @@ define <4 x float> @global_load_saddr_i8_zext_vgpr_ptrtoint_commute_add_imm_offs
; GFX1310-SDAG-NEXT: s_wait_bvhcnt 0x0
; GFX1310-SDAG-NEXT: s_wait_kmcnt 0x0
; GFX1310-SDAG-NEXT: v_add_co_u32 v0, s0, s0, v0
-; GFX1310-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1310-SDAG-NEXT: v_add_co_ci_u32_e64 v1, null, s1, 0, s0
; GFX1310-SDAG-NEXT: global_load_b128 v[0:3], v[0:1], off offset:128 scope:SCOPE_SE
; GFX1310-SDAG-NEXT: s_wait_loadcnt 0x0
@@ -14320,7 +14119,7 @@ define <4 x float> @global_load_saddr_i8_zext_vgpr_ptrtoint_commute_add_imm_offs
; GFX1250-ISEL-NEXT: s_wait_loadcnt_dscnt 0x0
; GFX1250-ISEL-NEXT: s_wait_kmcnt 0x0
; GFX1250-ISEL-NEXT: v_mov_b64_e32 v[2:3], s[0:1]
-; GFX1250-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-ISEL-NEXT: v_add_co_u32 v0, vcc_lo, v2, v0
; GFX1250-ISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v3, vcc_lo
; GFX1250-ISEL-NEXT: global_load_b128 v[0:3], v[0:1], off offset:128
@@ -14335,7 +14134,7 @@ define <4 x float> @global_load_saddr_i8_zext_vgpr_ptrtoint_commute_add_imm_offs
; GFX1310-ISEL-NEXT: s_wait_bvhcnt 0x0
; GFX1310-ISEL-NEXT: s_wait_kmcnt 0x0
; GFX1310-ISEL-NEXT: v_dual_mov_b32 v2, s1 :: v_dual_mov_b32 v1, s0
-; GFX1310-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1310-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1310-ISEL-NEXT: v_add_co_u32 v0, vcc_lo, v1, v0
; GFX1310-ISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v2, vcc_lo
; GFX1310-ISEL-NEXT: global_load_b128 v[0:3], v[0:1], off offset:128 scope:SCOPE_SE
@@ -15074,7 +14873,6 @@ define <4 x float> @global_load_saddr_i8_vgpr64_sgpr32(ptr addrspace(1) %vbase,
; GFX1100-SDAG: ; %bb.0:
; GFX1100-SDAG-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX1100-SDAG-NEXT: v_add_co_u32 v0, vcc_lo, v0, s0
-; GFX1100-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1100-SDAG-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1100-SDAG-NEXT: global_load_b128 v[0:3], v[0:1], off glc
; GFX1100-SDAG-NEXT: s_waitcnt vmcnt(0)
@@ -15099,7 +14897,6 @@ define <4 x float> @global_load_saddr_i8_vgpr64_sgpr32(ptr addrspace(1) %vbase,
; GFX1310-SDAG-NEXT: s_wait_bvhcnt 0x0
; GFX1310-SDAG-NEXT: s_wait_kmcnt 0x0
; GFX1310-SDAG-NEXT: v_add_co_u32 v0, vcc_lo, v0, s0
-; GFX1310-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1310-SDAG-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1310-SDAG-NEXT: global_load_b128 v[0:3], v[0:1], off scope:SCOPE_DEV
; GFX1310-SDAG-NEXT: s_wait_loadcnt 0x0
@@ -15148,7 +14945,7 @@ define <4 x float> @global_load_saddr_i8_vgpr64_sgpr32(ptr addrspace(1) %vbase,
; GFX1100-ISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1100-ISEL-NEXT: v_dual_mov_b32 v2, s0 :: v_dual_mov_b32 v3, s1
; GFX1100-ISEL-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
-; GFX1100-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1100-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX1100-ISEL-NEXT: v_add_co_ci_u32_e64 v1, null, v1, v3, vcc_lo
; GFX1100-ISEL-NEXT: global_load_b128 v[0:3], v[0:1], off glc
; GFX1100-ISEL-NEXT: s_waitcnt vmcnt(0)
@@ -15162,7 +14959,7 @@ define <4 x float> @global_load_saddr_i8_vgpr64_sgpr32(ptr addrspace(1) %vbase,
; GFX1250-ISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1250-ISEL-NEXT: v_mov_b64_e32 v[2:3], s[0:1]
; GFX1250-ISEL-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
-; GFX1250-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1250-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX1250-ISEL-NEXT: v_add_co_ci_u32_e64 v1, null, v1, v3, vcc_lo
; GFX1250-ISEL-NEXT: global_load_b128 v[0:3], v[0:1], off scope:SCOPE_DEV
; GFX1250-ISEL-NEXT: s_wait_loadcnt 0x0
@@ -15179,7 +14976,7 @@ define <4 x float> @global_load_saddr_i8_vgpr64_sgpr32(ptr addrspace(1) %vbase,
; GFX1310-ISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1310-ISEL-NEXT: v_dual_mov_b32 v2, s0 :: v_dual_mov_b32 v3, s1
; GFX1310-ISEL-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
-; GFX1310-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1310-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX1310-ISEL-NEXT: v_add_co_ci_u32_e64 v1, null, v1, v3, vcc_lo
; GFX1310-ISEL-NEXT: global_load_b128 v[0:3], v[0:1], off scope:SCOPE_DEV
; GFX1310-ISEL-NEXT: s_wait_loadcnt 0x0
@@ -15226,7 +15023,6 @@ define <4 x float> @global_load_saddr_i8_vgpr64_sgpr32_offset_4095(ptr addrspace
; GFX1100-SDAG: ; %bb.0:
; GFX1100-SDAG-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX1100-SDAG-NEXT: v_add_co_u32 v0, vcc_lo, v0, s0
-; GFX1100-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1100-SDAG-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1100-SDAG-NEXT: global_load_b128 v[0:3], v[0:1], off offset:4095 glc
; GFX1100-SDAG-NEXT: s_waitcnt vmcnt(0)
@@ -15251,7 +15047,6 @@ define <4 x float> @global_load_saddr_i8_vgpr64_sgpr32_offset_4095(ptr addrspace
; GFX1310-SDAG-NEXT: s_wait_bvhcnt 0x0
; GFX1310-SDAG-NEXT: s_wait_kmcnt 0x0
; GFX1310-SDAG-NEXT: v_add_co_u32 v0, vcc_lo, v0, s0
-; GFX1310-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1310-SDAG-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1310-SDAG-NEXT: global_load_b128 v[0:3], v[0:1], off offset:4095 scope:SCOPE_SYS
; GFX1310-SDAG-NEXT: s_wait_loadcnt 0x0
@@ -15302,7 +15097,7 @@ define <4 x float> @global_load_saddr_i8_vgpr64_sgpr32_offset_4095(ptr addrspace
; GFX1100-ISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1100-ISEL-NEXT: v_dual_mov_b32 v2, s0 :: v_dual_mov_b32 v3, s1
; GFX1100-ISEL-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
-; GFX1100-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1100-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX1100-ISEL-NEXT: v_add_co_ci_u32_e64 v1, null, v1, v3, vcc_lo
; GFX1100-ISEL-NEXT: global_load_b128 v[0:3], v[0:1], off offset:4095 glc
; GFX1100-ISEL-NEXT: s_waitcnt vmcnt(0)
@@ -15316,7 +15111,7 @@ define <4 x float> @global_load_saddr_i8_vgpr64_sgpr32_offset_4095(ptr addrspace
; GFX1250-ISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1250-ISEL-NEXT: v_mov_b64_e32 v[2:3], s[0:1]
; GFX1250-ISEL-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
-; GFX1250-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1250-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX1250-ISEL-NEXT: v_add_co_ci_u32_e64 v1, null, v1, v3, vcc_lo
; GFX1250-ISEL-NEXT: global_load_b128 v[0:3], v[0:1], off offset:4095 scope:SCOPE_SYS
; GFX1250-ISEL-NEXT: s_wait_loadcnt 0x0
@@ -15333,7 +15128,7 @@ define <4 x float> @global_load_saddr_i8_vgpr64_sgpr32_offset_4095(ptr addrspace
; GFX1310-ISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1310-ISEL-NEXT: v_dual_mov_b32 v2, s0 :: v_dual_mov_b32 v3, s1
; GFX1310-ISEL-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
-; GFX1310-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1310-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX1310-ISEL-NEXT: v_add_co_ci_u32_e64 v1, null, v1, v3, vcc_lo
; GFX1310-ISEL-NEXT: global_load_b128 v[0:3], v[0:1], off offset:4095 scope:SCOPE_SYS
; GFX1310-ISEL-NEXT: s_wait_loadcnt 0x0
@@ -15401,7 +15196,7 @@ define <4 x float> @global_load_saddr_f32_natural_addressing(ptr addrspace(1) in
; GFX1100-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1100-SDAG-NEXT: v_lshlrev_b64 v[0:1], 2, v[0:1]
; GFX1100-SDAG-NEXT: v_add_co_u32 v0, vcc_lo, s0, v0
-; GFX1100-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1100-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX1100-SDAG-NEXT: v_add_co_ci_u32_e64 v1, null, s1, v1, vcc_lo
; GFX1100-SDAG-NEXT: global_load_b128 v[0:3], v[0:1], off
; GFX1100-SDAG-NEXT: s_waitcnt vmcnt(0)
@@ -15434,7 +15229,7 @@ define <4 x float> @global_load_saddr_f32_natural_addressing(ptr addrspace(1) in
; GFX1310-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1310-SDAG-NEXT: v_lshlrev_b64_e32 v[0:1], 2, v[0:1]
; GFX1310-SDAG-NEXT: v_add_co_u32 v0, vcc_lo, s0, v0
-; GFX1310-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1310-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX1310-SDAG-NEXT: v_add_co_ci_u32_e64 v1, null, s1, v1, vcc_lo
; GFX1310-SDAG-NEXT: global_load_b128 v[0:3], v[0:1], off
; GFX1310-SDAG-NEXT: s_wait_loadcnt 0x0
@@ -15491,7 +15286,7 @@ define <4 x float> @global_load_saddr_f32_natural_addressing(ptr addrspace(1) in
; GFX1100-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1100-ISEL-NEXT: v_lshlrev_b64 v[0:1], 2, v[0:1]
; GFX1100-ISEL-NEXT: v_add_co_u32 v0, vcc_lo, v2, v0
-; GFX1100-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1100-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX1100-ISEL-NEXT: v_add_co_ci_u32_e64 v1, null, v3, v1, vcc_lo
; GFX1100-ISEL-NEXT: global_load_b128 v[0:3], v[0:1], off
; GFX1100-ISEL-NEXT: s_waitcnt vmcnt(0)
@@ -15525,7 +15320,7 @@ define <4 x float> @global_load_saddr_f32_natural_addressing(ptr addrspace(1) in
; GFX1310-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1310-ISEL-NEXT: v_lshlrev_b64_e32 v[0:1], 2, v[0:1]
; GFX1310-ISEL-NEXT: v_add_co_u32 v0, vcc_lo, v2, v0
-; GFX1310-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1310-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX1310-ISEL-NEXT: v_add_co_ci_u32_e64 v1, null, v3, v1, vcc_lo
; GFX1310-ISEL-NEXT: global_load_b128 v[0:3], v[0:1], off
; GFX1310-ISEL-NEXT: s_wait_loadcnt 0x0
@@ -15600,7 +15395,6 @@ define <4 x float> @global_load_saddr_f32_natural_addressing_immoffset(ptr addrs
; GFX1310-SDAG-NEXT: global_load_b32 v0, v[0:1], off
; GFX1310-SDAG-NEXT: s_wait_loadcnt 0x0
; GFX1310-SDAG-NEXT: v_add_co_u32 v0, s0, s0, v0
-; GFX1310-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1310-SDAG-NEXT: v_add_co_ci_u32_e64 v1, null, s1, 0, s0
; GFX1310-SDAG-NEXT: global_load_b128 v[0:3], v[0:1], off offset:128 scope:SCOPE_SE
; GFX1310-SDAG-NEXT: s_wait_loadcnt 0x0
@@ -15650,7 +15444,7 @@ define <4 x float> @global_load_saddr_f32_natural_addressing_immoffset(ptr addrs
; GFX1250-ISEL-NEXT: s_wait_xcnt 0x0
; GFX1250-ISEL-NEXT: v_mov_b64_e32 v[0:1], s[0:1]
; GFX1250-ISEL-NEXT: s_wait_loadcnt 0x0
-; GFX1250-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-ISEL-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
; GFX1250-ISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1250-ISEL-NEXT: global_load_b128 v[0:3], v[0:1], off offset:128
@@ -15667,7 +15461,7 @@ define <4 x float> @global_load_saddr_f32_natural_addressing_immoffset(ptr addrs
; GFX1310-ISEL-NEXT: global_load_b32 v2, v[0:1], off
; GFX1310-ISEL-NEXT: v_dual_mov_b32 v0, s0 :: v_dual_mov_b32 v1, s1
; GFX1310-ISEL-NEXT: s_wait_loadcnt 0x0
-; GFX1310-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1310-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1310-ISEL-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
; GFX1310-ISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1310-ISEL-NEXT: global_load_b128 v[0:3], v[0:1], off offset:128 scope:SCOPE_SE
@@ -15748,7 +15542,7 @@ define <4 x float> @global_load_f32_saddr_zext_vgpr_range(ptr addrspace(1) inreg
; GFX1310-SDAG-NEXT: global_load_b32 v0, v[0:1], off
; GFX1310-SDAG-NEXT: s_wait_loadcnt 0x0
; GFX1310-SDAG-NEXT: v_lshlrev_b32_e32 v0, 2, v0
-; GFX1310-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1310-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1310-SDAG-NEXT: v_add_co_u32 v0, s0, s0, v0
; GFX1310-SDAG-NEXT: v_add_co_ci_u32_e64 v1, null, s1, 0, s0
; GFX1310-SDAG-NEXT: global_load_b128 v[0:3], v[0:1], off scope:SCOPE_DEV
@@ -15804,7 +15598,7 @@ define <4 x float> @global_load_f32_saddr_zext_vgpr_range(ptr addrspace(1) inreg
; GFX1250-ISEL-NEXT: v_mov_b64_e32 v[0:1], s[0:1]
; GFX1250-ISEL-NEXT: s_wait_loadcnt 0x0
; GFX1250-ISEL-NEXT: v_lshlrev_b32_e32 v2, 2, v2
-; GFX1250-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_3)
; GFX1250-ISEL-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
; GFX1250-ISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1250-ISEL-NEXT: global_load_b128 v[0:3], v[0:1], off scope:SCOPE_DEV
@@ -15822,7 +15616,7 @@ define <4 x float> @global_load_f32_saddr_zext_vgpr_range(ptr addrspace(1) inreg
; GFX1310-ISEL-NEXT: v_dual_mov_b32 v0, s0 :: v_dual_mov_b32 v1, s1
; GFX1310-ISEL-NEXT: s_wait_loadcnt 0x0
; GFX1310-ISEL-NEXT: v_lshlrev_b32_e32 v2, 2, v2
-; GFX1310-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1310-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_3)
; GFX1310-ISEL-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
; GFX1310-ISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1310-ISEL-NEXT: global_load_b128 v[0:3], v[0:1], off scope:SCOPE_DEV
@@ -15902,7 +15696,7 @@ define <4 x float> @global_load_f32_saddr_zext_vgpr_range_imm_offset(ptr addrspa
; GFX1310-SDAG-NEXT: global_load_b32 v0, v[0:1], off
; GFX1310-SDAG-NEXT: s_wait_loadcnt 0x0
; GFX1310-SDAG-NEXT: v_lshlrev_b32_e32 v0, 2, v0
-; GFX1310-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1310-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1310-SDAG-NEXT: v_add_co_u32 v0, s0, s0, v0
; GFX1310-SDAG-NEXT: v_add_co_ci_u32_e64 v1, null, s1, 0, s0
; GFX1310-SDAG-NEXT: global_load_b128 v[0:3], v[0:1], off offset:400 scope:SCOPE_SYS
@@ -15958,7 +15752,7 @@ define <4 x float> @global_load_f32_saddr_zext_vgpr_range_imm_offset(ptr addrspa
; GFX1250-ISEL-NEXT: v_mov_b64_e32 v[0:1], s[0:1]
; GFX1250-ISEL-NEXT: s_wait_loadcnt 0x0
; GFX1250-ISEL-NEXT: v_lshlrev_b32_e32 v2, 2, v2
-; GFX1250-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_3)
; GFX1250-ISEL-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
; GFX1250-ISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1250-ISEL-NEXT: global_load_b128 v[0:3], v[0:1], off offset:400 scope:SCOPE_SYS
@@ -15976,7 +15770,7 @@ define <4 x float> @global_load_f32_saddr_zext_vgpr_range_imm_offset(ptr addrspa
; GFX1310-ISEL-NEXT: v_dual_mov_b32 v0, s0 :: v_dual_mov_b32 v1, s1
; GFX1310-ISEL-NEXT: s_wait_loadcnt 0x0
; GFX1310-ISEL-NEXT: v_lshlrev_b32_e32 v2, 2, v2
-; GFX1310-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1310-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_3)
; GFX1310-ISEL-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
; GFX1310-ISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1310-ISEL-NEXT: global_load_b128 v[0:3], v[0:1], off offset:400 scope:SCOPE_SYS
@@ -16042,7 +15836,7 @@ define <4 x float> @global_load_f32_saddr_zext_vgpr_range_too_large(ptr addrspac
; GFX1100-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1100-SDAG-NEXT: v_lshlrev_b64 v[0:1], 2, v[0:1]
; GFX1100-SDAG-NEXT: v_add_co_u32 v0, vcc_lo, s0, v0
-; GFX1100-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1100-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX1100-SDAG-NEXT: v_add_co_ci_u32_e64 v1, null, s1, v1, vcc_lo
; GFX1100-SDAG-NEXT: global_load_b128 v[0:3], v[0:1], off
; GFX1100-SDAG-NEXT: s_waitcnt vmcnt(0)
@@ -16075,7 +15869,7 @@ define <4 x float> @global_load_f32_saddr_zext_vgpr_range_too_large(ptr addrspac
; GFX1310-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1310-SDAG-NEXT: v_lshlrev_b64_e32 v[0:1], 2, v[0:1]
; GFX1310-SDAG-NEXT: v_add_co_u32 v0, vcc_lo, s0, v0
-; GFX1310-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1310-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX1310-SDAG-NEXT: v_add_co_ci_u32_e64 v1, null, s1, v1, vcc_lo
; GFX1310-SDAG-NEXT: global_load_b128 v[0:3], v[0:1], off
; GFX1310-SDAG-NEXT: s_wait_loadcnt 0x0
@@ -16132,7 +15926,7 @@ define <4 x float> @global_load_f32_saddr_zext_vgpr_range_too_large(ptr addrspac
; GFX1100-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1100-ISEL-NEXT: v_lshlrev_b64 v[0:1], 2, v[0:1]
; GFX1100-ISEL-NEXT: v_add_co_u32 v0, vcc_lo, v2, v0
-; GFX1100-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1100-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX1100-ISEL-NEXT: v_add_co_ci_u32_e64 v1, null, v3, v1, vcc_lo
; GFX1100-ISEL-NEXT: global_load_b128 v[0:3], v[0:1], off
; GFX1100-ISEL-NEXT: s_waitcnt vmcnt(0)
@@ -16166,7 +15960,7 @@ define <4 x float> @global_load_f32_saddr_zext_vgpr_range_too_large(ptr addrspac
; GFX1310-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1310-ISEL-NEXT: v_lshlrev_b64_e32 v[0:1], 2, v[0:1]
; GFX1310-ISEL-NEXT: v_add_co_u32 v0, vcc_lo, v2, v0
-; GFX1310-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1310-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX1310-ISEL-NEXT: v_add_co_ci_u32_e64 v1, null, v3, v1, vcc_lo
; GFX1310-ISEL-NEXT: global_load_b128 v[0:3], v[0:1], off
; GFX1310-ISEL-NEXT: s_wait_loadcnt 0x0
@@ -16503,6 +16297,7 @@ define <4 x float> @global_saddr_64bit_lsr_iv(ptr addrspace(1) inreg %arg) {
; GFX1100-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1100-SDAG-NEXT: s_add_i32 s2, s2, 1
; GFX1100-SDAG-NEXT: s_cmpk_eq_i32 s2, 0xff
+; GFX1100-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1100-SDAG-NEXT: s_cbranch_scc0 .LBB114_1
; GFX1100-SDAG-NEXT: ; %bb.2: ; %bb2
; GFX1100-SDAG-NEXT: s_mov_b32 s3, 0
@@ -16525,6 +16320,7 @@ define <4 x float> @global_saddr_64bit_lsr_iv(ptr addrspace(1) inreg %arg) {
; GFX1250-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1250-SDAG-NEXT: s_add_co_i32 s2, s2, 1
; GFX1250-SDAG-NEXT: s_cmp_eq_u32 s2, 0xff
+; GFX1250-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1250-SDAG-NEXT: s_cbranch_scc0 .LBB114_1
; GFX1250-SDAG-NEXT: ; %bb.2: ; %bb2
; GFX1250-SDAG-NEXT: s_mov_b32 s3, 0
@@ -16549,6 +16345,7 @@ define <4 x float> @global_saddr_64bit_lsr_iv(ptr addrspace(1) inreg %arg) {
; GFX1310-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1310-SDAG-NEXT: s_add_co_i32 s2, s2, 1
; GFX1310-SDAG-NEXT: s_cmp_eq_u32 s2, 0xff
+; GFX1310-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1310-SDAG-NEXT: s_cbranch_scc0 .LBB114_1
; GFX1310-SDAG-NEXT: ; %bb.2: ; %bb2
; GFX1310-SDAG-NEXT: s_mov_b32 s3, 0
@@ -16650,6 +16447,7 @@ define <4 x float> @global_saddr_64bit_lsr_iv(ptr addrspace(1) inreg %arg) {
; GFX1100-ISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1100-ISEL-NEXT: s_add_i32 s2, s2, 1
; GFX1100-ISEL-NEXT: s_cmpk_eq_i32 s2, 0xff
+; GFX1100-ISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1100-ISEL-NEXT: s_cbranch_scc0 .LBB114_1
; GFX1100-ISEL-NEXT: ; %bb.2: ; %bb2
; GFX1100-ISEL-NEXT: s_mov_b32 s3, 0
@@ -16679,6 +16477,7 @@ define <4 x float> @global_saddr_64bit_lsr_iv(ptr addrspace(1) inreg %arg) {
; GFX1250-ISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1250-ISEL-NEXT: s_add_co_i32 s2, s2, 1
; GFX1250-ISEL-NEXT: s_cmp_eq_u32 s2, 0xff
+; GFX1250-ISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1250-ISEL-NEXT: s_cbranch_scc0 .LBB114_1
; GFX1250-ISEL-NEXT: ; %bb.2: ; %bb2
; GFX1250-ISEL-NEXT: s_mov_b32 s3, 0
@@ -16712,6 +16511,7 @@ define <4 x float> @global_saddr_64bit_lsr_iv(ptr addrspace(1) inreg %arg) {
; GFX1310-ISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1310-ISEL-NEXT: s_add_co_i32 s2, s2, 1
; GFX1310-ISEL-NEXT: s_cmp_eq_u32 s2, 0xff
+; GFX1310-ISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1310-ISEL-NEXT: s_cbranch_scc0 .LBB114_1
; GFX1310-ISEL-NEXT: ; %bb.2: ; %bb2
; GFX1310-ISEL-NEXT: s_mov_b32 s3, 0
@@ -16817,6 +16617,7 @@ define <4 x float> @global_saddr_64bit_lsr_iv_multiload(ptr addrspace(1) inreg %
; GFX1100-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1100-SDAG-NEXT: s_add_i32 s2, s2, 1
; GFX1100-SDAG-NEXT: s_cmpk_eq_i32 s2, 0xff
+; GFX1100-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1100-SDAG-NEXT: s_cbranch_scc0 .LBB115_1
; GFX1100-SDAG-NEXT: ; %bb.2: ; %bb2
; GFX1100-SDAG-NEXT: s_mov_b32 s3, 0
@@ -16839,6 +16640,7 @@ define <4 x float> @global_saddr_64bit_lsr_iv_multiload(ptr addrspace(1) inreg %
; GFX1250-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1250-SDAG-NEXT: s_add_co_i32 s2, s2, 1
; GFX1250-SDAG-NEXT: s_cmp_eq_u32 s2, 0xff
+; GFX1250-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1250-SDAG-NEXT: s_cbranch_scc0 .LBB115_1
; GFX1250-SDAG-NEXT: ; %bb.2: ; %bb2
; GFX1250-SDAG-NEXT: s_mov_b32 s3, 0
@@ -16863,6 +16665,7 @@ define <4 x float> @global_saddr_64bit_lsr_iv_multiload(ptr addrspace(1) inreg %
; GFX1310-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1310-SDAG-NEXT: s_add_co_i32 s2, s2, 1
; GFX1310-SDAG-NEXT: s_cmp_eq_u32 s2, 0xff
+; GFX1310-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1310-SDAG-NEXT: s_cbranch_scc0 .LBB115_1
; GFX1310-SDAG-NEXT: ; %bb.2: ; %bb2
; GFX1310-SDAG-NEXT: s_mov_b32 s3, 0
@@ -16964,6 +16767,7 @@ define <4 x float> @global_saddr_64bit_lsr_iv_multiload(ptr addrspace(1) inreg %
; GFX1100-ISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1100-ISEL-NEXT: s_add_i32 s2, s2, 1
; GFX1100-ISEL-NEXT: s_cmpk_eq_i32 s2, 0xff
+; GFX1100-ISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1100-ISEL-NEXT: s_cbranch_scc0 .LBB115_1
; GFX1100-ISEL-NEXT: ; %bb.2: ; %bb2
; GFX1100-ISEL-NEXT: s_mov_b32 s3, 0
@@ -16993,6 +16797,7 @@ define <4 x float> @global_saddr_64bit_lsr_iv_multiload(ptr addrspace(1) inreg %
; GFX1250-ISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1250-ISEL-NEXT: s_add_co_i32 s2, s2, 1
; GFX1250-ISEL-NEXT: s_cmp_eq_u32 s2, 0xff
+; GFX1250-ISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1250-ISEL-NEXT: s_cbranch_scc0 .LBB115_1
; GFX1250-ISEL-NEXT: ; %bb.2: ; %bb2
; GFX1250-ISEL-NEXT: s_mov_b32 s3, 0
@@ -17026,6 +16831,7 @@ define <4 x float> @global_saddr_64bit_lsr_iv_multiload(ptr addrspace(1) inreg %
; GFX1310-ISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1310-ISEL-NEXT: s_add_co_i32 s2, s2, 1
; GFX1310-ISEL-NEXT: s_cmp_eq_u32 s2, 0xff
+; GFX1310-ISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1310-ISEL-NEXT: s_cbranch_scc0 .LBB115_1
; GFX1310-ISEL-NEXT: ; %bb.2: ; %bb2
; GFX1310-ISEL-NEXT: s_mov_b32 s3, 0
diff --git a/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.av.store.b128.ll b/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.av.store.b128.ll
index f0909a0cfb30b3..a5ecf2d555676d 100644
--- a/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.av.store.b128.ll
+++ b/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.av.store.b128.ll
@@ -430,7 +430,6 @@ define void @global_store_i8_zext_vgpr(ptr addrspace(1) %sbase, ptr addrspace(1)
; GFX1100-SDAG-NEXT: global_load_b32 v2, v[2:3], off
; GFX1100-SDAG-NEXT: s_waitcnt vmcnt(0)
; GFX1100-SDAG-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
-; GFX1100-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1100-SDAG-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1100-SDAG-NEXT: global_store_b128 v[0:1], v[4:7], off
; GFX1100-SDAG-NEXT: s_setpc_b64 s[30:31]
@@ -458,7 +457,6 @@ define void @global_store_i8_zext_vgpr(ptr addrspace(1) %sbase, ptr addrspace(1)
; GFX1310-SDAG-NEXT: global_load_b32 v2, v[2:3], off
; GFX1310-SDAG-NEXT: s_wait_loadcnt 0x0
; GFX1310-SDAG-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
-; GFX1310-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1310-SDAG-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1310-SDAG-NEXT: global_store_b128 v[0:1], v[4:7], off
; GFX1310-SDAG-NEXT: s_set_pc_i64 s[30:31]
@@ -502,7 +500,6 @@ define void @global_store_i8_zext_vgpr(ptr addrspace(1) %sbase, ptr addrspace(1)
; GFX1100-ISEL-NEXT: global_load_b32 v2, v[2:3], off
; GFX1100-ISEL-NEXT: s_waitcnt vmcnt(0)
; GFX1100-ISEL-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
-; GFX1100-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1100-ISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1100-ISEL-NEXT: global_store_b128 v[0:1], v[4:7], off
; GFX1100-ISEL-NEXT: s_setpc_b64 s[30:31]
@@ -514,7 +511,6 @@ define void @global_store_i8_zext_vgpr(ptr addrspace(1) %sbase, ptr addrspace(1)
; GFX1250-ISEL-NEXT: global_load_b32 v2, v[2:3], off
; GFX1250-ISEL-NEXT: s_wait_loadcnt 0x0
; GFX1250-ISEL-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
-; GFX1250-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1250-ISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1250-ISEL-NEXT: s_wait_xcnt 0x0
; GFX1250-ISEL-NEXT: global_store_b128 v[0:1], v[4:7], off
@@ -530,7 +526,6 @@ define void @global_store_i8_zext_vgpr(ptr addrspace(1) %sbase, ptr addrspace(1)
; GFX1310-ISEL-NEXT: global_load_b32 v2, v[2:3], off
; GFX1310-ISEL-NEXT: s_wait_loadcnt 0x0
; GFX1310-ISEL-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
-; GFX1310-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1310-ISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1310-ISEL-NEXT: global_store_b128 v[0:1], v[4:7], off
; GFX1310-ISEL-NEXT: s_set_pc_i64 s[30:31]
@@ -577,7 +572,6 @@ define void @global_store_v4i32_zext_vgpr_offset_neg128(ptr addrspace(1) %sbase,
; GFX1100-SDAG: ; %bb.0:
; GFX1100-SDAG-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX1100-SDAG-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
-; GFX1100-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1100-SDAG-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1100-SDAG-NEXT: global_store_b128 v[0:1], v[3:6], off offset:-128
; GFX1100-SDAG-NEXT: s_setpc_b64 s[30:31]
@@ -602,7 +596,6 @@ define void @global_store_v4i32_zext_vgpr_offset_neg128(ptr addrspace(1) %sbase,
; GFX1310-SDAG-NEXT: s_wait_bvhcnt 0x0
; GFX1310-SDAG-NEXT: s_wait_kmcnt 0x0
; GFX1310-SDAG-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
-; GFX1310-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1310-SDAG-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1310-SDAG-NEXT: global_store_b128 v[0:1], v[3:6], off offset:-128 scope:SCOPE_SE
; GFX1310-SDAG-NEXT: s_set_pc_i64 s[30:31]
@@ -641,7 +634,6 @@ define void @global_store_v4i32_zext_vgpr_offset_neg128(ptr addrspace(1) %sbase,
; GFX1100-ISEL: ; %bb.0:
; GFX1100-ISEL-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX1100-ISEL-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
-; GFX1100-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1100-ISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1100-ISEL-NEXT: global_store_b128 v[0:1], v[3:6], off offset:-128
; GFX1100-ISEL-NEXT: s_setpc_b64 s[30:31]
@@ -665,7 +657,6 @@ define void @global_store_v4i32_zext_vgpr_offset_neg128(ptr addrspace(1) %sbase,
; GFX1310-ISEL-NEXT: s_wait_bvhcnt 0x0
; GFX1310-ISEL-NEXT: s_wait_kmcnt 0x0
; GFX1310-ISEL-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
-; GFX1310-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1310-ISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1310-ISEL-NEXT: global_store_b128 v[0:1], v[3:6], off offset:-128 scope:SCOPE_SE
; GFX1310-ISEL-NEXT: s_set_pc_i64 s[30:31]
@@ -716,7 +707,6 @@ define void @global_store_i8_zext_vgpr_offset_2047(ptr addrspace(1) %sbase, ptr
; GFX1100-SDAG-NEXT: global_load_b32 v2, v[2:3], off
; GFX1100-SDAG-NEXT: s_waitcnt vmcnt(0)
; GFX1100-SDAG-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
-; GFX1100-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1100-SDAG-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1100-SDAG-NEXT: global_store_b128 v[0:1], v[4:7], off offset:2047
; GFX1100-SDAG-NEXT: s_setpc_b64 s[30:31]
@@ -744,7 +734,6 @@ define void @global_store_i8_zext_vgpr_offset_2047(ptr addrspace(1) %sbase, ptr
; GFX1310-SDAG-NEXT: global_load_b32 v2, v[2:3], off
; GFX1310-SDAG-NEXT: s_wait_loadcnt 0x0
; GFX1310-SDAG-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
-; GFX1310-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1310-SDAG-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1310-SDAG-NEXT: global_store_b128 v[0:1], v[4:7], off offset:2047 scope:SCOPE_DEV
; GFX1310-SDAG-NEXT: s_set_pc_i64 s[30:31]
@@ -788,7 +777,6 @@ define void @global_store_i8_zext_vgpr_offset_2047(ptr addrspace(1) %sbase, ptr
; GFX1100-ISEL-NEXT: global_load_b32 v2, v[2:3], off
; GFX1100-ISEL-NEXT: s_waitcnt vmcnt(0)
; GFX1100-ISEL-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
-; GFX1100-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1100-ISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1100-ISEL-NEXT: global_store_b128 v[0:1], v[4:7], off offset:2047
; GFX1100-ISEL-NEXT: s_setpc_b64 s[30:31]
@@ -800,7 +788,6 @@ define void @global_store_i8_zext_vgpr_offset_2047(ptr addrspace(1) %sbase, ptr
; GFX1250-ISEL-NEXT: global_load_b32 v2, v[2:3], off
; GFX1250-ISEL-NEXT: s_wait_loadcnt 0x0
; GFX1250-ISEL-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
-; GFX1250-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1250-ISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1250-ISEL-NEXT: s_wait_xcnt 0x0
; GFX1250-ISEL-NEXT: global_store_b128 v[0:1], v[4:7], off offset:2047 scope:SCOPE_DEV
@@ -816,7 +803,6 @@ define void @global_store_i8_zext_vgpr_offset_2047(ptr addrspace(1) %sbase, ptr
; GFX1310-ISEL-NEXT: global_load_b32 v2, v[2:3], off
; GFX1310-ISEL-NEXT: s_wait_loadcnt 0x0
; GFX1310-ISEL-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
-; GFX1310-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1310-ISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1310-ISEL-NEXT: global_store_b128 v[0:1], v[4:7], off offset:2047 scope:SCOPE_DEV
; GFX1310-ISEL-NEXT: s_set_pc_i64 s[30:31]
@@ -868,7 +854,6 @@ define void @global_store_i8_zext_vgpr_offset_neg2048(ptr addrspace(1) %sbase, p
; GFX1100-SDAG-NEXT: global_load_b32 v2, v[2:3], off
; GFX1100-SDAG-NEXT: s_waitcnt vmcnt(0)
; GFX1100-SDAG-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
-; GFX1100-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1100-SDAG-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1100-SDAG-NEXT: global_store_b128 v[0:1], v[4:7], off offset:-2048
; GFX1100-SDAG-NEXT: s_setpc_b64 s[30:31]
@@ -896,7 +881,6 @@ define void @global_store_i8_zext_vgpr_offset_neg2048(ptr addrspace(1) %sbase, p
; GFX1310-SDAG-NEXT: global_load_b32 v2, v[2:3], off
; GFX1310-SDAG-NEXT: s_wait_loadcnt 0x0
; GFX1310-SDAG-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
-; GFX1310-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1310-SDAG-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1310-SDAG-NEXT: global_store_b128 v[0:1], v[4:7], off offset:-2048 scope:SCOPE_SYS
; GFX1310-SDAG-NEXT: s_set_pc_i64 s[30:31]
@@ -940,7 +924,6 @@ define void @global_store_i8_zext_vgpr_offset_neg2048(ptr addrspace(1) %sbase, p
; GFX1100-ISEL-NEXT: global_load_b32 v2, v[2:3], off
; GFX1100-ISEL-NEXT: s_waitcnt vmcnt(0)
; GFX1100-ISEL-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
-; GFX1100-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1100-ISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1100-ISEL-NEXT: global_store_b128 v[0:1], v[4:7], off offset:-2048
; GFX1100-ISEL-NEXT: s_setpc_b64 s[30:31]
@@ -952,7 +935,6 @@ define void @global_store_i8_zext_vgpr_offset_neg2048(ptr addrspace(1) %sbase, p
; GFX1250-ISEL-NEXT: global_load_b32 v2, v[2:3], off
; GFX1250-ISEL-NEXT: s_wait_loadcnt 0x0
; GFX1250-ISEL-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
-; GFX1250-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1250-ISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1250-ISEL-NEXT: s_wait_xcnt 0x0
; GFX1250-ISEL-NEXT: global_store_b128 v[0:1], v[4:7], off offset:-2048 scope:SCOPE_SYS
@@ -968,7 +950,6 @@ define void @global_store_i8_zext_vgpr_offset_neg2048(ptr addrspace(1) %sbase, p
; GFX1310-ISEL-NEXT: global_load_b32 v2, v[2:3], off
; GFX1310-ISEL-NEXT: s_wait_loadcnt 0x0
; GFX1310-ISEL-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
-; GFX1310-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1310-ISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1310-ISEL-NEXT: global_store_b128 v[0:1], v[4:7], off offset:-2048 scope:SCOPE_SYS
; GFX1310-ISEL-NEXT: s_set_pc_i64 s[30:31]
@@ -1043,7 +1024,6 @@ define void @global_store_saddr_i8_zext_vgpr(ptr addrspace(1) inreg %sbase, ptr
; GFX1310-SDAG-NEXT: global_load_b32 v0, v[0:1], off
; GFX1310-SDAG-NEXT: s_wait_loadcnt 0x0
; GFX1310-SDAG-NEXT: v_add_co_u32 v0, s0, s0, v0
-; GFX1310-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1310-SDAG-NEXT: v_add_co_ci_u32_e64 v1, null, s1, 0, s0
; GFX1310-SDAG-NEXT: global_store_b128 v[0:1], v[2:5], off
; GFX1310-SDAG-NEXT: s_set_pc_i64 s[30:31]
@@ -1090,7 +1070,7 @@ define void @global_store_saddr_i8_zext_vgpr(ptr addrspace(1) inreg %sbase, ptr
; GFX1250-ISEL-NEXT: s_wait_xcnt 0x0
; GFX1250-ISEL-NEXT: v_mov_b64_e32 v[0:1], s[0:1]
; GFX1250-ISEL-NEXT: s_wait_loadcnt 0x0
-; GFX1250-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-ISEL-NEXT: v_add_co_u32 v0, vcc_lo, v0, v6
; GFX1250-ISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1250-ISEL-NEXT: global_store_b128 v[0:1], v[2:5], off
@@ -1106,7 +1086,7 @@ define void @global_store_saddr_i8_zext_vgpr(ptr addrspace(1) inreg %sbase, ptr
; GFX1310-ISEL-NEXT: global_load_b32 v6, v[0:1], off
; GFX1310-ISEL-NEXT: v_dual_mov_b32 v0, s0 :: v_dual_mov_b32 v1, s1
; GFX1310-ISEL-NEXT: s_wait_loadcnt 0x0
-; GFX1310-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1310-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1310-ISEL-NEXT: v_add_co_u32 v0, vcc_lo, v0, v6
; GFX1310-ISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1310-ISEL-NEXT: global_store_b128 v[0:1], v[2:5], off
@@ -1170,7 +1150,6 @@ define void @global_store_saddr_v4i32_zext_vgpr_offset_neg128(ptr addrspace(1) i
; GFX1310-SDAG-NEXT: s_wait_bvhcnt 0x0
; GFX1310-SDAG-NEXT: s_wait_kmcnt 0x0
; GFX1310-SDAG-NEXT: v_add_co_u32 v5, s0, s0, v0
-; GFX1310-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1310-SDAG-NEXT: v_add_co_ci_u32_e64 v6, null, s1, 0, s0
; GFX1310-SDAG-NEXT: global_store_b128 v[5:6], v[1:4], off offset:-128 scope:SCOPE_SE
; GFX1310-SDAG-NEXT: s_set_pc_i64 s[30:31]
@@ -1212,7 +1191,7 @@ define void @global_store_saddr_v4i32_zext_vgpr_offset_neg128(ptr addrspace(1) i
; GFX1250-ISEL-NEXT: v_dual_mov_b32 v6, v1 :: v_dual_mov_b32 v7, v2
; GFX1250-ISEL-NEXT: v_dual_mov_b32 v8, v3 :: v_dual_mov_b32 v9, v4
; GFX1250-ISEL-NEXT: v_mov_b64_e32 v[2:3], s[0:1]
-; GFX1250-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-ISEL-NEXT: v_add_co_u32 v0, vcc_lo, v2, v0
; GFX1250-ISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v3, vcc_lo
; GFX1250-ISEL-NEXT: global_store_b128 v[0:1], v[6:9], off offset:-128
@@ -1226,7 +1205,7 @@ define void @global_store_saddr_v4i32_zext_vgpr_offset_neg128(ptr addrspace(1) i
; GFX1310-ISEL-NEXT: s_wait_bvhcnt 0x0
; GFX1310-ISEL-NEXT: s_wait_kmcnt 0x0
; GFX1310-ISEL-NEXT: v_dual_mov_b32 v6, s1 :: v_dual_mov_b32 v5, s0
-; GFX1310-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1310-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1310-ISEL-NEXT: v_add_co_u32 v5, vcc_lo, v5, v0
; GFX1310-ISEL-NEXT: v_add_co_ci_u32_e64 v6, null, 0, v6, vcc_lo
; GFX1310-ISEL-NEXT: global_store_b128 v[5:6], v[1:4], off offset:-128 scope:SCOPE_SE
@@ -1297,7 +1276,6 @@ define void @global_store_saddr_i8_zext_vgpr_offset_2047(ptr addrspace(1) inreg
; GFX1310-SDAG-NEXT: global_load_b32 v0, v[0:1], off
; GFX1310-SDAG-NEXT: s_wait_loadcnt 0x0
; GFX1310-SDAG-NEXT: v_add_co_u32 v0, s0, s0, v0
-; GFX1310-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1310-SDAG-NEXT: v_add_co_ci_u32_e64 v1, null, s1, 0, s0
; GFX1310-SDAG-NEXT: global_store_b128 v[0:1], v[2:5], off offset:2047 scope:SCOPE_DEV
; GFX1310-SDAG-NEXT: s_set_pc_i64 s[30:31]
@@ -1344,7 +1322,7 @@ define void @global_store_saddr_i8_zext_vgpr_offset_2047(ptr addrspace(1) inreg
; GFX1250-ISEL-NEXT: s_wait_xcnt 0x0
; GFX1250-ISEL-NEXT: v_mov_b64_e32 v[0:1], s[0:1]
; GFX1250-ISEL-NEXT: s_wait_loadcnt 0x0
-; GFX1250-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-ISEL-NEXT: v_add_co_u32 v0, vcc_lo, v0, v6
; GFX1250-ISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1250-ISEL-NEXT: global_store_b128 v[0:1], v[2:5], off offset:2047 scope:SCOPE_DEV
@@ -1360,7 +1338,7 @@ define void @global_store_saddr_i8_zext_vgpr_offset_2047(ptr addrspace(1) inreg
; GFX1310-ISEL-NEXT: global_load_b32 v6, v[0:1], off
; GFX1310-ISEL-NEXT: v_dual_mov_b32 v0, s0 :: v_dual_mov_b32 v1, s1
; GFX1310-ISEL-NEXT: s_wait_loadcnt 0x0
-; GFX1310-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1310-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1310-ISEL-NEXT: v_add_co_u32 v0, vcc_lo, v0, v6
; GFX1310-ISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1310-ISEL-NEXT: global_store_b128 v[0:1], v[2:5], off offset:2047 scope:SCOPE_DEV
@@ -1432,7 +1410,6 @@ define void @global_store_saddr_i8_zext_vgpr_offset_neg2048(ptr addrspace(1) inr
; GFX1310-SDAG-NEXT: global_load_b32 v0, v[0:1], off
; GFX1310-SDAG-NEXT: s_wait_loadcnt 0x0
; GFX1310-SDAG-NEXT: v_add_co_u32 v0, s0, s0, v0
-; GFX1310-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1310-SDAG-NEXT: v_add_co_ci_u32_e64 v1, null, s1, 0, s0
; GFX1310-SDAG-NEXT: global_store_b128 v[0:1], v[2:5], off offset:-2048 scope:SCOPE_SYS
; GFX1310-SDAG-NEXT: s_set_pc_i64 s[30:31]
@@ -1479,7 +1456,7 @@ define void @global_store_saddr_i8_zext_vgpr_offset_neg2048(ptr addrspace(1) inr
; GFX1250-ISEL-NEXT: s_wait_xcnt 0x0
; GFX1250-ISEL-NEXT: v_mov_b64_e32 v[0:1], s[0:1]
; GFX1250-ISEL-NEXT: s_wait_loadcnt 0x0
-; GFX1250-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-ISEL-NEXT: v_add_co_u32 v0, vcc_lo, v0, v6
; GFX1250-ISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1250-ISEL-NEXT: global_store_b128 v[0:1], v[2:5], off offset:-2048 scope:SCOPE_SYS
@@ -1495,7 +1472,7 @@ define void @global_store_saddr_i8_zext_vgpr_offset_neg2048(ptr addrspace(1) inr
; GFX1310-ISEL-NEXT: global_load_b32 v6, v[0:1], off
; GFX1310-ISEL-NEXT: v_dual_mov_b32 v0, s0 :: v_dual_mov_b32 v1, s1
; GFX1310-ISEL-NEXT: s_wait_loadcnt 0x0
-; GFX1310-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1310-ISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1310-ISEL-NEXT: v_add_co_u32 v0, vcc_lo, v0, v6
; GFX1310-ISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1310-ISEL-NEXT: global_store_b128 v[0:1], v[2:5], off offset:-2048 scope:SCOPE_SYS
@@ -1615,7 +1592,6 @@ define amdgpu_kernel void @global_store_saddr_uniform_ptr_in_vgprs(i32 %voffset,
; GFX1310-SDAG-NEXT: v_dual_mov_b32 v2, s2 :: v_dual_mov_b32 v3, s3
; GFX1310-SDAG-NEXT: s_wait_dscnt 0x0
; GFX1310-SDAG-NEXT: v_add_co_u32 v4, vcc_lo, v0, s6
-; GFX1310-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1310-SDAG-NEXT: v_add_co_ci_u32_e64 v5, null, 0, v1, vcc_lo
; GFX1310-SDAG-NEXT: v_dual_mov_b32 v0, s0 :: v_dual_mov_b32 v1, s1
; GFX1310-SDAG-NEXT: global_store_b128 v[4:5], v[0:3], off
@@ -1835,7 +1811,6 @@ define amdgpu_kernel void @global_store_saddr_uniform_ptr_in_vgprs_immoffset(i32
; GFX1310-SDAG-NEXT: v_dual_mov_b32 v2, s2 :: v_dual_mov_b32 v3, s3
; GFX1310-SDAG-NEXT: s_wait_dscnt 0x0
; GFX1310-SDAG-NEXT: v_add_co_u32 v4, vcc_lo, v0, s6
-; GFX1310-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1310-SDAG-NEXT: v_add_co_ci_u32_e64 v5, null, 0, v1, vcc_lo
; GFX1310-SDAG-NEXT: v_dual_mov_b32 v0, s0 :: v_dual_mov_b32 v1, s1
; GFX1310-SDAG-NEXT: global_store_b128 v[4:5], v[0:3], off offset:-120 scope:SCOPE_SE
diff --git a/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.bvh8_intersect_ray.ll b/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.bvh8_intersect_ray.ll
index 7aa4816d264133..dfe2a84e790b9e 100644
--- a/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.bvh8_intersect_ray.ll
+++ b/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.bvh8_intersect_ray.ll
@@ -180,6 +180,7 @@ define amdgpu_ps <10 x float> @image_bvh8_intersect_ray_vvvvvv(i64 %node_ptr, fl
; GFX12-SDAG-NEXT: v_cmpx_eq_u64_e32 s[2:3], v[12:13]
; GFX12-SDAG-NEXT: s_wait_bvhcnt 0x0
; GFX12-SDAG-NEXT: image_bvh8_intersect_ray v[0:9], [v[24:25], v[26:27], v[21:23], v[18:20], v28], s[0:3]
+; GFX12-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-SDAG-NEXT: s_and_not1_wrexec_b32 s5, s5
; GFX12-SDAG-NEXT: ; implicit-def: $vgpr10_vgpr11_vgpr12_vgpr13
; GFX12-SDAG-NEXT: ; implicit-def: $vgpr24_vgpr25
@@ -212,7 +213,7 @@ define amdgpu_ps <10 x float> @image_bvh8_intersect_ray_vvvvvv(i64 %node_ptr, fl
; GFX12-GISEL-NEXT: v_cmp_eq_u64_e32 vcc_lo, s[4:5], v[10:11]
; GFX12-GISEL-NEXT: v_cmp_eq_u64_e64 s0, s[6:7], v[12:13]
; GFX12-GISEL-NEXT: s_and_b32 s0, vcc_lo, s0
-; GFX12-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX12-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_3) | instid1(SALU_CYCLE_1)
; GFX12-GISEL-NEXT: s_and_saveexec_b32 s0, s0
; GFX12-GISEL-NEXT: s_wait_bvhcnt 0x0
; GFX12-GISEL-NEXT: image_bvh8_intersect_ray v[0:9], [v[24:25], v[26:27], v[18:20], v[21:23], v28], s[4:7]
diff --git a/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.cluster.load.async.to.lds.ll b/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.cluster.load.async.to.lds.ll
index a54ab6563a61b4..ade4fabdd7a460 100644
--- a/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.cluster.load.async.to.lds.ll
+++ b/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.cluster.load.async.to.lds.ll
@@ -30,7 +30,6 @@ define amdgpu_ps void @cluster_load_async_to_lds_b8_vaddr(ptr addrspace(1) %gadd
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_readfirstlane_b32 s0, v3
; GFX1250-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, v0, 32
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1250-GISEL-NEXT: s_mov_b32 m0, s0
; GFX1250-GISEL-NEXT: cluster_load_async_to_lds_b8 v2, v[0:1], off offset:16 th:TH_LOAD_NT
@@ -55,7 +54,6 @@ define amdgpu_ps void @cluster_load_async_to_lds_b8_vaddr(ptr addrspace(1) %gadd
; GFX1250S-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250S-GISEL-NEXT: v_readfirstlane_b32 s0, v3
; GFX1250S-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, v0, 32
-; GFX1250S-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1250S-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1250S-GISEL-NEXT: s_mov_b32 m0, s0
; GFX1250S-GISEL-NEXT: cluster_load_async_to_lds_b8 v2, v[0:1], off offset:16 th:TH_LOAD_NT
@@ -85,7 +83,6 @@ define amdgpu_ps void @cluster_load_async_to_lds_b8_vaddr_imm_mask(ptr addrspace
; GFX1250-GISEL-NEXT: v_nop
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, v0, 32
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1250-GISEL-NEXT: s_mov_b32 m0, 15
; GFX1250-GISEL-NEXT: cluster_load_async_to_lds_b8 v2, v[0:1], off offset:16
@@ -109,7 +106,6 @@ define amdgpu_ps void @cluster_load_async_to_lds_b8_vaddr_imm_mask(ptr addrspace
; GFX1250S-GISEL-NEXT: v_nop
; GFX1250S-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250S-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, v0, 32
-; GFX1250S-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1250S-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1250S-GISEL-NEXT: s_mov_b32 m0, 15
; GFX1250S-GISEL-NEXT: cluster_load_async_to_lds_b8 v2, v[0:1], off offset:16
@@ -191,7 +187,6 @@ define amdgpu_ps void @cluster_load_async_to_lds_b32_vaddr(ptr addrspace(1) %gad
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_readfirstlane_b32 s0, v3
; GFX1250-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, v0, 32
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1250-GISEL-NEXT: s_mov_b32 m0, s0
; GFX1250-GISEL-NEXT: cluster_load_async_to_lds_b32 v2, v[0:1], off offset:16 th:TH_LOAD_HT scope:SCOPE_SE
@@ -216,7 +211,6 @@ define amdgpu_ps void @cluster_load_async_to_lds_b32_vaddr(ptr addrspace(1) %gad
; GFX1250S-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250S-GISEL-NEXT: v_readfirstlane_b32 s0, v3
; GFX1250S-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, v0, 32
-; GFX1250S-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1250S-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1250S-GISEL-NEXT: s_mov_b32 m0, s0
; GFX1250S-GISEL-NEXT: cluster_load_async_to_lds_b32 v2, v[0:1], off offset:16 th:TH_LOAD_HT scope:SCOPE_SE
@@ -246,7 +240,6 @@ define amdgpu_ps void @cluster_load_async_to_lds_b32_vaddr_imm_mask(ptr addrspac
; GFX1250-GISEL-NEXT: v_nop
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, v0, 32
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1250-GISEL-NEXT: s_mov_b32 m0, 15
; GFX1250-GISEL-NEXT: cluster_load_async_to_lds_b32 v2, v[0:1], off offset:16
@@ -270,7 +263,6 @@ define amdgpu_ps void @cluster_load_async_to_lds_b32_vaddr_imm_mask(ptr addrspac
; GFX1250S-GISEL-NEXT: v_nop
; GFX1250S-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250S-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, v0, 32
-; GFX1250S-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1250S-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1250S-GISEL-NEXT: s_mov_b32 m0, 15
; GFX1250S-GISEL-NEXT: cluster_load_async_to_lds_b32 v2, v[0:1], off offset:16
@@ -352,7 +344,6 @@ define amdgpu_ps void @cluster_load_async_to_lds_b64_vaddr(ptr addrspace(1) %gad
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_readfirstlane_b32 s0, v3
; GFX1250-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, v0, 32
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1250-GISEL-NEXT: s_mov_b32 m0, s0
; GFX1250-GISEL-NEXT: cluster_load_async_to_lds_b64 v2, v[0:1], off offset:16 th:TH_LOAD_NT_HT scope:SCOPE_DEV
@@ -377,7 +368,6 @@ define amdgpu_ps void @cluster_load_async_to_lds_b64_vaddr(ptr addrspace(1) %gad
; GFX1250S-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250S-GISEL-NEXT: v_readfirstlane_b32 s0, v3
; GFX1250S-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, v0, 32
-; GFX1250S-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1250S-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1250S-GISEL-NEXT: s_mov_b32 m0, s0
; GFX1250S-GISEL-NEXT: cluster_load_async_to_lds_b64 v2, v[0:1], off offset:16 th:TH_LOAD_NT_HT scope:SCOPE_DEV
@@ -407,7 +397,6 @@ define amdgpu_ps void @cluster_load_async_to_lds_b64_vaddr_imm_mask( ptr addrspa
; GFX1250-GISEL-NEXT: v_nop
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, v0, 32
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1250-GISEL-NEXT: s_movk_i32 m0, 0x7f
; GFX1250-GISEL-NEXT: cluster_load_async_to_lds_b64 v2, v[0:1], off offset:16
@@ -431,7 +420,6 @@ define amdgpu_ps void @cluster_load_async_to_lds_b64_vaddr_imm_mask( ptr addrspa
; GFX1250S-GISEL-NEXT: v_nop
; GFX1250S-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250S-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, v0, 32
-; GFX1250S-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1250S-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1250S-GISEL-NEXT: s_movk_i32 m0, 0x7f
; GFX1250S-GISEL-NEXT: cluster_load_async_to_lds_b64 v2, v[0:1], off offset:16
@@ -513,7 +501,6 @@ define amdgpu_ps void @cluster_load_async_to_lds_b128_vaddr(ptr addrspace(1) %ga
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_readfirstlane_b32 s0, v3
; GFX1250-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, v0, 32
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1250-GISEL-NEXT: s_mov_b32 m0, s0
; GFX1250-GISEL-NEXT: cluster_load_async_to_lds_b128 v2, v[0:1], off offset:16 th:TH_LOAD_BYPASS scope:SCOPE_SYS
@@ -538,7 +525,6 @@ define amdgpu_ps void @cluster_load_async_to_lds_b128_vaddr(ptr addrspace(1) %ga
; GFX1250S-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250S-GISEL-NEXT: v_readfirstlane_b32 s0, v3
; GFX1250S-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, v0, 32
-; GFX1250S-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1250S-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1250S-GISEL-NEXT: s_mov_b32 m0, s0
; GFX1250S-GISEL-NEXT: cluster_load_async_to_lds_b128 v2, v[0:1], off offset:16 th:TH_LOAD_BYPASS scope:SCOPE_SYS
@@ -568,7 +554,6 @@ define amdgpu_ps void @cluster_load_async_to_lds_b128_vaddr_imm_mask(ptr addrspa
; GFX1250-GISEL-NEXT: v_nop
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, v0, 32
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1250-GISEL-NEXT: s_movk_i32 m0, 0x7f
; GFX1250-GISEL-NEXT: cluster_load_async_to_lds_b128 v2, v[0:1], off offset:16
@@ -592,7 +577,6 @@ define amdgpu_ps void @cluster_load_async_to_lds_b128_vaddr_imm_mask(ptr addrspa
; GFX1250S-GISEL-NEXT: v_nop
; GFX1250S-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250S-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, v0, 32
-; GFX1250S-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1250S-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1250S-GISEL-NEXT: s_movk_i32 m0, 0x7f
; GFX1250S-GISEL-NEXT: cluster_load_async_to_lds_b128 v2, v[0:1], off offset:16
diff --git a/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.dead.ll b/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.dead.ll
index edd9543d861cc7..2d9d7cbd585b17 100644
--- a/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.dead.ll
+++ b/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.dead.ll
@@ -19,6 +19,7 @@ define i32 @dead_i32(i1 %cond, i32 %x, ptr addrspace(1) %ptr1) #0 {
; ASM-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; ASM-SDAG-NEXT: v_and_b32_e32 v1, 1, v4
; ASM-SDAG-NEXT: v_cmpx_eq_u32_e32 1, v1
+; ASM-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; ASM-SDAG-NEXT: s_cbranch_execz .LBB0_2
; ASM-SDAG-NEXT: ; %bb.1: ; %if.then
; ASM-SDAG-NEXT: v_add_nc_u32_e32 v0, 1, v0
@@ -42,6 +43,7 @@ define i32 @dead_i32(i1 %cond, i32 %x, ptr addrspace(1) %ptr1) #0 {
; ASM-GISEL-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; ASM-GISEL-TRUE16-NEXT: v_and_b32_e32 v1, 1, v4
; ASM-GISEL-TRUE16-NEXT: v_cmpx_ne_u32_e32 0, v1
+; ASM-GISEL-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; ASM-GISEL-TRUE16-NEXT: s_cbranch_execz .LBB0_2
; ASM-GISEL-TRUE16-NEXT: ; %bb.1: ; %if.then
; ASM-GISEL-TRUE16-NEXT: v_add_nc_u32_e32 v0, 1, v0
@@ -65,6 +67,7 @@ define i32 @dead_i32(i1 %cond, i32 %x, ptr addrspace(1) %ptr1) #0 {
; ASM-GISEL-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; ASM-GISEL-FAKE16-NEXT: v_and_b32_e32 v1, 1, v4
; ASM-GISEL-FAKE16-NEXT: v_cmpx_ne_u32_e32 0, v1
+; ASM-GISEL-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; ASM-GISEL-FAKE16-NEXT: s_cbranch_execz .LBB0_2
; ASM-GISEL-FAKE16-NEXT: ; %bb.1: ; %if.then
; ASM-GISEL-FAKE16-NEXT: v_add_nc_u32_e32 v0, 1, v0
@@ -104,6 +107,7 @@ define %trivial_types @dead_struct(i1 %cond, %trivial_types %x, ptr addrspace(1)
; ASM-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; ASM-SDAG-NEXT: v_and_b32_e32 v1, 1, v20
; ASM-SDAG-NEXT: v_cmpx_eq_u32_e32 1, v1
+; ASM-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; ASM-SDAG-NEXT: s_cbranch_execz .LBB1_2
; ASM-SDAG-NEXT: ; %bb.1: ; %if.then
; ASM-SDAG-NEXT: v_dual_mov_b32 v11, 0 :: v_dual_add_nc_u32 v0, 15, v19
@@ -122,6 +126,7 @@ define %trivial_types @dead_struct(i1 %cond, %trivial_types %x, ptr addrspace(1)
; ASM-SDAG-NEXT: .LBB1_2: ; %if.end
; ASM-SDAG-NEXT: s_wait_alu depctr_sa_sdst(0)
; ASM-SDAG-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; ASM-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; ASM-SDAG-NEXT: v_dual_mov_b32 v1, v2 :: v_dual_mov_b32 v2, v3
; ASM-SDAG-NEXT: v_dual_mov_b32 v3, v4 :: v_dual_mov_b32 v4, v5
; ASM-SDAG-NEXT: v_dual_mov_b32 v5, v6 :: v_dual_mov_b32 v6, v7
@@ -145,6 +150,7 @@ define %trivial_types @dead_struct(i1 %cond, %trivial_types %x, ptr addrspace(1)
; ASM-GISEL-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; ASM-GISEL-TRUE16-NEXT: v_and_b32_e32 v2, 1, v20
; ASM-GISEL-TRUE16-NEXT: v_cmpx_ne_u32_e32 0, v2
+; ASM-GISEL-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; ASM-GISEL-TRUE16-NEXT: s_cbranch_execz .LBB1_2
; ASM-GISEL-TRUE16-NEXT: ; %bb.1: ; %if.then
; ASM-GISEL-TRUE16-NEXT: s_mov_b32 s4, 0
@@ -168,6 +174,7 @@ define %trivial_types @dead_struct(i1 %cond, %trivial_types %x, ptr addrspace(1)
; ASM-GISEL-TRUE16-NEXT: .LBB1_2: ; %if.end
; ASM-GISEL-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; ASM-GISEL-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; ASM-GISEL-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; ASM-GISEL-TRUE16-NEXT: v_dual_mov_b32 v2, v3 :: v_dual_mov_b32 v3, v4
; ASM-GISEL-TRUE16-NEXT: v_dual_mov_b32 v4, v5 :: v_dual_mov_b32 v5, v6
; ASM-GISEL-TRUE16-NEXT: v_dual_mov_b32 v6, v7 :: v_dual_mov_b32 v7, v8
@@ -190,6 +197,7 @@ define %trivial_types @dead_struct(i1 %cond, %trivial_types %x, ptr addrspace(1)
; ASM-GISEL-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; ASM-GISEL-FAKE16-NEXT: v_and_b32_e32 v2, 1, v20
; ASM-GISEL-FAKE16-NEXT: v_cmpx_ne_u32_e32 0, v2
+; ASM-GISEL-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; ASM-GISEL-FAKE16-NEXT: s_cbranch_execz .LBB1_2
; ASM-GISEL-FAKE16-NEXT: ; %bb.1: ; %if.then
; ASM-GISEL-FAKE16-NEXT: s_mov_b32 s4, 0
@@ -213,6 +221,7 @@ define %trivial_types @dead_struct(i1 %cond, %trivial_types %x, ptr addrspace(1)
; ASM-GISEL-FAKE16-NEXT: .LBB1_2: ; %if.end
; ASM-GISEL-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; ASM-GISEL-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; ASM-GISEL-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; ASM-GISEL-FAKE16-NEXT: v_dual_mov_b32 v2, v3 :: v_dual_mov_b32 v3, v4
; ASM-GISEL-FAKE16-NEXT: v_dual_mov_b32 v4, v5 :: v_dual_mov_b32 v5, v6
; ASM-GISEL-FAKE16-NEXT: v_dual_mov_b32 v6, v7 :: v_dual_mov_b32 v7, v8
@@ -301,7 +310,7 @@ define [32 x i32] @dead_array(i1 %cond, [32 x i32] %x, ptr addrspace(1) %ptr1, i
; ASM-SDAG-NEXT: scratch_load_b32 v1, off, s32 offset:16
; ASM-SDAG-NEXT: s_mov_b32 s0, exec_lo
; ASM-SDAG-NEXT: v_and_b32_e32 v33, 1, v33
-; ASM-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; ASM-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; ASM-SDAG-NEXT: v_cmpx_eq_u32_e32 1, v33
; ASM-SDAG-NEXT: s_cbranch_execz .LBB2_2
; ASM-SDAG-NEXT: ; %bb.1: ; %if.then
@@ -391,7 +400,7 @@ define [32 x i32] @dead_array(i1 %cond, [32 x i32] %x, ptr addrspace(1) %ptr1, i
; ASM-GISEL-TRUE16-NEXT: scratch_load_b32 v35, off, s32 offset:16
; ASM-GISEL-TRUE16-NEXT: v_and_b32_e32 v32, 1, v32
; ASM-GISEL-TRUE16-NEXT: s_mov_b32 s0, exec_lo
-; ASM-GISEL-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; ASM-GISEL-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; ASM-GISEL-TRUE16-NEXT: v_cmpx_ne_u32_e32 0, v32
; ASM-GISEL-TRUE16-NEXT: s_cbranch_execz .LBB2_2
; ASM-GISEL-TRUE16-NEXT: ; %bb.1: ; %if.then
@@ -469,7 +478,7 @@ define [32 x i32] @dead_array(i1 %cond, [32 x i32] %x, ptr addrspace(1) %ptr1, i
; ASM-GISEL-FAKE16-NEXT: scratch_load_b32 v35, off, s32 offset:16
; ASM-GISEL-FAKE16-NEXT: v_and_b32_e32 v32, 1, v32
; ASM-GISEL-FAKE16-NEXT: s_mov_b32 s0, exec_lo
-; ASM-GISEL-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; ASM-GISEL-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; ASM-GISEL-FAKE16-NEXT: v_cmpx_ne_u32_e32 0, v32
; ASM-GISEL-FAKE16-NEXT: s_cbranch_execz .LBB2_2
; ASM-GISEL-FAKE16-NEXT: ; %bb.1: ; %if.then
@@ -645,7 +654,7 @@ define %non_trivial_types @dead_non_trivial(i1 %cond, %non_trivial_types %x, ptr
; ASM-GISEL-TRUE16-NEXT: scratch_load_b32 v68, off, s32 offset:84
; ASM-GISEL-TRUE16-NEXT: v_and_b32_e32 v1, 1, v1
; ASM-GISEL-TRUE16-NEXT: s_mov_b32 s0, exec_lo
-; ASM-GISEL-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; ASM-GISEL-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; ASM-GISEL-TRUE16-NEXT: v_cmpx_ne_u32_e32 0, v1
; ASM-GISEL-TRUE16-NEXT: s_cbranch_execz .LBB3_2
; ASM-GISEL-TRUE16-NEXT: ; %bb.1: ; %if.then
@@ -795,7 +804,7 @@ define %non_trivial_types @dead_non_trivial(i1 %cond, %non_trivial_types %x, ptr
; ASM-GISEL-FAKE16-NEXT: scratch_load_b32 v68, off, s32 offset:84
; ASM-GISEL-FAKE16-NEXT: v_and_b32_e32 v1, 1, v1
; ASM-GISEL-FAKE16-NEXT: s_mov_b32 s0, exec_lo
-; ASM-GISEL-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; ASM-GISEL-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; ASM-GISEL-FAKE16-NEXT: v_cmpx_ne_u32_e32 0, v1
; ASM-GISEL-FAKE16-NEXT: s_cbranch_execz .LBB3_2
; ASM-GISEL-FAKE16-NEXT: ; %bb.1: ; %if.then
diff --git a/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.dual_intersect_ray.ll b/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.dual_intersect_ray.ll
index b843d8394f2282..6cede3181652cb 100644
--- a/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.dual_intersect_ray.ll
+++ b/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.dual_intersect_ray.ll
@@ -185,6 +185,7 @@ define amdgpu_ps <10 x float> @image_bvh_dual_intersect_ray_vvvvvv(i64 %node_ptr
; GFX12-SDAG-NEXT: v_cmpx_eq_u64_e32 s[2:3], v[13:14]
; GFX12-SDAG-NEXT: s_wait_bvhcnt 0x0
; GFX12-SDAG-NEXT: image_bvh_dual_intersect_ray v[0:9], [v[27:28], v[29:30], v[22:24], v[19:21], v[25:26]], s[0:3]
+; GFX12-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-SDAG-NEXT: s_and_not1_wrexec_b32 s5, s5
; GFX12-SDAG-NEXT: ; implicit-def: $vgpr11_vgpr12_vgpr13_vgpr14
; GFX12-SDAG-NEXT: ; implicit-def: $vgpr27_vgpr28
@@ -217,7 +218,7 @@ define amdgpu_ps <10 x float> @image_bvh_dual_intersect_ray_vvvvvv(i64 %node_ptr
; GFX12-GISEL-NEXT: v_cmp_eq_u64_e32 vcc_lo, s[4:5], v[11:12]
; GFX12-GISEL-NEXT: v_cmp_eq_u64_e64 s0, s[6:7], v[13:14]
; GFX12-GISEL-NEXT: s_and_b32 s0, vcc_lo, s0
-; GFX12-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX12-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_3) | instid1(SALU_CYCLE_1)
; GFX12-GISEL-NEXT: s_and_saveexec_b32 s0, s0
; GFX12-GISEL-NEXT: s_wait_bvhcnt 0x0
; GFX12-GISEL-NEXT: image_bvh_dual_intersect_ray v[0:9], [v[25:26], v[27:28], v[19:21], v[22:24], v[29:30]], s[4:7]
diff --git a/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.global.load.async.to.lds.ll b/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.global.load.async.to.lds.ll
index ecfcffc73753a7..3239616d10d770 100644
--- a/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.global.load.async.to.lds.ll
+++ b/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.global.load.async.to.lds.ll
@@ -27,7 +27,6 @@ define amdgpu_ps void @global_load_async_to_lds_b8_vaddr(ptr addrspace(1) %gaddr
; GFX1250-GISEL-NEXT: v_nop
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, v0, 32
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1250-GISEL-NEXT: global_load_async_to_lds_b8 v2, v[0:1], off offset:16 th:TH_LOAD_NT
; GFX1250-GISEL-NEXT: s_endpgm
@@ -35,7 +34,6 @@ define amdgpu_ps void @global_load_async_to_lds_b8_vaddr(ptr addrspace(1) %gaddr
; GFX13-LABEL: global_load_async_to_lds_b8_vaddr:
; GFX13: ; %bb.0: ; %entry
; GFX13-NEXT: v_add_co_u32 v0, vcc_lo, v0, 32
-; GFX13-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX13-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX13-NEXT: global_load_async_to_lds_b8 v2, v[0:1], off offset:16 th:TH_LOAD_NT
; GFX13-NEXT: s_endpgm
@@ -85,7 +83,6 @@ define amdgpu_ps void @global_load_async_to_lds_b32_vaddr(ptr addrspace(1) %gadd
; GFX1250-GISEL-NEXT: v_nop
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, v0, 32
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1250-GISEL-NEXT: global_load_async_to_lds_b32 v2, v[0:1], off offset:16 th:TH_LOAD_HT scope:SCOPE_SE
; GFX1250-GISEL-NEXT: s_endpgm
@@ -93,7 +90,6 @@ define amdgpu_ps void @global_load_async_to_lds_b32_vaddr(ptr addrspace(1) %gadd
; GFX13-LABEL: global_load_async_to_lds_b32_vaddr:
; GFX13: ; %bb.0: ; %entry
; GFX13-NEXT: v_add_co_u32 v0, vcc_lo, v0, 32
-; GFX13-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX13-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX13-NEXT: global_load_async_to_lds_b32 v2, v[0:1], off offset:16 th:TH_LOAD_HT scope:SCOPE_SE
; GFX13-NEXT: s_endpgm
@@ -143,7 +139,6 @@ define amdgpu_ps void @global_load_async_to_lds_b64_vaddr(ptr addrspace(1) %gadd
; GFX1250-GISEL-NEXT: v_nop
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, v0, 32
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1250-GISEL-NEXT: global_load_async_to_lds_b64 v2, v[0:1], off offset:16 th:TH_LOAD_NT_HT scope:SCOPE_DEV
; GFX1250-GISEL-NEXT: s_endpgm
@@ -151,7 +146,6 @@ define amdgpu_ps void @global_load_async_to_lds_b64_vaddr(ptr addrspace(1) %gadd
; GFX13-LABEL: global_load_async_to_lds_b64_vaddr:
; GFX13: ; %bb.0: ; %entry
; GFX13-NEXT: v_add_co_u32 v0, vcc_lo, v0, 32
-; GFX13-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX13-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX13-NEXT: global_load_async_to_lds_b64 v2, v[0:1], off offset:16 th:TH_LOAD_NT_HT scope:SCOPE_DEV
; GFX13-NEXT: s_endpgm
@@ -201,7 +195,6 @@ define amdgpu_ps void @global_load_async_to_lds_b128_vaddr(ptr addrspace(1) %gad
; GFX1250-GISEL-NEXT: v_nop
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, v0, 32
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1250-GISEL-NEXT: global_load_async_to_lds_b128 v2, v[0:1], off offset:16 th:TH_LOAD_BYPASS scope:SCOPE_SYS
; GFX1250-GISEL-NEXT: s_endpgm
@@ -209,7 +202,6 @@ define amdgpu_ps void @global_load_async_to_lds_b128_vaddr(ptr addrspace(1) %gad
; GFX13-LABEL: global_load_async_to_lds_b128_vaddr:
; GFX13: ; %bb.0: ; %entry
; GFX13-NEXT: v_add_co_u32 v0, vcc_lo, v0, 32
-; GFX13-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX13-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX13-NEXT: global_load_async_to_lds_b128 v2, v[0:1], off offset:16 th:TH_LOAD_BYPASS scope:SCOPE_SYS
; GFX13-NEXT: s_endpgm
@@ -303,7 +295,7 @@ define amdgpu_ps void @global_load_async_to_lds_b64_saddr_no_scale_offset(ptr ad
; GFX13-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX13-SDAG-NEXT: v_lshlrev_b64_e32 v[1:2], 2, v[1:2]
; GFX13-SDAG-NEXT: v_add_co_u32 v1, vcc_lo, s0, v1
-; GFX13-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX13-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX13-SDAG-NEXT: v_add_co_ci_u32_e64 v2, null, s1, v2, vcc_lo
; GFX13-SDAG-NEXT: global_load_async_to_lds_b64 v0, v[1:2], off offset:16 th:TH_LOAD_NT
; GFX13-SDAG-NEXT: s_endpgm
@@ -315,7 +307,7 @@ define amdgpu_ps void @global_load_async_to_lds_b64_saddr_no_scale_offset(ptr ad
; GFX13-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX13-GISEL-NEXT: v_lshlrev_b64_e32 v[1:2], 2, v[1:2]
; GFX13-GISEL-NEXT: v_add_co_u32 v1, vcc_lo, v3, v1
-; GFX13-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX13-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX13-GISEL-NEXT: v_add_co_ci_u32_e64 v2, null, v4, v2, vcc_lo
; GFX13-GISEL-NEXT: global_load_async_to_lds_b64 v0, v[1:2], off offset:16 th:TH_LOAD_NT
; GFX13-GISEL-NEXT: s_endpgm
diff --git a/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.global.store.async.from.lds.ll b/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.global.store.async.from.lds.ll
index 68a28b60688bc8..3ba59297f20e0a 100644
--- a/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.global.store.async.from.lds.ll
+++ b/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.global.store.async.from.lds.ll
@@ -25,7 +25,6 @@ define amdgpu_ps void @global_store_async_from_lds_b8_vaddr(ptr addrspace(1) %ga
; GFX1250-GISEL-NEXT: v_nop
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, v0, 32
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1250-GISEL-NEXT: global_store_async_from_lds_b8 v[0:1], v2, off offset:16 th:TH_STORE_NT
; GFX1250-GISEL-NEXT: s_endpgm
@@ -69,7 +68,6 @@ define amdgpu_ps void @global_store_async_from_lds_b32(ptr addrspace(1) %gaddr,
; GFX1250-GISEL-NEXT: v_nop
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, v0, 32
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1250-GISEL-NEXT: global_store_async_from_lds_b32 v[0:1], v2, off offset:16 th:TH_STORE_HT scope:SCOPE_SE
; GFX1250-GISEL-NEXT: s_endpgm
@@ -113,7 +111,6 @@ define amdgpu_ps void @global_store_async_from_lds_b64_vaddr(ptr addrspace(1) %g
; GFX1250-GISEL-NEXT: v_nop
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, v0, 32
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1250-GISEL-NEXT: global_store_async_from_lds_b64 v[0:1], v2, off offset:16 th:TH_STORE_NT_HT scope:SCOPE_DEV
; GFX1250-GISEL-NEXT: s_endpgm
@@ -157,7 +154,6 @@ define amdgpu_ps void @global_store_async_from_lds_b128_vaddr(ptr addrspace(1) %
; GFX1250-GISEL-NEXT: v_nop
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, v0, 32
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX1250-GISEL-NEXT: global_store_async_from_lds_b128 v[0:1], v2, off offset:16 th:TH_STORE_BYPASS scope:SCOPE_SYS
; GFX1250-GISEL-NEXT: s_endpgm
diff --git a/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.init.whole.wave-w32.ll b/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.init.whole.wave-w32.ll
index d48a2f85724adf..d94af1c90ff0fd 100644
--- a/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.init.whole.wave-w32.ll
+++ b/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.init.whole.wave-w32.ll
@@ -23,7 +23,7 @@ define amdgpu_cs_chain void @basic(<3 x i32> inreg %sgpr, ptr inreg %callee, i32
; GISEL12-NEXT: ; %bb.2: ; %tail
; GISEL12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GISEL12-NEXT: s_or_b32 exec_lo, exec_lo, s3
-; GISEL12-NEXT: s_delay_alu instid0(VALU_DEP_2)
+; GISEL12-NEXT: s_delay_alu instid0(VALU_DEP_2) | instid1(SALU_CYCLE_1)
; GISEL12-NEXT: v_add_nc_u32_e32 v11, 32, v12
; GISEL12-NEXT: s_mov_b32 exec_lo, s5
; GISEL12-NEXT: s_setpc_b64 s[6:7]
@@ -46,7 +46,7 @@ define amdgpu_cs_chain void @basic(<3 x i32> inreg %sgpr, ptr inreg %callee, i32
; DAGISEL12-NEXT: ; %bb.2: ; %tail
; DAGISEL12-NEXT: s_wait_alu depctr_sa_sdst(0)
; DAGISEL12-NEXT: s_or_b32 exec_lo, exec_lo, s3
-; DAGISEL12-NEXT: s_delay_alu instid0(VALU_DEP_2)
+; DAGISEL12-NEXT: s_delay_alu instid0(VALU_DEP_2) | instid1(SALU_CYCLE_1)
; DAGISEL12-NEXT: v_add_nc_u32_e32 v11, 32, v12
; DAGISEL12-NEXT: s_mov_b32 exec_lo, s5
; DAGISEL12-NEXT: s_setpc_b64 s[6:7]
@@ -121,7 +121,7 @@ define amdgpu_cs_chain void @wwm_in_shader(<3 x i32> inreg %sgpr, ptr inreg %cal
; GISEL12-NEXT: s_or_saveexec_b32 s4, -1
; GISEL12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GISEL12-NEXT: v_cndmask_b32_e64 v0, 0x47, v10, s4
-; GISEL12-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GISEL12-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GISEL12-NEXT: v_cmp_ne_u32_e64 s8, 0, v0
; GISEL12-NEXT: s_mov_b32 exec_lo, s4
; GISEL12-NEXT: v_dual_mov_b32 v11, s8 :: v_dual_add_nc_u32 v10, 42, v10
@@ -147,7 +147,7 @@ define amdgpu_cs_chain void @wwm_in_shader(<3 x i32> inreg %sgpr, ptr inreg %cal
; DAGISEL12-NEXT: s_or_saveexec_b32 s4, -1
; DAGISEL12-NEXT: s_wait_alu depctr_sa_sdst(0)
; DAGISEL12-NEXT: v_cndmask_b32_e64 v0, 0x47, v10, s4
-; DAGISEL12-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; DAGISEL12-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; DAGISEL12-NEXT: v_cmp_ne_u32_e64 s8, 0, v0
; DAGISEL12-NEXT: s_mov_b32 exec_lo, s4
; DAGISEL12-NEXT: v_dual_mov_b32 v11, s8 :: v_dual_add_nc_u32 v10, 42, v10
@@ -237,7 +237,7 @@ define amdgpu_cs_chain void @phi_whole_struct(<3 x i32> inreg %sgpr, ptr inreg %
; GISEL12-NEXT: s_or_saveexec_b32 s4, -1
; GISEL12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GISEL12-NEXT: v_cndmask_b32_e64 v0, 0x47, v12, s4
-; GISEL12-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GISEL12-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GISEL12-NEXT: v_cmp_ne_u32_e64 s8, 0, v0
; GISEL12-NEXT: s_mov_b32 exec_lo, s4
; GISEL12-NEXT: v_dual_mov_b32 v11, s8 :: v_dual_add_nc_u32 v10, 42, v12
@@ -262,7 +262,7 @@ define amdgpu_cs_chain void @phi_whole_struct(<3 x i32> inreg %sgpr, ptr inreg %
; DAGISEL12-NEXT: s_or_saveexec_b32 s4, -1
; DAGISEL12-NEXT: s_wait_alu depctr_sa_sdst(0)
; DAGISEL12-NEXT: v_cndmask_b32_e64 v0, 0x47, v12, s4
-; DAGISEL12-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; DAGISEL12-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; DAGISEL12-NEXT: v_cmp_ne_u32_e64 s8, 0, v0
; DAGISEL12-NEXT: s_mov_b32 exec_lo, s4
; DAGISEL12-NEXT: v_dual_mov_b32 v11, s8 :: v_dual_add_nc_u32 v10, 42, v12
@@ -359,6 +359,7 @@ define amdgpu_cs_chain void @control_flow(<3 x i32> inreg %sgpr, ptr inreg %call
; GISEL12-NEXT: v_cndmask_b32_e64 v0, 0x47, v1, s8
; GISEL12-NEXT: v_cmp_ne_u32_e64 s9, 0, v0
; GISEL12-NEXT: s_mov_b32 exec_lo, s8
+; GISEL12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_4) | instid1(SALU_CYCLE_1)
; GISEL12-NEXT: v_cmp_eq_u32_e32 vcc_lo, v13, v1
; GISEL12-NEXT: v_mov_b32_e32 v11, s9
; GISEL12-NEXT: s_or_b32 s4, vcc_lo, s4
@@ -367,11 +368,12 @@ define amdgpu_cs_chain void @control_flow(<3 x i32> inreg %sgpr, ptr inreg %call
; GISEL12-NEXT: s_cbranch_execnz .LBB3_2
; GISEL12-NEXT: ; %bb.3: ; %tail.loopexit
; GISEL12-NEXT: s_or_b32 exec_lo, exec_lo, s4
+; GISEL12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GISEL12-NEXT: v_add_nc_u32_e32 v10, 43, v2
; GISEL12-NEXT: .LBB3_4: ; %Flow1
; GISEL12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GISEL12-NEXT: s_or_b32 exec_lo, exec_lo, s3
-; GISEL12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GISEL12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(VALU_DEP_1)
; GISEL12-NEXT: s_mov_b32 s4, exec_lo
; GISEL12-NEXT: ; implicit-def: $sgpr3
; GISEL12-NEXT: v_cmpx_lt_i32_e32 v12, v13
@@ -417,7 +419,7 @@ define amdgpu_cs_chain void @control_flow(<3 x i32> inreg %sgpr, ptr inreg %call
; DAGISEL12-NEXT: s_or_saveexec_b32 s8, -1
; DAGISEL12-NEXT: s_wait_alu depctr_sa_sdst(0)
; DAGISEL12-NEXT: v_cndmask_b32_e64 v0, 0x47, v1, s8
-; DAGISEL12-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; DAGISEL12-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; DAGISEL12-NEXT: v_cmp_ne_u32_e64 s9, 0, v0
; DAGISEL12-NEXT: s_mov_b32 exec_lo, s8
; DAGISEL12-NEXT: v_cmp_eq_u32_e32 vcc_lo, v13, v1
@@ -425,14 +427,16 @@ define amdgpu_cs_chain void @control_flow(<3 x i32> inreg %sgpr, ptr inreg %call
; DAGISEL12-NEXT: s_or_b32 s4, vcc_lo, s4
; DAGISEL12-NEXT: s_wait_alu depctr_sa_sdst(0)
; DAGISEL12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
+; DAGISEL12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; DAGISEL12-NEXT: s_cbranch_execnz .LBB3_2
; DAGISEL12-NEXT: ; %bb.3: ; %tail.loopexit
; DAGISEL12-NEXT: s_or_b32 exec_lo, exec_lo, s4
+; DAGISEL12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; DAGISEL12-NEXT: v_add_nc_u32_e32 v10, 42, v1
; DAGISEL12-NEXT: .LBB3_4: ; %Flow1
; DAGISEL12-NEXT: s_wait_alu depctr_sa_sdst(0)
; DAGISEL12-NEXT: s_or_b32 exec_lo, exec_lo, s3
-; DAGISEL12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; DAGISEL12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(VALU_DEP_1)
; DAGISEL12-NEXT: s_mov_b32 s3, exec_lo
; DAGISEL12-NEXT: ; implicit-def: $vgpr8
; DAGISEL12-NEXT: v_cmpx_lt_i32_e32 v12, v13
@@ -602,7 +606,7 @@ define amdgpu_cs_chain void @use_v0_7(<3 x i32> inreg %sgpr, ptr inreg %callee,
; GISEL12-NEXT: s_or_saveexec_b32 s4, -1
; GISEL12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GISEL12-NEXT: v_cndmask_b32_e64 v13, 0x47, v12, s4
-; GISEL12-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GISEL12-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GISEL12-NEXT: v_cmp_ne_u32_e64 s8, 0, v13
; GISEL12-NEXT: s_mov_b32 exec_lo, s4
; GISEL12-NEXT: v_dual_mov_b32 v11, s8 :: v_dual_add_nc_u32 v10, 42, v12
@@ -632,7 +636,7 @@ define amdgpu_cs_chain void @use_v0_7(<3 x i32> inreg %sgpr, ptr inreg %callee,
; DAGISEL12-NEXT: s_or_saveexec_b32 s4, -1
; DAGISEL12-NEXT: s_wait_alu depctr_sa_sdst(0)
; DAGISEL12-NEXT: v_cndmask_b32_e64 v13, 0x47, v12, s4
-; DAGISEL12-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; DAGISEL12-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; DAGISEL12-NEXT: v_cmp_ne_u32_e64 s8, 0, v13
; DAGISEL12-NEXT: s_mov_b32 exec_lo, s4
; DAGISEL12-NEXT: v_dual_mov_b32 v11, s8 :: v_dual_add_nc_u32 v10, 42, v12
@@ -743,6 +747,7 @@ define amdgpu_cs_chain void @wwm_write_to_arg_reg(<3 x i32> inreg %sgpr, ptr inr
; GISEL12-NEXT: v_dual_mov_b32 v38, v22 :: v_dual_mov_b32 v39, v23
; GISEL12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GISEL12-NEXT: s_mov_b32 exec_lo, s12
+; GISEL12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GISEL12-NEXT: s_and_saveexec_b32 s4, s9
; GISEL12-NEXT: s_cbranch_execz .LBB5_2
; GISEL12-NEXT: ; %bb.1: ; %shader
@@ -773,6 +778,7 @@ define amdgpu_cs_chain void @wwm_write_to_arg_reg(<3 x i32> inreg %sgpr, ptr inr
; GISEL12-NEXT: v_dual_mov_b32 v52, v12 :: v_dual_mov_b32 v53, v13
; GISEL12-NEXT: v_dual_mov_b32 v54, v14 :: v_dual_mov_b32 v55, v15
; GISEL12-NEXT: s_mov_b32 exec_lo, s9
+; GISEL12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GISEL12-NEXT: v_dual_mov_b32 v24, v40 :: v_dual_mov_b32 v25, v41
; GISEL12-NEXT: v_dual_mov_b32 v26, v42 :: v_dual_mov_b32 v27, v43
; GISEL12-NEXT: v_dual_mov_b32 v28, v44 :: v_dual_mov_b32 v29, v45
@@ -784,6 +790,7 @@ define amdgpu_cs_chain void @wwm_write_to_arg_reg(<3 x i32> inreg %sgpr, ptr inr
; GISEL12-NEXT: .LBB5_2: ; %tail
; GISEL12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GISEL12-NEXT: s_or_b32 exec_lo, exec_lo, s4
+; GISEL12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GISEL12-NEXT: v_dual_mov_b32 v8, v24 :: v_dual_mov_b32 v9, v25
; GISEL12-NEXT: v_dual_mov_b32 v10, v26 :: v_dual_mov_b32 v11, v27
; GISEL12-NEXT: v_dual_mov_b32 v12, v28 :: v_dual_mov_b32 v13, v29
@@ -853,6 +860,7 @@ define amdgpu_cs_chain void @wwm_write_to_arg_reg(<3 x i32> inreg %sgpr, ptr inr
; DAGISEL12-NEXT: v_dual_mov_b32 v52, v12 :: v_dual_mov_b32 v53, v13
; DAGISEL12-NEXT: v_dual_mov_b32 v54, v14 :: v_dual_mov_b32 v55, v15
; DAGISEL12-NEXT: s_mov_b32 exec_lo, s11
+; DAGISEL12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; DAGISEL12-NEXT: v_dual_mov_b32 v24, v40 :: v_dual_mov_b32 v25, v41
; DAGISEL12-NEXT: v_dual_mov_b32 v26, v42 :: v_dual_mov_b32 v27, v43
; DAGISEL12-NEXT: v_dual_mov_b32 v28, v44 :: v_dual_mov_b32 v29, v45
@@ -864,6 +872,7 @@ define amdgpu_cs_chain void @wwm_write_to_arg_reg(<3 x i32> inreg %sgpr, ptr inr
; DAGISEL12-NEXT: .LBB5_2: ; %tail
; DAGISEL12-NEXT: s_wait_alu depctr_sa_sdst(0)
; DAGISEL12-NEXT: s_or_b32 exec_lo, exec_lo, s10
+; DAGISEL12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; DAGISEL12-NEXT: v_dual_mov_b32 v8, v24 :: v_dual_mov_b32 v9, v25
; DAGISEL12-NEXT: v_dual_mov_b32 v10, v26 :: v_dual_mov_b32 v11, v27
; DAGISEL12-NEXT: v_dual_mov_b32 v12, v28 :: v_dual_mov_b32 v13, v29
diff --git a/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.init.whole.wave-w64.ll b/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.init.whole.wave-w64.ll
index 71708bd8705145..bb86cb3dc35b5c 100644
--- a/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.init.whole.wave-w64.ll
+++ b/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.init.whole.wave-w64.ll
@@ -25,7 +25,7 @@ define amdgpu_cs_chain void @basic(<3 x i32> inreg %sgpr, ptr inreg %callee, i64
; GISEL12-NEXT: s_or_saveexec_b64 s[10:11], -1
; GISEL12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GISEL12-NEXT: v_cndmask_b32_e64 v0, 0x47, v13, s[10:11]
-; GISEL12-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GISEL12-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GISEL12-NEXT: v_cmp_ne_u32_e64 s[12:13], 0, v0
; GISEL12-NEXT: s_mov_b64 exec, s[10:11]
; GISEL12-NEXT: v_add_nc_u32_e32 v10, 42, v13
@@ -54,7 +54,7 @@ define amdgpu_cs_chain void @basic(<3 x i32> inreg %sgpr, ptr inreg %callee, i64
; DAGISEL12-NEXT: s_or_saveexec_b64 s[10:11], -1
; DAGISEL12-NEXT: s_wait_alu depctr_sa_sdst(0)
; DAGISEL12-NEXT: v_cndmask_b32_e64 v0, 0x47, v13, s[10:11]
-; DAGISEL12-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; DAGISEL12-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; DAGISEL12-NEXT: v_cmp_ne_u32_e64 s[12:13], 0, v0
; DAGISEL12-NEXT: s_mov_b64 exec, s[10:11]
; DAGISEL12-NEXT: v_add_nc_u32_e32 v10, 42, v13
diff --git a/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.interp.inreg.ll b/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.interp.inreg.ll
index 8efa1133bc8c17..53aacbd99ca65c 100644
--- a/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.interp.inreg.ll
+++ b/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.interp.inreg.ll
@@ -17,13 +17,13 @@ define amdgpu_ps void @v_interp_f32(float inreg %i, float inreg %j, i32 inreg %m
; GFX11-NEXT: lds_param_load v0, attr0.y wait_vdst:15
; GFX11-NEXT: lds_param_load v1, attr1.x wait_vdst:15
; GFX11-NEXT: s_mov_b32 exec_lo, s3
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX11-NEXT: v_mov_b32_e32 v2, s0
; GFX11-NEXT: v_mov_b32_e32 v4, s1
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX11-NEXT: v_interp_p10_f32 v3, v0, v2, v0 wait_exp:1
; GFX11-NEXT: v_interp_p10_f32 v2, v1, v2, v1 wait_exp:0
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_interp_p2_f32 v5, v0, v4, v3 wait_exp:7
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_interp_p2_f32 v4, v1, v4, v5 wait_exp:7
; GFX11-NEXT: exp mrt0, v3, v2, v5, v4 done
; GFX11-NEXT: s_endpgm
@@ -36,13 +36,13 @@ define amdgpu_ps void @v_interp_f32(float inreg %i, float inreg %j, i32 inreg %m
; GFX12-NEXT: ds_param_load v0, attr0.y wait_va_vdst:15 wait_vm_vsrc:1
; GFX12-NEXT: ds_param_load v1, attr1.x wait_va_vdst:15 wait_vm_vsrc:1
; GFX12-NEXT: s_mov_b32 exec_lo, s3
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX12-NEXT: v_mov_b32_e32 v2, s0
; GFX12-NEXT: v_mov_b32_e32 v4, s1
-; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX12-NEXT: v_interp_p10_f32 v3, v0, v2, v0 wait_exp:1
; GFX12-NEXT: v_interp_p10_f32 v2, v1, v2, v1 wait_exp:0
+; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX12-NEXT: v_interp_p2_f32 v5, v0, v4, v3 wait_exp:7
-; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-NEXT: v_interp_p2_f32 v4, v1, v4, v5 wait_exp:7
; GFX12-NEXT: export mrt0, v3, v2, v5, v4 done
; GFX12-NEXT: s_endpgm
@@ -68,17 +68,17 @@ define amdgpu_ps void @v_interp_f32_many(float inreg %i, float inreg %j, i32 inr
; GFX11-NEXT: lds_param_load v2, attr2.x wait_vdst:15
; GFX11-NEXT: lds_param_load v3, attr3.x wait_vdst:15
; GFX11-NEXT: s_mov_b32 exec_lo, s3
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_dual_mov_b32 v4, s0 :: v_dual_mov_b32 v5, s1
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_3) | instid1(VALU_DEP_4)
; GFX11-NEXT: v_interp_p10_f32 v6, v0, v4, v0 wait_exp:3
; GFX11-NEXT: v_interp_p10_f32 v7, v1, v4, v1 wait_exp:2
; GFX11-NEXT: v_interp_p10_f32 v8, v2, v4, v2 wait_exp:1
; GFX11-NEXT: v_interp_p10_f32 v4, v3, v4, v3 wait_exp:0
-; GFX11-NEXT: v_interp_p2_f32 v6, v0, v5, v6 wait_exp:7
; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
+; GFX11-NEXT: v_interp_p2_f32 v6, v0, v5, v6 wait_exp:7
; GFX11-NEXT: v_interp_p2_f32 v7, v1, v5, v7 wait_exp:7
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
; GFX11-NEXT: v_interp_p2_f32 v8, v2, v5, v8 wait_exp:7
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_4)
; GFX11-NEXT: v_interp_p2_f32 v4, v3, v5, v4 wait_exp:7
; GFX11-NEXT: exp mrt0, v6, v7, v8, v4 done
; GFX11-NEXT: s_endpgm
@@ -93,17 +93,17 @@ define amdgpu_ps void @v_interp_f32_many(float inreg %i, float inreg %j, i32 inr
; GFX12-NEXT: ds_param_load v2, attr2.x wait_va_vdst:15 wait_vm_vsrc:1
; GFX12-NEXT: ds_param_load v3, attr3.x wait_va_vdst:15 wait_vm_vsrc:1
; GFX12-NEXT: s_mov_b32 exec_lo, s3
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX12-NEXT: v_dual_mov_b32 v4, s0 :: v_dual_mov_b32 v5, s1
-; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_3) | instid1(VALU_DEP_4)
; GFX12-NEXT: v_interp_p10_f32 v6, v0, v4, v0 wait_exp:3
; GFX12-NEXT: v_interp_p10_f32 v7, v1, v4, v1 wait_exp:2
; GFX12-NEXT: v_interp_p10_f32 v8, v2, v4, v2 wait_exp:1
; GFX12-NEXT: v_interp_p10_f32 v4, v3, v4, v3 wait_exp:0
-; GFX12-NEXT: v_interp_p2_f32 v6, v0, v5, v6 wait_exp:7
; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
+; GFX12-NEXT: v_interp_p2_f32 v6, v0, v5, v6 wait_exp:7
; GFX12-NEXT: v_interp_p2_f32 v7, v1, v5, v7 wait_exp:7
+; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
; GFX12-NEXT: v_interp_p2_f32 v8, v2, v5, v8 wait_exp:7
-; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_4)
; GFX12-NEXT: v_interp_p2_f32 v4, v3, v5, v4 wait_exp:7
; GFX12-NEXT: export mrt0, v6, v7, v8, v4 done
; GFX12-NEXT: s_endpgm
@@ -203,14 +203,15 @@ define amdgpu_ps half @v_interp_f16(float inreg %i, float inreg %j, i32 inreg %m
; GFX11-TRUE16-NEXT: s_mov_b32 m0, s2
; GFX11-TRUE16-NEXT: lds_param_load v1, attr0.x wait_vdst:15
; GFX11-TRUE16-NEXT: s_mov_b32 exec_lo, s3
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX11-TRUE16-NEXT: v_mov_b32_e32 v0, s0
; GFX11-TRUE16-NEXT: v_mov_b32_e32 v2, s1
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX11-TRUE16-NEXT: v_interp_p10_f16_f32 v3, v1.l, v0, v1.l wait_exp:0
; GFX11-TRUE16-NEXT: v_interp_p10_f16_f32 v4, v1.h, v0, v1.h wait_exp:7
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-TRUE16-NEXT: v_interp_p2_f16_f32 v0.l, v1.l, v2, v3 wait_exp:7
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_interp_p2_f16_f32 v0.h, v1.h, v2, v4 wait_exp:7
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_add_f16_e32 v0.l, v0.l, v0.h
; GFX11-TRUE16-NEXT: ; return to shader part epilog
;
@@ -221,14 +222,15 @@ define amdgpu_ps half @v_interp_f16(float inreg %i, float inreg %j, i32 inreg %m
; GFX11-FAKE16-NEXT: s_mov_b32 m0, s2
; GFX11-FAKE16-NEXT: lds_param_load v1, attr0.x wait_vdst:15
; GFX11-FAKE16-NEXT: s_mov_b32 exec_lo, s3
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v0, s0
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v2, s1
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX11-FAKE16-NEXT: v_interp_p10_f16_f32 v3, v1, v0, v1 wait_exp:0
; GFX11-FAKE16-NEXT: v_interp_p10_f16_f32 v0, v1, v0, v1 op_sel:[1,0,1,0] wait_exp:7
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-FAKE16-NEXT: v_interp_p2_f16_f32 v3, v1, v2, v3 wait_exp:7
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_interp_p2_f16_f32 v0, v1, v2, v0 op_sel:[1,0,0,0] wait_exp:7
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_add_f16_e32 v0, v3, v0
; GFX11-FAKE16-NEXT: ; return to shader part epilog
;
@@ -239,14 +241,15 @@ define amdgpu_ps half @v_interp_f16(float inreg %i, float inreg %j, i32 inreg %m
; GFX12-TRUE16-NEXT: s_mov_b32 m0, s2
; GFX12-TRUE16-NEXT: ds_param_load v1, attr0.x wait_va_vdst:15 wait_vm_vsrc:1
; GFX12-TRUE16-NEXT: s_mov_b32 exec_lo, s3
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX12-TRUE16-NEXT: v_mov_b32_e32 v0, s0
; GFX12-TRUE16-NEXT: v_mov_b32_e32 v2, s1
-; GFX12-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX12-TRUE16-NEXT: v_interp_p10_f16_f32 v3, v1.l, v0, v1.l wait_exp:0
; GFX12-TRUE16-NEXT: v_interp_p10_f16_f32 v4, v1.h, v0, v1.h wait_exp:7
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX12-TRUE16-NEXT: v_interp_p2_f16_f32 v0.l, v1.l, v2, v3 wait_exp:7
-; GFX12-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX12-TRUE16-NEXT: v_interp_p2_f16_f32 v0.h, v1.h, v2, v4 wait_exp:7
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-TRUE16-NEXT: v_add_f16_e32 v0.l, v0.l, v0.h
; GFX12-TRUE16-NEXT: ; return to shader part epilog
;
@@ -257,14 +260,15 @@ define amdgpu_ps half @v_interp_f16(float inreg %i, float inreg %j, i32 inreg %m
; GFX12-FAKE16-NEXT: s_mov_b32 m0, s2
; GFX12-FAKE16-NEXT: ds_param_load v1, attr0.x wait_va_vdst:15 wait_vm_vsrc:1
; GFX12-FAKE16-NEXT: s_mov_b32 exec_lo, s3
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX12-FAKE16-NEXT: v_mov_b32_e32 v0, s0
; GFX12-FAKE16-NEXT: v_mov_b32_e32 v2, s1
-; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX12-FAKE16-NEXT: v_interp_p10_f16_f32 v3, v1, v0, v1 wait_exp:0
; GFX12-FAKE16-NEXT: v_interp_p10_f16_f32 v0, v1, v0, v1 op_sel:[1,0,1,0] wait_exp:7
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX12-FAKE16-NEXT: v_interp_p2_f16_f32 v3, v1, v2, v3 wait_exp:7
-; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX12-FAKE16-NEXT: v_interp_p2_f16_f32 v0, v1, v2, v0 op_sel:[1,0,0,0] wait_exp:7
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-FAKE16-NEXT: v_add_f16_e32 v0, v3, v0
; GFX12-FAKE16-NEXT: ; return to shader part epilog
main_body:
@@ -285,14 +289,15 @@ define amdgpu_ps half @v_interp_rtz_f16(float inreg %i, float inreg %j, i32 inre
; GFX11-TRUE16-NEXT: s_mov_b32 m0, s2
; GFX11-TRUE16-NEXT: lds_param_load v1, attr0.x wait_vdst:15
; GFX11-TRUE16-NEXT: s_mov_b32 exec_lo, s3
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX11-TRUE16-NEXT: v_mov_b32_e32 v0, s0
; GFX11-TRUE16-NEXT: v_mov_b32_e32 v2, s1
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX11-TRUE16-NEXT: v_interp_p10_rtz_f16_f32 v3, v1.l, v0, v1.l wait_exp:0
; GFX11-TRUE16-NEXT: v_interp_p10_rtz_f16_f32 v4, v1.h, v0, v1.h wait_exp:7
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-TRUE16-NEXT: v_interp_p2_rtz_f16_f32 v0.l, v1.l, v2, v3 wait_exp:7
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_interp_p2_rtz_f16_f32 v0.h, v1.h, v2, v4 wait_exp:7
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_add_f16_e32 v0.l, v0.l, v0.h
; GFX11-TRUE16-NEXT: ; return to shader part epilog
;
@@ -303,14 +308,15 @@ define amdgpu_ps half @v_interp_rtz_f16(float inreg %i, float inreg %j, i32 inre
; GFX11-FAKE16-NEXT: s_mov_b32 m0, s2
; GFX11-FAKE16-NEXT: lds_param_load v1, attr0.x wait_vdst:15
; GFX11-FAKE16-NEXT: s_mov_b32 exec_lo, s3
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v0, s0
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v2, s1
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX11-FAKE16-NEXT: v_interp_p10_rtz_f16_f32 v3, v1, v0, v1 wait_exp:0
; GFX11-FAKE16-NEXT: v_interp_p10_rtz_f16_f32 v0, v1, v0, v1 op_sel:[1,0,1,0] wait_exp:7
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-FAKE16-NEXT: v_interp_p2_rtz_f16_f32 v3, v1, v2, v3 wait_exp:7
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_interp_p2_rtz_f16_f32 v0, v1, v2, v0 op_sel:[1,0,0,0] wait_exp:7
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_add_f16_e32 v0, v3, v0
; GFX11-FAKE16-NEXT: ; return to shader part epilog
;
@@ -321,14 +327,15 @@ define amdgpu_ps half @v_interp_rtz_f16(float inreg %i, float inreg %j, i32 inre
; GFX12-TRUE16-NEXT: s_mov_b32 m0, s2
; GFX12-TRUE16-NEXT: ds_param_load v1, attr0.x wait_va_vdst:15 wait_vm_vsrc:1
; GFX12-TRUE16-NEXT: s_mov_b32 exec_lo, s3
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX12-TRUE16-NEXT: v_mov_b32_e32 v0, s0
; GFX12-TRUE16-NEXT: v_mov_b32_e32 v2, s1
-; GFX12-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX12-TRUE16-NEXT: v_interp_p10_rtz_f16_f32 v3, v1.l, v0, v1.l wait_exp:0
; GFX12-TRUE16-NEXT: v_interp_p10_rtz_f16_f32 v4, v1.h, v0, v1.h wait_exp:7
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX12-TRUE16-NEXT: v_interp_p2_rtz_f16_f32 v0.l, v1.l, v2, v3 wait_exp:7
-; GFX12-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX12-TRUE16-NEXT: v_interp_p2_rtz_f16_f32 v0.h, v1.h, v2, v4 wait_exp:7
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-TRUE16-NEXT: v_add_f16_e32 v0.l, v0.l, v0.h
; GFX12-TRUE16-NEXT: ; return to shader part epilog
;
@@ -339,14 +346,15 @@ define amdgpu_ps half @v_interp_rtz_f16(float inreg %i, float inreg %j, i32 inre
; GFX12-FAKE16-NEXT: s_mov_b32 m0, s2
; GFX12-FAKE16-NEXT: ds_param_load v1, attr0.x wait_va_vdst:15 wait_vm_vsrc:1
; GFX12-FAKE16-NEXT: s_mov_b32 exec_lo, s3
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX12-FAKE16-NEXT: v_mov_b32_e32 v0, s0
; GFX12-FAKE16-NEXT: v_mov_b32_e32 v2, s1
-; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX12-FAKE16-NEXT: v_interp_p10_rtz_f16_f32 v3, v1, v0, v1 wait_exp:0
; GFX12-FAKE16-NEXT: v_interp_p10_rtz_f16_f32 v0, v1, v0, v1 op_sel:[1,0,1,0] wait_exp:7
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX12-FAKE16-NEXT: v_interp_p2_rtz_f16_f32 v3, v1, v2, v3 wait_exp:7
-; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX12-FAKE16-NEXT: v_interp_p2_rtz_f16_f32 v0, v1, v2, v0 op_sel:[1,0,0,0] wait_exp:7
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-FAKE16-NEXT: v_add_f16_e32 v0, v3, v0
; GFX12-FAKE16-NEXT: ; return to shader part epilog
main_body:
diff --git a/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.intersect_ray.ll b/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.intersect_ray.ll
index 173739fa8f35f4..eebfadb7be059d 100644
--- a/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.intersect_ray.ll
+++ b/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.intersect_ray.ll
@@ -779,7 +779,6 @@ define amdgpu_kernel void @image_bvh_intersect_ray_nsa_reassign(ptr %p_node_ptr,
; GFX11-SDAG-NEXT: v_mov_b32_e32 v4, 4.0
; GFX11-SDAG-NEXT: s_waitcnt lgkmcnt(0)
; GFX11-SDAG-NEXT: v_add_co_u32 v0, s0, s0, v2
-; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-SDAG-NEXT: v_add_co_ci_u32_e64 v1, null, s1, 0, s0
; GFX11-SDAG-NEXT: v_add_co_u32 v2, s0, s2, v2
; GFX11-SDAG-NEXT: v_add_co_ci_u32_e64 v3, null, s3, 0, s0
@@ -815,11 +814,10 @@ define amdgpu_kernel void @image_bvh_intersect_ray_nsa_reassign(ptr %p_node_ptr,
; GFX11-GISEL-NEXT: v_mov_b32_e32 v0, s0
; GFX11-GISEL-NEXT: s_mov_b32 s1, 1.0
; GFX11-GISEL-NEXT: s_mov_b32 s0, 0
-; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, v0, v4
; GFX11-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-GISEL-NEXT: v_add_co_u32 v2, vcc_lo, v2, v4
-; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-GISEL-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX11-GISEL-NEXT: flat_load_b32 v9, v[0:1]
; GFX11-GISEL-NEXT: flat_load_b32 v10, v[2:3]
@@ -880,7 +878,7 @@ define amdgpu_kernel void @image_bvh_intersect_ray_nsa_reassign(ptr %p_node_ptr,
; GFX12-GISEL-NEXT: v_mov_b32_e32 v0, s0
; GFX12-GISEL-NEXT: s_mov_b32 s1, 1.0
; GFX12-GISEL-NEXT: s_mov_b32 s0, 0
-; GFX12-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX12-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, v0, v4
; GFX12-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX12-GISEL-NEXT: v_add_co_u32 v2, vcc_lo, v2, v4
@@ -1027,7 +1025,6 @@ define amdgpu_kernel void @image_bvh_intersect_ray_a16_nsa_reassign(ptr %p_node_
; GFX11-SDAG-NEXT: v_lshlrev_b32_e32 v2, 2, v0
; GFX11-SDAG-NEXT: s_waitcnt lgkmcnt(0)
; GFX11-SDAG-NEXT: v_add_co_u32 v0, s0, s0, v2
-; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-SDAG-NEXT: v_add_co_ci_u32_e64 v1, null, s1, 0, s0
; GFX11-SDAG-NEXT: v_add_co_u32 v2, s0, s2, v2
; GFX11-SDAG-NEXT: v_add_co_ci_u32_e64 v3, null, s3, 0, s0
@@ -1214,7 +1211,6 @@ define amdgpu_kernel void @image_bvh64_intersect_ray_nsa_reassign(ptr %p_ray, <4
; GFX11-SDAG-NEXT: v_bfrev_b32_e32 v10, 4.0
; GFX11-SDAG-NEXT: s_waitcnt lgkmcnt(0)
; GFX11-SDAG-NEXT: v_add_co_u32 v0, s4, s6, v0
-; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-SDAG-NEXT: v_add_co_ci_u32_e64 v1, null, s7, 0, s4
; GFX11-SDAG-NEXT: flat_load_b32 v11, v[0:1]
; GFX11-SDAG-NEXT: v_mov_b32_e32 v0, 0x40c00000
@@ -1249,7 +1245,7 @@ define amdgpu_kernel void @image_bvh64_intersect_ray_nsa_reassign(ptr %p_ray, <4
; GFX11-GISEL-NEXT: v_dual_mov_b32 v1, s7 :: v_dual_lshlrev_b32 v2, 2, v0
; GFX11-GISEL-NEXT: v_mov_b32_e32 v0, s6
; GFX11-GISEL-NEXT: s_mov_b32 s6, 2.0
-; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_3)
; GFX11-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
; GFX11-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-GISEL-NEXT: v_mov_b32_e32 v2, s6
@@ -1276,7 +1272,6 @@ define amdgpu_kernel void @image_bvh64_intersect_ray_nsa_reassign(ptr %p_ray, <4
; GFX12-SDAG-NEXT: v_bfrev_b32_e32 v10, 4.0
; GFX12-SDAG-NEXT: s_wait_kmcnt 0x0
; GFX12-SDAG-NEXT: v_add_co_u32 v0, s4, s6, v0
-; GFX12-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-SDAG-NEXT: v_add_co_ci_u32_e64 v1, null, s7, 0, s4
; GFX12-SDAG-NEXT: flat_load_b32 v11, v[0:1]
; GFX12-SDAG-NEXT: v_mov_b32_e32 v0, 0x40c00000
@@ -1311,7 +1306,7 @@ define amdgpu_kernel void @image_bvh64_intersect_ray_nsa_reassign(ptr %p_ray, <4
; GFX12-GISEL-NEXT: v_dual_mov_b32 v1, s7 :: v_dual_lshlrev_b32 v2, 2, v0
; GFX12-GISEL-NEXT: v_mov_b32_e32 v0, s6
; GFX12-GISEL-NEXT: s_mov_b32 s6, 2.0
-; GFX12-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX12-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_3)
; GFX12-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
; GFX12-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX12-GISEL-NEXT: s_wait_alu depctr_sa_sdst(0)
@@ -1457,7 +1452,6 @@ define amdgpu_kernel void @image_bvh64_intersect_ray_a16_nsa_reassign(ptr %p_ray
; GFX11-SDAG-NEXT: v_mov_b32_e32 v5, 2.0
; GFX11-SDAG-NEXT: s_waitcnt lgkmcnt(0)
; GFX11-SDAG-NEXT: v_add_co_u32 v0, s4, s6, v0
-; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-SDAG-NEXT: v_add_co_ci_u32_e64 v1, null, s7, 0, s4
; GFX11-SDAG-NEXT: flat_load_b32 v8, v[0:1]
; GFX11-SDAG-NEXT: v_mov_b32_e32 v0, 0x46004200
@@ -1483,7 +1477,6 @@ define amdgpu_kernel void @image_bvh64_intersect_ray_a16_nsa_reassign(ptr %p_ray
; GFX12-SDAG-NEXT: v_mov_b32_e32 v5, 2.0
; GFX12-SDAG-NEXT: s_wait_kmcnt 0x0
; GFX12-SDAG-NEXT: v_add_co_u32 v0, s4, s6, v0
-; GFX12-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-SDAG-NEXT: v_add_co_ci_u32_e64 v1, null, s7, 0, s4
; GFX12-SDAG-NEXT: flat_load_b32 v8, v[0:1]
; GFX12-SDAG-NEXT: v_mov_b32_e32 v0, 0x46004200
diff --git a/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.is.private.ll b/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.is.private.ll
index c9a6e388e4aaa5..0d1595c3f7bafd 100644
--- a/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.is.private.ll
+++ b/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.is.private.ll
@@ -198,9 +198,10 @@ define amdgpu_kernel void @is_private_sgpr(ptr %ptr) {
; GFX1250-SDAG-NEXT: s_load_b32 s0, s[4:5], 0x4 nv
; GFX1250-SDAG-NEXT: s_wait_kmcnt 0x0
; GFX1250-SDAG-NEXT: s_xor_b32 s0, s0, src_flat_scratch_base_hi
-; GFX1250-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
+; GFX1250-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1250-SDAG-NEXT: s_cmp_lt_u32 s0, 0x4000000
; GFX1250-SDAG-NEXT: s_cselect_b32 s0, 1, 0
+; GFX1250-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1250-SDAG-NEXT: s_cmp_lg_u32 s0, 1
; GFX1250-SDAG-NEXT: s_cbranch_scc1 .LBB1_2
; GFX1250-SDAG-NEXT: ; %bb.1: ; %bb0
@@ -262,6 +263,7 @@ define amdgpu_kernel void @is_private_sgpr(ptr %ptr) {
; GFX11-NEXT: s_mov_b64 s[0:1], src_private_base
; GFX11-NEXT: s_waitcnt lgkmcnt(0)
; GFX11-NEXT: s_cmp_lg_u32 s3, s1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB1_2
; GFX11-NEXT: ; %bb.1: ; %bb0
; GFX11-NEXT: v_mov_b32_e32 v0, 0
@@ -279,7 +281,7 @@ define amdgpu_kernel void @is_private_sgpr(ptr %ptr) {
; GFX1250-GISEL-NEXT: s_load_b64 s[0:1], s[4:5], 0x0 nv
; GFX1250-GISEL-NEXT: s_wait_kmcnt 0x0
; GFX1250-GISEL-NEXT: s_xor_b32 s0, s1, src_flat_scratch_base_hi
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1250-GISEL-NEXT: s_cmp_ge_u32 s0, 0x4000000
; GFX1250-GISEL-NEXT: s_cbranch_scc1 .LBB1_2
; GFX1250-GISEL-NEXT: ; %bb.1: ; %bb0
diff --git a/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.is.shared.ll b/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.is.shared.ll
index 273b534f600080..9b4af41e43faf6 100644
--- a/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.is.shared.ll
+++ b/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.is.shared.ll
@@ -265,9 +265,10 @@ define amdgpu_kernel void @is_local_sgpr(ptr %ptr) {
; GFX1250-SDAG-NEXT: s_load_b32 s0, s[4:5], 0x4 nv
; GFX1250-SDAG-NEXT: s_wait_kmcnt 0x0
; GFX1250-SDAG-NEXT: s_cmp_eq_u32 s0, s1
+; GFX1250-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1250-SDAG-NEXT: s_cselect_b32 s0, 1, 0
-; GFX1250-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1250-SDAG-NEXT: s_cmp_lg_u32 s0, 1
+; GFX1250-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1250-SDAG-NEXT: s_cbranch_scc1 .LBB1_2
; GFX1250-SDAG-NEXT: ; %bb.1: ; %bb0
; GFX1250-SDAG-NEXT: v_mov_b32_e32 v0, 0
@@ -328,6 +329,7 @@ define amdgpu_kernel void @is_local_sgpr(ptr %ptr) {
; GFX11-NEXT: s_mov_b64 s[0:1], src_shared_base
; GFX11-NEXT: s_waitcnt lgkmcnt(0)
; GFX11-NEXT: s_cmp_lg_u32 s3, s1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB1_2
; GFX11-NEXT: ; %bb.1: ; %bb0
; GFX11-NEXT: v_mov_b32_e32 v0, 0
@@ -346,6 +348,7 @@ define amdgpu_kernel void @is_local_sgpr(ptr %ptr) {
; GFX1250-GISEL-NEXT: s_mov_b64 s[0:1], src_shared_base
; GFX1250-GISEL-NEXT: s_wait_kmcnt 0x0
; GFX1250-GISEL-NEXT: s_cmp_lg_u32 s3, s1
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1250-GISEL-NEXT: s_cbranch_scc1 .LBB1_2
; GFX1250-GISEL-NEXT: ; %bb.1: ; %bb0
; GFX1250-GISEL-NEXT: v_mov_b32_e32 v0, 0
diff --git a/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.permlane.ll b/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.permlane.ll
index 99240b29d2a9ec..9871c95991e0c4 100644
--- a/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.permlane.ll
+++ b/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.permlane.ll
@@ -14175,6 +14175,7 @@ define amdgpu_kernel void @v_permlanex16_convergent(ptr addrspace(1) %out, i32 %
; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-NEXT: v_permlanex16_b32 v1, v1, s1, s2
; GFX11-NEXT: v_cmpx_eq_u32_e32 0, v0
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_cbranch_execz .LBB142_2
; GFX11-NEXT: ; %bb.1: ; %t
; GFX11-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
@@ -14193,6 +14194,7 @@ define amdgpu_kernel void @v_permlanex16_convergent(ptr addrspace(1) %out, i32 %
; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX12-NEXT: v_permlanex16_b32 v1, v1, s1, s2
; GFX12-NEXT: v_cmpx_eq_u32_e32 0, v0
+; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-NEXT: s_cbranch_execz .LBB142_2
; GFX12-NEXT: ; %bb.1: ; %t
; GFX12-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
@@ -14209,7 +14211,7 @@ define amdgpu_kernel void @v_permlanex16_convergent(ptr addrspace(1) %out, i32 %
; GFX13-NEXT: s_wait_kmcnt 0x0
; GFX13-NEXT: v_mov_b32_e32 v1, s0
; GFX13-NEXT: s_mov_b32 s0, exec_lo
-; GFX13-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX13-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX13-NEXT: v_permlanex16_b32 v1, v1, s1, s2
; GFX13-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX13-NEXT: s_cbranch_execz .LBB142_2
diff --git a/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.raw.atomic.buffer.load.ll b/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.raw.atomic.buffer.load.ll
index 37647c5acec4e9..61f6d3397ab3e4 100644
--- a/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.raw.atomic.buffer.load.ll
+++ b/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.raw.atomic.buffer.load.ll
@@ -21,7 +21,7 @@ define amdgpu_kernel void @raw_atomic_buffer_load_i32(<4 x i32> %addr) {
; GFX11-NEXT: s_waitcnt vmcnt(0)
; GFX11-NEXT: v_cmp_ne_u32_e32 vcc_lo, v1, v0
; GFX11-NEXT: s_or_b32 s4, vcc_lo, s4
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
; GFX11-NEXT: s_cbranch_execnz .LBB0_1
; GFX11-NEXT: ; %bb.2: ; %bb2
@@ -44,7 +44,7 @@ define amdgpu_kernel void @raw_atomic_buffer_load_i32(<4 x i32> %addr) {
; GFX12-NEXT: s_wait_loadcnt 0x0
; GFX12-NEXT: v_cmp_ne_u32_e32 vcc_lo, v1, v0
; GFX12-NEXT: s_or_b32 s4, vcc_lo, s4
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
; GFX12-NEXT: s_cbranch_execnz .LBB0_1
; GFX12-NEXT: ; %bb.2: ; %bb2
@@ -111,7 +111,7 @@ define amdgpu_kernel void @raw_atomic_buffer_load_i32_off(<4 x i32> %addr) {
; GFX11-NEXT: s_waitcnt vmcnt(0)
; GFX11-NEXT: v_cmp_ne_u32_e32 vcc_lo, v1, v0
; GFX11-NEXT: s_or_b32 s4, vcc_lo, s4
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
; GFX11-NEXT: s_cbranch_execnz .LBB1_1
; GFX11-NEXT: ; %bb.2: ; %bb2
@@ -134,7 +134,7 @@ define amdgpu_kernel void @raw_atomic_buffer_load_i32_off(<4 x i32> %addr) {
; GFX12-NEXT: s_wait_loadcnt 0x0
; GFX12-NEXT: v_cmp_ne_u32_e32 vcc_lo, v1, v0
; GFX12-NEXT: s_or_b32 s4, vcc_lo, s4
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
; GFX12-NEXT: s_cbranch_execnz .LBB1_1
; GFX12-NEXT: ; %bb.2: ; %bb2
@@ -200,7 +200,7 @@ define amdgpu_kernel void @raw_atomic_buffer_load_i32_soff(<4 x i32> %addr) {
; GFX11-NEXT: s_waitcnt vmcnt(0)
; GFX11-NEXT: v_cmp_ne_u32_e32 vcc_lo, v1, v0
; GFX11-NEXT: s_or_b32 s4, vcc_lo, s4
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
; GFX11-NEXT: s_cbranch_execnz .LBB2_1
; GFX11-NEXT: ; %bb.2: ; %bb2
@@ -224,7 +224,7 @@ define amdgpu_kernel void @raw_atomic_buffer_load_i32_soff(<4 x i32> %addr) {
; GFX12-NEXT: s_wait_loadcnt 0x0
; GFX12-NEXT: v_cmp_ne_u32_e32 vcc_lo, v1, v0
; GFX12-NEXT: s_or_b32 s4, vcc_lo, s4
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
; GFX12-NEXT: s_cbranch_execnz .LBB2_1
; GFX12-NEXT: ; %bb.2: ; %bb2
@@ -292,7 +292,7 @@ define amdgpu_kernel void @raw_atomic_buffer_load_i32_dlc(<4 x i32> %addr) {
; GFX11-NEXT: s_waitcnt vmcnt(0)
; GFX11-NEXT: v_cmp_ne_u32_e32 vcc_lo, v1, v0
; GFX11-NEXT: s_or_b32 s4, vcc_lo, s4
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
; GFX11-NEXT: s_cbranch_execnz .LBB3_1
; GFX11-NEXT: ; %bb.2: ; %bb2
@@ -315,7 +315,7 @@ define amdgpu_kernel void @raw_atomic_buffer_load_i32_dlc(<4 x i32> %addr) {
; GFX12-NEXT: s_wait_loadcnt 0x0
; GFX12-NEXT: v_cmp_ne_u32_e32 vcc_lo, v1, v0
; GFX12-NEXT: s_or_b32 s4, vcc_lo, s4
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
; GFX12-NEXT: s_cbranch_execnz .LBB3_1
; GFX12-NEXT: ; %bb.2: ; %bb2
@@ -385,6 +385,7 @@ define amdgpu_kernel void @raw_nonatomic_buffer_load_i32(<4 x i32> %addr) {
; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_or_b32 s0, s1, s0
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_execnz .LBB4_1
; GFX11-NEXT: ; %bb.2: ; %bb2
; GFX11-NEXT: s_endpgm
@@ -409,6 +410,7 @@ define amdgpu_kernel void @raw_nonatomic_buffer_load_i32(<4 x i32> %addr) {
; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_or_b32 s0, s1, s0
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB4_1
; GFX12-NEXT: ; %bb.2: ; %bb2
; GFX12-NEXT: s_endpgm
@@ -476,7 +478,7 @@ define amdgpu_kernel void @raw_atomic_buffer_load_i64(<4 x i32> %addr) {
; GFX11-NEXT: s_waitcnt vmcnt(0)
; GFX11-NEXT: v_cmp_ne_u64_e32 vcc_lo, v[2:3], v[0:1]
; GFX11-NEXT: s_or_b32 s4, vcc_lo, s4
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
; GFX11-NEXT: s_cbranch_execnz .LBB5_1
; GFX11-NEXT: ; %bb.2: ; %bb2
@@ -500,7 +502,7 @@ define amdgpu_kernel void @raw_atomic_buffer_load_i64(<4 x i32> %addr) {
; GFX12-SDAG-TRUE16-NEXT: s_wait_loadcnt 0x0
; GFX12-SDAG-TRUE16-NEXT: v_cmp_ne_u64_e32 vcc_lo, v[2:3], v[0:1]
; GFX12-SDAG-TRUE16-NEXT: s_or_b32 s4, vcc_lo, s4
-; GFX12-SDAG-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX12-SDAG-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-SDAG-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
; GFX12-SDAG-TRUE16-NEXT: s_cbranch_execnz .LBB5_1
; GFX12-SDAG-TRUE16-NEXT: ; %bb.2: ; %bb2
@@ -524,7 +526,7 @@ define amdgpu_kernel void @raw_atomic_buffer_load_i64(<4 x i32> %addr) {
; GFX12-SDAG-FAKE16-NEXT: s_wait_loadcnt 0x0
; GFX12-SDAG-FAKE16-NEXT: v_cmp_ne_u64_e32 vcc_lo, v[2:3], v[0:1]
; GFX12-SDAG-FAKE16-NEXT: s_or_b32 s4, vcc_lo, s4
-; GFX12-SDAG-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX12-SDAG-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-SDAG-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
; GFX12-SDAG-FAKE16-NEXT: s_cbranch_execnz .LBB5_1
; GFX12-SDAG-FAKE16-NEXT: ; %bb.2: ; %bb2
@@ -548,7 +550,7 @@ define amdgpu_kernel void @raw_atomic_buffer_load_i64(<4 x i32> %addr) {
; GFX12-GISEL-TRUE16-NEXT: s_wait_loadcnt 0x0
; GFX12-GISEL-TRUE16-NEXT: v_cmp_ne_u64_e32 vcc_lo, v[2:3], v[0:1]
; GFX12-GISEL-TRUE16-NEXT: s_or_b32 s4, vcc_lo, s4
-; GFX12-GISEL-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX12-GISEL-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-GISEL-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
; GFX12-GISEL-TRUE16-NEXT: s_cbranch_execnz .LBB5_1
; GFX12-GISEL-TRUE16-NEXT: ; %bb.2: ; %bb2
@@ -572,7 +574,7 @@ define amdgpu_kernel void @raw_atomic_buffer_load_i64(<4 x i32> %addr) {
; GFX12-GISEL-FAKE16-NEXT: s_wait_loadcnt 0x0
; GFX12-GISEL-FAKE16-NEXT: v_cmp_ne_u64_e32 vcc_lo, v[2:3], v[0:1]
; GFX12-GISEL-FAKE16-NEXT: s_or_b32 s4, vcc_lo, s4
-; GFX12-GISEL-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX12-GISEL-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-GISEL-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
; GFX12-GISEL-FAKE16-NEXT: s_cbranch_execnz .LBB5_1
; GFX12-GISEL-FAKE16-NEXT: ; %bb.2: ; %bb2
@@ -602,7 +604,7 @@ define amdgpu_kernel void @raw_atomic_buffer_load_v2i16(<4 x i32> %addr) {
; GFX11-NEXT: s_waitcnt vmcnt(0)
; GFX11-NEXT: v_cmp_ne_u32_e32 vcc_lo, v1, v0
; GFX11-NEXT: s_or_b32 s4, vcc_lo, s4
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
; GFX11-NEXT: s_cbranch_execnz .LBB6_1
; GFX11-NEXT: ; %bb.2: ; %bb2
@@ -625,7 +627,7 @@ define amdgpu_kernel void @raw_atomic_buffer_load_v2i16(<4 x i32> %addr) {
; GFX12-NEXT: s_wait_loadcnt 0x0
; GFX12-NEXT: v_cmp_ne_u32_e32 vcc_lo, v1, v0
; GFX12-NEXT: s_or_b32 s4, vcc_lo, s4
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
; GFX12-NEXT: s_cbranch_execnz .LBB6_1
; GFX12-NEXT: ; %bb.2: ; %bb2
@@ -696,6 +698,7 @@ define amdgpu_kernel void @raw_atomic_buffer_load_v4i16(<4 x i32> %addr) {
; GFX11-SDAG-TRUE16-NEXT: v_cmp_ne_u32_e32 vcc_lo, v1, v0
; GFX11-SDAG-TRUE16-NEXT: s_or_b32 s4, vcc_lo, s4
; GFX11-SDAG-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
+; GFX11-SDAG-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-SDAG-TRUE16-NEXT: s_cbranch_execnz .LBB7_1
; GFX11-SDAG-TRUE16-NEXT: ; %bb.2: ; %bb2
; GFX11-SDAG-TRUE16-NEXT: s_endpgm
@@ -715,7 +718,7 @@ define amdgpu_kernel void @raw_atomic_buffer_load_v4i16(<4 x i32> %addr) {
; GFX11-SDAG-FAKE16-NEXT: v_lshl_or_b32 v1, v2, 16, v1
; GFX11-SDAG-FAKE16-NEXT: v_cmp_ne_u32_e32 vcc_lo, v1, v0
; GFX11-SDAG-FAKE16-NEXT: s_or_b32 s4, vcc_lo, s4
-; GFX11-SDAG-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-SDAG-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-SDAG-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
; GFX11-SDAG-FAKE16-NEXT: s_cbranch_execnz .LBB7_1
; GFX11-SDAG-FAKE16-NEXT: ; %bb.2: ; %bb2
@@ -738,6 +741,7 @@ define amdgpu_kernel void @raw_atomic_buffer_load_v4i16(<4 x i32> %addr) {
; GFX11-GISEL-NEXT: v_cmp_ne_u32_e32 vcc_lo, s5, v0
; GFX11-GISEL-NEXT: s_or_b32 s4, vcc_lo, s4
; GFX11-GISEL-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
+; GFX11-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-GISEL-NEXT: s_cbranch_execnz .LBB7_1
; GFX11-GISEL-NEXT: ; %bb.2: ; %bb2
; GFX11-GISEL-NEXT: s_endpgm
@@ -762,6 +766,7 @@ define amdgpu_kernel void @raw_atomic_buffer_load_v4i16(<4 x i32> %addr) {
; GFX12-SDAG-TRUE16-NEXT: v_cmp_ne_u32_e32 vcc_lo, v2, v0
; GFX12-SDAG-TRUE16-NEXT: s_or_b32 s4, vcc_lo, s4
; GFX12-SDAG-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
+; GFX12-SDAG-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-SDAG-TRUE16-NEXT: s_cbranch_execnz .LBB7_1
; GFX12-SDAG-TRUE16-NEXT: ; %bb.2: ; %bb2
; GFX12-SDAG-TRUE16-NEXT: s_endpgm
@@ -786,7 +791,7 @@ define amdgpu_kernel void @raw_atomic_buffer_load_v4i16(<4 x i32> %addr) {
; GFX12-SDAG-FAKE16-NEXT: v_lshl_or_b32 v1, v3, 16, v1
; GFX12-SDAG-FAKE16-NEXT: v_cmp_ne_u32_e32 vcc_lo, v1, v0
; GFX12-SDAG-FAKE16-NEXT: s_or_b32 s4, vcc_lo, s4
-; GFX12-SDAG-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX12-SDAG-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-SDAG-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
; GFX12-SDAG-FAKE16-NEXT: s_cbranch_execnz .LBB7_1
; GFX12-SDAG-FAKE16-NEXT: ; %bb.2: ; %bb2
@@ -814,6 +819,7 @@ define amdgpu_kernel void @raw_atomic_buffer_load_v4i16(<4 x i32> %addr) {
; GFX12-GISEL-TRUE16-NEXT: v_cmp_ne_u32_e32 vcc_lo, s5, v0
; GFX12-GISEL-TRUE16-NEXT: s_or_b32 s4, vcc_lo, s4
; GFX12-GISEL-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
+; GFX12-GISEL-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-GISEL-TRUE16-NEXT: s_cbranch_execnz .LBB7_1
; GFX12-GISEL-TRUE16-NEXT: ; %bb.2: ; %bb2
; GFX12-GISEL-TRUE16-NEXT: s_endpgm
@@ -840,6 +846,7 @@ define amdgpu_kernel void @raw_atomic_buffer_load_v4i16(<4 x i32> %addr) {
; GFX12-GISEL-FAKE16-NEXT: v_cmp_ne_u32_e32 vcc_lo, s5, v0
; GFX12-GISEL-FAKE16-NEXT: s_or_b32 s4, vcc_lo, s4
; GFX12-GISEL-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
+; GFX12-GISEL-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-GISEL-FAKE16-NEXT: s_cbranch_execnz .LBB7_1
; GFX12-GISEL-FAKE16-NEXT: ; %bb.2: ; %bb2
; GFX12-GISEL-FAKE16-NEXT: s_endpgm
@@ -869,7 +876,7 @@ define amdgpu_kernel void @raw_atomic_buffer_load_v4i32(<4 x i32> %addr) {
; GFX11-NEXT: s_waitcnt vmcnt(0)
; GFX11-NEXT: v_cmp_ne_u32_e32 vcc_lo, v4, v0
; GFX11-NEXT: s_or_b32 s4, vcc_lo, s4
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
; GFX11-NEXT: s_cbranch_execnz .LBB8_1
; GFX11-NEXT: ; %bb.2: ; %bb2
@@ -892,7 +899,7 @@ define amdgpu_kernel void @raw_atomic_buffer_load_v4i32(<4 x i32> %addr) {
; GFX12-NEXT: s_wait_loadcnt 0x0
; GFX12-NEXT: v_cmp_ne_u32_e32 vcc_lo, v5, v0
; GFX12-NEXT: s_or_b32 s4, vcc_lo, s4
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
; GFX12-NEXT: s_cbranch_execnz .LBB8_1
; GFX12-NEXT: ; %bb.2: ; %bb2
@@ -962,7 +969,7 @@ define amdgpu_kernel void @raw_atomic_buffer_load_ptr(<4 x i32> %addr) {
; GFX11-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
; GFX11-NEXT: v_cmp_ne_u32_e32 vcc_lo, v1, v0
; GFX11-NEXT: s_or_b32 s4, vcc_lo, s4
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
; GFX11-NEXT: s_cbranch_execnz .LBB9_1
; GFX11-NEXT: ; %bb.2: ; %bb2
@@ -987,7 +994,7 @@ define amdgpu_kernel void @raw_atomic_buffer_load_ptr(<4 x i32> %addr) {
; GFX12-NEXT: s_wait_loadcnt_dscnt 0x0
; GFX12-NEXT: v_cmp_ne_u32_e32 vcc_lo, v1, v0
; GFX12-NEXT: s_or_b32 s4, vcc_lo, s4
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
; GFX12-NEXT: s_cbranch_execnz .LBB9_1
; GFX12-NEXT: ; %bb.2: ; %bb2
diff --git a/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.raw.ptr.atomic.buffer.load.ll b/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.raw.ptr.atomic.buffer.load.ll
index ca2679f6b155d7..a348f8ea7b9d61 100644
--- a/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.raw.ptr.atomic.buffer.load.ll
+++ b/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.raw.ptr.atomic.buffer.load.ll
@@ -21,7 +21,7 @@ define amdgpu_kernel void @raw_ptr_atomic_buffer_ptr_load_i32(ptr addrspace(8) %
; GFX11-NEXT: s_waitcnt vmcnt(0)
; GFX11-NEXT: v_cmp_ne_u32_e32 vcc_lo, v1, v0
; GFX11-NEXT: s_or_b32 s4, vcc_lo, s4
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
; GFX11-NEXT: s_cbranch_execnz .LBB0_1
; GFX11-NEXT: ; %bb.2: ; %bb2
@@ -44,7 +44,7 @@ define amdgpu_kernel void @raw_ptr_atomic_buffer_ptr_load_i32(ptr addrspace(8) %
; GFX12-NEXT: s_wait_loadcnt 0x0
; GFX12-NEXT: v_cmp_ne_u32_e32 vcc_lo, v1, v0
; GFX12-NEXT: s_or_b32 s4, vcc_lo, s4
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
; GFX12-NEXT: s_cbranch_execnz .LBB0_1
; GFX12-NEXT: ; %bb.2: ; %bb2
@@ -111,7 +111,7 @@ define amdgpu_kernel void @raw_ptr_atomic_buffer_load_i32_off(ptr addrspace(8) %
; GFX11-NEXT: s_waitcnt vmcnt(0)
; GFX11-NEXT: v_cmp_ne_u32_e32 vcc_lo, v1, v0
; GFX11-NEXT: s_or_b32 s4, vcc_lo, s4
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
; GFX11-NEXT: s_cbranch_execnz .LBB1_1
; GFX11-NEXT: ; %bb.2: ; %bb2
@@ -134,7 +134,7 @@ define amdgpu_kernel void @raw_ptr_atomic_buffer_load_i32_off(ptr addrspace(8) %
; GFX12-NEXT: s_wait_loadcnt 0x0
; GFX12-NEXT: v_cmp_ne_u32_e32 vcc_lo, v1, v0
; GFX12-NEXT: s_or_b32 s4, vcc_lo, s4
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
; GFX12-NEXT: s_cbranch_execnz .LBB1_1
; GFX12-NEXT: ; %bb.2: ; %bb2
@@ -200,7 +200,7 @@ define amdgpu_kernel void @raw_ptr_atomic_buffer_load_i32_soff(ptr addrspace(8)
; GFX11-NEXT: s_waitcnt vmcnt(0)
; GFX11-NEXT: v_cmp_ne_u32_e32 vcc_lo, v1, v0
; GFX11-NEXT: s_or_b32 s4, vcc_lo, s4
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
; GFX11-NEXT: s_cbranch_execnz .LBB2_1
; GFX11-NEXT: ; %bb.2: ; %bb2
@@ -224,7 +224,7 @@ define amdgpu_kernel void @raw_ptr_atomic_buffer_load_i32_soff(ptr addrspace(8)
; GFX12-NEXT: s_wait_loadcnt 0x0
; GFX12-NEXT: v_cmp_ne_u32_e32 vcc_lo, v1, v0
; GFX12-NEXT: s_or_b32 s4, vcc_lo, s4
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
; GFX12-NEXT: s_cbranch_execnz .LBB2_1
; GFX12-NEXT: ; %bb.2: ; %bb2
@@ -292,7 +292,7 @@ define amdgpu_kernel void @raw_ptr_atomic_buffer_load_i32_dlc(ptr addrspace(8) %
; GFX11-NEXT: s_waitcnt vmcnt(0)
; GFX11-NEXT: v_cmp_ne_u32_e32 vcc_lo, v1, v0
; GFX11-NEXT: s_or_b32 s4, vcc_lo, s4
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
; GFX11-NEXT: s_cbranch_execnz .LBB3_1
; GFX11-NEXT: ; %bb.2: ; %bb2
@@ -315,7 +315,7 @@ define amdgpu_kernel void @raw_ptr_atomic_buffer_load_i32_dlc(ptr addrspace(8) %
; GFX12-NEXT: s_wait_loadcnt 0x0
; GFX12-NEXT: v_cmp_ne_u32_e32 vcc_lo, v1, v0
; GFX12-NEXT: s_or_b32 s4, vcc_lo, s4
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
; GFX12-NEXT: s_cbranch_execnz .LBB3_1
; GFX12-NEXT: ; %bb.2: ; %bb2
@@ -385,6 +385,7 @@ define amdgpu_kernel void @raw_nonptr_atomic_buffer_load_i32(ptr addrspace(8) %p
; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_or_b32 s0, s1, s0
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_execnz .LBB4_1
; GFX11-NEXT: ; %bb.2: ; %bb2
; GFX11-NEXT: s_endpgm
@@ -409,6 +410,7 @@ define amdgpu_kernel void @raw_nonptr_atomic_buffer_load_i32(ptr addrspace(8) %p
; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_or_b32 s0, s1, s0
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB4_1
; GFX12-NEXT: ; %bb.2: ; %bb2
; GFX12-NEXT: s_endpgm
@@ -476,7 +478,7 @@ define amdgpu_kernel void @raw_ptr_atomic_buffer_load_i64(ptr addrspace(8) %ptr)
; GFX11-NEXT: s_waitcnt vmcnt(0)
; GFX11-NEXT: v_cmp_ne_u64_e32 vcc_lo, v[2:3], v[0:1]
; GFX11-NEXT: s_or_b32 s4, vcc_lo, s4
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
; GFX11-NEXT: s_cbranch_execnz .LBB5_1
; GFX11-NEXT: ; %bb.2: ; %bb2
@@ -500,7 +502,7 @@ define amdgpu_kernel void @raw_ptr_atomic_buffer_load_i64(ptr addrspace(8) %ptr)
; GFX12-SDAG-TRUE16-NEXT: s_wait_loadcnt 0x0
; GFX12-SDAG-TRUE16-NEXT: v_cmp_ne_u64_e32 vcc_lo, v[2:3], v[0:1]
; GFX12-SDAG-TRUE16-NEXT: s_or_b32 s4, vcc_lo, s4
-; GFX12-SDAG-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX12-SDAG-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-SDAG-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
; GFX12-SDAG-TRUE16-NEXT: s_cbranch_execnz .LBB5_1
; GFX12-SDAG-TRUE16-NEXT: ; %bb.2: ; %bb2
@@ -524,7 +526,7 @@ define amdgpu_kernel void @raw_ptr_atomic_buffer_load_i64(ptr addrspace(8) %ptr)
; GFX12-SDAG-FAKE16-NEXT: s_wait_loadcnt 0x0
; GFX12-SDAG-FAKE16-NEXT: v_cmp_ne_u64_e32 vcc_lo, v[2:3], v[0:1]
; GFX12-SDAG-FAKE16-NEXT: s_or_b32 s4, vcc_lo, s4
-; GFX12-SDAG-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX12-SDAG-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-SDAG-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
; GFX12-SDAG-FAKE16-NEXT: s_cbranch_execnz .LBB5_1
; GFX12-SDAG-FAKE16-NEXT: ; %bb.2: ; %bb2
@@ -548,7 +550,7 @@ define amdgpu_kernel void @raw_ptr_atomic_buffer_load_i64(ptr addrspace(8) %ptr)
; GFX12-GISEL-TRUE16-NEXT: s_wait_loadcnt 0x0
; GFX12-GISEL-TRUE16-NEXT: v_cmp_ne_u64_e32 vcc_lo, v[2:3], v[0:1]
; GFX12-GISEL-TRUE16-NEXT: s_or_b32 s4, vcc_lo, s4
-; GFX12-GISEL-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX12-GISEL-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-GISEL-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
; GFX12-GISEL-TRUE16-NEXT: s_cbranch_execnz .LBB5_1
; GFX12-GISEL-TRUE16-NEXT: ; %bb.2: ; %bb2
@@ -572,7 +574,7 @@ define amdgpu_kernel void @raw_ptr_atomic_buffer_load_i64(ptr addrspace(8) %ptr)
; GFX12-GISEL-FAKE16-NEXT: s_wait_loadcnt 0x0
; GFX12-GISEL-FAKE16-NEXT: v_cmp_ne_u64_e32 vcc_lo, v[2:3], v[0:1]
; GFX12-GISEL-FAKE16-NEXT: s_or_b32 s4, vcc_lo, s4
-; GFX12-GISEL-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX12-GISEL-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-GISEL-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
; GFX12-GISEL-FAKE16-NEXT: s_cbranch_execnz .LBB5_1
; GFX12-GISEL-FAKE16-NEXT: ; %bb.2: ; %bb2
@@ -602,7 +604,7 @@ define amdgpu_kernel void @raw_ptr_atomic_buffer_load_v2i16(ptr addrspace(8) %pt
; GFX11-NEXT: s_waitcnt vmcnt(0)
; GFX11-NEXT: v_cmp_ne_u32_e32 vcc_lo, v1, v0
; GFX11-NEXT: s_or_b32 s4, vcc_lo, s4
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
; GFX11-NEXT: s_cbranch_execnz .LBB6_1
; GFX11-NEXT: ; %bb.2: ; %bb2
@@ -625,7 +627,7 @@ define amdgpu_kernel void @raw_ptr_atomic_buffer_load_v2i16(ptr addrspace(8) %pt
; GFX12-NEXT: s_wait_loadcnt 0x0
; GFX12-NEXT: v_cmp_ne_u32_e32 vcc_lo, v1, v0
; GFX12-NEXT: s_or_b32 s4, vcc_lo, s4
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
; GFX12-NEXT: s_cbranch_execnz .LBB6_1
; GFX12-NEXT: ; %bb.2: ; %bb2
@@ -696,6 +698,7 @@ define amdgpu_kernel void @raw_ptr_atomic_buffer_load_v4i16(ptr addrspace(8) %pt
; GFX11-SDAG-TRUE16-NEXT: v_cmp_ne_u32_e32 vcc_lo, v1, v0
; GFX11-SDAG-TRUE16-NEXT: s_or_b32 s4, vcc_lo, s4
; GFX11-SDAG-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
+; GFX11-SDAG-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-SDAG-TRUE16-NEXT: s_cbranch_execnz .LBB7_1
; GFX11-SDAG-TRUE16-NEXT: ; %bb.2: ; %bb2
; GFX11-SDAG-TRUE16-NEXT: s_endpgm
@@ -715,7 +718,7 @@ define amdgpu_kernel void @raw_ptr_atomic_buffer_load_v4i16(ptr addrspace(8) %pt
; GFX11-SDAG-FAKE16-NEXT: v_lshl_or_b32 v1, v2, 16, v1
; GFX11-SDAG-FAKE16-NEXT: v_cmp_ne_u32_e32 vcc_lo, v1, v0
; GFX11-SDAG-FAKE16-NEXT: s_or_b32 s4, vcc_lo, s4
-; GFX11-SDAG-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-SDAG-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-SDAG-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
; GFX11-SDAG-FAKE16-NEXT: s_cbranch_execnz .LBB7_1
; GFX11-SDAG-FAKE16-NEXT: ; %bb.2: ; %bb2
@@ -738,6 +741,7 @@ define amdgpu_kernel void @raw_ptr_atomic_buffer_load_v4i16(ptr addrspace(8) %pt
; GFX11-GISEL-NEXT: v_cmp_ne_u32_e32 vcc_lo, s5, v0
; GFX11-GISEL-NEXT: s_or_b32 s4, vcc_lo, s4
; GFX11-GISEL-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
+; GFX11-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-GISEL-NEXT: s_cbranch_execnz .LBB7_1
; GFX11-GISEL-NEXT: ; %bb.2: ; %bb2
; GFX11-GISEL-NEXT: s_endpgm
@@ -762,6 +766,7 @@ define amdgpu_kernel void @raw_ptr_atomic_buffer_load_v4i16(ptr addrspace(8) %pt
; GFX12-SDAG-TRUE16-NEXT: v_cmp_ne_u32_e32 vcc_lo, v2, v0
; GFX12-SDAG-TRUE16-NEXT: s_or_b32 s4, vcc_lo, s4
; GFX12-SDAG-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
+; GFX12-SDAG-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-SDAG-TRUE16-NEXT: s_cbranch_execnz .LBB7_1
; GFX12-SDAG-TRUE16-NEXT: ; %bb.2: ; %bb2
; GFX12-SDAG-TRUE16-NEXT: s_endpgm
@@ -786,7 +791,7 @@ define amdgpu_kernel void @raw_ptr_atomic_buffer_load_v4i16(ptr addrspace(8) %pt
; GFX12-SDAG-FAKE16-NEXT: v_lshl_or_b32 v1, v3, 16, v1
; GFX12-SDAG-FAKE16-NEXT: v_cmp_ne_u32_e32 vcc_lo, v1, v0
; GFX12-SDAG-FAKE16-NEXT: s_or_b32 s4, vcc_lo, s4
-; GFX12-SDAG-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX12-SDAG-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-SDAG-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
; GFX12-SDAG-FAKE16-NEXT: s_cbranch_execnz .LBB7_1
; GFX12-SDAG-FAKE16-NEXT: ; %bb.2: ; %bb2
@@ -814,6 +819,7 @@ define amdgpu_kernel void @raw_ptr_atomic_buffer_load_v4i16(ptr addrspace(8) %pt
; GFX12-GISEL-TRUE16-NEXT: v_cmp_ne_u32_e32 vcc_lo, s5, v0
; GFX12-GISEL-TRUE16-NEXT: s_or_b32 s4, vcc_lo, s4
; GFX12-GISEL-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
+; GFX12-GISEL-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-GISEL-TRUE16-NEXT: s_cbranch_execnz .LBB7_1
; GFX12-GISEL-TRUE16-NEXT: ; %bb.2: ; %bb2
; GFX12-GISEL-TRUE16-NEXT: s_endpgm
@@ -840,6 +846,7 @@ define amdgpu_kernel void @raw_ptr_atomic_buffer_load_v4i16(ptr addrspace(8) %pt
; GFX12-GISEL-FAKE16-NEXT: v_cmp_ne_u32_e32 vcc_lo, s5, v0
; GFX12-GISEL-FAKE16-NEXT: s_or_b32 s4, vcc_lo, s4
; GFX12-GISEL-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
+; GFX12-GISEL-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-GISEL-FAKE16-NEXT: s_cbranch_execnz .LBB7_1
; GFX12-GISEL-FAKE16-NEXT: ; %bb.2: ; %bb2
; GFX12-GISEL-FAKE16-NEXT: s_endpgm
@@ -869,7 +876,7 @@ define amdgpu_kernel void @raw_ptr_atomic_buffer_load_v4i32(ptr addrspace(8) %pt
; GFX11-NEXT: s_waitcnt vmcnt(0)
; GFX11-NEXT: v_cmp_ne_u32_e32 vcc_lo, v4, v0
; GFX11-NEXT: s_or_b32 s4, vcc_lo, s4
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
; GFX11-NEXT: s_cbranch_execnz .LBB8_1
; GFX11-NEXT: ; %bb.2: ; %bb2
@@ -892,7 +899,7 @@ define amdgpu_kernel void @raw_ptr_atomic_buffer_load_v4i32(ptr addrspace(8) %pt
; GFX12-NEXT: s_wait_loadcnt 0x0
; GFX12-NEXT: v_cmp_ne_u32_e32 vcc_lo, v5, v0
; GFX12-NEXT: s_or_b32 s4, vcc_lo, s4
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
; GFX12-NEXT: s_cbranch_execnz .LBB8_1
; GFX12-NEXT: ; %bb.2: ; %bb2
@@ -962,7 +969,7 @@ define amdgpu_kernel void @raw_ptr_atomic_buffer_load_ptr(ptr addrspace(8) %ptr)
; GFX11-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
; GFX11-NEXT: v_cmp_ne_u32_e32 vcc_lo, v1, v0
; GFX11-NEXT: s_or_b32 s4, vcc_lo, s4
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
; GFX11-NEXT: s_cbranch_execnz .LBB9_1
; GFX11-NEXT: ; %bb.2: ; %bb2
@@ -987,7 +994,7 @@ define amdgpu_kernel void @raw_ptr_atomic_buffer_load_ptr(ptr addrspace(8) %ptr)
; GFX12-NEXT: s_wait_loadcnt_dscnt 0x0
; GFX12-NEXT: v_cmp_ne_u32_e32 vcc_lo, v1, v0
; GFX12-NEXT: s_or_b32 s4, vcc_lo, s4
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
; GFX12-NEXT: s_cbranch_execnz .LBB9_1
; GFX12-NEXT: ; %bb.2: ; %bb2
diff --git a/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.readfirstlane.m0.ll b/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.readfirstlane.m0.ll
index aa01a0ffb57826..8a3627e4202598 100644
--- a/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.readfirstlane.m0.ll
+++ b/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.readfirstlane.m0.ll
@@ -23,6 +23,7 @@ define void @test_readfirstlane_m0(i32 %arg) {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: v_readfirstlane_b32 s0, v0
; GFX11-NEXT: s_mov_b32 m0, s0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_sendmsg sendmsg(MSG_INTERRUPT)
; GFX11-NEXT: s_waitcnt lgkmcnt(0)
; GFX11-NEXT: s_setpc_b64 s[30:31]
diff --git a/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.reduce.add.ll b/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.reduce.add.ll
index 4d367c1eb3c06b..201604b9b2e986 100644
--- a/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.reduce.add.ll
+++ b/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.reduce.add.ll
@@ -489,6 +489,7 @@ define void @divergent_value_i16(ptr addrspace(1) %out, i16 %in) {
; GFX1164DAGISEL-NEXT: s_bitset0_b64 s[0:1], s3
; GFX1164DAGISEL-NEXT: s_add_i32 s2, s2, s4
; GFX1164DAGISEL-NEXT: s_cmp_lg_u64 s[0:1], 0
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164DAGISEL-NEXT: s_cbranch_scc1 .LBB1_1
; GFX1164DAGISEL-NEXT: ; %bb.2:
; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v2, s2
@@ -508,6 +509,7 @@ define void @divergent_value_i16(ptr addrspace(1) %out, i16 %in) {
; GFX1164GISEL-NEXT: s_bitset0_b64 s[0:1], s3
; GFX1164GISEL-NEXT: s_add_i32 s2, s2, s4
; GFX1164GISEL-NEXT: s_cmp_lg_u64 s[0:1], 0
+; GFX1164GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164GISEL-NEXT: s_cbranch_scc1 .LBB1_1
; GFX1164GISEL-NEXT: ; %bb.2:
; GFX1164GISEL-NEXT: v_mov_b32_e32 v2, s2
@@ -527,6 +529,7 @@ define void @divergent_value_i16(ptr addrspace(1) %out, i16 %in) {
; GFX1132DAGISEL-NEXT: s_bitset0_b32 s1, s2
; GFX1132DAGISEL-NEXT: s_add_i32 s0, s0, s3
; GFX1132DAGISEL-NEXT: s_cmp_lg_u32 s1, 0
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1132DAGISEL-NEXT: s_cbranch_scc1 .LBB1_1
; GFX1132DAGISEL-NEXT: ; %bb.2:
; GFX1132DAGISEL-NEXT: v_mov_b32_e32 v2, s0
@@ -546,6 +549,7 @@ define void @divergent_value_i16(ptr addrspace(1) %out, i16 %in) {
; GFX1132GISEL-NEXT: s_bitset0_b32 s1, s2
; GFX1132GISEL-NEXT: s_add_i32 s0, s0, s3
; GFX1132GISEL-NEXT: s_cmp_lg_u32 s1, 0
+; GFX1132GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1132GISEL-NEXT: s_cbranch_scc1 .LBB1_1
; GFX1132GISEL-NEXT: ; %bb.2:
; GFX1132GISEL-NEXT: v_mov_b32_e32 v2, s0
@@ -1004,6 +1008,7 @@ define amdgpu_kernel void @divergent_value(ptr addrspace(1) %out) #0 {
; GFX1164DAGISEL-NEXT: s_bitset0_b64 s[2:3], s5
; GFX1164DAGISEL-NEXT: s_add_i32 s4, s4, s6
; GFX1164DAGISEL-NEXT: s_cmp_lg_u64 s[2:3], 0
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164DAGISEL-NEXT: s_cbranch_scc1 .LBB3_1
; GFX1164DAGISEL-NEXT: ; %bb.2:
; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v0, s4
@@ -1024,6 +1029,7 @@ define amdgpu_kernel void @divergent_value(ptr addrspace(1) %out) #0 {
; GFX1164GISEL-NEXT: s_bitset0_b64 s[2:3], s5
; GFX1164GISEL-NEXT: s_add_i32 s4, s4, s6
; GFX1164GISEL-NEXT: s_cmp_lg_u64 s[2:3], 0
+; GFX1164GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164GISEL-NEXT: s_cbranch_scc1 .LBB3_1
; GFX1164GISEL-NEXT: ; %bb.2:
; GFX1164GISEL-NEXT: v_mov_b32_e32 v0, s4
@@ -1045,6 +1051,7 @@ define amdgpu_kernel void @divergent_value(ptr addrspace(1) %out) #0 {
; GFX1132DAGISEL-NEXT: s_bitset0_b32 s3, s4
; GFX1132DAGISEL-NEXT: s_add_i32 s2, s2, s5
; GFX1132DAGISEL-NEXT: s_cmp_lg_u32 s3, 0
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1132DAGISEL-NEXT: s_cbranch_scc1 .LBB3_1
; GFX1132DAGISEL-NEXT: ; %bb.2:
; GFX1132DAGISEL-NEXT: v_mov_b32_e32 v0, s2
@@ -1065,6 +1072,7 @@ define amdgpu_kernel void @divergent_value(ptr addrspace(1) %out) #0 {
; GFX1132GISEL-NEXT: s_bitset0_b32 s3, s4
; GFX1132GISEL-NEXT: s_add_i32 s2, s2, s5
; GFX1132GISEL-NEXT: s_cmp_lg_u32 s3, 0
+; GFX1132GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1132GISEL-NEXT: s_cbranch_scc1 .LBB3_1
; GFX1132GISEL-NEXT: ; %bb.2:
; GFX1132GISEL-NEXT: v_dual_mov_b32 v0, s2 :: v_dual_mov_b32 v1, 0
@@ -1583,7 +1591,7 @@ define amdgpu_kernel void @divergent_value_dpp(ptr addrspace(1) %out) #0 {
; GFX1164DAGISEL-NEXT: s_or_saveexec_b64 s[2:3], -1
; GFX1164DAGISEL-NEXT: s_waitcnt lgkmcnt(0)
; GFX1164DAGISEL-NEXT: v_add_nc_u32_e32 v1, v1, v2
-; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX1164DAGISEL-NEXT: v_readlane_b32 s4, v1, 63
; GFX1164DAGISEL-NEXT: s_mov_b64 exec, s[2:3]
; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v0, 0
@@ -1618,7 +1626,7 @@ define amdgpu_kernel void @divergent_value_dpp(ptr addrspace(1) %out) #0 {
; GFX1164GISEL-NEXT: s_mov_b64 exec, s[0:1]
; GFX1164GISEL-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
; GFX1164GISEL-NEXT: s_or_saveexec_b64 s[2:3], -1
-; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX1164GISEL-NEXT: v_readlane_b32 s4, v1, 63
; GFX1164GISEL-NEXT: s_mov_b64 exec, s[2:3]
; GFX1164GISEL-NEXT: v_mov_b32_e32 v0, s4
@@ -1645,7 +1653,7 @@ define amdgpu_kernel void @divergent_value_dpp(ptr addrspace(1) %out) #0 {
; GFX1132DAGISEL-NEXT: s_or_saveexec_b32 s2, -1
; GFX1132DAGISEL-NEXT: s_waitcnt lgkmcnt(0)
; GFX1132DAGISEL-NEXT: v_add_nc_u32_e32 v1, v1, v2
-; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX1132DAGISEL-NEXT: v_readlane_b32 s3, v1, 31
; GFX1132DAGISEL-NEXT: s_mov_b32 exec_lo, s2
; GFX1132DAGISEL-NEXT: v_dual_mov_b32 v0, 0 :: v_dual_mov_b32 v3, s3
@@ -1670,7 +1678,7 @@ define amdgpu_kernel void @divergent_value_dpp(ptr addrspace(1) %out) #0 {
; GFX1132GISEL-NEXT: s_mov_b32 exec_lo, s0
; GFX1132GISEL-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
; GFX1132GISEL-NEXT: s_or_saveexec_b32 s2, -1
-; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX1132GISEL-NEXT: v_readlane_b32 s3, v1, 31
; GFX1132GISEL-NEXT: s_mov_b32 exec_lo, s2
; GFX1132GISEL-NEXT: v_dual_mov_b32 v0, s3 :: v_dual_mov_b32 v3, 0
@@ -2311,50 +2319,51 @@ define void @divergent_value_dpp_i64(ptr addrspace(1) %out, i64 %in) #0 {
; GFX1164DAGISEL-NEXT: scratch_store_b32 off, v7, s32 offset:12
; GFX1164DAGISEL-NEXT: scratch_store_b32 off, v8, s32 offset:16
; GFX1164DAGISEL-NEXT: s_mov_b64 exec, s[0:1]
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1164DAGISEL-NEXT: s_or_saveexec_b64 s[0:1], -1
-; GFX1164DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(VALU_DEP_3)
; GFX1164DAGISEL-NEXT: v_cndmask_b32_e64 v4, 0, v2, s[0:1]
; GFX1164DAGISEL-NEXT: v_cndmask_b32_e64 v5, 0, v3, s[0:1]
; GFX1164DAGISEL-NEXT: v_mbcnt_lo_u32_b32 v8, -1, 0
-; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v6, v4
; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v6, v4
; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v7, v5
-; GFX1164DAGISEL-NEXT: v_mbcnt_hi_u32_b32 v8, -1, v8
; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX1164DAGISEL-NEXT: v_mbcnt_hi_u32_b32 v8, -1, v8
; GFX1164DAGISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:1 row_mask:0xf bank_mask:0xf
-; GFX1164DAGISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:1 row_mask:0xf bank_mask:0xf
; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX1164DAGISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:1 row_mask:0xf bank_mask:0xf
; GFX1164DAGISEL-NEXT: v_add_nc_u32_e32 v8, 32, v8
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1164DAGISEL-NEXT: v_add_co_u32 v4, vcc, v4, v6
-; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1164DAGISEL-NEXT: v_mul_lo_u32 v8, 4, v8
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v6, v4
; GFX1164DAGISEL-NEXT: v_add_co_ci_u32_e64 v5, null, v5, v7, vcc
-; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1164DAGISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:2 row_mask:0xf bank_mask:0xf
-; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v7, v5
; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v7, v5
; GFX1164DAGISEL-NEXT: v_add_co_u32 v4, vcc, v4, v6
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1164DAGISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:2 row_mask:0xf bank_mask:0xf
-; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v6, v4
; GFX1164DAGISEL-NEXT: s_waitcnt_depctr depctr_va_vcc(0)
-; GFX1164DAGISEL-NEXT: v_add_co_ci_u32_e64 v5, null, v5, v7, vcc
; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX1164DAGISEL-NEXT: v_add_co_ci_u32_e64 v5, null, v5, v7, vcc
; GFX1164DAGISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:4 row_mask:0xf bank_mask:0xf
-; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v7, v5
; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v7, v5
; GFX1164DAGISEL-NEXT: v_add_co_u32 v4, vcc, v4, v6
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1164DAGISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:4 row_mask:0xf bank_mask:0xf
-; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v6, v4
; GFX1164DAGISEL-NEXT: s_waitcnt_depctr depctr_va_vcc(0)
-; GFX1164DAGISEL-NEXT: v_add_co_ci_u32_e64 v5, null, v5, v7, vcc
; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX1164DAGISEL-NEXT: v_add_co_ci_u32_e64 v5, null, v5, v7, vcc
; GFX1164DAGISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:8 row_mask:0xf bank_mask:0xf
-; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v7, v5
; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v7, v5
; GFX1164DAGISEL-NEXT: v_add_co_u32 v4, vcc, v4, v6
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX1164DAGISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:8 row_mask:0xf bank_mask:0xf
; GFX1164DAGISEL-NEXT: ds_swizzle_b32 v6, v4 offset:swizzle(BROADCAST,32,15)
; GFX1164DAGISEL-NEXT: s_waitcnt_depctr depctr_va_vcc(0)
@@ -2376,6 +2385,7 @@ define void @divergent_value_dpp_i64(ptr addrspace(1) %out, i64 %in) #0 {
; GFX1164DAGISEL-NEXT: v_add_co_ci_u32_e64 v5, null, v5, v7, vcc
; GFX1164DAGISEL-NEXT: v_readlane_b32 s3, v5, 63
; GFX1164DAGISEL-NEXT: s_mov_b64 exec, s[0:1]
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v2, s2
; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v3, s3
; GFX1164DAGISEL-NEXT: global_store_b64 v[0:1], v[2:3], off
@@ -2402,50 +2412,51 @@ define void @divergent_value_dpp_i64(ptr addrspace(1) %out, i64 %in) #0 {
; GFX1164GISEL-NEXT: scratch_store_b32 off, v7, s32 offset:12
; GFX1164GISEL-NEXT: scratch_store_b32 off, v8, s32 offset:16
; GFX1164GISEL-NEXT: s_mov_b64 exec, s[0:1]
+; GFX1164GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1164GISEL-NEXT: s_or_saveexec_b64 s[0:1], -1
-; GFX1164GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(VALU_DEP_3)
; GFX1164GISEL-NEXT: v_cndmask_b32_e64 v4, 0, v2, s[0:1]
; GFX1164GISEL-NEXT: v_cndmask_b32_e64 v5, 0, v3, s[0:1]
; GFX1164GISEL-NEXT: v_mbcnt_lo_u32_b32 v8, -1, 0
-; GFX1164GISEL-NEXT: v_mov_b32_e32 v6, v4
; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX1164GISEL-NEXT: v_mov_b32_e32 v6, v4
; GFX1164GISEL-NEXT: v_mov_b32_e32 v7, v5
-; GFX1164GISEL-NEXT: v_mbcnt_hi_u32_b32 v8, -1, v8
; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX1164GISEL-NEXT: v_mbcnt_hi_u32_b32 v8, -1, v8
; GFX1164GISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:1 row_mask:0xf bank_mask:0xf
-; GFX1164GISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:1 row_mask:0xf bank_mask:0xf
; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX1164GISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:1 row_mask:0xf bank_mask:0xf
; GFX1164GISEL-NEXT: v_add_nc_u32_e32 v8, 32, v8
+; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1164GISEL-NEXT: v_add_co_u32 v4, vcc, v4, v6
-; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1164GISEL-NEXT: v_mul_lo_u32 v8, 4, v8
+; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX1164GISEL-NEXT: v_mov_b32_e32 v6, v4
; GFX1164GISEL-NEXT: v_add_co_ci_u32_e64 v5, null, v5, v7, vcc
-; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1164GISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:2 row_mask:0xf bank_mask:0xf
-; GFX1164GISEL-NEXT: v_mov_b32_e32 v7, v5
; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX1164GISEL-NEXT: v_mov_b32_e32 v7, v5
; GFX1164GISEL-NEXT: v_add_co_u32 v4, vcc, v4, v6
+; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1164GISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:2 row_mask:0xf bank_mask:0xf
-; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX1164GISEL-NEXT: v_mov_b32_e32 v6, v4
; GFX1164GISEL-NEXT: s_waitcnt_depctr depctr_va_vcc(0)
-; GFX1164GISEL-NEXT: v_add_co_ci_u32_e64 v5, null, v5, v7, vcc
; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX1164GISEL-NEXT: v_add_co_ci_u32_e64 v5, null, v5, v7, vcc
; GFX1164GISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:4 row_mask:0xf bank_mask:0xf
-; GFX1164GISEL-NEXT: v_mov_b32_e32 v7, v5
; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX1164GISEL-NEXT: v_mov_b32_e32 v7, v5
; GFX1164GISEL-NEXT: v_add_co_u32 v4, vcc, v4, v6
+; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1164GISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:4 row_mask:0xf bank_mask:0xf
-; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX1164GISEL-NEXT: v_mov_b32_e32 v6, v4
; GFX1164GISEL-NEXT: s_waitcnt_depctr depctr_va_vcc(0)
-; GFX1164GISEL-NEXT: v_add_co_ci_u32_e64 v5, null, v5, v7, vcc
; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX1164GISEL-NEXT: v_add_co_ci_u32_e64 v5, null, v5, v7, vcc
; GFX1164GISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:8 row_mask:0xf bank_mask:0xf
-; GFX1164GISEL-NEXT: v_mov_b32_e32 v7, v5
; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX1164GISEL-NEXT: v_mov_b32_e32 v7, v5
; GFX1164GISEL-NEXT: v_add_co_u32 v4, vcc, v4, v6
+; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX1164GISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:8 row_mask:0xf bank_mask:0xf
; GFX1164GISEL-NEXT: ds_swizzle_b32 v6, v4 offset:swizzle(BROADCAST,32,15)
; GFX1164GISEL-NEXT: s_waitcnt_depctr depctr_va_vcc(0)
@@ -2467,6 +2478,7 @@ define void @divergent_value_dpp_i64(ptr addrspace(1) %out, i64 %in) #0 {
; GFX1164GISEL-NEXT: v_add_co_ci_u32_e64 v5, null, v5, v7, vcc
; GFX1164GISEL-NEXT: v_readlane_b32 s3, v5, 63
; GFX1164GISEL-NEXT: s_mov_b64 exec, s[0:1]
+; GFX1164GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164GISEL-NEXT: v_mov_b32_e32 v2, s2
; GFX1164GISEL-NEXT: v_mov_b32_e32 v3, s3
; GFX1164GISEL-NEXT: global_store_b64 v[0:1], v[2:3], off
@@ -2492,39 +2504,39 @@ define void @divergent_value_dpp_i64(ptr addrspace(1) %out, i64 %in) #0 {
; GFX1132DAGISEL-NEXT: scratch_store_b32 off, v6, s32 offset:8
; GFX1132DAGISEL-NEXT: scratch_store_b32 off, v7, s32 offset:12
; GFX1132DAGISEL-NEXT: s_mov_b32 exec_lo, s0
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1132DAGISEL-NEXT: s_or_saveexec_b32 s2, -1
-; GFX1132DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX1132DAGISEL-NEXT: v_cndmask_b32_e64 v4, 0, v2, s2
; GFX1132DAGISEL-NEXT: v_cndmask_b32_e64 v5, 0, v3, s2
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132DAGISEL-NEXT: v_dual_mov_b32 v6, v4 :: v_dual_mov_b32 v7, v5
-; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1132DAGISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:1 row_mask:0xf bank_mask:0xf
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1132DAGISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:1 row_mask:0xf bank_mask:0xf
-; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132DAGISEL-NEXT: v_add_co_u32 v4, vcc_lo, v4, v6
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132DAGISEL-NEXT: v_add_co_ci_u32_e64 v5, null, v5, v7, vcc_lo
-; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132DAGISEL-NEXT: v_dual_mov_b32 v6, v4 :: v_dual_mov_b32 v7, v5
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1132DAGISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:2 row_mask:0xf bank_mask:0xf
-; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1132DAGISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:2 row_mask:0xf bank_mask:0xf
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1132DAGISEL-NEXT: v_add_co_u32 v4, vcc_lo, v4, v6
-; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132DAGISEL-NEXT: v_add_co_ci_u32_e64 v5, null, v5, v7, vcc_lo
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132DAGISEL-NEXT: v_dual_mov_b32 v6, v4 :: v_dual_mov_b32 v7, v5
-; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1132DAGISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:4 row_mask:0xf bank_mask:0xf
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1132DAGISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:4 row_mask:0xf bank_mask:0xf
-; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132DAGISEL-NEXT: v_add_co_u32 v4, vcc_lo, v4, v6
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132DAGISEL-NEXT: v_add_co_ci_u32_e64 v5, null, v5, v7, vcc_lo
-; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132DAGISEL-NEXT: v_dual_mov_b32 v6, v4 :: v_dual_mov_b32 v7, v5
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1132DAGISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:8 row_mask:0xf bank_mask:0xf
-; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1132DAGISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:8 row_mask:0xf bank_mask:0xf
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1132DAGISEL-NEXT: v_add_co_u32 v4, vcc_lo, v4, v6
-; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1132DAGISEL-NEXT: v_add_co_ci_u32_e64 v5, null, v5, v7, vcc_lo
; GFX1132DAGISEL-NEXT: ds_swizzle_b32 v6, v4 offset:swizzle(BROADCAST,32,15)
; GFX1132DAGISEL-NEXT: ds_swizzle_b32 v7, v5 offset:swizzle(BROADCAST,32,15)
@@ -2536,6 +2548,7 @@ define void @divergent_value_dpp_i64(ptr addrspace(1) %out, i64 %in) #0 {
; GFX1132DAGISEL-NEXT: v_readlane_b32 s0, v4, 31
; GFX1132DAGISEL-NEXT: v_readlane_b32 s1, v5, 31
; GFX1132DAGISEL-NEXT: s_mov_b32 exec_lo, s2
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1132DAGISEL-NEXT: v_dual_mov_b32 v3, s1 :: v_dual_mov_b32 v2, s0
; GFX1132DAGISEL-NEXT: global_store_b64 v[0:1], v[2:3], off
; GFX1132DAGISEL-NEXT: s_xor_saveexec_b32 s0, -1
@@ -2558,39 +2571,39 @@ define void @divergent_value_dpp_i64(ptr addrspace(1) %out, i64 %in) #0 {
; GFX1132GISEL-NEXT: scratch_store_b32 off, v6, s32 offset:8
; GFX1132GISEL-NEXT: scratch_store_b32 off, v7, s32 offset:12
; GFX1132GISEL-NEXT: s_mov_b32 exec_lo, s0
+; GFX1132GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1132GISEL-NEXT: s_or_saveexec_b32 s2, -1
-; GFX1132GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX1132GISEL-NEXT: v_cndmask_b32_e64 v4, 0, v2, s2
; GFX1132GISEL-NEXT: v_cndmask_b32_e64 v5, 0, v3, s2
+; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132GISEL-NEXT: v_dual_mov_b32 v6, v4 :: v_dual_mov_b32 v7, v5
-; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1132GISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:1 row_mask:0xf bank_mask:0xf
+; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1132GISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:1 row_mask:0xf bank_mask:0xf
-; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132GISEL-NEXT: v_add_co_u32 v4, vcc_lo, v4, v6
+; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132GISEL-NEXT: v_add_co_ci_u32_e64 v5, null, v5, v7, vcc_lo
-; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132GISEL-NEXT: v_dual_mov_b32 v6, v4 :: v_dual_mov_b32 v7, v5
+; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1132GISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:2 row_mask:0xf bank_mask:0xf
-; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1132GISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:2 row_mask:0xf bank_mask:0xf
+; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1132GISEL-NEXT: v_add_co_u32 v4, vcc_lo, v4, v6
-; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132GISEL-NEXT: v_add_co_ci_u32_e64 v5, null, v5, v7, vcc_lo
+; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132GISEL-NEXT: v_dual_mov_b32 v6, v4 :: v_dual_mov_b32 v7, v5
-; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1132GISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:4 row_mask:0xf bank_mask:0xf
+; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1132GISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:4 row_mask:0xf bank_mask:0xf
-; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132GISEL-NEXT: v_add_co_u32 v4, vcc_lo, v4, v6
+; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132GISEL-NEXT: v_add_co_ci_u32_e64 v5, null, v5, v7, vcc_lo
-; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132GISEL-NEXT: v_dual_mov_b32 v6, v4 :: v_dual_mov_b32 v7, v5
+; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1132GISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:8 row_mask:0xf bank_mask:0xf
-; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1132GISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:8 row_mask:0xf bank_mask:0xf
+; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1132GISEL-NEXT: v_add_co_u32 v4, vcc_lo, v4, v6
-; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1132GISEL-NEXT: v_add_co_ci_u32_e64 v5, null, v5, v7, vcc_lo
; GFX1132GISEL-NEXT: ds_swizzle_b32 v6, v4 offset:swizzle(BROADCAST,32,15)
; GFX1132GISEL-NEXT: ds_swizzle_b32 v7, v5 offset:swizzle(BROADCAST,32,15)
@@ -2602,6 +2615,7 @@ define void @divergent_value_dpp_i64(ptr addrspace(1) %out, i64 %in) #0 {
; GFX1132GISEL-NEXT: v_readlane_b32 s0, v4, 31
; GFX1132GISEL-NEXT: v_readlane_b32 s1, v5, 31
; GFX1132GISEL-NEXT: s_mov_b32 exec_lo, s2
+; GFX1132GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1132GISEL-NEXT: v_dual_mov_b32 v3, s1 :: v_dual_mov_b32 v2, s0
; GFX1132GISEL-NEXT: global_store_b64 v[0:1], v[2:3], off
; GFX1132GISEL-NEXT: s_xor_saveexec_b32 s0, -1
@@ -2967,7 +2981,7 @@ define amdgpu_kernel void @default_stratergy(ptr addrspace(1) %out) #0 {
; GFX1164DAGISEL-NEXT: s_or_saveexec_b64 s[2:3], -1
; GFX1164DAGISEL-NEXT: s_waitcnt lgkmcnt(0)
; GFX1164DAGISEL-NEXT: v_add_nc_u32_e32 v1, v1, v2
-; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX1164DAGISEL-NEXT: v_readlane_b32 s4, v1, 63
; GFX1164DAGISEL-NEXT: s_mov_b64 exec, s[2:3]
; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v0, 0
@@ -3002,7 +3016,7 @@ define amdgpu_kernel void @default_stratergy(ptr addrspace(1) %out) #0 {
; GFX1164GISEL-NEXT: s_mov_b64 exec, s[0:1]
; GFX1164GISEL-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
; GFX1164GISEL-NEXT: s_or_saveexec_b64 s[2:3], -1
-; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX1164GISEL-NEXT: v_readlane_b32 s4, v1, 63
; GFX1164GISEL-NEXT: s_mov_b64 exec, s[2:3]
; GFX1164GISEL-NEXT: v_mov_b32_e32 v0, s4
@@ -3029,7 +3043,7 @@ define amdgpu_kernel void @default_stratergy(ptr addrspace(1) %out) #0 {
; GFX1132DAGISEL-NEXT: s_or_saveexec_b32 s2, -1
; GFX1132DAGISEL-NEXT: s_waitcnt lgkmcnt(0)
; GFX1132DAGISEL-NEXT: v_add_nc_u32_e32 v1, v1, v2
-; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX1132DAGISEL-NEXT: v_readlane_b32 s3, v1, 31
; GFX1132DAGISEL-NEXT: s_mov_b32 exec_lo, s2
; GFX1132DAGISEL-NEXT: v_dual_mov_b32 v0, 0 :: v_dual_mov_b32 v3, s3
@@ -3054,7 +3068,7 @@ define amdgpu_kernel void @default_stratergy(ptr addrspace(1) %out) #0 {
; GFX1132GISEL-NEXT: s_mov_b32 exec_lo, s0
; GFX1132GISEL-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
; GFX1132GISEL-NEXT: s_or_saveexec_b32 s2, -1
-; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX1132GISEL-NEXT: v_readlane_b32 s3, v1, 31
; GFX1132GISEL-NEXT: s_mov_b32 exec_lo, s2
; GFX1132GISEL-NEXT: v_dual_mov_b32 v0, s3 :: v_dual_mov_b32 v3, 0
@@ -3493,7 +3507,7 @@ define amdgpu_kernel void @divergent_cfg(ptr addrspace(1) %out, i32 %in) #0 {
; GFX1164DAGISEL-NEXT: v_and_b32_e32 v0, 0x3ff, v0
; GFX1164DAGISEL-NEXT: s_mov_b64 s[0:1], exec
; GFX1164DAGISEL-NEXT: ; implicit-def: $sgpr2
-; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164DAGISEL-NEXT: v_cmpx_lt_u32_e32 15, v0
; GFX1164DAGISEL-NEXT: s_xor_b64 s[0:1], exec, s[0:1]
; GFX1164DAGISEL-NEXT: s_cbranch_execz .LBB8_2
@@ -3509,13 +3523,14 @@ define amdgpu_kernel void @divergent_cfg(ptr addrspace(1) %out, i32 %in) #0 {
; GFX1164DAGISEL-NEXT: s_or_saveexec_b64 s[0:1], s[0:1]
; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v1, s2
; GFX1164DAGISEL-NEXT: s_xor_b64 exec, exec, s[0:1]
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164DAGISEL-NEXT: s_cbranch_execz .LBB8_6
; GFX1164DAGISEL-NEXT: ; %bb.3: ; %if
; GFX1164DAGISEL-NEXT: s_mov_b64 s[2:3], exec
; GFX1164DAGISEL-NEXT: s_mov_b32 s6, 0
; GFX1164DAGISEL-NEXT: .LBB8_4: ; =>This Inner Loop Header: Depth=1
; GFX1164DAGISEL-NEXT: s_ctz_i32_b64 s7, s[2:3]
-; GFX1164DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_3) | instid1(SALU_CYCLE_1)
; GFX1164DAGISEL-NEXT: v_readlane_b32 s8, v0, s7
; GFX1164DAGISEL-NEXT: s_bitset0_b64 s[2:3], s7
; GFX1164DAGISEL-NEXT: s_add_i32 s6, s6, s8
@@ -3536,7 +3551,7 @@ define amdgpu_kernel void @divergent_cfg(ptr addrspace(1) %out, i32 %in) #0 {
; GFX1164GISEL-NEXT: v_and_b32_e32 v0, 0x3ff, v0
; GFX1164GISEL-NEXT: s_mov_b64 s[0:1], exec
; GFX1164GISEL-NEXT: ; implicit-def: $sgpr2
-; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164GISEL-NEXT: v_cmpx_le_u32_e32 16, v0
; GFX1164GISEL-NEXT: s_xor_b64 s[0:1], exec, s[0:1]
; GFX1164GISEL-NEXT: s_cbranch_execz .LBB8_2
@@ -3552,13 +3567,14 @@ define amdgpu_kernel void @divergent_cfg(ptr addrspace(1) %out, i32 %in) #0 {
; GFX1164GISEL-NEXT: s_or_saveexec_b64 s[0:1], s[0:1]
; GFX1164GISEL-NEXT: v_mov_b32_e32 v1, s2
; GFX1164GISEL-NEXT: s_xor_b64 exec, exec, s[0:1]
+; GFX1164GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164GISEL-NEXT: s_cbranch_execz .LBB8_6
; GFX1164GISEL-NEXT: ; %bb.3: ; %if
; GFX1164GISEL-NEXT: s_mov_b64 s[2:3], exec
; GFX1164GISEL-NEXT: s_mov_b32 s6, 0
; GFX1164GISEL-NEXT: .LBB8_4: ; =>This Inner Loop Header: Depth=1
; GFX1164GISEL-NEXT: s_ctz_i32_b64 s7, s[2:3]
-; GFX1164GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1164GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_3) | instid1(SALU_CYCLE_1)
; GFX1164GISEL-NEXT: v_readlane_b32 s8, v0, s7
; GFX1164GISEL-NEXT: s_bitset0_b64 s[2:3], s7
; GFX1164GISEL-NEXT: s_add_i32 s6, s6, s8
@@ -3579,9 +3595,10 @@ define amdgpu_kernel void @divergent_cfg(ptr addrspace(1) %out, i32 %in) #0 {
; GFX1132DAGISEL-NEXT: v_and_b32_e32 v0, 0x3ff, v0
; GFX1132DAGISEL-NEXT: s_mov_b32 s0, exec_lo
; GFX1132DAGISEL-NEXT: ; implicit-def: $sgpr1
-; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132DAGISEL-NEXT: v_cmpx_lt_u32_e32 15, v0
; GFX1132DAGISEL-NEXT: s_xor_b32 s0, exec_lo, s0
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1132DAGISEL-NEXT: s_cbranch_execz .LBB8_2
; GFX1132DAGISEL-NEXT: ; %bb.1: ; %else
; GFX1132DAGISEL-NEXT: s_load_b32 s1, s[4:5], 0x2c
@@ -3595,13 +3612,14 @@ define amdgpu_kernel void @divergent_cfg(ptr addrspace(1) %out, i32 %in) #0 {
; GFX1132DAGISEL-NEXT: s_or_saveexec_b32 s0, s0
; GFX1132DAGISEL-NEXT: v_mov_b32_e32 v1, s1
; GFX1132DAGISEL-NEXT: s_xor_b32 exec_lo, exec_lo, s0
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1132DAGISEL-NEXT: s_cbranch_execz .LBB8_6
; GFX1132DAGISEL-NEXT: ; %bb.3: ; %if
; GFX1132DAGISEL-NEXT: s_mov_b32 s2, exec_lo
; GFX1132DAGISEL-NEXT: s_mov_b32 s1, 0
; GFX1132DAGISEL-NEXT: .LBB8_4: ; =>This Inner Loop Header: Depth=1
; GFX1132DAGISEL-NEXT: s_ctz_i32_b32 s3, s2
-; GFX1132DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_3) | instid1(SALU_CYCLE_1)
; GFX1132DAGISEL-NEXT: v_readlane_b32 s6, v0, s3
; GFX1132DAGISEL-NEXT: s_bitset0_b32 s2, s3
; GFX1132DAGISEL-NEXT: s_add_i32 s1, s1, s6
@@ -3622,9 +3640,10 @@ define amdgpu_kernel void @divergent_cfg(ptr addrspace(1) %out, i32 %in) #0 {
; GFX1132GISEL-NEXT: v_and_b32_e32 v0, 0x3ff, v0
; GFX1132GISEL-NEXT: s_mov_b32 s0, exec_lo
; GFX1132GISEL-NEXT: ; implicit-def: $sgpr1
-; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132GISEL-NEXT: v_cmpx_le_u32_e32 16, v0
; GFX1132GISEL-NEXT: s_xor_b32 s0, exec_lo, s0
+; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1132GISEL-NEXT: s_cbranch_execz .LBB8_2
; GFX1132GISEL-NEXT: ; %bb.1: ; %else
; GFX1132GISEL-NEXT: s_load_b32 s1, s[4:5], 0x2c
@@ -3638,13 +3657,14 @@ define amdgpu_kernel void @divergent_cfg(ptr addrspace(1) %out, i32 %in) #0 {
; GFX1132GISEL-NEXT: s_or_saveexec_b32 s0, s0
; GFX1132GISEL-NEXT: v_mov_b32_e32 v1, s1
; GFX1132GISEL-NEXT: s_xor_b32 exec_lo, exec_lo, s0
+; GFX1132GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1132GISEL-NEXT: s_cbranch_execz .LBB8_6
; GFX1132GISEL-NEXT: ; %bb.3: ; %if
; GFX1132GISEL-NEXT: s_mov_b32 s2, exec_lo
; GFX1132GISEL-NEXT: s_mov_b32 s1, 0
; GFX1132GISEL-NEXT: .LBB8_4: ; =>This Inner Loop Header: Depth=1
; GFX1132GISEL-NEXT: s_ctz_i32_b32 s3, s2
-; GFX1132GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1132GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_3) | instid1(SALU_CYCLE_1)
; GFX1132GISEL-NEXT: v_readlane_b32 s6, v0, s3
; GFX1132GISEL-NEXT: s_bitset0_b32 s2, s3
; GFX1132GISEL-NEXT: s_add_i32 s1, s1, s6
@@ -4213,6 +4233,7 @@ define void @divergent_value_i64(ptr addrspace(1) %out, i64 %id.x) #0 {
; GFX1164DAGISEL-NEXT: s_add_u32 s0, s0, s5
; GFX1164DAGISEL-NEXT: s_addc_u32 s1, s1, s6
; GFX1164DAGISEL-NEXT: s_cmp_lg_u64 s[2:3], 0
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164DAGISEL-NEXT: s_cbranch_scc1 .LBB10_1
; GFX1164DAGISEL-NEXT: ; %bb.2:
; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v3, s1
@@ -4234,6 +4255,7 @@ define void @divergent_value_i64(ptr addrspace(1) %out, i64 %id.x) #0 {
; GFX1164GISEL-NEXT: s_add_u32 s0, s0, s5
; GFX1164GISEL-NEXT: s_addc_u32 s1, s1, s6
; GFX1164GISEL-NEXT: s_cmp_lg_u64 s[2:3], 0
+; GFX1164GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164GISEL-NEXT: s_cbranch_scc1 .LBB10_1
; GFX1164GISEL-NEXT: ; %bb.2:
; GFX1164GISEL-NEXT: v_mov_b32_e32 v3, s1
@@ -4255,6 +4277,7 @@ define void @divergent_value_i64(ptr addrspace(1) %out, i64 %id.x) #0 {
; GFX1132DAGISEL-NEXT: s_add_u32 s0, s0, s4
; GFX1132DAGISEL-NEXT: s_addc_u32 s1, s1, s5
; GFX1132DAGISEL-NEXT: s_cmp_lg_u32 s2, 0
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1132DAGISEL-NEXT: s_cbranch_scc1 .LBB10_1
; GFX1132DAGISEL-NEXT: ; %bb.2:
; GFX1132DAGISEL-NEXT: v_dual_mov_b32 v3, s1 :: v_dual_mov_b32 v2, s0
@@ -4275,6 +4298,7 @@ define void @divergent_value_i64(ptr addrspace(1) %out, i64 %id.x) #0 {
; GFX1132GISEL-NEXT: s_add_u32 s0, s0, s4
; GFX1132GISEL-NEXT: s_addc_u32 s1, s1, s5
; GFX1132GISEL-NEXT: s_cmp_lg_u32 s2, 0
+; GFX1132GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1132GISEL-NEXT: s_cbranch_scc1 .LBB10_1
; GFX1132GISEL-NEXT: ; %bb.2:
; GFX1132GISEL-NEXT: v_dual_mov_b32 v3, s1 :: v_dual_mov_b32 v2, s0
@@ -4733,7 +4757,7 @@ define amdgpu_kernel void @divergent_cfg_i64(ptr addrspace(1) %out, i64 %in, i64
; GFX1164DAGISEL-NEXT: v_and_b32_e32 v0, 0x3ff, v0
; GFX1164DAGISEL-NEXT: s_mov_b64 s[6:7], exec
; GFX1164DAGISEL-NEXT: ; implicit-def: $sgpr8_sgpr9
-; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164DAGISEL-NEXT: v_cmpx_lt_u32_e32 15, v0
; GFX1164DAGISEL-NEXT: s_xor_b64 s[6:7], exec, s[6:7]
; GFX1164DAGISEL-NEXT: s_cbranch_execz .LBB11_2
@@ -4765,6 +4789,7 @@ define amdgpu_kernel void @divergent_cfg_i64(ptr addrspace(1) %out, i64 %in, i64
; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v1, s5
; GFX1164DAGISEL-NEXT: ; %bb.4: ; %endif
; GFX1164DAGISEL-NEXT: s_or_b64 exec, exec, s[2:3]
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v2, 0
; GFX1164DAGISEL-NEXT: global_store_b64 v2, v[0:1], s[0:1]
; GFX1164DAGISEL-NEXT: s_endpgm
@@ -4775,7 +4800,7 @@ define amdgpu_kernel void @divergent_cfg_i64(ptr addrspace(1) %out, i64 %in, i64
; GFX1164GISEL-NEXT: v_and_b32_e32 v0, 0x3ff, v0
; GFX1164GISEL-NEXT: s_mov_b64 s[6:7], exec
; GFX1164GISEL-NEXT: ; implicit-def: $sgpr8_sgpr9
-; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164GISEL-NEXT: v_cmpx_le_u32_e32 16, v0
; GFX1164GISEL-NEXT: s_xor_b64 s[6:7], exec, s[6:7]
; GFX1164GISEL-NEXT: s_cbranch_execz .LBB11_2
@@ -4794,6 +4819,7 @@ define amdgpu_kernel void @divergent_cfg_i64(ptr addrspace(1) %out, i64 %in, i64
; GFX1164GISEL-NEXT: v_mov_b32_e32 v0, s8
; GFX1164GISEL-NEXT: v_mov_b32_e32 v1, s9
; GFX1164GISEL-NEXT: s_xor_b64 exec, exec, s[2:3]
+; GFX1164GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164GISEL-NEXT: s_cbranch_execz .LBB11_4
; GFX1164GISEL-NEXT: ; %bb.3: ; %if
; GFX1164GISEL-NEXT: s_load_b64 s[4:5], s[4:5], 0x34
@@ -4809,6 +4835,7 @@ define amdgpu_kernel void @divergent_cfg_i64(ptr addrspace(1) %out, i64 %in, i64
; GFX1164GISEL-NEXT: v_mov_b32_e32 v1, s5
; GFX1164GISEL-NEXT: .LBB11_4: ; %endif
; GFX1164GISEL-NEXT: s_or_b64 exec, exec, s[2:3]
+; GFX1164GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164GISEL-NEXT: v_mov_b32_e32 v2, 0
; GFX1164GISEL-NEXT: global_store_b64 v2, v[0:1], s[0:1]
; GFX1164GISEL-NEXT: s_endpgm
@@ -4821,9 +4848,10 @@ define amdgpu_kernel void @divergent_cfg_i64(ptr addrspace(1) %out, i64 %in, i64
; GFX1132DAGISEL-NEXT: v_and_b32_e32 v0, 0x3ff, v0
; GFX1132DAGISEL-NEXT: s_mov_b32 s8, exec_lo
; GFX1132DAGISEL-NEXT: ; implicit-def: $sgpr6_sgpr7
-; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132DAGISEL-NEXT: v_cmpx_lt_u32_e32 15, v0
; GFX1132DAGISEL-NEXT: s_xor_b32 s8, exec_lo, s8
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1132DAGISEL-NEXT: s_cbranch_execz .LBB11_2
; GFX1132DAGISEL-NEXT: ; %bb.1: ; %else
; GFX1132DAGISEL-NEXT: s_mov_b32 s6, exec_lo
@@ -4851,6 +4879,7 @@ define amdgpu_kernel void @divergent_cfg_i64(ptr addrspace(1) %out, i64 %in, i64
; GFX1132DAGISEL-NEXT: v_dual_mov_b32 v0, s4 :: v_dual_mov_b32 v1, s5
; GFX1132DAGISEL-NEXT: ; %bb.4: ; %endif
; GFX1132DAGISEL-NEXT: s_or_b32 exec_lo, exec_lo, s2
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1132DAGISEL-NEXT: v_mov_b32_e32 v2, 0
; GFX1132DAGISEL-NEXT: global_store_b64 v2, v[0:1], s[0:1]
; GFX1132DAGISEL-NEXT: s_endpgm
@@ -4861,9 +4890,10 @@ define amdgpu_kernel void @divergent_cfg_i64(ptr addrspace(1) %out, i64 %in, i64
; GFX1132GISEL-NEXT: v_and_b32_e32 v0, 0x3ff, v0
; GFX1132GISEL-NEXT: s_mov_b32 s8, exec_lo
; GFX1132GISEL-NEXT: ; implicit-def: $sgpr6_sgpr7
-; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132GISEL-NEXT: v_cmpx_le_u32_e32 16, v0
; GFX1132GISEL-NEXT: s_xor_b32 s8, exec_lo, s8
+; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1132GISEL-NEXT: s_cbranch_execz .LBB11_2
; GFX1132GISEL-NEXT: ; %bb.1: ; %else
; GFX1132GISEL-NEXT: s_mov_b32 s6, exec_lo
@@ -4879,6 +4909,7 @@ define amdgpu_kernel void @divergent_cfg_i64(ptr addrspace(1) %out, i64 %in, i64
; GFX1132GISEL-NEXT: s_or_saveexec_b32 s2, s8
; GFX1132GISEL-NEXT: v_dual_mov_b32 v0, s6 :: v_dual_mov_b32 v1, s7
; GFX1132GISEL-NEXT: s_xor_b32 exec_lo, exec_lo, s2
+; GFX1132GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1132GISEL-NEXT: s_cbranch_execz .LBB11_4
; GFX1132GISEL-NEXT: ; %bb.3: ; %if
; GFX1132GISEL-NEXT: s_load_b64 s[4:5], s[4:5], 0x34
@@ -4894,6 +4925,7 @@ define amdgpu_kernel void @divergent_cfg_i64(ptr addrspace(1) %out, i64 %in, i64
; GFX1132GISEL-NEXT: v_dual_mov_b32 v0, s4 :: v_dual_mov_b32 v1, s5
; GFX1132GISEL-NEXT: .LBB11_4: ; %endif
; GFX1132GISEL-NEXT: s_or_b32 exec_lo, exec_lo, s2
+; GFX1132GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1132GISEL-NEXT: v_mov_b32_e32 v2, 0
; GFX1132GISEL-NEXT: global_store_b64 v2, v[0:1], s[0:1]
; GFX1132GISEL-NEXT: s_endpgm
diff --git a/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.reduce.and.ll b/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.reduce.and.ll
index c876aa657338de..fcccdf81444340 100644
--- a/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.reduce.and.ll
+++ b/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.reduce.and.ll
@@ -403,6 +403,7 @@ define void @divergent_value_i16(ptr addrspace(1) %out, i16 %in) {
; GFX1164DAGISEL-FAKE16-NEXT: s_bitset0_b64 s[0:1], s3
; GFX1164DAGISEL-FAKE16-NEXT: s_and_b32 s2, s2, s4
; GFX1164DAGISEL-FAKE16-NEXT: s_cmp_lg_u64 s[0:1], 0
+; GFX1164DAGISEL-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164DAGISEL-FAKE16-NEXT: s_cbranch_scc1 .LBB1_1
; GFX1164DAGISEL-FAKE16-NEXT: ; %bb.2:
; GFX1164DAGISEL-FAKE16-NEXT: v_mov_b32_e32 v2, s2
@@ -422,6 +423,7 @@ define void @divergent_value_i16(ptr addrspace(1) %out, i16 %in) {
; GFX1164GISEL-NEXT: s_bitset0_b64 s[0:1], s3
; GFX1164GISEL-NEXT: s_and_b32 s2, s2, s4
; GFX1164GISEL-NEXT: s_cmp_lg_u64 s[0:1], 0
+; GFX1164GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164GISEL-NEXT: s_cbranch_scc1 .LBB1_1
; GFX1164GISEL-NEXT: ; %bb.2:
; GFX1164GISEL-NEXT: v_mov_b32_e32 v2, s2
@@ -441,6 +443,7 @@ define void @divergent_value_i16(ptr addrspace(1) %out, i16 %in) {
; GFX1132DAGISEL-FAKE16-NEXT: s_bitset0_b32 s1, s2
; GFX1132DAGISEL-FAKE16-NEXT: s_and_b32 s0, s0, s3
; GFX1132DAGISEL-FAKE16-NEXT: s_cmp_lg_u32 s1, 0
+; GFX1132DAGISEL-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1132DAGISEL-FAKE16-NEXT: s_cbranch_scc1 .LBB1_1
; GFX1132DAGISEL-FAKE16-NEXT: ; %bb.2:
; GFX1132DAGISEL-FAKE16-NEXT: v_mov_b32_e32 v2, s0
@@ -460,6 +463,7 @@ define void @divergent_value_i16(ptr addrspace(1) %out, i16 %in) {
; GFX1132GISEL-NEXT: s_bitset0_b32 s1, s2
; GFX1132GISEL-NEXT: s_and_b32 s0, s0, s3
; GFX1132GISEL-NEXT: s_cmp_lg_u32 s1, 0
+; GFX1132GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1132GISEL-NEXT: s_cbranch_scc1 .LBB1_1
; GFX1132GISEL-NEXT: ; %bb.2:
; GFX1132GISEL-NEXT: v_mov_b32_e32 v2, s0
@@ -479,6 +483,7 @@ define void @divergent_value_i16(ptr addrspace(1) %out, i16 %in) {
; GFX1164DAGISEL-TRUE16-NEXT: s_bitset0_b64 s[0:1], s3
; GFX1164DAGISEL-TRUE16-NEXT: s_and_b32 s2, s2, s4
; GFX1164DAGISEL-TRUE16-NEXT: s_cmp_lg_u64 s[0:1], 0
+; GFX1164DAGISEL-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164DAGISEL-TRUE16-NEXT: s_cbranch_scc1 .LBB1_1
; GFX1164DAGISEL-TRUE16-NEXT: ; %bb.2:
; GFX1164DAGISEL-TRUE16-NEXT: v_mov_b32_e32 v2, s2
@@ -498,6 +503,7 @@ define void @divergent_value_i16(ptr addrspace(1) %out, i16 %in) {
; GFX1132DAGISEL-TRUE16-NEXT: s_bitset0_b32 s1, s2
; GFX1132DAGISEL-TRUE16-NEXT: s_and_b32 s0, s0, s3
; GFX1132DAGISEL-TRUE16-NEXT: s_cmp_lg_u32 s1, 0
+; GFX1132DAGISEL-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1132DAGISEL-TRUE16-NEXT: s_cbranch_scc1 .LBB1_1
; GFX1132DAGISEL-TRUE16-NEXT: ; %bb.2:
; GFX1132DAGISEL-TRUE16-NEXT: v_mov_b32_e32 v2, s0
@@ -852,6 +858,7 @@ define amdgpu_kernel void @divergent_value(ptr addrspace(1) %out) #0 {
; GFX1164DAGISEL-NEXT: s_bitset0_b64 s[2:3], s5
; GFX1164DAGISEL-NEXT: s_and_b32 s4, s4, s6
; GFX1164DAGISEL-NEXT: s_cmp_lg_u64 s[2:3], 0
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164DAGISEL-NEXT: s_cbranch_scc1 .LBB3_1
; GFX1164DAGISEL-NEXT: ; %bb.2:
; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v0, s4
@@ -872,6 +879,7 @@ define amdgpu_kernel void @divergent_value(ptr addrspace(1) %out) #0 {
; GFX1164GISEL-NEXT: s_bitset0_b64 s[2:3], s5
; GFX1164GISEL-NEXT: s_and_b32 s4, s4, s6
; GFX1164GISEL-NEXT: s_cmp_lg_u64 s[2:3], 0
+; GFX1164GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164GISEL-NEXT: s_cbranch_scc1 .LBB3_1
; GFX1164GISEL-NEXT: ; %bb.2:
; GFX1164GISEL-NEXT: v_mov_b32_e32 v0, s4
@@ -893,6 +901,7 @@ define amdgpu_kernel void @divergent_value(ptr addrspace(1) %out) #0 {
; GFX1132DAGISEL-NEXT: s_bitset0_b32 s3, s4
; GFX1132DAGISEL-NEXT: s_and_b32 s2, s2, s5
; GFX1132DAGISEL-NEXT: s_cmp_lg_u32 s3, 0
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1132DAGISEL-NEXT: s_cbranch_scc1 .LBB3_1
; GFX1132DAGISEL-NEXT: ; %bb.2:
; GFX1132DAGISEL-NEXT: v_mov_b32_e32 v0, s2
@@ -913,6 +922,7 @@ define amdgpu_kernel void @divergent_value(ptr addrspace(1) %out) #0 {
; GFX1132GISEL-NEXT: s_bitset0_b32 s3, s4
; GFX1132GISEL-NEXT: s_and_b32 s2, s2, s5
; GFX1132GISEL-NEXT: s_cmp_lg_u32 s3, 0
+; GFX1132GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1132GISEL-NEXT: s_cbranch_scc1 .LBB3_1
; GFX1132GISEL-NEXT: ; %bb.2:
; GFX1132GISEL-NEXT: v_dual_mov_b32 v0, s2 :: v_dual_mov_b32 v1, 0
@@ -1332,7 +1342,7 @@ define amdgpu_kernel void @divergent_value_dpp(ptr addrspace(1) %out) #0 {
; GFX1164DAGISEL-NEXT: s_or_saveexec_b64 s[2:3], -1
; GFX1164DAGISEL-NEXT: s_waitcnt lgkmcnt(0)
; GFX1164DAGISEL-NEXT: v_and_b32_e32 v1, v1, v2
-; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX1164DAGISEL-NEXT: v_readlane_b32 s4, v1, 63
; GFX1164DAGISEL-NEXT: s_mov_b64 exec, s[2:3]
; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v0, 0
@@ -1367,7 +1377,7 @@ define amdgpu_kernel void @divergent_value_dpp(ptr addrspace(1) %out) #0 {
; GFX1164GISEL-NEXT: s_mov_b64 exec, s[0:1]
; GFX1164GISEL-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
; GFX1164GISEL-NEXT: s_or_saveexec_b64 s[2:3], -1
-; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX1164GISEL-NEXT: v_readlane_b32 s4, v1, 63
; GFX1164GISEL-NEXT: s_mov_b64 exec, s[2:3]
; GFX1164GISEL-NEXT: v_mov_b32_e32 v0, s4
@@ -1394,7 +1404,7 @@ define amdgpu_kernel void @divergent_value_dpp(ptr addrspace(1) %out) #0 {
; GFX1132DAGISEL-NEXT: s_or_saveexec_b32 s2, -1
; GFX1132DAGISEL-NEXT: s_waitcnt lgkmcnt(0)
; GFX1132DAGISEL-NEXT: v_and_b32_e32 v1, v1, v2
-; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX1132DAGISEL-NEXT: v_readlane_b32 s3, v1, 31
; GFX1132DAGISEL-NEXT: s_mov_b32 exec_lo, s2
; GFX1132DAGISEL-NEXT: v_dual_mov_b32 v0, 0 :: v_dual_mov_b32 v3, s3
@@ -1419,7 +1429,7 @@ define amdgpu_kernel void @divergent_value_dpp(ptr addrspace(1) %out) #0 {
; GFX1132GISEL-NEXT: s_mov_b32 exec_lo, s0
; GFX1132GISEL-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
; GFX1132GISEL-NEXT: s_or_saveexec_b32 s2, -1
-; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX1132GISEL-NEXT: v_readlane_b32 s3, v1, 31
; GFX1132GISEL-NEXT: s_mov_b32 exec_lo, s2
; GFX1132GISEL-NEXT: v_dual_mov_b32 v0, s3 :: v_dual_mov_b32 v3, 0
@@ -2033,50 +2043,51 @@ define void @divergent_value_dpp_i64(ptr addrspace(1) %out, i64 %in) #0 {
; GFX1164DAGISEL-NEXT: scratch_store_b32 off, v7, s32 offset:12
; GFX1164DAGISEL-NEXT: scratch_store_b32 off, v8, s32 offset:16
; GFX1164DAGISEL-NEXT: s_mov_b64 exec, s[0:1]
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1164DAGISEL-NEXT: s_or_saveexec_b64 s[0:1], -1
-; GFX1164DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(VALU_DEP_3)
; GFX1164DAGISEL-NEXT: v_cndmask_b32_e64 v4, -1, v2, s[0:1]
; GFX1164DAGISEL-NEXT: v_cndmask_b32_e64 v5, -1, v3, s[0:1]
; GFX1164DAGISEL-NEXT: v_mbcnt_lo_u32_b32 v8, -1, 0
-; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v6, v4
; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v6, v4
; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v7, v5
-; GFX1164DAGISEL-NEXT: v_mbcnt_hi_u32_b32 v8, -1, v8
; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX1164DAGISEL-NEXT: v_mbcnt_hi_u32_b32 v8, -1, v8
; GFX1164DAGISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:1 row_mask:0xf bank_mask:0xf
-; GFX1164DAGISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:1 row_mask:0xf bank_mask:0xf
; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX1164DAGISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:1 row_mask:0xf bank_mask:0xf
; GFX1164DAGISEL-NEXT: v_add_nc_u32_e32 v8, 32, v8
-; GFX1164DAGISEL-NEXT: v_and_b32_e32 v4, v4, v6
; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX1164DAGISEL-NEXT: v_and_b32_e32 v4, v4, v6
; GFX1164DAGISEL-NEXT: v_and_b32_e32 v5, v5, v7
-; GFX1164DAGISEL-NEXT: v_mul_lo_u32 v8, 4, v8
; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX1164DAGISEL-NEXT: v_mul_lo_u32 v8, 4, v8
; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v6, v4
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v7, v5
-; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1164DAGISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:2 row_mask:0xf bank_mask:0xf
-; GFX1164DAGISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:2 row_mask:0xf bank_mask:0xf
; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX1164DAGISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:2 row_mask:0xf bank_mask:0xf
; GFX1164DAGISEL-NEXT: v_and_b32_e32 v4, v4, v6
-; GFX1164DAGISEL-NEXT: v_and_b32_e32 v5, v5, v7
; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX1164DAGISEL-NEXT: v_and_b32_e32 v5, v5, v7
; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v6, v4
-; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v7, v5
; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v7, v5
; GFX1164DAGISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:4 row_mask:0xf bank_mask:0xf
-; GFX1164DAGISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:4 row_mask:0xf bank_mask:0xf
; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX1164DAGISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:4 row_mask:0xf bank_mask:0xf
; GFX1164DAGISEL-NEXT: v_and_b32_e32 v4, v4, v6
-; GFX1164DAGISEL-NEXT: v_and_b32_e32 v5, v5, v7
; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX1164DAGISEL-NEXT: v_and_b32_e32 v5, v5, v7
; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v6, v4
-; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v7, v5
; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v7, v5
; GFX1164DAGISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:8 row_mask:0xf bank_mask:0xf
-; GFX1164DAGISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:8 row_mask:0xf bank_mask:0xf
; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX1164DAGISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:8 row_mask:0xf bank_mask:0xf
; GFX1164DAGISEL-NEXT: v_and_b32_e32 v4, v4, v6
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX1164DAGISEL-NEXT: v_and_b32_e32 v5, v5, v7
; GFX1164DAGISEL-NEXT: ds_swizzle_b32 v6, v4 offset:swizzle(BROADCAST,32,15)
; GFX1164DAGISEL-NEXT: ds_swizzle_b32 v7, v5 offset:swizzle(BROADCAST,32,15)
@@ -2094,6 +2105,7 @@ define void @divergent_value_dpp_i64(ptr addrspace(1) %out, i64 %in) #0 {
; GFX1164DAGISEL-NEXT: v_readlane_b32 s2, v4, 63
; GFX1164DAGISEL-NEXT: v_readlane_b32 s3, v5, 63
; GFX1164DAGISEL-NEXT: s_mov_b64 exec, s[0:1]
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v2, s2
; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v3, s3
; GFX1164DAGISEL-NEXT: global_store_b64 v[0:1], v[2:3], off
@@ -2120,50 +2132,51 @@ define void @divergent_value_dpp_i64(ptr addrspace(1) %out, i64 %in) #0 {
; GFX1164GISEL-NEXT: scratch_store_b32 off, v7, s32 offset:12
; GFX1164GISEL-NEXT: scratch_store_b32 off, v8, s32 offset:16
; GFX1164GISEL-NEXT: s_mov_b64 exec, s[0:1]
+; GFX1164GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1164GISEL-NEXT: s_or_saveexec_b64 s[0:1], -1
-; GFX1164GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(VALU_DEP_3)
; GFX1164GISEL-NEXT: v_cndmask_b32_e64 v4, -1, v2, s[0:1]
; GFX1164GISEL-NEXT: v_cndmask_b32_e64 v5, -1, v3, s[0:1]
; GFX1164GISEL-NEXT: v_mbcnt_lo_u32_b32 v8, -1, 0
-; GFX1164GISEL-NEXT: v_mov_b32_e32 v6, v4
; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX1164GISEL-NEXT: v_mov_b32_e32 v6, v4
; GFX1164GISEL-NEXT: v_mov_b32_e32 v7, v5
-; GFX1164GISEL-NEXT: v_mbcnt_hi_u32_b32 v8, -1, v8
; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX1164GISEL-NEXT: v_mbcnt_hi_u32_b32 v8, -1, v8
; GFX1164GISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:1 row_mask:0xf bank_mask:0xf
-; GFX1164GISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:1 row_mask:0xf bank_mask:0xf
; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX1164GISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:1 row_mask:0xf bank_mask:0xf
; GFX1164GISEL-NEXT: v_add_nc_u32_e32 v8, 32, v8
-; GFX1164GISEL-NEXT: v_and_b32_e32 v4, v4, v6
; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX1164GISEL-NEXT: v_and_b32_e32 v4, v4, v6
; GFX1164GISEL-NEXT: v_and_b32_e32 v5, v5, v7
-; GFX1164GISEL-NEXT: v_mul_lo_u32 v8, 4, v8
; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX1164GISEL-NEXT: v_mul_lo_u32 v8, 4, v8
; GFX1164GISEL-NEXT: v_mov_b32_e32 v6, v4
+; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1164GISEL-NEXT: v_mov_b32_e32 v7, v5
-; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1164GISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:2 row_mask:0xf bank_mask:0xf
-; GFX1164GISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:2 row_mask:0xf bank_mask:0xf
; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX1164GISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:2 row_mask:0xf bank_mask:0xf
; GFX1164GISEL-NEXT: v_and_b32_e32 v4, v4, v6
-; GFX1164GISEL-NEXT: v_and_b32_e32 v5, v5, v7
; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX1164GISEL-NEXT: v_and_b32_e32 v5, v5, v7
; GFX1164GISEL-NEXT: v_mov_b32_e32 v6, v4
-; GFX1164GISEL-NEXT: v_mov_b32_e32 v7, v5
; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX1164GISEL-NEXT: v_mov_b32_e32 v7, v5
; GFX1164GISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:4 row_mask:0xf bank_mask:0xf
-; GFX1164GISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:4 row_mask:0xf bank_mask:0xf
; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX1164GISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:4 row_mask:0xf bank_mask:0xf
; GFX1164GISEL-NEXT: v_and_b32_e32 v4, v4, v6
-; GFX1164GISEL-NEXT: v_and_b32_e32 v5, v5, v7
; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX1164GISEL-NEXT: v_and_b32_e32 v5, v5, v7
; GFX1164GISEL-NEXT: v_mov_b32_e32 v6, v4
-; GFX1164GISEL-NEXT: v_mov_b32_e32 v7, v5
; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX1164GISEL-NEXT: v_mov_b32_e32 v7, v5
; GFX1164GISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:8 row_mask:0xf bank_mask:0xf
-; GFX1164GISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:8 row_mask:0xf bank_mask:0xf
; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX1164GISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:8 row_mask:0xf bank_mask:0xf
; GFX1164GISEL-NEXT: v_and_b32_e32 v4, v4, v6
+; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX1164GISEL-NEXT: v_and_b32_e32 v5, v5, v7
; GFX1164GISEL-NEXT: ds_swizzle_b32 v6, v4 offset:swizzle(BROADCAST,32,15)
; GFX1164GISEL-NEXT: ds_swizzle_b32 v7, v5 offset:swizzle(BROADCAST,32,15)
@@ -2181,6 +2194,7 @@ define void @divergent_value_dpp_i64(ptr addrspace(1) %out, i64 %in) #0 {
; GFX1164GISEL-NEXT: v_readlane_b32 s2, v4, 63
; GFX1164GISEL-NEXT: v_readlane_b32 s3, v5, 63
; GFX1164GISEL-NEXT: s_mov_b64 exec, s[0:1]
+; GFX1164GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164GISEL-NEXT: v_mov_b32_e32 v2, s2
; GFX1164GISEL-NEXT: v_mov_b32_e32 v3, s3
; GFX1164GISEL-NEXT: global_store_b64 v[0:1], v[2:3], off
@@ -2206,39 +2220,39 @@ define void @divergent_value_dpp_i64(ptr addrspace(1) %out, i64 %in) #0 {
; GFX1132DAGISEL-NEXT: scratch_store_b32 off, v6, s32 offset:8
; GFX1132DAGISEL-NEXT: scratch_store_b32 off, v7, s32 offset:12
; GFX1132DAGISEL-NEXT: s_mov_b32 exec_lo, s0
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1132DAGISEL-NEXT: s_or_saveexec_b32 s2, -1
-; GFX1132DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX1132DAGISEL-NEXT: v_cndmask_b32_e64 v4, -1, v2, s2
; GFX1132DAGISEL-NEXT: v_cndmask_b32_e64 v5, -1, v3, s2
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132DAGISEL-NEXT: v_dual_mov_b32 v6, v4 :: v_dual_mov_b32 v7, v5
-; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1132DAGISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:1 row_mask:0xf bank_mask:0xf
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132DAGISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:1 row_mask:0xf bank_mask:0xf
-; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132DAGISEL-NEXT: v_and_b32_e32 v5, v5, v7
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132DAGISEL-NEXT: v_dual_mov_b32 v7, v5 :: v_dual_and_b32 v4, v4, v6
-; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1132DAGISEL-NEXT: v_mov_b32_e32 v6, v4
-; GFX1132DAGISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:2 row_mask:0xf bank_mask:0xf
; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX1132DAGISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:2 row_mask:0xf bank_mask:0xf
; GFX1132DAGISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:2 row_mask:0xf bank_mask:0xf
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132DAGISEL-NEXT: v_and_b32_e32 v5, v5, v7
-; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132DAGISEL-NEXT: v_dual_mov_b32 v7, v5 :: v_dual_and_b32 v4, v4, v6
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1132DAGISEL-NEXT: v_mov_b32_e32 v6, v4
-; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1132DAGISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:4 row_mask:0xf bank_mask:0xf
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1132DAGISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:4 row_mask:0xf bank_mask:0xf
-; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132DAGISEL-NEXT: v_and_b32_e32 v5, v5, v7
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132DAGISEL-NEXT: v_dual_mov_b32 v7, v5 :: v_dual_and_b32 v4, v4, v6
-; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1132DAGISEL-NEXT: v_mov_b32_e32 v6, v4
-; GFX1132DAGISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:8 row_mask:0xf bank_mask:0xf
; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX1132DAGISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:8 row_mask:0xf bank_mask:0xf
; GFX1132DAGISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:8 row_mask:0xf bank_mask:0xf
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1132DAGISEL-NEXT: v_and_b32_e32 v5, v5, v7
-; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX1132DAGISEL-NEXT: v_and_b32_e32 v4, v4, v6
; GFX1132DAGISEL-NEXT: ds_swizzle_b32 v7, v5 offset:swizzle(BROADCAST,32,15)
; GFX1132DAGISEL-NEXT: ds_swizzle_b32 v6, v4 offset:swizzle(BROADCAST,32,15)
@@ -2250,6 +2264,7 @@ define void @divergent_value_dpp_i64(ptr addrspace(1) %out, i64 %in) #0 {
; GFX1132DAGISEL-NEXT: v_readlane_b32 s1, v5, 31
; GFX1132DAGISEL-NEXT: v_readlane_b32 s0, v4, 31
; GFX1132DAGISEL-NEXT: s_mov_b32 exec_lo, s2
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1132DAGISEL-NEXT: v_dual_mov_b32 v3, s1 :: v_dual_mov_b32 v2, s0
; GFX1132DAGISEL-NEXT: global_store_b64 v[0:1], v[2:3], off
; GFX1132DAGISEL-NEXT: s_xor_saveexec_b32 s0, -1
@@ -2272,39 +2287,39 @@ define void @divergent_value_dpp_i64(ptr addrspace(1) %out, i64 %in) #0 {
; GFX1132GISEL-NEXT: scratch_store_b32 off, v6, s32 offset:8
; GFX1132GISEL-NEXT: scratch_store_b32 off, v7, s32 offset:12
; GFX1132GISEL-NEXT: s_mov_b32 exec_lo, s0
+; GFX1132GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1132GISEL-NEXT: s_or_saveexec_b32 s2, -1
-; GFX1132GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX1132GISEL-NEXT: v_cndmask_b32_e64 v4, -1, v2, s2
; GFX1132GISEL-NEXT: v_cndmask_b32_e64 v5, -1, v3, s2
+; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132GISEL-NEXT: v_dual_mov_b32 v6, v4 :: v_dual_mov_b32 v7, v5
-; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1132GISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:1 row_mask:0xf bank_mask:0xf
+; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132GISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:1 row_mask:0xf bank_mask:0xf
-; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132GISEL-NEXT: v_and_b32_e32 v5, v5, v7
+; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132GISEL-NEXT: v_dual_mov_b32 v7, v5 :: v_dual_and_b32 v4, v4, v6
-; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1132GISEL-NEXT: v_mov_b32_e32 v6, v4
-; GFX1132GISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:2 row_mask:0xf bank_mask:0xf
; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX1132GISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:2 row_mask:0xf bank_mask:0xf
; GFX1132GISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:2 row_mask:0xf bank_mask:0xf
+; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132GISEL-NEXT: v_and_b32_e32 v5, v5, v7
-; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132GISEL-NEXT: v_dual_mov_b32 v7, v5 :: v_dual_and_b32 v4, v4, v6
+; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1132GISEL-NEXT: v_mov_b32_e32 v6, v4
-; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1132GISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:4 row_mask:0xf bank_mask:0xf
+; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1132GISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:4 row_mask:0xf bank_mask:0xf
-; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132GISEL-NEXT: v_and_b32_e32 v5, v5, v7
+; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132GISEL-NEXT: v_dual_mov_b32 v7, v5 :: v_dual_and_b32 v4, v4, v6
-; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1132GISEL-NEXT: v_mov_b32_e32 v6, v4
-; GFX1132GISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:8 row_mask:0xf bank_mask:0xf
; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX1132GISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:8 row_mask:0xf bank_mask:0xf
; GFX1132GISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:8 row_mask:0xf bank_mask:0xf
+; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1132GISEL-NEXT: v_and_b32_e32 v5, v5, v7
-; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX1132GISEL-NEXT: v_and_b32_e32 v4, v4, v6
; GFX1132GISEL-NEXT: ds_swizzle_b32 v7, v5 offset:swizzle(BROADCAST,32,15)
; GFX1132GISEL-NEXT: ds_swizzle_b32 v6, v4 offset:swizzle(BROADCAST,32,15)
@@ -2316,6 +2331,7 @@ define void @divergent_value_dpp_i64(ptr addrspace(1) %out, i64 %in) #0 {
; GFX1132GISEL-NEXT: v_readlane_b32 s1, v5, 31
; GFX1132GISEL-NEXT: v_readlane_b32 s0, v4, 31
; GFX1132GISEL-NEXT: s_mov_b32 exec_lo, s2
+; GFX1132GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1132GISEL-NEXT: v_dual_mov_b32 v3, s1 :: v_dual_mov_b32 v2, s0
; GFX1132GISEL-NEXT: global_store_b64 v[0:1], v[2:3], off
; GFX1132GISEL-NEXT: s_xor_saveexec_b32 s0, -1
@@ -2605,7 +2621,7 @@ define amdgpu_kernel void @default_stratergy(ptr addrspace(1) %out) #0 {
; GFX1164DAGISEL-NEXT: s_or_saveexec_b64 s[2:3], -1
; GFX1164DAGISEL-NEXT: s_waitcnt lgkmcnt(0)
; GFX1164DAGISEL-NEXT: v_and_b32_e32 v1, v1, v2
-; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX1164DAGISEL-NEXT: v_readlane_b32 s4, v1, 63
; GFX1164DAGISEL-NEXT: s_mov_b64 exec, s[2:3]
; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v0, 0
@@ -2640,7 +2656,7 @@ define amdgpu_kernel void @default_stratergy(ptr addrspace(1) %out) #0 {
; GFX1164GISEL-NEXT: s_mov_b64 exec, s[0:1]
; GFX1164GISEL-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
; GFX1164GISEL-NEXT: s_or_saveexec_b64 s[2:3], -1
-; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX1164GISEL-NEXT: v_readlane_b32 s4, v1, 63
; GFX1164GISEL-NEXT: s_mov_b64 exec, s[2:3]
; GFX1164GISEL-NEXT: v_mov_b32_e32 v0, s4
@@ -2667,7 +2683,7 @@ define amdgpu_kernel void @default_stratergy(ptr addrspace(1) %out) #0 {
; GFX1132DAGISEL-NEXT: s_or_saveexec_b32 s2, -1
; GFX1132DAGISEL-NEXT: s_waitcnt lgkmcnt(0)
; GFX1132DAGISEL-NEXT: v_and_b32_e32 v1, v1, v2
-; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX1132DAGISEL-NEXT: v_readlane_b32 s3, v1, 31
; GFX1132DAGISEL-NEXT: s_mov_b32 exec_lo, s2
; GFX1132DAGISEL-NEXT: v_dual_mov_b32 v0, 0 :: v_dual_mov_b32 v3, s3
@@ -2692,7 +2708,7 @@ define amdgpu_kernel void @default_stratergy(ptr addrspace(1) %out) #0 {
; GFX1132GISEL-NEXT: s_mov_b32 exec_lo, s0
; GFX1132GISEL-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
; GFX1132GISEL-NEXT: s_or_saveexec_b32 s2, -1
-; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX1132GISEL-NEXT: v_readlane_b32 s3, v1, 31
; GFX1132GISEL-NEXT: s_mov_b32 exec_lo, s2
; GFX1132GISEL-NEXT: v_dual_mov_b32 v0, s3 :: v_dual_mov_b32 v3, 0
@@ -3074,7 +3090,7 @@ define amdgpu_kernel void @divergent_cfg(ptr addrspace(1) %out, i32 %in) #0 {
; GFX1164DAGISEL-NEXT: v_and_b32_e32 v0, 0x3ff, v0
; GFX1164DAGISEL-NEXT: s_mov_b64 s[0:1], exec
; GFX1164DAGISEL-NEXT: ; implicit-def: $sgpr2
-; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164DAGISEL-NEXT: v_cmpx_lt_u32_e32 15, v0
; GFX1164DAGISEL-NEXT: s_xor_b64 s[0:1], exec, s[0:1]
; GFX1164DAGISEL-NEXT: ; %bb.1: ; %else
@@ -3085,13 +3101,14 @@ define amdgpu_kernel void @divergent_cfg(ptr addrspace(1) %out, i32 %in) #0 {
; GFX1164DAGISEL-NEXT: s_waitcnt lgkmcnt(0)
; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v1, s2
; GFX1164DAGISEL-NEXT: s_xor_b64 exec, exec, s[0:1]
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164DAGISEL-NEXT: s_cbranch_execz .LBB8_6
; GFX1164DAGISEL-NEXT: ; %bb.3: ; %if
; GFX1164DAGISEL-NEXT: s_mov_b64 s[2:3], exec
; GFX1164DAGISEL-NEXT: s_mov_b32 s6, -1
; GFX1164DAGISEL-NEXT: .LBB8_4: ; =>This Inner Loop Header: Depth=1
; GFX1164DAGISEL-NEXT: s_ctz_i32_b64 s7, s[2:3]
-; GFX1164DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_3) | instid1(SALU_CYCLE_1)
; GFX1164DAGISEL-NEXT: v_readlane_b32 s8, v0, s7
; GFX1164DAGISEL-NEXT: s_bitset0_b64 s[2:3], s7
; GFX1164DAGISEL-NEXT: s_and_b32 s6, s6, s8
@@ -3112,7 +3129,7 @@ define amdgpu_kernel void @divergent_cfg(ptr addrspace(1) %out, i32 %in) #0 {
; GFX1164GISEL-NEXT: v_and_b32_e32 v0, 0x3ff, v0
; GFX1164GISEL-NEXT: s_mov_b64 s[0:1], exec
; GFX1164GISEL-NEXT: ; implicit-def: $sgpr2
-; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164GISEL-NEXT: v_cmpx_le_u32_e32 16, v0
; GFX1164GISEL-NEXT: s_xor_b64 s[0:1], exec, s[0:1]
; GFX1164GISEL-NEXT: s_cbranch_execz .LBB8_2
@@ -3125,13 +3142,14 @@ define amdgpu_kernel void @divergent_cfg(ptr addrspace(1) %out, i32 %in) #0 {
; GFX1164GISEL-NEXT: s_or_saveexec_b64 s[0:1], s[0:1]
; GFX1164GISEL-NEXT: v_mov_b32_e32 v1, s2
; GFX1164GISEL-NEXT: s_xor_b64 exec, exec, s[0:1]
+; GFX1164GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164GISEL-NEXT: s_cbranch_execz .LBB8_6
; GFX1164GISEL-NEXT: ; %bb.3: ; %if
; GFX1164GISEL-NEXT: s_mov_b64 s[2:3], exec
; GFX1164GISEL-NEXT: s_mov_b32 s6, -1
; GFX1164GISEL-NEXT: .LBB8_4: ; =>This Inner Loop Header: Depth=1
; GFX1164GISEL-NEXT: s_ctz_i32_b64 s7, s[2:3]
-; GFX1164GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1164GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_3) | instid1(SALU_CYCLE_1)
; GFX1164GISEL-NEXT: v_readlane_b32 s8, v0, s7
; GFX1164GISEL-NEXT: s_bitset0_b64 s[2:3], s7
; GFX1164GISEL-NEXT: s_and_b32 s6, s6, s8
@@ -3152,13 +3170,14 @@ define amdgpu_kernel void @divergent_cfg(ptr addrspace(1) %out, i32 %in) #0 {
; GFX1132DAGISEL-NEXT: v_and_b32_e32 v0, 0x3ff, v0
; GFX1132DAGISEL-NEXT: s_mov_b32 s0, exec_lo
; GFX1132DAGISEL-NEXT: ; implicit-def: $sgpr1
-; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132DAGISEL-NEXT: v_cmpx_lt_u32_e32 15, v0
; GFX1132DAGISEL-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX1132DAGISEL-NEXT: ; %bb.1: ; %else
; GFX1132DAGISEL-NEXT: s_load_b32 s1, s[4:5], 0x2c
; GFX1132DAGISEL-NEXT: ; implicit-def: $vgpr0
; GFX1132DAGISEL-NEXT: ; %bb.2: ; %Flow
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_3) | instid1(SALU_CYCLE_1)
; GFX1132DAGISEL-NEXT: s_or_saveexec_b32 s0, s0
; GFX1132DAGISEL-NEXT: s_waitcnt lgkmcnt(0)
; GFX1132DAGISEL-NEXT: v_mov_b32_e32 v1, s1
@@ -3169,7 +3188,7 @@ define amdgpu_kernel void @divergent_cfg(ptr addrspace(1) %out, i32 %in) #0 {
; GFX1132DAGISEL-NEXT: s_mov_b32 s1, -1
; GFX1132DAGISEL-NEXT: .LBB8_4: ; =>This Inner Loop Header: Depth=1
; GFX1132DAGISEL-NEXT: s_ctz_i32_b32 s3, s2
-; GFX1132DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_3) | instid1(SALU_CYCLE_1)
; GFX1132DAGISEL-NEXT: v_readlane_b32 s6, v0, s3
; GFX1132DAGISEL-NEXT: s_bitset0_b32 s2, s3
; GFX1132DAGISEL-NEXT: s_and_b32 s1, s1, s6
@@ -3190,9 +3209,10 @@ define amdgpu_kernel void @divergent_cfg(ptr addrspace(1) %out, i32 %in) #0 {
; GFX1132GISEL-NEXT: v_and_b32_e32 v0, 0x3ff, v0
; GFX1132GISEL-NEXT: s_mov_b32 s0, exec_lo
; GFX1132GISEL-NEXT: ; implicit-def: $sgpr1
-; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132GISEL-NEXT: v_cmpx_le_u32_e32 16, v0
; GFX1132GISEL-NEXT: s_xor_b32 s0, exec_lo, s0
+; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1132GISEL-NEXT: s_cbranch_execz .LBB8_2
; GFX1132GISEL-NEXT: ; %bb.1: ; %else
; GFX1132GISEL-NEXT: s_load_b32 s1, s[4:5], 0x2c
@@ -3203,13 +3223,14 @@ define amdgpu_kernel void @divergent_cfg(ptr addrspace(1) %out, i32 %in) #0 {
; GFX1132GISEL-NEXT: s_or_saveexec_b32 s0, s0
; GFX1132GISEL-NEXT: v_mov_b32_e32 v1, s1
; GFX1132GISEL-NEXT: s_xor_b32 exec_lo, exec_lo, s0
+; GFX1132GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1132GISEL-NEXT: s_cbranch_execz .LBB8_6
; GFX1132GISEL-NEXT: ; %bb.3: ; %if
; GFX1132GISEL-NEXT: s_mov_b32 s2, exec_lo
; GFX1132GISEL-NEXT: s_mov_b32 s1, -1
; GFX1132GISEL-NEXT: .LBB8_4: ; =>This Inner Loop Header: Depth=1
; GFX1132GISEL-NEXT: s_ctz_i32_b32 s3, s2
-; GFX1132GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1132GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_3) | instid1(SALU_CYCLE_1)
; GFX1132GISEL-NEXT: v_readlane_b32 s6, v0, s3
; GFX1132GISEL-NEXT: s_bitset0_b32 s2, s3
; GFX1132GISEL-NEXT: s_and_b32 s1, s1, s6
@@ -3593,6 +3614,7 @@ define void @divergent_value_i64(ptr addrspace(1) %out, i64 %id.x) #0 {
; GFX1164DAGISEL-NEXT: s_bitset0_b64 s[2:3], s6
; GFX1164DAGISEL-NEXT: s_and_b64 s[0:1], s[0:1], s[4:5]
; GFX1164DAGISEL-NEXT: s_cmp_lg_u64 s[2:3], 0
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164DAGISEL-NEXT: s_cbranch_scc1 .LBB10_1
; GFX1164DAGISEL-NEXT: ; %bb.2:
; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v3, s1
@@ -3613,6 +3635,7 @@ define void @divergent_value_i64(ptr addrspace(1) %out, i64 %id.x) #0 {
; GFX1164GISEL-NEXT: s_bitset0_b64 s[2:3], s6
; GFX1164GISEL-NEXT: s_and_b64 s[0:1], s[0:1], s[4:5]
; GFX1164GISEL-NEXT: s_cmp_lg_u64 s[2:3], 0
+; GFX1164GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164GISEL-NEXT: s_cbranch_scc1 .LBB10_1
; GFX1164GISEL-NEXT: ; %bb.2:
; GFX1164GISEL-NEXT: v_mov_b32_e32 v3, s1
@@ -3633,6 +3656,7 @@ define void @divergent_value_i64(ptr addrspace(1) %out, i64 %id.x) #0 {
; GFX1132DAGISEL-NEXT: s_bitset0_b32 s2, s3
; GFX1132DAGISEL-NEXT: s_and_b64 s[0:1], s[0:1], s[4:5]
; GFX1132DAGISEL-NEXT: s_cmp_lg_u32 s2, 0
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1132DAGISEL-NEXT: s_cbranch_scc1 .LBB10_1
; GFX1132DAGISEL-NEXT: ; %bb.2:
; GFX1132DAGISEL-NEXT: v_dual_mov_b32 v3, s1 :: v_dual_mov_b32 v2, s0
@@ -3652,6 +3676,7 @@ define void @divergent_value_i64(ptr addrspace(1) %out, i64 %id.x) #0 {
; GFX1132GISEL-NEXT: s_bitset0_b32 s2, s3
; GFX1132GISEL-NEXT: s_and_b64 s[0:1], s[0:1], s[4:5]
; GFX1132GISEL-NEXT: s_cmp_lg_u32 s2, 0
+; GFX1132GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1132GISEL-NEXT: s_cbranch_scc1 .LBB10_1
; GFX1132GISEL-NEXT: ; %bb.2:
; GFX1132GISEL-NEXT: v_dual_mov_b32 v3, s1 :: v_dual_mov_b32 v2, s0
@@ -3923,19 +3948,22 @@ define amdgpu_kernel void @divergent_cfg_i64(ptr addrspace(1) %out, i64 %in, i64
; GFX1164DAGISEL-NEXT: s_load_b64 s[4:5], s[4:5], 0x34
; GFX1164DAGISEL-NEXT: v_and_b32_e32 v0, 0x3ff, v0
; GFX1164DAGISEL-NEXT: s_mov_b64 s[6:7], exec
-; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164DAGISEL-NEXT: v_cmpx_lt_u32_e32 15, v0
; GFX1164DAGISEL-NEXT: s_xor_b64 s[6:7], exec, s[6:7]
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164DAGISEL-NEXT: s_or_saveexec_b64 s[6:7], s[6:7]
; GFX1164DAGISEL-NEXT: s_waitcnt lgkmcnt(0)
; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v0, s2
; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v1, s3
; GFX1164DAGISEL-NEXT: s_xor_b64 exec, exec, s[6:7]
; GFX1164DAGISEL-NEXT: ; %bb.1: ; %if
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v0, s4
; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v1, s5
; GFX1164DAGISEL-NEXT: ; %bb.2: ; %endif
; GFX1164DAGISEL-NEXT: s_or_b64 exec, exec, s[6:7]
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v2, 0
; GFX1164DAGISEL-NEXT: global_store_b64 v2, v[0:1], s[0:1]
; GFX1164DAGISEL-NEXT: s_endpgm
@@ -3946,7 +3974,7 @@ define amdgpu_kernel void @divergent_cfg_i64(ptr addrspace(1) %out, i64 %in, i64
; GFX1164GISEL-NEXT: v_and_b32_e32 v0, 0x3ff, v0
; GFX1164GISEL-NEXT: s_mov_b64 s[8:9], exec
; GFX1164GISEL-NEXT: ; implicit-def: $sgpr6_sgpr7
-; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164GISEL-NEXT: v_cmpx_le_u32_e32 16, v0
; GFX1164GISEL-NEXT: s_xor_b64 s[8:9], exec, s[8:9]
; GFX1164GISEL-NEXT: s_cbranch_execz .LBB11_2
@@ -3959,6 +3987,7 @@ define amdgpu_kernel void @divergent_cfg_i64(ptr addrspace(1) %out, i64 %in, i64
; GFX1164GISEL-NEXT: v_mov_b32_e32 v0, s6
; GFX1164GISEL-NEXT: v_mov_b32_e32 v1, s7
; GFX1164GISEL-NEXT: s_xor_b64 exec, exec, s[2:3]
+; GFX1164GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164GISEL-NEXT: s_cbranch_execz .LBB11_4
; GFX1164GISEL-NEXT: ; %bb.3: ; %if
; GFX1164GISEL-NEXT: s_load_b64 s[4:5], s[4:5], 0x34
@@ -3969,6 +3998,7 @@ define amdgpu_kernel void @divergent_cfg_i64(ptr addrspace(1) %out, i64 %in, i64
; GFX1164GISEL-NEXT: v_mov_b32_e32 v1, s5
; GFX1164GISEL-NEXT: .LBB11_4: ; %endif
; GFX1164GISEL-NEXT: s_or_b64 exec, exec, s[2:3]
+; GFX1164GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164GISEL-NEXT: v_mov_b32_e32 v2, 0
; GFX1164GISEL-NEXT: global_store_b64 v2, v[0:1], s[0:1]
; GFX1164GISEL-NEXT: s_endpgm
@@ -3980,17 +4010,20 @@ define amdgpu_kernel void @divergent_cfg_i64(ptr addrspace(1) %out, i64 %in, i64
; GFX1132DAGISEL-NEXT: s_load_b64 s[4:5], s[4:5], 0x34
; GFX1132DAGISEL-NEXT: v_and_b32_e32 v0, 0x3ff, v0
; GFX1132DAGISEL-NEXT: s_mov_b32 s6, exec_lo
-; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132DAGISEL-NEXT: v_cmpx_lt_u32_e32 15, v0
; GFX1132DAGISEL-NEXT: s_xor_b32 s6, exec_lo, s6
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX1132DAGISEL-NEXT: s_or_saveexec_b32 s6, s6
; GFX1132DAGISEL-NEXT: s_waitcnt lgkmcnt(0)
; GFX1132DAGISEL-NEXT: v_dual_mov_b32 v0, s2 :: v_dual_mov_b32 v1, s3
; GFX1132DAGISEL-NEXT: s_xor_b32 exec_lo, exec_lo, s6
; GFX1132DAGISEL-NEXT: ; %bb.1: ; %if
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1132DAGISEL-NEXT: v_dual_mov_b32 v0, s4 :: v_dual_mov_b32 v1, s5
; GFX1132DAGISEL-NEXT: ; %bb.2: ; %endif
; GFX1132DAGISEL-NEXT: s_or_b32 exec_lo, exec_lo, s6
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1132DAGISEL-NEXT: v_mov_b32_e32 v2, 0
; GFX1132DAGISEL-NEXT: global_store_b64 v2, v[0:1], s[0:1]
; GFX1132DAGISEL-NEXT: s_endpgm
@@ -4001,9 +4034,10 @@ define amdgpu_kernel void @divergent_cfg_i64(ptr addrspace(1) %out, i64 %in, i64
; GFX1132GISEL-NEXT: v_and_b32_e32 v0, 0x3ff, v0
; GFX1132GISEL-NEXT: s_mov_b32 s8, exec_lo
; GFX1132GISEL-NEXT: ; implicit-def: $sgpr6_sgpr7
-; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132GISEL-NEXT: v_cmpx_le_u32_e32 16, v0
; GFX1132GISEL-NEXT: s_xor_b32 s8, exec_lo, s8
+; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1132GISEL-NEXT: s_cbranch_execz .LBB11_2
; GFX1132GISEL-NEXT: ; %bb.1: ; %else
; GFX1132GISEL-NEXT: s_waitcnt lgkmcnt(0)
@@ -4013,6 +4047,7 @@ define amdgpu_kernel void @divergent_cfg_i64(ptr addrspace(1) %out, i64 %in, i64
; GFX1132GISEL-NEXT: s_or_saveexec_b32 s2, s8
; GFX1132GISEL-NEXT: v_dual_mov_b32 v0, s6 :: v_dual_mov_b32 v1, s7
; GFX1132GISEL-NEXT: s_xor_b32 exec_lo, exec_lo, s2
+; GFX1132GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1132GISEL-NEXT: s_cbranch_execz .LBB11_4
; GFX1132GISEL-NEXT: ; %bb.3: ; %if
; GFX1132GISEL-NEXT: s_load_b64 s[4:5], s[4:5], 0x34
@@ -4022,6 +4057,7 @@ define amdgpu_kernel void @divergent_cfg_i64(ptr addrspace(1) %out, i64 %in, i64
; GFX1132GISEL-NEXT: v_dual_mov_b32 v0, s4 :: v_dual_mov_b32 v1, s5
; GFX1132GISEL-NEXT: .LBB11_4: ; %endif
; GFX1132GISEL-NEXT: s_or_b32 exec_lo, exec_lo, s2
+; GFX1132GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1132GISEL-NEXT: v_mov_b32_e32 v2, 0
; GFX1132GISEL-NEXT: global_store_b64 v2, v[0:1], s[0:1]
; GFX1132GISEL-NEXT: s_endpgm
diff --git a/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.reduce.fadd.ll b/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.reduce.fadd.ll
index 1f950e2ea82941..21788dca6ecb0e 100644
--- a/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.reduce.fadd.ll
+++ b/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.reduce.fadd.ll
@@ -1502,19 +1502,20 @@ define void @divergent_value_float_dpp(ptr addrspace(1) %out, float %id.x) #0 {
; GFX1164DAGISEL-NEXT: scratch_store_b32 off, v4, s32 offset:4
; GFX1164DAGISEL-NEXT: scratch_store_b32 off, v5, s32 offset:8
; GFX1164DAGISEL-NEXT: s_mov_b64 exec, s[0:1]
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1164DAGISEL-NEXT: s_or_saveexec_b64 s[0:1], -1
-; GFX1164DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX1164DAGISEL-NEXT: v_cndmask_b32_e64 v3, 0x80000000, v2, s[0:1]
; GFX1164DAGISEL-NEXT: v_mbcnt_lo_u32_b32 v5, -1, 0
-; GFX1164DAGISEL-NEXT: v_add_f32_dpp v3, v3, v3 row_shr:1 row_mask:0xf bank_mask:0xf
; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX1164DAGISEL-NEXT: v_add_f32_dpp v3, v3, v3 row_shr:1 row_mask:0xf bank_mask:0xf
; GFX1164DAGISEL-NEXT: v_mbcnt_hi_u32_b32 v5, -1, v5
-; GFX1164DAGISEL-NEXT: v_add_f32_dpp v3, v3, v3 row_shr:2 row_mask:0xf bank_mask:0xf
; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX1164DAGISEL-NEXT: v_add_f32_dpp v3, v3, v3 row_shr:2 row_mask:0xf bank_mask:0xf
; GFX1164DAGISEL-NEXT: v_add_nc_u32_e32 v5, 32, v5
-; GFX1164DAGISEL-NEXT: v_add_f32_dpp v3, v3, v3 row_shr:4 row_mask:0xf bank_mask:0xf
; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX1164DAGISEL-NEXT: v_add_f32_dpp v3, v3, v3 row_shr:4 row_mask:0xf bank_mask:0xf
; GFX1164DAGISEL-NEXT: v_mul_lo_u32 v5, 4, v5
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX1164DAGISEL-NEXT: v_add_f32_dpp v3, v3, v3 row_shr:8 row_mask:0xf bank_mask:0xf
; GFX1164DAGISEL-NEXT: ds_swizzle_b32 v4, v3 offset:swizzle(BROADCAST,32,15)
; GFX1164DAGISEL-NEXT: s_waitcnt lgkmcnt(0)
@@ -1522,7 +1523,7 @@ define void @divergent_value_float_dpp(ptr addrspace(1) %out, float %id.x) #0 {
; GFX1164DAGISEL-NEXT: ds_permute_b32 v4, v5, v3
; GFX1164DAGISEL-NEXT: s_waitcnt lgkmcnt(0)
; GFX1164DAGISEL-NEXT: v_add_f32_e32 v3, v3, v4
-; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX1164DAGISEL-NEXT: v_readlane_b32 s2, v3, 63
; GFX1164DAGISEL-NEXT: s_mov_b64 exec, s[0:1]
; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v2, s2
@@ -1546,19 +1547,20 @@ define void @divergent_value_float_dpp(ptr addrspace(1) %out, float %id.x) #0 {
; GFX1164GISEL-NEXT: scratch_store_b32 off, v4, s32 offset:4
; GFX1164GISEL-NEXT: scratch_store_b32 off, v5, s32 offset:8
; GFX1164GISEL-NEXT: s_mov_b64 exec, s[0:1]
+; GFX1164GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1164GISEL-NEXT: s_or_saveexec_b64 s[0:1], -1
-; GFX1164GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX1164GISEL-NEXT: v_cndmask_b32_e64 v3, 0x80000000, v2, s[0:1]
; GFX1164GISEL-NEXT: v_mbcnt_lo_u32_b32 v5, -1, 0
-; GFX1164GISEL-NEXT: v_add_f32_dpp v3, v3, v3 row_shr:1 row_mask:0xf bank_mask:0xf
; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX1164GISEL-NEXT: v_add_f32_dpp v3, v3, v3 row_shr:1 row_mask:0xf bank_mask:0xf
; GFX1164GISEL-NEXT: v_mbcnt_hi_u32_b32 v5, -1, v5
-; GFX1164GISEL-NEXT: v_add_f32_dpp v3, v3, v3 row_shr:2 row_mask:0xf bank_mask:0xf
; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX1164GISEL-NEXT: v_add_f32_dpp v3, v3, v3 row_shr:2 row_mask:0xf bank_mask:0xf
; GFX1164GISEL-NEXT: v_add_nc_u32_e32 v5, 32, v5
-; GFX1164GISEL-NEXT: v_add_f32_dpp v3, v3, v3 row_shr:4 row_mask:0xf bank_mask:0xf
; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX1164GISEL-NEXT: v_add_f32_dpp v3, v3, v3 row_shr:4 row_mask:0xf bank_mask:0xf
; GFX1164GISEL-NEXT: v_mul_lo_u32 v5, 4, v5
+; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX1164GISEL-NEXT: v_add_f32_dpp v3, v3, v3 row_shr:8 row_mask:0xf bank_mask:0xf
; GFX1164GISEL-NEXT: ds_swizzle_b32 v4, v3 offset:swizzle(BROADCAST,32,15)
; GFX1164GISEL-NEXT: s_waitcnt lgkmcnt(0)
@@ -1566,7 +1568,7 @@ define void @divergent_value_float_dpp(ptr addrspace(1) %out, float %id.x) #0 {
; GFX1164GISEL-NEXT: ds_permute_b32 v4, v5, v3
; GFX1164GISEL-NEXT: s_waitcnt lgkmcnt(0)
; GFX1164GISEL-NEXT: v_add_f32_e32 v3, v3, v4
-; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX1164GISEL-NEXT: v_readlane_b32 s2, v3, 63
; GFX1164GISEL-NEXT: s_mov_b64 exec, s[0:1]
; GFX1164GISEL-NEXT: v_mov_b32_e32 v2, s2
@@ -1589,18 +1591,19 @@ define void @divergent_value_float_dpp(ptr addrspace(1) %out, float %id.x) #0 {
; GFX1132DAGISEL-NEXT: scratch_store_b32 off, v3, s32
; GFX1132DAGISEL-NEXT: scratch_store_b32 off, v4, s32 offset:4
; GFX1132DAGISEL-NEXT: s_mov_b32 exec_lo, s0
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1132DAGISEL-NEXT: s_or_saveexec_b32 s0, -1
-; GFX1132DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132DAGISEL-NEXT: v_cndmask_b32_e64 v3, 0x80000000, v2, s0
-; GFX1132DAGISEL-NEXT: v_add_f32_dpp v3, v3, v3 row_shr:1 row_mask:0xf bank_mask:0xf
; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1132DAGISEL-NEXT: v_add_f32_dpp v3, v3, v3 row_shr:1 row_mask:0xf bank_mask:0xf
; GFX1132DAGISEL-NEXT: v_add_f32_dpp v3, v3, v3 row_shr:2 row_mask:0xf bank_mask:0xf
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132DAGISEL-NEXT: v_add_f32_dpp v3, v3, v3 row_shr:4 row_mask:0xf bank_mask:0xf
-; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_3) | instid1(VALU_DEP_1)
; GFX1132DAGISEL-NEXT: v_add_f32_dpp v3, v3, v3 row_shr:8 row_mask:0xf bank_mask:0xf
; GFX1132DAGISEL-NEXT: ds_swizzle_b32 v4, v3 offset:swizzle(BROADCAST,32,15)
; GFX1132DAGISEL-NEXT: s_waitcnt lgkmcnt(0)
; GFX1132DAGISEL-NEXT: v_add_f32_e32 v3, v3, v4
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX1132DAGISEL-NEXT: v_readlane_b32 s1, v3, 31
; GFX1132DAGISEL-NEXT: s_mov_b32 exec_lo, s0
; GFX1132DAGISEL-NEXT: v_mov_b32_e32 v2, s1
@@ -1621,18 +1624,19 @@ define void @divergent_value_float_dpp(ptr addrspace(1) %out, float %id.x) #0 {
; GFX1132GISEL-NEXT: scratch_store_b32 off, v3, s32
; GFX1132GISEL-NEXT: scratch_store_b32 off, v4, s32 offset:4
; GFX1132GISEL-NEXT: s_mov_b32 exec_lo, s0
+; GFX1132GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1132GISEL-NEXT: s_or_saveexec_b32 s0, -1
-; GFX1132GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132GISEL-NEXT: v_cndmask_b32_e64 v3, 0x80000000, v2, s0
-; GFX1132GISEL-NEXT: v_add_f32_dpp v3, v3, v3 row_shr:1 row_mask:0xf bank_mask:0xf
; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1132GISEL-NEXT: v_add_f32_dpp v3, v3, v3 row_shr:1 row_mask:0xf bank_mask:0xf
; GFX1132GISEL-NEXT: v_add_f32_dpp v3, v3, v3 row_shr:2 row_mask:0xf bank_mask:0xf
+; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132GISEL-NEXT: v_add_f32_dpp v3, v3, v3 row_shr:4 row_mask:0xf bank_mask:0xf
-; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_3) | instid1(VALU_DEP_1)
; GFX1132GISEL-NEXT: v_add_f32_dpp v3, v3, v3 row_shr:8 row_mask:0xf bank_mask:0xf
; GFX1132GISEL-NEXT: ds_swizzle_b32 v4, v3 offset:swizzle(BROADCAST,32,15)
; GFX1132GISEL-NEXT: s_waitcnt lgkmcnt(0)
; GFX1132GISEL-NEXT: v_add_f32_e32 v3, v3, v4
+; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX1132GISEL-NEXT: v_readlane_b32 s1, v3, 31
; GFX1132GISEL-NEXT: s_mov_b32 exec_lo, s0
; GFX1132GISEL-NEXT: v_mov_b32_e32 v2, s1
@@ -1658,21 +1662,22 @@ define void @divergent_value_float_dpp(ptr addrspace(1) %out, float %id.x) #0 {
; GFX12DAGISEL-NEXT: scratch_store_b32 off, v4, s32 offset:4
; GFX12DAGISEL-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12DAGISEL-NEXT: s_mov_b32 exec_lo, s0
+; GFX12DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(VALU_DEP_1)
; GFX12DAGISEL-NEXT: s_or_saveexec_b32 s0, -1
; GFX12DAGISEL-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12DAGISEL-NEXT: v_cndmask_b32_e64 v3, 0x80000000, v2, s0
-; GFX12DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX12DAGISEL-NEXT: v_add_f32_dpp v3, v3, v3 row_shr:1 row_mask:0xf bank_mask:0xf
-; GFX12DAGISEL-NEXT: v_add_f32_dpp v3, v3, v3 row_shr:2 row_mask:0xf bank_mask:0xf
; GFX12DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX12DAGISEL-NEXT: v_add_f32_dpp v3, v3, v3 row_shr:2 row_mask:0xf bank_mask:0xf
; GFX12DAGISEL-NEXT: v_add_f32_dpp v3, v3, v3 row_shr:4 row_mask:0xf bank_mask:0xf
+; GFX12DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_3) | instid1(VALU_DEP_1)
; GFX12DAGISEL-NEXT: v_add_f32_dpp v3, v3, v3 row_shr:8 row_mask:0xf bank_mask:0xf
; GFX12DAGISEL-NEXT: ds_swizzle_b32 v4, v3 offset:swizzle(BROADCAST,32,15)
; GFX12DAGISEL-NEXT: s_wait_dscnt 0x0
; GFX12DAGISEL-NEXT: v_add_f32_e32 v3, v3, v4
-; GFX12DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12DAGISEL-NEXT: v_readlane_b32 s1, v3, 31
; GFX12DAGISEL-NEXT: s_mov_b32 exec_lo, s0
+; GFX12DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12DAGISEL-NEXT: v_mov_b32_e32 v2, s1
; GFX12DAGISEL-NEXT: global_store_b32 v[0:1], v2, off
; GFX12DAGISEL-NEXT: s_xor_saveexec_b32 s0, -1
@@ -2252,39 +2257,39 @@ define void @divergent_value_double_dpp(ptr addrspace(1) %out, double %in) #0 {
; GFX1164DAGISEL-NEXT: scratch_store_b32 off, v6, s32 offset:16
; GFX1164DAGISEL-NEXT: scratch_store_b64 off, v[7:8], s32 offset:20
; GFX1164DAGISEL-NEXT: s_mov_b64 exec, s[0:1]
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1164DAGISEL-NEXT: s_or_saveexec_b64 s[0:1], -1
-; GFX1164DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX1164DAGISEL-NEXT: v_cndmask_b32_e64 v4, 0, v2, s[0:1]
; GFX1164DAGISEL-NEXT: v_cndmask_b32_e64 v5, 0x80000000, v3, s[0:1]
-; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v6, v4
; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v6, v4
; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v7, v5
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1164DAGISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:1 row_mask:0xf bank_mask:0xf
-; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164DAGISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:1 row_mask:0xf bank_mask:0xf
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164DAGISEL-NEXT: v_add_f64 v[4:5], v[4:5], v[6:7]
-; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v6, v4
-; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v7, v5
; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v7, v5
; GFX1164DAGISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:2 row_mask:0xf bank_mask:0xf
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164DAGISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:2 row_mask:0xf bank_mask:0xf
-; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164DAGISEL-NEXT: v_add_f64 v[4:5], v[4:5], v[6:7]
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v6, v4
-; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v7, v5
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1164DAGISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:4 row_mask:0xf bank_mask:0xf
-; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164DAGISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:4 row_mask:0xf bank_mask:0xf
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164DAGISEL-NEXT: v_add_f64 v[4:5], v[4:5], v[6:7]
-; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v6, v4
-; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v7, v5
; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v7, v5
; GFX1164DAGISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:8 row_mask:0xf bank_mask:0xf
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164DAGISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:8 row_mask:0xf bank_mask:0xf
-; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1164DAGISEL-NEXT: v_add_f64 v[4:5], v[4:5], v[6:7]
; GFX1164DAGISEL-NEXT: ds_swizzle_b32 v6, v4 offset:swizzle(BROADCAST,32,15)
; GFX1164DAGISEL-NEXT: ds_swizzle_b32 v7, v5 offset:swizzle(BROADCAST,32,15)
@@ -2301,7 +2306,7 @@ define void @divergent_value_double_dpp(ptr addrspace(1) %out, double %in) #0 {
; GFX1164DAGISEL-NEXT: s_waitcnt lgkmcnt(0)
; GFX1164DAGISEL-NEXT: v_add_f64 v[4:5], v[4:5], v[7:8]
; GFX1164DAGISEL-NEXT: v_readlane_b32 s2, v4, 63
-; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2)
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX1164DAGISEL-NEXT: v_readlane_b32 s3, v5, 63
; GFX1164DAGISEL-NEXT: s_mov_b64 exec, s[0:1]
; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v2, s2
@@ -2328,39 +2333,39 @@ define void @divergent_value_double_dpp(ptr addrspace(1) %out, double %in) #0 {
; GFX1164GISEL-NEXT: scratch_store_b32 off, v6, s32 offset:16
; GFX1164GISEL-NEXT: scratch_store_b64 off, v[7:8], s32 offset:20
; GFX1164GISEL-NEXT: s_mov_b64 exec, s[0:1]
+; GFX1164GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1164GISEL-NEXT: s_or_saveexec_b64 s[0:1], -1
-; GFX1164GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX1164GISEL-NEXT: v_cndmask_b32_e64 v4, 0, v2, s[0:1]
; GFX1164GISEL-NEXT: v_cndmask_b32_e64 v5, 0x80000000, v3, s[0:1]
-; GFX1164GISEL-NEXT: v_mov_b32_e32 v6, v4
; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX1164GISEL-NEXT: v_mov_b32_e32 v6, v4
; GFX1164GISEL-NEXT: v_mov_b32_e32 v7, v5
+; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1164GISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:1 row_mask:0xf bank_mask:0xf
-; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164GISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:1 row_mask:0xf bank_mask:0xf
+; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164GISEL-NEXT: v_add_f64 v[4:5], v[4:5], v[6:7]
-; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1164GISEL-NEXT: v_mov_b32_e32 v6, v4
-; GFX1164GISEL-NEXT: v_mov_b32_e32 v7, v5
; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX1164GISEL-NEXT: v_mov_b32_e32 v7, v5
; GFX1164GISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:2 row_mask:0xf bank_mask:0xf
+; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164GISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:2 row_mask:0xf bank_mask:0xf
-; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164GISEL-NEXT: v_add_f64 v[4:5], v[4:5], v[6:7]
+; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1164GISEL-NEXT: v_mov_b32_e32 v6, v4
-; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1164GISEL-NEXT: v_mov_b32_e32 v7, v5
+; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1164GISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:4 row_mask:0xf bank_mask:0xf
-; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164GISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:4 row_mask:0xf bank_mask:0xf
+; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164GISEL-NEXT: v_add_f64 v[4:5], v[4:5], v[6:7]
-; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1164GISEL-NEXT: v_mov_b32_e32 v6, v4
-; GFX1164GISEL-NEXT: v_mov_b32_e32 v7, v5
; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX1164GISEL-NEXT: v_mov_b32_e32 v7, v5
; GFX1164GISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:8 row_mask:0xf bank_mask:0xf
+; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164GISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:8 row_mask:0xf bank_mask:0xf
-; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1164GISEL-NEXT: v_add_f64 v[4:5], v[4:5], v[6:7]
; GFX1164GISEL-NEXT: ds_swizzle_b32 v6, v4 offset:swizzle(BROADCAST,32,15)
; GFX1164GISEL-NEXT: ds_swizzle_b32 v7, v5 offset:swizzle(BROADCAST,32,15)
@@ -2377,7 +2382,7 @@ define void @divergent_value_double_dpp(ptr addrspace(1) %out, double %in) #0 {
; GFX1164GISEL-NEXT: s_waitcnt lgkmcnt(0)
; GFX1164GISEL-NEXT: v_add_f64 v[4:5], v[4:5], v[7:8]
; GFX1164GISEL-NEXT: v_readlane_b32 s2, v4, 63
-; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2)
+; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX1164GISEL-NEXT: v_readlane_b32 s3, v5, 63
; GFX1164GISEL-NEXT: s_mov_b64 exec, s[0:1]
; GFX1164GISEL-NEXT: v_mov_b32_e32 v2, s2
@@ -2402,42 +2407,43 @@ define void @divergent_value_double_dpp(ptr addrspace(1) %out, double %in) #0 {
; GFX1132DAGISEL-NEXT: scratch_store_b64 off, v[4:5], s32
; GFX1132DAGISEL-NEXT: scratch_store_b64 off, v[6:7], s32 offset:8
; GFX1132DAGISEL-NEXT: s_mov_b32 exec_lo, s0
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1132DAGISEL-NEXT: s_or_saveexec_b32 s2, -1
-; GFX1132DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX1132DAGISEL-NEXT: v_cndmask_b32_e64 v4, 0, v2, s2
; GFX1132DAGISEL-NEXT: v_cndmask_b32_e64 v5, 0x80000000, v3, s2
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132DAGISEL-NEXT: v_dual_mov_b32 v6, v4 :: v_dual_mov_b32 v7, v5
-; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1132DAGISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:1 row_mask:0xf bank_mask:0xf
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132DAGISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:1 row_mask:0xf bank_mask:0xf
-; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132DAGISEL-NEXT: v_add_f64 v[4:5], v[4:5], v[6:7]
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132DAGISEL-NEXT: v_dual_mov_b32 v6, v4 :: v_dual_mov_b32 v7, v5
-; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1132DAGISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:2 row_mask:0xf bank_mask:0xf
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132DAGISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:2 row_mask:0xf bank_mask:0xf
-; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132DAGISEL-NEXT: v_add_f64 v[4:5], v[4:5], v[6:7]
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132DAGISEL-NEXT: v_dual_mov_b32 v6, v4 :: v_dual_mov_b32 v7, v5
-; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1132DAGISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:4 row_mask:0xf bank_mask:0xf
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132DAGISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:4 row_mask:0xf bank_mask:0xf
-; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132DAGISEL-NEXT: v_add_f64 v[4:5], v[4:5], v[6:7]
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132DAGISEL-NEXT: v_dual_mov_b32 v6, v4 :: v_dual_mov_b32 v7, v5
-; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1132DAGISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:8 row_mask:0xf bank_mask:0xf
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132DAGISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:8 row_mask:0xf bank_mask:0xf
-; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_4) | instid1(VALU_DEP_1)
; GFX1132DAGISEL-NEXT: v_add_f64 v[4:5], v[4:5], v[6:7]
; GFX1132DAGISEL-NEXT: ds_swizzle_b32 v6, v4 offset:swizzle(BROADCAST,32,15)
; GFX1132DAGISEL-NEXT: ds_swizzle_b32 v7, v5 offset:swizzle(BROADCAST,32,15)
; GFX1132DAGISEL-NEXT: s_waitcnt lgkmcnt(0)
; GFX1132DAGISEL-NEXT: v_add_f64 v[4:5], v[4:5], v[6:7]
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1132DAGISEL-NEXT: v_readlane_b32 s0, v4, 31
-; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX1132DAGISEL-NEXT: v_readlane_b32 s1, v5, 31
; GFX1132DAGISEL-NEXT: s_mov_b32 exec_lo, s2
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1132DAGISEL-NEXT: v_dual_mov_b32 v3, s1 :: v_dual_mov_b32 v2, s0
; GFX1132DAGISEL-NEXT: global_store_b64 v[0:1], v[2:3], off
; GFX1132DAGISEL-NEXT: s_xor_saveexec_b32 s0, -1
@@ -2456,42 +2462,43 @@ define void @divergent_value_double_dpp(ptr addrspace(1) %out, double %in) #0 {
; GFX1132GISEL-NEXT: scratch_store_b64 off, v[4:5], s32
; GFX1132GISEL-NEXT: scratch_store_b64 off, v[6:7], s32 offset:8
; GFX1132GISEL-NEXT: s_mov_b32 exec_lo, s0
+; GFX1132GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1132GISEL-NEXT: s_or_saveexec_b32 s2, -1
-; GFX1132GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX1132GISEL-NEXT: v_cndmask_b32_e64 v4, 0, v2, s2
; GFX1132GISEL-NEXT: v_cndmask_b32_e64 v5, 0x80000000, v3, s2
+; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132GISEL-NEXT: v_dual_mov_b32 v6, v4 :: v_dual_mov_b32 v7, v5
-; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1132GISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:1 row_mask:0xf bank_mask:0xf
+; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132GISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:1 row_mask:0xf bank_mask:0xf
-; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132GISEL-NEXT: v_add_f64 v[4:5], v[4:5], v[6:7]
+; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132GISEL-NEXT: v_dual_mov_b32 v6, v4 :: v_dual_mov_b32 v7, v5
-; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1132GISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:2 row_mask:0xf bank_mask:0xf
+; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132GISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:2 row_mask:0xf bank_mask:0xf
-; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132GISEL-NEXT: v_add_f64 v[4:5], v[4:5], v[6:7]
+; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132GISEL-NEXT: v_dual_mov_b32 v6, v4 :: v_dual_mov_b32 v7, v5
-; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1132GISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:4 row_mask:0xf bank_mask:0xf
+; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132GISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:4 row_mask:0xf bank_mask:0xf
-; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132GISEL-NEXT: v_add_f64 v[4:5], v[4:5], v[6:7]
+; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132GISEL-NEXT: v_dual_mov_b32 v6, v4 :: v_dual_mov_b32 v7, v5
-; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1132GISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:8 row_mask:0xf bank_mask:0xf
+; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132GISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:8 row_mask:0xf bank_mask:0xf
-; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_4) | instid1(VALU_DEP_1)
; GFX1132GISEL-NEXT: v_add_f64 v[4:5], v[4:5], v[6:7]
; GFX1132GISEL-NEXT: ds_swizzle_b32 v6, v4 offset:swizzle(BROADCAST,32,15)
; GFX1132GISEL-NEXT: ds_swizzle_b32 v7, v5 offset:swizzle(BROADCAST,32,15)
; GFX1132GISEL-NEXT: s_waitcnt lgkmcnt(0)
; GFX1132GISEL-NEXT: v_add_f64 v[4:5], v[4:5], v[6:7]
+; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1132GISEL-NEXT: v_readlane_b32 s0, v4, 31
-; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX1132GISEL-NEXT: v_readlane_b32 s1, v5, 31
; GFX1132GISEL-NEXT: s_mov_b32 exec_lo, s2
+; GFX1132GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1132GISEL-NEXT: v_dual_mov_b32 v3, s1 :: v_dual_mov_b32 v2, s0
; GFX1132GISEL-NEXT: global_store_b64 v[0:1], v[2:3], off
; GFX1132GISEL-NEXT: s_xor_saveexec_b32 s0, -1
@@ -2515,40 +2522,41 @@ define void @divergent_value_double_dpp(ptr addrspace(1) %out, double %in) #0 {
; GFX12DAGISEL-NEXT: scratch_store_b64 off, v[6:7], s32 offset:8
; GFX12DAGISEL-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12DAGISEL-NEXT: s_mov_b32 exec_lo, s0
+; GFX12DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_3) | instid1(VALU_DEP_1)
; GFX12DAGISEL-NEXT: s_or_saveexec_b32 s2, -1
; GFX12DAGISEL-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12DAGISEL-NEXT: v_cndmask_b32_e64 v4, 0, v2, s2
; GFX12DAGISEL-NEXT: v_cndmask_b32_e64 v5, 0x80000000, v3, s2
-; GFX12DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX12DAGISEL-NEXT: v_dual_mov_b32 v6, v4 :: v_dual_mov_b32 v7, v5
+; GFX12DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX12DAGISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:1 row_mask:0xf bank_mask:0xf
-; GFX12DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX12DAGISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:1 row_mask:0xf bank_mask:0xf
-; GFX12DAGISEL-NEXT: v_add_f64_e32 v[4:5], v[4:5], v[6:7]
; GFX12DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX12DAGISEL-NEXT: v_add_f64_e32 v[4:5], v[4:5], v[6:7]
; GFX12DAGISEL-NEXT: v_dual_mov_b32 v6, v4 :: v_dual_mov_b32 v7, v5
+; GFX12DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX12DAGISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:2 row_mask:0xf bank_mask:0xf
-; GFX12DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX12DAGISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:2 row_mask:0xf bank_mask:0xf
-; GFX12DAGISEL-NEXT: v_add_f64_e32 v[4:5], v[4:5], v[6:7]
; GFX12DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX12DAGISEL-NEXT: v_add_f64_e32 v[4:5], v[4:5], v[6:7]
; GFX12DAGISEL-NEXT: v_dual_mov_b32 v6, v4 :: v_dual_mov_b32 v7, v5
+; GFX12DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX12DAGISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:4 row_mask:0xf bank_mask:0xf
-; GFX12DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX12DAGISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:4 row_mask:0xf bank_mask:0xf
-; GFX12DAGISEL-NEXT: v_add_f64_e32 v[4:5], v[4:5], v[6:7]
; GFX12DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX12DAGISEL-NEXT: v_add_f64_e32 v[4:5], v[4:5], v[6:7]
; GFX12DAGISEL-NEXT: v_dual_mov_b32 v6, v4 :: v_dual_mov_b32 v7, v5
+; GFX12DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX12DAGISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:8 row_mask:0xf bank_mask:0xf
-; GFX12DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX12DAGISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:8 row_mask:0xf bank_mask:0xf
+; GFX12DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_4) | instid1(VALU_DEP_1)
; GFX12DAGISEL-NEXT: v_add_f64_e32 v[4:5], v[4:5], v[6:7]
; GFX12DAGISEL-NEXT: ds_swizzle_b32 v6, v4 offset:swizzle(BROADCAST,32,15)
; GFX12DAGISEL-NEXT: ds_swizzle_b32 v7, v5 offset:swizzle(BROADCAST,32,15)
; GFX12DAGISEL-NEXT: s_wait_dscnt 0x0
; GFX12DAGISEL-NEXT: v_add_f64_e32 v[4:5], v[4:5], v[6:7]
-; GFX12DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX12DAGISEL-NEXT: v_readlane_b32 s0, v4, 31
+; GFX12DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX12DAGISEL-NEXT: v_readlane_b32 s1, v5, 31
; GFX12DAGISEL-NEXT: s_mov_b32 exec_lo, s2
; GFX12DAGISEL-NEXT: v_dual_mov_b32 v3, s1 :: v_dual_mov_b32 v2, s0
@@ -2870,7 +2878,7 @@ define amdgpu_kernel void @divergent_cfg_float(ptr addrspace(1) %out, float %in,
; GFX1164DAGISEL-NEXT: v_and_b32_e32 v0, 0x3ff, v0
; GFX1164DAGISEL-NEXT: s_mov_b64 s[2:3], exec
; GFX1164DAGISEL-NEXT: ; implicit-def: $sgpr6
-; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164DAGISEL-NEXT: v_cmpx_lt_u32_e32 15, v0
; GFX1164DAGISEL-NEXT: s_xor_b64 s[2:3], exec, s[2:3]
; GFX1164DAGISEL-NEXT: s_cbranch_execz .LBB6_2
@@ -2885,7 +2893,7 @@ define amdgpu_kernel void @divergent_cfg_float(ptr addrspace(1) %out, float %in,
; GFX1164DAGISEL-NEXT: v_readfirstlane_b32 s6, v0
; GFX1164DAGISEL-NEXT: .LBB6_2: ; %Flow
; GFX1164DAGISEL-NEXT: s_or_saveexec_b64 s[2:3], s[2:3]
-; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v0, s6
; GFX1164DAGISEL-NEXT: s_xor_b64 exec, exec, s[2:3]
; GFX1164DAGISEL-NEXT: s_cbranch_execz .LBB6_4
@@ -2914,7 +2922,7 @@ define amdgpu_kernel void @divergent_cfg_float(ptr addrspace(1) %out, float %in,
; GFX1164GISEL-NEXT: v_and_b32_e32 v0, 0x3ff, v0
; GFX1164GISEL-NEXT: s_mov_b64 s[2:3], exec
; GFX1164GISEL-NEXT: ; implicit-def: $sgpr6
-; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164GISEL-NEXT: v_cmpx_le_u32_e32 16, v0
; GFX1164GISEL-NEXT: s_xor_b64 s[2:3], exec, s[2:3]
; GFX1164GISEL-NEXT: s_cbranch_execz .LBB6_2
@@ -2929,7 +2937,7 @@ define amdgpu_kernel void @divergent_cfg_float(ptr addrspace(1) %out, float %in,
; GFX1164GISEL-NEXT: v_readfirstlane_b32 s6, v0
; GFX1164GISEL-NEXT: .LBB6_2: ; %Flow
; GFX1164GISEL-NEXT: s_or_saveexec_b64 s[2:3], s[2:3]
-; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX1164GISEL-NEXT: v_mov_b32_e32 v0, s6
; GFX1164GISEL-NEXT: s_xor_b64 exec, exec, s[2:3]
; GFX1164GISEL-NEXT: s_cbranch_execz .LBB6_4
@@ -2958,9 +2966,10 @@ define amdgpu_kernel void @divergent_cfg_float(ptr addrspace(1) %out, float %in,
; GFX1132DAGISEL-NEXT: v_and_b32_e32 v0, 0x3ff, v0
; GFX1132DAGISEL-NEXT: s_mov_b32 s2, exec_lo
; GFX1132DAGISEL-NEXT: ; implicit-def: $sgpr3
-; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132DAGISEL-NEXT: v_cmpx_lt_u32_e32 15, v0
; GFX1132DAGISEL-NEXT: s_xor_b32 s2, exec_lo, s2
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1132DAGISEL-NEXT: s_cbranch_execz .LBB6_2
; GFX1132DAGISEL-NEXT: ; %bb.1: ; %else
; GFX1132DAGISEL-NEXT: s_mov_b32 s3, exec_lo
@@ -2974,7 +2983,7 @@ define amdgpu_kernel void @divergent_cfg_float(ptr addrspace(1) %out, float %in,
; GFX1132DAGISEL-NEXT: .LBB6_2: ; %Flow
; GFX1132DAGISEL-NEXT: s_waitcnt lgkmcnt(0)
; GFX1132DAGISEL-NEXT: s_or_saveexec_b32 s0, s2
-; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX1132DAGISEL-NEXT: v_mov_b32_e32 v0, s3
; GFX1132DAGISEL-NEXT: s_xor_b32 exec_lo, exec_lo, s0
; GFX1132DAGISEL-NEXT: s_cbranch_execz .LBB6_4
@@ -3002,9 +3011,10 @@ define amdgpu_kernel void @divergent_cfg_float(ptr addrspace(1) %out, float %in,
; GFX1132GISEL-NEXT: v_and_b32_e32 v0, 0x3ff, v0
; GFX1132GISEL-NEXT: s_mov_b32 s2, exec_lo
; GFX1132GISEL-NEXT: ; implicit-def: $sgpr3
-; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132GISEL-NEXT: v_cmpx_le_u32_e32 16, v0
; GFX1132GISEL-NEXT: s_xor_b32 s2, exec_lo, s2
+; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1132GISEL-NEXT: s_cbranch_execz .LBB6_2
; GFX1132GISEL-NEXT: ; %bb.1: ; %else
; GFX1132GISEL-NEXT: s_mov_b32 s3, exec_lo
@@ -3018,7 +3028,7 @@ define amdgpu_kernel void @divergent_cfg_float(ptr addrspace(1) %out, float %in,
; GFX1132GISEL-NEXT: .LBB6_2: ; %Flow
; GFX1132GISEL-NEXT: s_waitcnt lgkmcnt(0)
; GFX1132GISEL-NEXT: s_or_saveexec_b32 s0, s2
-; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX1132GISEL-NEXT: v_mov_b32_e32 v0, s3
; GFX1132GISEL-NEXT: s_xor_b32 exec_lo, exec_lo, s0
; GFX1132GISEL-NEXT: s_cbranch_execz .LBB6_4
@@ -3046,9 +3056,10 @@ define amdgpu_kernel void @divergent_cfg_float(ptr addrspace(1) %out, float %in,
; GFX12DAGISEL-NEXT: v_and_b32_e32 v0, 0x3ff, v0
; GFX12DAGISEL-NEXT: s_mov_b32 s2, exec_lo
; GFX12DAGISEL-NEXT: ; implicit-def: $sgpr3
-; GFX12DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX12DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX12DAGISEL-NEXT: v_cmpx_lt_u32_e32 15, v0
; GFX12DAGISEL-NEXT: s_xor_b32 s2, exec_lo, s2
+; GFX12DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12DAGISEL-NEXT: s_cbranch_execz .LBB6_2
; GFX12DAGISEL-NEXT: ; %bb.1: ; %else
; GFX12DAGISEL-NEXT: s_mov_b32 s3, exec_lo
@@ -3062,7 +3073,7 @@ define amdgpu_kernel void @divergent_cfg_float(ptr addrspace(1) %out, float %in,
; GFX12DAGISEL-NEXT: .LBB6_2: ; %Flow
; GFX12DAGISEL-NEXT: s_wait_kmcnt 0x0
; GFX12DAGISEL-NEXT: s_or_saveexec_b32 s0, s2
-; GFX12DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX12DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
; GFX12DAGISEL-NEXT: v_mov_b32_e32 v0, s3
; GFX12DAGISEL-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12DAGISEL-NEXT: s_xor_b32 exec_lo, exec_lo, s0
@@ -3985,7 +3996,7 @@ define amdgpu_kernel void @divergent_cfg_double(ptr addrspace(1) %out, double %i
; GFX1164DAGISEL-NEXT: v_and_b32_e32 v0, 0x3ff, v0
; GFX1164DAGISEL-NEXT: s_mov_b64 s[6:7], exec
; GFX1164DAGISEL-NEXT: ; implicit-def: $sgpr8_sgpr9
-; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164DAGISEL-NEXT: v_cmpx_lt_u32_e32 15, v0
; GFX1164DAGISEL-NEXT: s_xor_b64 s[6:7], exec, s[6:7]
; GFX1164DAGISEL-NEXT: s_cbranch_execz .LBB9_2
@@ -4007,6 +4018,7 @@ define amdgpu_kernel void @divergent_cfg_double(ptr addrspace(1) %out, double %i
; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v0, s8
; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v1, s9
; GFX1164DAGISEL-NEXT: s_xor_b64 exec, exec, s[2:3]
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164DAGISEL-NEXT: s_cbranch_execz .LBB9_4
; GFX1164DAGISEL-NEXT: ; %bb.3: ; %if
; GFX1164DAGISEL-NEXT: s_mov_b64 s[6:7], exec
@@ -4023,6 +4035,7 @@ define amdgpu_kernel void @divergent_cfg_double(ptr addrspace(1) %out, double %i
; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v1, s5
; GFX1164DAGISEL-NEXT: .LBB9_4: ; %endif
; GFX1164DAGISEL-NEXT: s_or_b64 exec, exec, s[2:3]
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v2, 0
; GFX1164DAGISEL-NEXT: global_store_b64 v2, v[0:1], s[0:1]
; GFX1164DAGISEL-NEXT: s_endpgm
@@ -4033,7 +4046,7 @@ define amdgpu_kernel void @divergent_cfg_double(ptr addrspace(1) %out, double %i
; GFX1164GISEL-NEXT: v_and_b32_e32 v0, 0x3ff, v0
; GFX1164GISEL-NEXT: s_mov_b64 s[6:7], exec
; GFX1164GISEL-NEXT: ; implicit-def: $sgpr8_sgpr9
-; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164GISEL-NEXT: v_cmpx_le_u32_e32 16, v0
; GFX1164GISEL-NEXT: s_xor_b64 s[6:7], exec, s[6:7]
; GFX1164GISEL-NEXT: s_cbranch_execz .LBB9_2
@@ -4055,6 +4068,7 @@ define amdgpu_kernel void @divergent_cfg_double(ptr addrspace(1) %out, double %i
; GFX1164GISEL-NEXT: v_mov_b32_e32 v0, s8
; GFX1164GISEL-NEXT: v_mov_b32_e32 v1, s9
; GFX1164GISEL-NEXT: s_xor_b64 exec, exec, s[2:3]
+; GFX1164GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164GISEL-NEXT: s_cbranch_execz .LBB9_4
; GFX1164GISEL-NEXT: ; %bb.3: ; %if
; GFX1164GISEL-NEXT: s_mov_b64 s[6:7], exec
@@ -4072,6 +4086,7 @@ define amdgpu_kernel void @divergent_cfg_double(ptr addrspace(1) %out, double %i
; GFX1164GISEL-NEXT: v_mov_b32_e32 v1, s5
; GFX1164GISEL-NEXT: .LBB9_4: ; %endif
; GFX1164GISEL-NEXT: s_or_b64 exec, exec, s[2:3]
+; GFX1164GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164GISEL-NEXT: v_mov_b32_e32 v2, 0
; GFX1164GISEL-NEXT: global_store_b64 v2, v[0:1], s[0:1]
; GFX1164GISEL-NEXT: s_endpgm
@@ -4084,9 +4099,10 @@ define amdgpu_kernel void @divergent_cfg_double(ptr addrspace(1) %out, double %i
; GFX1132DAGISEL-NEXT: v_and_b32_e32 v0, 0x3ff, v0
; GFX1132DAGISEL-NEXT: s_mov_b32 s8, exec_lo
; GFX1132DAGISEL-NEXT: ; implicit-def: $sgpr6_sgpr7
-; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132DAGISEL-NEXT: v_cmpx_lt_u32_e32 15, v0
; GFX1132DAGISEL-NEXT: s_xor_b32 s8, exec_lo, s8
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1132DAGISEL-NEXT: s_cbranch_execz .LBB9_2
; GFX1132DAGISEL-NEXT: ; %bb.1: ; %else
; GFX1132DAGISEL-NEXT: s_mov_b32 s6, exec_lo
@@ -4102,7 +4118,7 @@ define amdgpu_kernel void @divergent_cfg_double(ptr addrspace(1) %out, double %i
; GFX1132DAGISEL-NEXT: .LBB9_2: ; %Flow
; GFX1132DAGISEL-NEXT: s_waitcnt lgkmcnt(0)
; GFX1132DAGISEL-NEXT: s_or_saveexec_b32 s2, s8
-; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX1132DAGISEL-NEXT: v_dual_mov_b32 v0, s6 :: v_dual_mov_b32 v1, s7
; GFX1132DAGISEL-NEXT: s_xor_b32 exec_lo, exec_lo, s2
; GFX1132DAGISEL-NEXT: s_cbranch_execz .LBB9_4
@@ -4119,6 +4135,7 @@ define amdgpu_kernel void @divergent_cfg_double(ptr addrspace(1) %out, double %i
; GFX1132DAGISEL-NEXT: v_dual_mov_b32 v0, s4 :: v_dual_mov_b32 v1, s5
; GFX1132DAGISEL-NEXT: .LBB9_4: ; %endif
; GFX1132DAGISEL-NEXT: s_or_b32 exec_lo, exec_lo, s2
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1132DAGISEL-NEXT: v_mov_b32_e32 v2, 0
; GFX1132DAGISEL-NEXT: global_store_b64 v2, v[0:1], s[0:1]
; GFX1132DAGISEL-NEXT: s_endpgm
@@ -4129,9 +4146,10 @@ define amdgpu_kernel void @divergent_cfg_double(ptr addrspace(1) %out, double %i
; GFX1132GISEL-NEXT: v_and_b32_e32 v0, 0x3ff, v0
; GFX1132GISEL-NEXT: s_mov_b32 s8, exec_lo
; GFX1132GISEL-NEXT: ; implicit-def: $sgpr6_sgpr7
-; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132GISEL-NEXT: v_cmpx_le_u32_e32 16, v0
; GFX1132GISEL-NEXT: s_xor_b32 s8, exec_lo, s8
+; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1132GISEL-NEXT: s_cbranch_execz .LBB9_2
; GFX1132GISEL-NEXT: ; %bb.1: ; %else
; GFX1132GISEL-NEXT: s_mov_b32 s6, exec_lo
@@ -4147,7 +4165,7 @@ define amdgpu_kernel void @divergent_cfg_double(ptr addrspace(1) %out, double %i
; GFX1132GISEL-NEXT: .LBB9_2: ; %Flow
; GFX1132GISEL-NEXT: s_waitcnt lgkmcnt(0)
; GFX1132GISEL-NEXT: s_or_saveexec_b32 s2, s8
-; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX1132GISEL-NEXT: v_dual_mov_b32 v0, s6 :: v_dual_mov_b32 v1, s7
; GFX1132GISEL-NEXT: s_xor_b32 exec_lo, exec_lo, s2
; GFX1132GISEL-NEXT: s_cbranch_execz .LBB9_4
@@ -4166,6 +4184,7 @@ define amdgpu_kernel void @divergent_cfg_double(ptr addrspace(1) %out, double %i
; GFX1132GISEL-NEXT: v_dual_mov_b32 v0, s4 :: v_dual_mov_b32 v1, s5
; GFX1132GISEL-NEXT: .LBB9_4: ; %endif
; GFX1132GISEL-NEXT: s_or_b32 exec_lo, exec_lo, s2
+; GFX1132GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1132GISEL-NEXT: v_mov_b32_e32 v2, 0
; GFX1132GISEL-NEXT: global_store_b64 v2, v[0:1], s[0:1]
; GFX1132GISEL-NEXT: s_endpgm
@@ -4178,9 +4197,10 @@ define amdgpu_kernel void @divergent_cfg_double(ptr addrspace(1) %out, double %i
; GFX12DAGISEL-NEXT: v_and_b32_e32 v0, 0x3ff, v0
; GFX12DAGISEL-NEXT: s_mov_b32 s8, exec_lo
; GFX12DAGISEL-NEXT: ; implicit-def: $sgpr6_sgpr7
-; GFX12DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX12DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX12DAGISEL-NEXT: v_cmpx_lt_u32_e32 15, v0
; GFX12DAGISEL-NEXT: s_xor_b32 s8, exec_lo, s8
+; GFX12DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12DAGISEL-NEXT: s_cbranch_execz .LBB9_2
; GFX12DAGISEL-NEXT: ; %bb.1: ; %else
; GFX12DAGISEL-NEXT: s_mov_b32 s6, exec_lo
@@ -4197,7 +4217,7 @@ define amdgpu_kernel void @divergent_cfg_double(ptr addrspace(1) %out, double %i
; GFX12DAGISEL-NEXT: s_wait_kmcnt 0x0
; GFX12DAGISEL-NEXT: s_or_saveexec_b32 s2, s8
; GFX12DAGISEL-NEXT: s_wait_alu depctr_va_sdst(0)
-; GFX12DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX12DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
; GFX12DAGISEL-NEXT: v_dual_mov_b32 v0, s6 :: v_dual_mov_b32 v1, s7
; GFX12DAGISEL-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12DAGISEL-NEXT: s_xor_b32 exec_lo, exec_lo, s2
@@ -4217,6 +4237,7 @@ define amdgpu_kernel void @divergent_cfg_double(ptr addrspace(1) %out, double %i
; GFX12DAGISEL-NEXT: v_dual_mov_b32 v0, s4 :: v_dual_mov_b32 v1, s5
; GFX12DAGISEL-NEXT: .LBB9_4: ; %endif
; GFX12DAGISEL-NEXT: s_or_b32 exec_lo, exec_lo, s2
+; GFX12DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12DAGISEL-NEXT: v_mov_b32_e32 v2, 0
; GFX12DAGISEL-NEXT: global_store_b64 v2, v[0:1], s[0:1]
; GFX12DAGISEL-NEXT: s_endpgm
diff --git a/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.reduce.fmax.ll b/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.reduce.fmax.ll
index 78e0dd4908f3fb..e78e37baa10eb0 100644
--- a/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.reduce.fmax.ll
+++ b/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.reduce.fmax.ll
@@ -1348,19 +1348,20 @@ define void @divergent_value_float_dpp(ptr addrspace(1) %out, float %in) #0 {
; GFX1164DAGISEL-NEXT: scratch_store_b32 off, v4, s32 offset:4
; GFX1164DAGISEL-NEXT: scratch_store_b32 off, v5, s32 offset:8
; GFX1164DAGISEL-NEXT: s_mov_b64 exec, s[0:1]
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1164DAGISEL-NEXT: s_or_saveexec_b64 s[0:1], -1
-; GFX1164DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX1164DAGISEL-NEXT: v_cndmask_b32_e64 v3, 0x7fc00000, v2, s[0:1]
; GFX1164DAGISEL-NEXT: v_mbcnt_lo_u32_b32 v5, -1, 0
-; GFX1164DAGISEL-NEXT: v_max_f32_dpp v3, v3, v3 row_shr:1 row_mask:0xf bank_mask:0xf
; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX1164DAGISEL-NEXT: v_max_f32_dpp v3, v3, v3 row_shr:1 row_mask:0xf bank_mask:0xf
; GFX1164DAGISEL-NEXT: v_mbcnt_hi_u32_b32 v5, -1, v5
-; GFX1164DAGISEL-NEXT: v_max_f32_dpp v3, v3, v3 row_shr:2 row_mask:0xf bank_mask:0xf
; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX1164DAGISEL-NEXT: v_max_f32_dpp v3, v3, v3 row_shr:2 row_mask:0xf bank_mask:0xf
; GFX1164DAGISEL-NEXT: v_add_nc_u32_e32 v5, 32, v5
-; GFX1164DAGISEL-NEXT: v_max_f32_dpp v3, v3, v3 row_shr:4 row_mask:0xf bank_mask:0xf
; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX1164DAGISEL-NEXT: v_max_f32_dpp v3, v3, v3 row_shr:4 row_mask:0xf bank_mask:0xf
; GFX1164DAGISEL-NEXT: v_mul_lo_u32 v5, 4, v5
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX1164DAGISEL-NEXT: v_max_f32_dpp v3, v3, v3 row_shr:8 row_mask:0xf bank_mask:0xf
; GFX1164DAGISEL-NEXT: ds_swizzle_b32 v4, v3 offset:swizzle(BROADCAST,32,15)
; GFX1164DAGISEL-NEXT: s_waitcnt lgkmcnt(0)
@@ -1368,7 +1369,7 @@ define void @divergent_value_float_dpp(ptr addrspace(1) %out, float %in) #0 {
; GFX1164DAGISEL-NEXT: ds_permute_b32 v4, v5, v3
; GFX1164DAGISEL-NEXT: s_waitcnt lgkmcnt(0)
; GFX1164DAGISEL-NEXT: v_max_f32_e32 v3, v3, v4
-; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX1164DAGISEL-NEXT: v_readlane_b32 s2, v3, 63
; GFX1164DAGISEL-NEXT: s_mov_b64 exec, s[0:1]
; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v2, s2
@@ -1392,19 +1393,20 @@ define void @divergent_value_float_dpp(ptr addrspace(1) %out, float %in) #0 {
; GFX1164GISEL-NEXT: scratch_store_b32 off, v4, s32 offset:4
; GFX1164GISEL-NEXT: scratch_store_b32 off, v5, s32 offset:8
; GFX1164GISEL-NEXT: s_mov_b64 exec, s[0:1]
+; GFX1164GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1164GISEL-NEXT: s_or_saveexec_b64 s[0:1], -1
-; GFX1164GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX1164GISEL-NEXT: v_cndmask_b32_e64 v3, 0x7fc00000, v2, s[0:1]
; GFX1164GISEL-NEXT: v_mbcnt_lo_u32_b32 v5, -1, 0
-; GFX1164GISEL-NEXT: v_max_f32_dpp v3, v3, v3 row_shr:1 row_mask:0xf bank_mask:0xf
; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX1164GISEL-NEXT: v_max_f32_dpp v3, v3, v3 row_shr:1 row_mask:0xf bank_mask:0xf
; GFX1164GISEL-NEXT: v_mbcnt_hi_u32_b32 v5, -1, v5
-; GFX1164GISEL-NEXT: v_max_f32_dpp v3, v3, v3 row_shr:2 row_mask:0xf bank_mask:0xf
; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX1164GISEL-NEXT: v_max_f32_dpp v3, v3, v3 row_shr:2 row_mask:0xf bank_mask:0xf
; GFX1164GISEL-NEXT: v_add_nc_u32_e32 v5, 32, v5
-; GFX1164GISEL-NEXT: v_max_f32_dpp v3, v3, v3 row_shr:4 row_mask:0xf bank_mask:0xf
; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX1164GISEL-NEXT: v_max_f32_dpp v3, v3, v3 row_shr:4 row_mask:0xf bank_mask:0xf
; GFX1164GISEL-NEXT: v_mul_lo_u32 v5, 4, v5
+; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX1164GISEL-NEXT: v_max_f32_dpp v3, v3, v3 row_shr:8 row_mask:0xf bank_mask:0xf
; GFX1164GISEL-NEXT: ds_swizzle_b32 v4, v3 offset:swizzle(BROADCAST,32,15)
; GFX1164GISEL-NEXT: s_waitcnt lgkmcnt(0)
@@ -1412,7 +1414,7 @@ define void @divergent_value_float_dpp(ptr addrspace(1) %out, float %in) #0 {
; GFX1164GISEL-NEXT: ds_permute_b32 v4, v5, v3
; GFX1164GISEL-NEXT: s_waitcnt lgkmcnt(0)
; GFX1164GISEL-NEXT: v_max_f32_e32 v3, v3, v4
-; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX1164GISEL-NEXT: v_readlane_b32 s2, v3, 63
; GFX1164GISEL-NEXT: s_mov_b64 exec, s[0:1]
; GFX1164GISEL-NEXT: v_mov_b32_e32 v2, s2
@@ -1435,18 +1437,19 @@ define void @divergent_value_float_dpp(ptr addrspace(1) %out, float %in) #0 {
; GFX1132DAGISEL-NEXT: scratch_store_b32 off, v3, s32
; GFX1132DAGISEL-NEXT: scratch_store_b32 off, v4, s32 offset:4
; GFX1132DAGISEL-NEXT: s_mov_b32 exec_lo, s0
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1132DAGISEL-NEXT: s_or_saveexec_b32 s0, -1
-; GFX1132DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132DAGISEL-NEXT: v_cndmask_b32_e64 v3, 0x7fc00000, v2, s0
-; GFX1132DAGISEL-NEXT: v_max_f32_dpp v3, v3, v3 row_shr:1 row_mask:0xf bank_mask:0xf
; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1132DAGISEL-NEXT: v_max_f32_dpp v3, v3, v3 row_shr:1 row_mask:0xf bank_mask:0xf
; GFX1132DAGISEL-NEXT: v_max_f32_dpp v3, v3, v3 row_shr:2 row_mask:0xf bank_mask:0xf
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132DAGISEL-NEXT: v_max_f32_dpp v3, v3, v3 row_shr:4 row_mask:0xf bank_mask:0xf
-; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_3) | instid1(VALU_DEP_1)
; GFX1132DAGISEL-NEXT: v_max_f32_dpp v3, v3, v3 row_shr:8 row_mask:0xf bank_mask:0xf
; GFX1132DAGISEL-NEXT: ds_swizzle_b32 v4, v3 offset:swizzle(BROADCAST,32,15)
; GFX1132DAGISEL-NEXT: s_waitcnt lgkmcnt(0)
; GFX1132DAGISEL-NEXT: v_max_f32_e32 v3, v3, v4
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX1132DAGISEL-NEXT: v_readlane_b32 s1, v3, 31
; GFX1132DAGISEL-NEXT: s_mov_b32 exec_lo, s0
; GFX1132DAGISEL-NEXT: v_mov_b32_e32 v2, s1
@@ -1467,18 +1470,19 @@ define void @divergent_value_float_dpp(ptr addrspace(1) %out, float %in) #0 {
; GFX1132GISEL-NEXT: scratch_store_b32 off, v3, s32
; GFX1132GISEL-NEXT: scratch_store_b32 off, v4, s32 offset:4
; GFX1132GISEL-NEXT: s_mov_b32 exec_lo, s0
+; GFX1132GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1132GISEL-NEXT: s_or_saveexec_b32 s0, -1
-; GFX1132GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132GISEL-NEXT: v_cndmask_b32_e64 v3, 0x7fc00000, v2, s0
-; GFX1132GISEL-NEXT: v_max_f32_dpp v3, v3, v3 row_shr:1 row_mask:0xf bank_mask:0xf
; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1132GISEL-NEXT: v_max_f32_dpp v3, v3, v3 row_shr:1 row_mask:0xf bank_mask:0xf
; GFX1132GISEL-NEXT: v_max_f32_dpp v3, v3, v3 row_shr:2 row_mask:0xf bank_mask:0xf
+; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132GISEL-NEXT: v_max_f32_dpp v3, v3, v3 row_shr:4 row_mask:0xf bank_mask:0xf
-; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_3) | instid1(VALU_DEP_1)
; GFX1132GISEL-NEXT: v_max_f32_dpp v3, v3, v3 row_shr:8 row_mask:0xf bank_mask:0xf
; GFX1132GISEL-NEXT: ds_swizzle_b32 v4, v3 offset:swizzle(BROADCAST,32,15)
; GFX1132GISEL-NEXT: s_waitcnt lgkmcnt(0)
; GFX1132GISEL-NEXT: v_max_f32_e32 v3, v3, v4
+; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX1132GISEL-NEXT: v_readlane_b32 s1, v3, 31
; GFX1132GISEL-NEXT: s_mov_b32 exec_lo, s0
; GFX1132GISEL-NEXT: v_mov_b32_e32 v2, s1
@@ -1504,21 +1508,22 @@ define void @divergent_value_float_dpp(ptr addrspace(1) %out, float %in) #0 {
; GFX12DAGISEL-NEXT: scratch_store_b32 off, v4, s32 offset:4
; GFX12DAGISEL-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12DAGISEL-NEXT: s_mov_b32 exec_lo, s0
+; GFX12DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(VALU_DEP_1)
; GFX12DAGISEL-NEXT: s_or_saveexec_b32 s0, -1
; GFX12DAGISEL-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12DAGISEL-NEXT: v_cndmask_b32_e64 v3, 0x7fc00000, v2, s0
-; GFX12DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX12DAGISEL-NEXT: v_max_num_f32_dpp v3, v3, v3 row_shr:1 row_mask:0xf bank_mask:0xf
-; GFX12DAGISEL-NEXT: v_max_num_f32_dpp v3, v3, v3 row_shr:2 row_mask:0xf bank_mask:0xf
; GFX12DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX12DAGISEL-NEXT: v_max_num_f32_dpp v3, v3, v3 row_shr:2 row_mask:0xf bank_mask:0xf
; GFX12DAGISEL-NEXT: v_max_num_f32_dpp v3, v3, v3 row_shr:4 row_mask:0xf bank_mask:0xf
+; GFX12DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_3) | instid1(VALU_DEP_1)
; GFX12DAGISEL-NEXT: v_max_num_f32_dpp v3, v3, v3 row_shr:8 row_mask:0xf bank_mask:0xf
; GFX12DAGISEL-NEXT: ds_swizzle_b32 v4, v3 offset:swizzle(BROADCAST,32,15)
; GFX12DAGISEL-NEXT: s_wait_dscnt 0x0
; GFX12DAGISEL-NEXT: v_max_num_f32_e32 v3, v3, v4
-; GFX12DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12DAGISEL-NEXT: v_readlane_b32 s1, v3, 31
; GFX12DAGISEL-NEXT: s_mov_b32 exec_lo, s0
+; GFX12DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12DAGISEL-NEXT: v_mov_b32_e32 v2, s1
; GFX12DAGISEL-NEXT: global_store_b32 v[0:1], v2, off
; GFX12DAGISEL-NEXT: s_xor_saveexec_b32 s0, -1
@@ -2098,39 +2103,39 @@ define void @divergent_value_double_dpp(ptr addrspace(1) %out, double %in) #0 {
; GFX1164DAGISEL-NEXT: scratch_store_b32 off, v6, s32 offset:16
; GFX1164DAGISEL-NEXT: scratch_store_b64 off, v[7:8], s32 offset:20
; GFX1164DAGISEL-NEXT: s_mov_b64 exec, s[0:1]
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1164DAGISEL-NEXT: s_or_saveexec_b64 s[0:1], -1
-; GFX1164DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX1164DAGISEL-NEXT: v_cndmask_b32_e64 v4, 0, v2, s[0:1]
; GFX1164DAGISEL-NEXT: v_cndmask_b32_e64 v5, 0x7ff80000, v3, s[0:1]
-; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v6, v4
; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v6, v4
; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v7, v5
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1164DAGISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:1 row_mask:0xf bank_mask:0xf
-; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164DAGISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:1 row_mask:0xf bank_mask:0xf
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164DAGISEL-NEXT: v_max_f64 v[4:5], v[4:5], v[6:7]
-; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v6, v4
-; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v7, v5
; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v7, v5
; GFX1164DAGISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:2 row_mask:0xf bank_mask:0xf
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164DAGISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:2 row_mask:0xf bank_mask:0xf
-; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164DAGISEL-NEXT: v_max_f64 v[4:5], v[4:5], v[6:7]
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v6, v4
-; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v7, v5
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1164DAGISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:4 row_mask:0xf bank_mask:0xf
-; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164DAGISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:4 row_mask:0xf bank_mask:0xf
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164DAGISEL-NEXT: v_max_f64 v[4:5], v[4:5], v[6:7]
-; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v6, v4
-; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v7, v5
; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v7, v5
; GFX1164DAGISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:8 row_mask:0xf bank_mask:0xf
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164DAGISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:8 row_mask:0xf bank_mask:0xf
-; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1164DAGISEL-NEXT: v_max_f64 v[4:5], v[4:5], v[6:7]
; GFX1164DAGISEL-NEXT: ds_swizzle_b32 v6, v4 offset:swizzle(BROADCAST,32,15)
; GFX1164DAGISEL-NEXT: ds_swizzle_b32 v7, v5 offset:swizzle(BROADCAST,32,15)
@@ -2147,7 +2152,7 @@ define void @divergent_value_double_dpp(ptr addrspace(1) %out, double %in) #0 {
; GFX1164DAGISEL-NEXT: s_waitcnt lgkmcnt(0)
; GFX1164DAGISEL-NEXT: v_max_f64 v[4:5], v[4:5], v[7:8]
; GFX1164DAGISEL-NEXT: v_readlane_b32 s2, v4, 63
-; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2)
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX1164DAGISEL-NEXT: v_readlane_b32 s3, v5, 63
; GFX1164DAGISEL-NEXT: s_mov_b64 exec, s[0:1]
; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v2, s2
@@ -2174,39 +2179,39 @@ define void @divergent_value_double_dpp(ptr addrspace(1) %out, double %in) #0 {
; GFX1164GISEL-NEXT: scratch_store_b32 off, v6, s32 offset:16
; GFX1164GISEL-NEXT: scratch_store_b64 off, v[7:8], s32 offset:20
; GFX1164GISEL-NEXT: s_mov_b64 exec, s[0:1]
+; GFX1164GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1164GISEL-NEXT: s_or_saveexec_b64 s[0:1], -1
-; GFX1164GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX1164GISEL-NEXT: v_cndmask_b32_e64 v4, 0, v2, s[0:1]
; GFX1164GISEL-NEXT: v_cndmask_b32_e64 v5, 0x7ff80000, v3, s[0:1]
-; GFX1164GISEL-NEXT: v_mov_b32_e32 v6, v4
; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX1164GISEL-NEXT: v_mov_b32_e32 v6, v4
; GFX1164GISEL-NEXT: v_mov_b32_e32 v7, v5
+; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1164GISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:1 row_mask:0xf bank_mask:0xf
-; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164GISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:1 row_mask:0xf bank_mask:0xf
+; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164GISEL-NEXT: v_max_f64 v[4:5], v[4:5], v[6:7]
-; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1164GISEL-NEXT: v_mov_b32_e32 v6, v4
-; GFX1164GISEL-NEXT: v_mov_b32_e32 v7, v5
; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX1164GISEL-NEXT: v_mov_b32_e32 v7, v5
; GFX1164GISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:2 row_mask:0xf bank_mask:0xf
+; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164GISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:2 row_mask:0xf bank_mask:0xf
-; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164GISEL-NEXT: v_max_f64 v[4:5], v[4:5], v[6:7]
+; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1164GISEL-NEXT: v_mov_b32_e32 v6, v4
-; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1164GISEL-NEXT: v_mov_b32_e32 v7, v5
+; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1164GISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:4 row_mask:0xf bank_mask:0xf
-; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164GISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:4 row_mask:0xf bank_mask:0xf
+; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164GISEL-NEXT: v_max_f64 v[4:5], v[4:5], v[6:7]
-; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1164GISEL-NEXT: v_mov_b32_e32 v6, v4
-; GFX1164GISEL-NEXT: v_mov_b32_e32 v7, v5
; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX1164GISEL-NEXT: v_mov_b32_e32 v7, v5
; GFX1164GISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:8 row_mask:0xf bank_mask:0xf
+; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164GISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:8 row_mask:0xf bank_mask:0xf
-; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1164GISEL-NEXT: v_max_f64 v[4:5], v[4:5], v[6:7]
; GFX1164GISEL-NEXT: ds_swizzle_b32 v6, v4 offset:swizzle(BROADCAST,32,15)
; GFX1164GISEL-NEXT: ds_swizzle_b32 v7, v5 offset:swizzle(BROADCAST,32,15)
@@ -2223,7 +2228,7 @@ define void @divergent_value_double_dpp(ptr addrspace(1) %out, double %in) #0 {
; GFX1164GISEL-NEXT: s_waitcnt lgkmcnt(0)
; GFX1164GISEL-NEXT: v_max_f64 v[4:5], v[4:5], v[7:8]
; GFX1164GISEL-NEXT: v_readlane_b32 s2, v4, 63
-; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2)
+; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX1164GISEL-NEXT: v_readlane_b32 s3, v5, 63
; GFX1164GISEL-NEXT: s_mov_b64 exec, s[0:1]
; GFX1164GISEL-NEXT: v_mov_b32_e32 v2, s2
@@ -2248,42 +2253,43 @@ define void @divergent_value_double_dpp(ptr addrspace(1) %out, double %in) #0 {
; GFX1132DAGISEL-NEXT: scratch_store_b64 off, v[4:5], s32
; GFX1132DAGISEL-NEXT: scratch_store_b64 off, v[6:7], s32 offset:8
; GFX1132DAGISEL-NEXT: s_mov_b32 exec_lo, s0
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1132DAGISEL-NEXT: s_or_saveexec_b32 s2, -1
-; GFX1132DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX1132DAGISEL-NEXT: v_cndmask_b32_e64 v4, 0, v2, s2
; GFX1132DAGISEL-NEXT: v_cndmask_b32_e64 v5, 0x7ff80000, v3, s2
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132DAGISEL-NEXT: v_dual_mov_b32 v6, v4 :: v_dual_mov_b32 v7, v5
-; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1132DAGISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:1 row_mask:0xf bank_mask:0xf
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132DAGISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:1 row_mask:0xf bank_mask:0xf
-; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132DAGISEL-NEXT: v_max_f64 v[4:5], v[4:5], v[6:7]
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132DAGISEL-NEXT: v_dual_mov_b32 v6, v4 :: v_dual_mov_b32 v7, v5
-; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1132DAGISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:2 row_mask:0xf bank_mask:0xf
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132DAGISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:2 row_mask:0xf bank_mask:0xf
-; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132DAGISEL-NEXT: v_max_f64 v[4:5], v[4:5], v[6:7]
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132DAGISEL-NEXT: v_dual_mov_b32 v6, v4 :: v_dual_mov_b32 v7, v5
-; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1132DAGISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:4 row_mask:0xf bank_mask:0xf
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132DAGISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:4 row_mask:0xf bank_mask:0xf
-; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132DAGISEL-NEXT: v_max_f64 v[4:5], v[4:5], v[6:7]
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132DAGISEL-NEXT: v_dual_mov_b32 v6, v4 :: v_dual_mov_b32 v7, v5
-; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1132DAGISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:8 row_mask:0xf bank_mask:0xf
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132DAGISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:8 row_mask:0xf bank_mask:0xf
-; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_4) | instid1(VALU_DEP_1)
; GFX1132DAGISEL-NEXT: v_max_f64 v[4:5], v[4:5], v[6:7]
; GFX1132DAGISEL-NEXT: ds_swizzle_b32 v6, v4 offset:swizzle(BROADCAST,32,15)
; GFX1132DAGISEL-NEXT: ds_swizzle_b32 v7, v5 offset:swizzle(BROADCAST,32,15)
; GFX1132DAGISEL-NEXT: s_waitcnt lgkmcnt(0)
; GFX1132DAGISEL-NEXT: v_max_f64 v[4:5], v[4:5], v[6:7]
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1132DAGISEL-NEXT: v_readlane_b32 s0, v4, 31
-; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX1132DAGISEL-NEXT: v_readlane_b32 s1, v5, 31
; GFX1132DAGISEL-NEXT: s_mov_b32 exec_lo, s2
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1132DAGISEL-NEXT: v_dual_mov_b32 v3, s1 :: v_dual_mov_b32 v2, s0
; GFX1132DAGISEL-NEXT: global_store_b64 v[0:1], v[2:3], off
; GFX1132DAGISEL-NEXT: s_xor_saveexec_b32 s0, -1
@@ -2302,42 +2308,43 @@ define void @divergent_value_double_dpp(ptr addrspace(1) %out, double %in) #0 {
; GFX1132GISEL-NEXT: scratch_store_b64 off, v[4:5], s32
; GFX1132GISEL-NEXT: scratch_store_b64 off, v[6:7], s32 offset:8
; GFX1132GISEL-NEXT: s_mov_b32 exec_lo, s0
+; GFX1132GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1132GISEL-NEXT: s_or_saveexec_b32 s2, -1
-; GFX1132GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX1132GISEL-NEXT: v_cndmask_b32_e64 v4, 0, v2, s2
; GFX1132GISEL-NEXT: v_cndmask_b32_e64 v5, 0x7ff80000, v3, s2
+; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132GISEL-NEXT: v_dual_mov_b32 v6, v4 :: v_dual_mov_b32 v7, v5
-; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1132GISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:1 row_mask:0xf bank_mask:0xf
+; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132GISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:1 row_mask:0xf bank_mask:0xf
-; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132GISEL-NEXT: v_max_f64 v[4:5], v[4:5], v[6:7]
+; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132GISEL-NEXT: v_dual_mov_b32 v6, v4 :: v_dual_mov_b32 v7, v5
-; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1132GISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:2 row_mask:0xf bank_mask:0xf
+; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132GISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:2 row_mask:0xf bank_mask:0xf
-; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132GISEL-NEXT: v_max_f64 v[4:5], v[4:5], v[6:7]
+; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132GISEL-NEXT: v_dual_mov_b32 v6, v4 :: v_dual_mov_b32 v7, v5
-; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1132GISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:4 row_mask:0xf bank_mask:0xf
+; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132GISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:4 row_mask:0xf bank_mask:0xf
-; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132GISEL-NEXT: v_max_f64 v[4:5], v[4:5], v[6:7]
+; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132GISEL-NEXT: v_dual_mov_b32 v6, v4 :: v_dual_mov_b32 v7, v5
-; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1132GISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:8 row_mask:0xf bank_mask:0xf
+; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132GISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:8 row_mask:0xf bank_mask:0xf
-; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_4) | instid1(VALU_DEP_1)
; GFX1132GISEL-NEXT: v_max_f64 v[4:5], v[4:5], v[6:7]
; GFX1132GISEL-NEXT: ds_swizzle_b32 v6, v4 offset:swizzle(BROADCAST,32,15)
; GFX1132GISEL-NEXT: ds_swizzle_b32 v7, v5 offset:swizzle(BROADCAST,32,15)
; GFX1132GISEL-NEXT: s_waitcnt lgkmcnt(0)
; GFX1132GISEL-NEXT: v_max_f64 v[4:5], v[4:5], v[6:7]
+; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1132GISEL-NEXT: v_readlane_b32 s0, v4, 31
-; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX1132GISEL-NEXT: v_readlane_b32 s1, v5, 31
; GFX1132GISEL-NEXT: s_mov_b32 exec_lo, s2
+; GFX1132GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1132GISEL-NEXT: v_dual_mov_b32 v3, s1 :: v_dual_mov_b32 v2, s0
; GFX1132GISEL-NEXT: global_store_b64 v[0:1], v[2:3], off
; GFX1132GISEL-NEXT: s_xor_saveexec_b32 s0, -1
@@ -2361,40 +2368,41 @@ define void @divergent_value_double_dpp(ptr addrspace(1) %out, double %in) #0 {
; GFX12DAGISEL-NEXT: scratch_store_b64 off, v[6:7], s32 offset:8
; GFX12DAGISEL-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12DAGISEL-NEXT: s_mov_b32 exec_lo, s0
+; GFX12DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_3) | instid1(VALU_DEP_1)
; GFX12DAGISEL-NEXT: s_or_saveexec_b32 s2, -1
; GFX12DAGISEL-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12DAGISEL-NEXT: v_cndmask_b32_e64 v4, 0, v2, s2
; GFX12DAGISEL-NEXT: v_cndmask_b32_e64 v5, 0x7ff80000, v3, s2
-; GFX12DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX12DAGISEL-NEXT: v_dual_mov_b32 v6, v4 :: v_dual_mov_b32 v7, v5
+; GFX12DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX12DAGISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:1 row_mask:0xf bank_mask:0xf
-; GFX12DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX12DAGISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:1 row_mask:0xf bank_mask:0xf
-; GFX12DAGISEL-NEXT: v_max_num_f64_e32 v[4:5], v[4:5], v[6:7]
; GFX12DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX12DAGISEL-NEXT: v_max_num_f64_e32 v[4:5], v[4:5], v[6:7]
; GFX12DAGISEL-NEXT: v_dual_mov_b32 v6, v4 :: v_dual_mov_b32 v7, v5
+; GFX12DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX12DAGISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:2 row_mask:0xf bank_mask:0xf
-; GFX12DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX12DAGISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:2 row_mask:0xf bank_mask:0xf
-; GFX12DAGISEL-NEXT: v_max_num_f64_e32 v[4:5], v[4:5], v[6:7]
; GFX12DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX12DAGISEL-NEXT: v_max_num_f64_e32 v[4:5], v[4:5], v[6:7]
; GFX12DAGISEL-NEXT: v_dual_mov_b32 v6, v4 :: v_dual_mov_b32 v7, v5
+; GFX12DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX12DAGISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:4 row_mask:0xf bank_mask:0xf
-; GFX12DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX12DAGISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:4 row_mask:0xf bank_mask:0xf
-; GFX12DAGISEL-NEXT: v_max_num_f64_e32 v[4:5], v[4:5], v[6:7]
; GFX12DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX12DAGISEL-NEXT: v_max_num_f64_e32 v[4:5], v[4:5], v[6:7]
; GFX12DAGISEL-NEXT: v_dual_mov_b32 v6, v4 :: v_dual_mov_b32 v7, v5
+; GFX12DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX12DAGISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:8 row_mask:0xf bank_mask:0xf
-; GFX12DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX12DAGISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:8 row_mask:0xf bank_mask:0xf
+; GFX12DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_4) | instid1(VALU_DEP_1)
; GFX12DAGISEL-NEXT: v_max_num_f64_e32 v[4:5], v[4:5], v[6:7]
; GFX12DAGISEL-NEXT: ds_swizzle_b32 v6, v4 offset:swizzle(BROADCAST,32,15)
; GFX12DAGISEL-NEXT: ds_swizzle_b32 v7, v5 offset:swizzle(BROADCAST,32,15)
; GFX12DAGISEL-NEXT: s_wait_dscnt 0x0
; GFX12DAGISEL-NEXT: v_max_num_f64_e32 v[4:5], v[4:5], v[6:7]
-; GFX12DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX12DAGISEL-NEXT: v_readlane_b32 s0, v4, 31
+; GFX12DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX12DAGISEL-NEXT: v_readlane_b32 s1, v5, 31
; GFX12DAGISEL-NEXT: s_mov_b32 exec_lo, s2
; GFX12DAGISEL-NEXT: v_dual_mov_b32 v3, s1 :: v_dual_mov_b32 v2, s0
@@ -2782,10 +2790,11 @@ define void @divergent_cfg_float(ptr addrspace(1) %out, float %in, float %in2) #
; GFX1164DAGISEL: ; %bb.0: ; %entry
; GFX1164DAGISEL-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX1164DAGISEL-NEXT: v_and_b32_e32 v4, 0x3ff, v31
-; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164DAGISEL-NEXT: v_cmp_lt_u32_e32 vcc, 15, v4
; GFX1164DAGISEL-NEXT: ; implicit-def: $vgpr4
; GFX1164DAGISEL-NEXT: s_and_saveexec_b64 s[0:1], vcc
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164DAGISEL-NEXT: s_xor_b64 s[0:1], exec, s[0:1]
; GFX1164DAGISEL-NEXT: s_cbranch_execz .LBB6_4
; GFX1164DAGISEL-NEXT: ; %bb.1: ; %else
@@ -2831,10 +2840,11 @@ define void @divergent_cfg_float(ptr addrspace(1) %out, float %in, float %in2) #
; GFX1164GISEL: ; %bb.0: ; %entry
; GFX1164GISEL-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX1164GISEL-NEXT: v_and_b32_e32 v4, 0x3ff, v31
-; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
+; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164GISEL-NEXT: v_cmp_le_u32_e32 vcc, 16, v4
; GFX1164GISEL-NEXT: ; implicit-def: $vgpr4
; GFX1164GISEL-NEXT: s_and_saveexec_b64 s[0:1], vcc
+; GFX1164GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164GISEL-NEXT: s_xor_b64 s[0:1], exec, s[0:1]
; GFX1164GISEL-NEXT: s_cbranch_execz .LBB6_4
; GFX1164GISEL-NEXT: ; %bb.1: ; %else
@@ -3940,10 +3950,11 @@ define void @divergent_cfg_double(ptr addrspace(1) %out, double %in, double %in2
; GFX1164DAGISEL: ; %bb.0: ; %entry
; GFX1164DAGISEL-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX1164DAGISEL-NEXT: v_and_b32_e32 v6, 0x3ff, v31
-; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164DAGISEL-NEXT: v_cmp_lt_u32_e32 vcc, 15, v6
; GFX1164DAGISEL-NEXT: ; implicit-def: $vgpr6_vgpr7
; GFX1164DAGISEL-NEXT: s_and_saveexec_b64 s[0:1], vcc
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164DAGISEL-NEXT: s_xor_b64 s[0:1], exec, s[0:1]
; GFX1164DAGISEL-NEXT: s_cbranch_execz .LBB9_4
; GFX1164DAGISEL-NEXT: ; %bb.1: ; %else
@@ -4000,10 +4011,11 @@ define void @divergent_cfg_double(ptr addrspace(1) %out, double %in, double %in2
; GFX1164GISEL: ; %bb.0: ; %entry
; GFX1164GISEL-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX1164GISEL-NEXT: v_and_b32_e32 v6, 0x3ff, v31
-; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
+; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164GISEL-NEXT: v_cmp_le_u32_e32 vcc, 16, v6
; GFX1164GISEL-NEXT: ; implicit-def: $vgpr6_vgpr7
; GFX1164GISEL-NEXT: s_and_saveexec_b64 s[0:1], vcc
+; GFX1164GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164GISEL-NEXT: s_xor_b64 s[0:1], exec, s[0:1]
; GFX1164GISEL-NEXT: s_cbranch_execz .LBB9_4
; GFX1164GISEL-NEXT: ; %bb.1: ; %else
diff --git a/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.reduce.fmin.ll b/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.reduce.fmin.ll
index e45588dac7ff89..107c097b1e3b63 100644
--- a/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.reduce.fmin.ll
+++ b/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.reduce.fmin.ll
@@ -1348,19 +1348,20 @@ define void @divergent_value_float_dpp(ptr addrspace(1) %out, float %in) #0 {
; GFX1164DAGISEL-NEXT: scratch_store_b32 off, v4, s32 offset:4
; GFX1164DAGISEL-NEXT: scratch_store_b32 off, v5, s32 offset:8
; GFX1164DAGISEL-NEXT: s_mov_b64 exec, s[0:1]
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1164DAGISEL-NEXT: s_or_saveexec_b64 s[0:1], -1
-; GFX1164DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX1164DAGISEL-NEXT: v_cndmask_b32_e64 v3, 0x7fc00000, v2, s[0:1]
; GFX1164DAGISEL-NEXT: v_mbcnt_lo_u32_b32 v5, -1, 0
-; GFX1164DAGISEL-NEXT: v_min_f32_dpp v3, v3, v3 row_shr:1 row_mask:0xf bank_mask:0xf
; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX1164DAGISEL-NEXT: v_min_f32_dpp v3, v3, v3 row_shr:1 row_mask:0xf bank_mask:0xf
; GFX1164DAGISEL-NEXT: v_mbcnt_hi_u32_b32 v5, -1, v5
-; GFX1164DAGISEL-NEXT: v_min_f32_dpp v3, v3, v3 row_shr:2 row_mask:0xf bank_mask:0xf
; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX1164DAGISEL-NEXT: v_min_f32_dpp v3, v3, v3 row_shr:2 row_mask:0xf bank_mask:0xf
; GFX1164DAGISEL-NEXT: v_add_nc_u32_e32 v5, 32, v5
-; GFX1164DAGISEL-NEXT: v_min_f32_dpp v3, v3, v3 row_shr:4 row_mask:0xf bank_mask:0xf
; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX1164DAGISEL-NEXT: v_min_f32_dpp v3, v3, v3 row_shr:4 row_mask:0xf bank_mask:0xf
; GFX1164DAGISEL-NEXT: v_mul_lo_u32 v5, 4, v5
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX1164DAGISEL-NEXT: v_min_f32_dpp v3, v3, v3 row_shr:8 row_mask:0xf bank_mask:0xf
; GFX1164DAGISEL-NEXT: ds_swizzle_b32 v4, v3 offset:swizzle(BROADCAST,32,15)
; GFX1164DAGISEL-NEXT: s_waitcnt lgkmcnt(0)
@@ -1368,7 +1369,7 @@ define void @divergent_value_float_dpp(ptr addrspace(1) %out, float %in) #0 {
; GFX1164DAGISEL-NEXT: ds_permute_b32 v4, v5, v3
; GFX1164DAGISEL-NEXT: s_waitcnt lgkmcnt(0)
; GFX1164DAGISEL-NEXT: v_min_f32_e32 v3, v3, v4
-; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX1164DAGISEL-NEXT: v_readlane_b32 s2, v3, 63
; GFX1164DAGISEL-NEXT: s_mov_b64 exec, s[0:1]
; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v2, s2
@@ -1392,19 +1393,20 @@ define void @divergent_value_float_dpp(ptr addrspace(1) %out, float %in) #0 {
; GFX1164GISEL-NEXT: scratch_store_b32 off, v4, s32 offset:4
; GFX1164GISEL-NEXT: scratch_store_b32 off, v5, s32 offset:8
; GFX1164GISEL-NEXT: s_mov_b64 exec, s[0:1]
+; GFX1164GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1164GISEL-NEXT: s_or_saveexec_b64 s[0:1], -1
-; GFX1164GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX1164GISEL-NEXT: v_cndmask_b32_e64 v3, 0x7fc00000, v2, s[0:1]
; GFX1164GISEL-NEXT: v_mbcnt_lo_u32_b32 v5, -1, 0
-; GFX1164GISEL-NEXT: v_min_f32_dpp v3, v3, v3 row_shr:1 row_mask:0xf bank_mask:0xf
; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX1164GISEL-NEXT: v_min_f32_dpp v3, v3, v3 row_shr:1 row_mask:0xf bank_mask:0xf
; GFX1164GISEL-NEXT: v_mbcnt_hi_u32_b32 v5, -1, v5
-; GFX1164GISEL-NEXT: v_min_f32_dpp v3, v3, v3 row_shr:2 row_mask:0xf bank_mask:0xf
; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX1164GISEL-NEXT: v_min_f32_dpp v3, v3, v3 row_shr:2 row_mask:0xf bank_mask:0xf
; GFX1164GISEL-NEXT: v_add_nc_u32_e32 v5, 32, v5
-; GFX1164GISEL-NEXT: v_min_f32_dpp v3, v3, v3 row_shr:4 row_mask:0xf bank_mask:0xf
; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX1164GISEL-NEXT: v_min_f32_dpp v3, v3, v3 row_shr:4 row_mask:0xf bank_mask:0xf
; GFX1164GISEL-NEXT: v_mul_lo_u32 v5, 4, v5
+; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX1164GISEL-NEXT: v_min_f32_dpp v3, v3, v3 row_shr:8 row_mask:0xf bank_mask:0xf
; GFX1164GISEL-NEXT: ds_swizzle_b32 v4, v3 offset:swizzle(BROADCAST,32,15)
; GFX1164GISEL-NEXT: s_waitcnt lgkmcnt(0)
@@ -1412,7 +1414,7 @@ define void @divergent_value_float_dpp(ptr addrspace(1) %out, float %in) #0 {
; GFX1164GISEL-NEXT: ds_permute_b32 v4, v5, v3
; GFX1164GISEL-NEXT: s_waitcnt lgkmcnt(0)
; GFX1164GISEL-NEXT: v_min_f32_e32 v3, v3, v4
-; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX1164GISEL-NEXT: v_readlane_b32 s2, v3, 63
; GFX1164GISEL-NEXT: s_mov_b64 exec, s[0:1]
; GFX1164GISEL-NEXT: v_mov_b32_e32 v2, s2
@@ -1435,18 +1437,19 @@ define void @divergent_value_float_dpp(ptr addrspace(1) %out, float %in) #0 {
; GFX1132DAGISEL-NEXT: scratch_store_b32 off, v3, s32
; GFX1132DAGISEL-NEXT: scratch_store_b32 off, v4, s32 offset:4
; GFX1132DAGISEL-NEXT: s_mov_b32 exec_lo, s0
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1132DAGISEL-NEXT: s_or_saveexec_b32 s0, -1
-; GFX1132DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132DAGISEL-NEXT: v_cndmask_b32_e64 v3, 0x7fc00000, v2, s0
-; GFX1132DAGISEL-NEXT: v_min_f32_dpp v3, v3, v3 row_shr:1 row_mask:0xf bank_mask:0xf
; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1132DAGISEL-NEXT: v_min_f32_dpp v3, v3, v3 row_shr:1 row_mask:0xf bank_mask:0xf
; GFX1132DAGISEL-NEXT: v_min_f32_dpp v3, v3, v3 row_shr:2 row_mask:0xf bank_mask:0xf
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132DAGISEL-NEXT: v_min_f32_dpp v3, v3, v3 row_shr:4 row_mask:0xf bank_mask:0xf
-; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_3) | instid1(VALU_DEP_1)
; GFX1132DAGISEL-NEXT: v_min_f32_dpp v3, v3, v3 row_shr:8 row_mask:0xf bank_mask:0xf
; GFX1132DAGISEL-NEXT: ds_swizzle_b32 v4, v3 offset:swizzle(BROADCAST,32,15)
; GFX1132DAGISEL-NEXT: s_waitcnt lgkmcnt(0)
; GFX1132DAGISEL-NEXT: v_min_f32_e32 v3, v3, v4
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX1132DAGISEL-NEXT: v_readlane_b32 s1, v3, 31
; GFX1132DAGISEL-NEXT: s_mov_b32 exec_lo, s0
; GFX1132DAGISEL-NEXT: v_mov_b32_e32 v2, s1
@@ -1467,18 +1470,19 @@ define void @divergent_value_float_dpp(ptr addrspace(1) %out, float %in) #0 {
; GFX1132GISEL-NEXT: scratch_store_b32 off, v3, s32
; GFX1132GISEL-NEXT: scratch_store_b32 off, v4, s32 offset:4
; GFX1132GISEL-NEXT: s_mov_b32 exec_lo, s0
+; GFX1132GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1132GISEL-NEXT: s_or_saveexec_b32 s0, -1
-; GFX1132GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132GISEL-NEXT: v_cndmask_b32_e64 v3, 0x7fc00000, v2, s0
-; GFX1132GISEL-NEXT: v_min_f32_dpp v3, v3, v3 row_shr:1 row_mask:0xf bank_mask:0xf
; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1132GISEL-NEXT: v_min_f32_dpp v3, v3, v3 row_shr:1 row_mask:0xf bank_mask:0xf
; GFX1132GISEL-NEXT: v_min_f32_dpp v3, v3, v3 row_shr:2 row_mask:0xf bank_mask:0xf
+; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132GISEL-NEXT: v_min_f32_dpp v3, v3, v3 row_shr:4 row_mask:0xf bank_mask:0xf
-; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_3) | instid1(VALU_DEP_1)
; GFX1132GISEL-NEXT: v_min_f32_dpp v3, v3, v3 row_shr:8 row_mask:0xf bank_mask:0xf
; GFX1132GISEL-NEXT: ds_swizzle_b32 v4, v3 offset:swizzle(BROADCAST,32,15)
; GFX1132GISEL-NEXT: s_waitcnt lgkmcnt(0)
; GFX1132GISEL-NEXT: v_min_f32_e32 v3, v3, v4
+; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX1132GISEL-NEXT: v_readlane_b32 s1, v3, 31
; GFX1132GISEL-NEXT: s_mov_b32 exec_lo, s0
; GFX1132GISEL-NEXT: v_mov_b32_e32 v2, s1
@@ -1504,21 +1508,22 @@ define void @divergent_value_float_dpp(ptr addrspace(1) %out, float %in) #0 {
; GFX12DAGISEL-NEXT: scratch_store_b32 off, v4, s32 offset:4
; GFX12DAGISEL-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12DAGISEL-NEXT: s_mov_b32 exec_lo, s0
+; GFX12DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(VALU_DEP_1)
; GFX12DAGISEL-NEXT: s_or_saveexec_b32 s0, -1
; GFX12DAGISEL-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12DAGISEL-NEXT: v_cndmask_b32_e64 v3, 0x7fc00000, v2, s0
-; GFX12DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX12DAGISEL-NEXT: v_min_num_f32_dpp v3, v3, v3 row_shr:1 row_mask:0xf bank_mask:0xf
-; GFX12DAGISEL-NEXT: v_min_num_f32_dpp v3, v3, v3 row_shr:2 row_mask:0xf bank_mask:0xf
; GFX12DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX12DAGISEL-NEXT: v_min_num_f32_dpp v3, v3, v3 row_shr:2 row_mask:0xf bank_mask:0xf
; GFX12DAGISEL-NEXT: v_min_num_f32_dpp v3, v3, v3 row_shr:4 row_mask:0xf bank_mask:0xf
+; GFX12DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_3) | instid1(VALU_DEP_1)
; GFX12DAGISEL-NEXT: v_min_num_f32_dpp v3, v3, v3 row_shr:8 row_mask:0xf bank_mask:0xf
; GFX12DAGISEL-NEXT: ds_swizzle_b32 v4, v3 offset:swizzle(BROADCAST,32,15)
; GFX12DAGISEL-NEXT: s_wait_dscnt 0x0
; GFX12DAGISEL-NEXT: v_min_num_f32_e32 v3, v3, v4
-; GFX12DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12DAGISEL-NEXT: v_readlane_b32 s1, v3, 31
; GFX12DAGISEL-NEXT: s_mov_b32 exec_lo, s0
+; GFX12DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12DAGISEL-NEXT: v_mov_b32_e32 v2, s1
; GFX12DAGISEL-NEXT: global_store_b32 v[0:1], v2, off
; GFX12DAGISEL-NEXT: s_xor_saveexec_b32 s0, -1
@@ -2098,39 +2103,39 @@ define void @divergent_value_double_dpp(ptr addrspace(1) %out, double %in) #0 {
; GFX1164DAGISEL-NEXT: scratch_store_b32 off, v6, s32 offset:16
; GFX1164DAGISEL-NEXT: scratch_store_b64 off, v[7:8], s32 offset:20
; GFX1164DAGISEL-NEXT: s_mov_b64 exec, s[0:1]
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1164DAGISEL-NEXT: s_or_saveexec_b64 s[0:1], -1
-; GFX1164DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX1164DAGISEL-NEXT: v_cndmask_b32_e64 v4, 0, v2, s[0:1]
; GFX1164DAGISEL-NEXT: v_cndmask_b32_e64 v5, 0x7ff80000, v3, s[0:1]
-; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v6, v4
; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v6, v4
; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v7, v5
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1164DAGISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:1 row_mask:0xf bank_mask:0xf
-; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164DAGISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:1 row_mask:0xf bank_mask:0xf
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164DAGISEL-NEXT: v_min_f64 v[4:5], v[4:5], v[6:7]
-; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v6, v4
-; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v7, v5
; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v7, v5
; GFX1164DAGISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:2 row_mask:0xf bank_mask:0xf
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164DAGISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:2 row_mask:0xf bank_mask:0xf
-; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164DAGISEL-NEXT: v_min_f64 v[4:5], v[4:5], v[6:7]
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v6, v4
-; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v7, v5
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1164DAGISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:4 row_mask:0xf bank_mask:0xf
-; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164DAGISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:4 row_mask:0xf bank_mask:0xf
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164DAGISEL-NEXT: v_min_f64 v[4:5], v[4:5], v[6:7]
-; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v6, v4
-; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v7, v5
; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v7, v5
; GFX1164DAGISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:8 row_mask:0xf bank_mask:0xf
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164DAGISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:8 row_mask:0xf bank_mask:0xf
-; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1164DAGISEL-NEXT: v_min_f64 v[4:5], v[4:5], v[6:7]
; GFX1164DAGISEL-NEXT: ds_swizzle_b32 v6, v4 offset:swizzle(BROADCAST,32,15)
; GFX1164DAGISEL-NEXT: ds_swizzle_b32 v7, v5 offset:swizzle(BROADCAST,32,15)
@@ -2147,7 +2152,7 @@ define void @divergent_value_double_dpp(ptr addrspace(1) %out, double %in) #0 {
; GFX1164DAGISEL-NEXT: s_waitcnt lgkmcnt(0)
; GFX1164DAGISEL-NEXT: v_min_f64 v[4:5], v[4:5], v[7:8]
; GFX1164DAGISEL-NEXT: v_readlane_b32 s2, v4, 63
-; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2)
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX1164DAGISEL-NEXT: v_readlane_b32 s3, v5, 63
; GFX1164DAGISEL-NEXT: s_mov_b64 exec, s[0:1]
; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v2, s2
@@ -2174,39 +2179,39 @@ define void @divergent_value_double_dpp(ptr addrspace(1) %out, double %in) #0 {
; GFX1164GISEL-NEXT: scratch_store_b32 off, v6, s32 offset:16
; GFX1164GISEL-NEXT: scratch_store_b64 off, v[7:8], s32 offset:20
; GFX1164GISEL-NEXT: s_mov_b64 exec, s[0:1]
+; GFX1164GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1164GISEL-NEXT: s_or_saveexec_b64 s[0:1], -1
-; GFX1164GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX1164GISEL-NEXT: v_cndmask_b32_e64 v4, 0, v2, s[0:1]
; GFX1164GISEL-NEXT: v_cndmask_b32_e64 v5, 0x7ff80000, v3, s[0:1]
-; GFX1164GISEL-NEXT: v_mov_b32_e32 v6, v4
; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX1164GISEL-NEXT: v_mov_b32_e32 v6, v4
; GFX1164GISEL-NEXT: v_mov_b32_e32 v7, v5
+; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1164GISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:1 row_mask:0xf bank_mask:0xf
-; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164GISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:1 row_mask:0xf bank_mask:0xf
+; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164GISEL-NEXT: v_min_f64 v[4:5], v[4:5], v[6:7]
-; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1164GISEL-NEXT: v_mov_b32_e32 v6, v4
-; GFX1164GISEL-NEXT: v_mov_b32_e32 v7, v5
; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX1164GISEL-NEXT: v_mov_b32_e32 v7, v5
; GFX1164GISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:2 row_mask:0xf bank_mask:0xf
+; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164GISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:2 row_mask:0xf bank_mask:0xf
-; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164GISEL-NEXT: v_min_f64 v[4:5], v[4:5], v[6:7]
+; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1164GISEL-NEXT: v_mov_b32_e32 v6, v4
-; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1164GISEL-NEXT: v_mov_b32_e32 v7, v5
+; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1164GISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:4 row_mask:0xf bank_mask:0xf
-; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164GISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:4 row_mask:0xf bank_mask:0xf
+; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164GISEL-NEXT: v_min_f64 v[4:5], v[4:5], v[6:7]
-; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1164GISEL-NEXT: v_mov_b32_e32 v6, v4
-; GFX1164GISEL-NEXT: v_mov_b32_e32 v7, v5
; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX1164GISEL-NEXT: v_mov_b32_e32 v7, v5
; GFX1164GISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:8 row_mask:0xf bank_mask:0xf
+; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164GISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:8 row_mask:0xf bank_mask:0xf
-; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1164GISEL-NEXT: v_min_f64 v[4:5], v[4:5], v[6:7]
; GFX1164GISEL-NEXT: ds_swizzle_b32 v6, v4 offset:swizzle(BROADCAST,32,15)
; GFX1164GISEL-NEXT: ds_swizzle_b32 v7, v5 offset:swizzle(BROADCAST,32,15)
@@ -2223,7 +2228,7 @@ define void @divergent_value_double_dpp(ptr addrspace(1) %out, double %in) #0 {
; GFX1164GISEL-NEXT: s_waitcnt lgkmcnt(0)
; GFX1164GISEL-NEXT: v_min_f64 v[4:5], v[4:5], v[7:8]
; GFX1164GISEL-NEXT: v_readlane_b32 s2, v4, 63
-; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2)
+; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX1164GISEL-NEXT: v_readlane_b32 s3, v5, 63
; GFX1164GISEL-NEXT: s_mov_b64 exec, s[0:1]
; GFX1164GISEL-NEXT: v_mov_b32_e32 v2, s2
@@ -2248,42 +2253,43 @@ define void @divergent_value_double_dpp(ptr addrspace(1) %out, double %in) #0 {
; GFX1132DAGISEL-NEXT: scratch_store_b64 off, v[4:5], s32
; GFX1132DAGISEL-NEXT: scratch_store_b64 off, v[6:7], s32 offset:8
; GFX1132DAGISEL-NEXT: s_mov_b32 exec_lo, s0
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1132DAGISEL-NEXT: s_or_saveexec_b32 s2, -1
-; GFX1132DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX1132DAGISEL-NEXT: v_cndmask_b32_e64 v4, 0, v2, s2
; GFX1132DAGISEL-NEXT: v_cndmask_b32_e64 v5, 0x7ff80000, v3, s2
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132DAGISEL-NEXT: v_dual_mov_b32 v6, v4 :: v_dual_mov_b32 v7, v5
-; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1132DAGISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:1 row_mask:0xf bank_mask:0xf
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132DAGISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:1 row_mask:0xf bank_mask:0xf
-; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132DAGISEL-NEXT: v_min_f64 v[4:5], v[4:5], v[6:7]
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132DAGISEL-NEXT: v_dual_mov_b32 v6, v4 :: v_dual_mov_b32 v7, v5
-; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1132DAGISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:2 row_mask:0xf bank_mask:0xf
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132DAGISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:2 row_mask:0xf bank_mask:0xf
-; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132DAGISEL-NEXT: v_min_f64 v[4:5], v[4:5], v[6:7]
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132DAGISEL-NEXT: v_dual_mov_b32 v6, v4 :: v_dual_mov_b32 v7, v5
-; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1132DAGISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:4 row_mask:0xf bank_mask:0xf
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132DAGISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:4 row_mask:0xf bank_mask:0xf
-; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132DAGISEL-NEXT: v_min_f64 v[4:5], v[4:5], v[6:7]
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132DAGISEL-NEXT: v_dual_mov_b32 v6, v4 :: v_dual_mov_b32 v7, v5
-; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1132DAGISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:8 row_mask:0xf bank_mask:0xf
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132DAGISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:8 row_mask:0xf bank_mask:0xf
-; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_4) | instid1(VALU_DEP_1)
; GFX1132DAGISEL-NEXT: v_min_f64 v[4:5], v[4:5], v[6:7]
; GFX1132DAGISEL-NEXT: ds_swizzle_b32 v6, v4 offset:swizzle(BROADCAST,32,15)
; GFX1132DAGISEL-NEXT: ds_swizzle_b32 v7, v5 offset:swizzle(BROADCAST,32,15)
; GFX1132DAGISEL-NEXT: s_waitcnt lgkmcnt(0)
; GFX1132DAGISEL-NEXT: v_min_f64 v[4:5], v[4:5], v[6:7]
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1132DAGISEL-NEXT: v_readlane_b32 s0, v4, 31
-; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX1132DAGISEL-NEXT: v_readlane_b32 s1, v5, 31
; GFX1132DAGISEL-NEXT: s_mov_b32 exec_lo, s2
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1132DAGISEL-NEXT: v_dual_mov_b32 v3, s1 :: v_dual_mov_b32 v2, s0
; GFX1132DAGISEL-NEXT: global_store_b64 v[0:1], v[2:3], off
; GFX1132DAGISEL-NEXT: s_xor_saveexec_b32 s0, -1
@@ -2302,42 +2308,43 @@ define void @divergent_value_double_dpp(ptr addrspace(1) %out, double %in) #0 {
; GFX1132GISEL-NEXT: scratch_store_b64 off, v[4:5], s32
; GFX1132GISEL-NEXT: scratch_store_b64 off, v[6:7], s32 offset:8
; GFX1132GISEL-NEXT: s_mov_b32 exec_lo, s0
+; GFX1132GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1132GISEL-NEXT: s_or_saveexec_b32 s2, -1
-; GFX1132GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX1132GISEL-NEXT: v_cndmask_b32_e64 v4, 0, v2, s2
; GFX1132GISEL-NEXT: v_cndmask_b32_e64 v5, 0x7ff80000, v3, s2
+; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132GISEL-NEXT: v_dual_mov_b32 v6, v4 :: v_dual_mov_b32 v7, v5
-; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1132GISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:1 row_mask:0xf bank_mask:0xf
+; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132GISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:1 row_mask:0xf bank_mask:0xf
-; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132GISEL-NEXT: v_min_f64 v[4:5], v[4:5], v[6:7]
+; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132GISEL-NEXT: v_dual_mov_b32 v6, v4 :: v_dual_mov_b32 v7, v5
-; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1132GISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:2 row_mask:0xf bank_mask:0xf
+; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132GISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:2 row_mask:0xf bank_mask:0xf
-; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132GISEL-NEXT: v_min_f64 v[4:5], v[4:5], v[6:7]
+; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132GISEL-NEXT: v_dual_mov_b32 v6, v4 :: v_dual_mov_b32 v7, v5
-; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1132GISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:4 row_mask:0xf bank_mask:0xf
+; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132GISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:4 row_mask:0xf bank_mask:0xf
-; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132GISEL-NEXT: v_min_f64 v[4:5], v[4:5], v[6:7]
+; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132GISEL-NEXT: v_dual_mov_b32 v6, v4 :: v_dual_mov_b32 v7, v5
-; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1132GISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:8 row_mask:0xf bank_mask:0xf
+; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132GISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:8 row_mask:0xf bank_mask:0xf
-; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_4) | instid1(VALU_DEP_1)
; GFX1132GISEL-NEXT: v_min_f64 v[4:5], v[4:5], v[6:7]
; GFX1132GISEL-NEXT: ds_swizzle_b32 v6, v4 offset:swizzle(BROADCAST,32,15)
; GFX1132GISEL-NEXT: ds_swizzle_b32 v7, v5 offset:swizzle(BROADCAST,32,15)
; GFX1132GISEL-NEXT: s_waitcnt lgkmcnt(0)
; GFX1132GISEL-NEXT: v_min_f64 v[4:5], v[4:5], v[6:7]
+; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1132GISEL-NEXT: v_readlane_b32 s0, v4, 31
-; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX1132GISEL-NEXT: v_readlane_b32 s1, v5, 31
; GFX1132GISEL-NEXT: s_mov_b32 exec_lo, s2
+; GFX1132GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1132GISEL-NEXT: v_dual_mov_b32 v3, s1 :: v_dual_mov_b32 v2, s0
; GFX1132GISEL-NEXT: global_store_b64 v[0:1], v[2:3], off
; GFX1132GISEL-NEXT: s_xor_saveexec_b32 s0, -1
@@ -2361,40 +2368,41 @@ define void @divergent_value_double_dpp(ptr addrspace(1) %out, double %in) #0 {
; GFX12DAGISEL-NEXT: scratch_store_b64 off, v[6:7], s32 offset:8
; GFX12DAGISEL-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12DAGISEL-NEXT: s_mov_b32 exec_lo, s0
+; GFX12DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_3) | instid1(VALU_DEP_1)
; GFX12DAGISEL-NEXT: s_or_saveexec_b32 s2, -1
; GFX12DAGISEL-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12DAGISEL-NEXT: v_cndmask_b32_e64 v4, 0, v2, s2
; GFX12DAGISEL-NEXT: v_cndmask_b32_e64 v5, 0x7ff80000, v3, s2
-; GFX12DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX12DAGISEL-NEXT: v_dual_mov_b32 v6, v4 :: v_dual_mov_b32 v7, v5
+; GFX12DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX12DAGISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:1 row_mask:0xf bank_mask:0xf
-; GFX12DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX12DAGISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:1 row_mask:0xf bank_mask:0xf
-; GFX12DAGISEL-NEXT: v_min_num_f64_e32 v[4:5], v[4:5], v[6:7]
; GFX12DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX12DAGISEL-NEXT: v_min_num_f64_e32 v[4:5], v[4:5], v[6:7]
; GFX12DAGISEL-NEXT: v_dual_mov_b32 v6, v4 :: v_dual_mov_b32 v7, v5
+; GFX12DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX12DAGISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:2 row_mask:0xf bank_mask:0xf
-; GFX12DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX12DAGISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:2 row_mask:0xf bank_mask:0xf
-; GFX12DAGISEL-NEXT: v_min_num_f64_e32 v[4:5], v[4:5], v[6:7]
; GFX12DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX12DAGISEL-NEXT: v_min_num_f64_e32 v[4:5], v[4:5], v[6:7]
; GFX12DAGISEL-NEXT: v_dual_mov_b32 v6, v4 :: v_dual_mov_b32 v7, v5
+; GFX12DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX12DAGISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:4 row_mask:0xf bank_mask:0xf
-; GFX12DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX12DAGISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:4 row_mask:0xf bank_mask:0xf
-; GFX12DAGISEL-NEXT: v_min_num_f64_e32 v[4:5], v[4:5], v[6:7]
; GFX12DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX12DAGISEL-NEXT: v_min_num_f64_e32 v[4:5], v[4:5], v[6:7]
; GFX12DAGISEL-NEXT: v_dual_mov_b32 v6, v4 :: v_dual_mov_b32 v7, v5
+; GFX12DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX12DAGISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:8 row_mask:0xf bank_mask:0xf
-; GFX12DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX12DAGISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:8 row_mask:0xf bank_mask:0xf
+; GFX12DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_4) | instid1(VALU_DEP_1)
; GFX12DAGISEL-NEXT: v_min_num_f64_e32 v[4:5], v[4:5], v[6:7]
; GFX12DAGISEL-NEXT: ds_swizzle_b32 v6, v4 offset:swizzle(BROADCAST,32,15)
; GFX12DAGISEL-NEXT: ds_swizzle_b32 v7, v5 offset:swizzle(BROADCAST,32,15)
; GFX12DAGISEL-NEXT: s_wait_dscnt 0x0
; GFX12DAGISEL-NEXT: v_min_num_f64_e32 v[4:5], v[4:5], v[6:7]
-; GFX12DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX12DAGISEL-NEXT: v_readlane_b32 s0, v4, 31
+; GFX12DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX12DAGISEL-NEXT: v_readlane_b32 s1, v5, 31
; GFX12DAGISEL-NEXT: s_mov_b32 exec_lo, s2
; GFX12DAGISEL-NEXT: v_dual_mov_b32 v3, s1 :: v_dual_mov_b32 v2, s0
@@ -2782,10 +2790,11 @@ define void @divergent_cfg_float(ptr addrspace(1) %out, float %in, float %in2) #
; GFX1164DAGISEL: ; %bb.0: ; %entry
; GFX1164DAGISEL-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX1164DAGISEL-NEXT: v_and_b32_e32 v4, 0x3ff, v31
-; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164DAGISEL-NEXT: v_cmp_lt_u32_e32 vcc, 15, v4
; GFX1164DAGISEL-NEXT: ; implicit-def: $vgpr4
; GFX1164DAGISEL-NEXT: s_and_saveexec_b64 s[0:1], vcc
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164DAGISEL-NEXT: s_xor_b64 s[0:1], exec, s[0:1]
; GFX1164DAGISEL-NEXT: s_cbranch_execz .LBB6_4
; GFX1164DAGISEL-NEXT: ; %bb.1: ; %else
@@ -2831,10 +2840,11 @@ define void @divergent_cfg_float(ptr addrspace(1) %out, float %in, float %in2) #
; GFX1164GISEL: ; %bb.0: ; %entry
; GFX1164GISEL-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX1164GISEL-NEXT: v_and_b32_e32 v4, 0x3ff, v31
-; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
+; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164GISEL-NEXT: v_cmp_le_u32_e32 vcc, 16, v4
; GFX1164GISEL-NEXT: ; implicit-def: $vgpr4
; GFX1164GISEL-NEXT: s_and_saveexec_b64 s[0:1], vcc
+; GFX1164GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164GISEL-NEXT: s_xor_b64 s[0:1], exec, s[0:1]
; GFX1164GISEL-NEXT: s_cbranch_execz .LBB6_4
; GFX1164GISEL-NEXT: ; %bb.1: ; %else
@@ -3940,10 +3950,11 @@ define void @divergent_cfg_double(ptr addrspace(1) %out, double %in, double %in2
; GFX1164DAGISEL: ; %bb.0: ; %entry
; GFX1164DAGISEL-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX1164DAGISEL-NEXT: v_and_b32_e32 v6, 0x3ff, v31
-; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164DAGISEL-NEXT: v_cmp_lt_u32_e32 vcc, 15, v6
; GFX1164DAGISEL-NEXT: ; implicit-def: $vgpr6_vgpr7
; GFX1164DAGISEL-NEXT: s_and_saveexec_b64 s[0:1], vcc
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164DAGISEL-NEXT: s_xor_b64 s[0:1], exec, s[0:1]
; GFX1164DAGISEL-NEXT: s_cbranch_execz .LBB9_4
; GFX1164DAGISEL-NEXT: ; %bb.1: ; %else
@@ -4000,10 +4011,11 @@ define void @divergent_cfg_double(ptr addrspace(1) %out, double %in, double %in2
; GFX1164GISEL: ; %bb.0: ; %entry
; GFX1164GISEL-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX1164GISEL-NEXT: v_and_b32_e32 v6, 0x3ff, v31
-; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
+; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164GISEL-NEXT: v_cmp_le_u32_e32 vcc, 16, v6
; GFX1164GISEL-NEXT: ; implicit-def: $vgpr6_vgpr7
; GFX1164GISEL-NEXT: s_and_saveexec_b64 s[0:1], vcc
+; GFX1164GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164GISEL-NEXT: s_xor_b64 s[0:1], exec, s[0:1]
; GFX1164GISEL-NEXT: s_cbranch_execz .LBB9_4
; GFX1164GISEL-NEXT: ; %bb.1: ; %else
diff --git a/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.reduce.fsub.ll b/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.reduce.fsub.ll
index ede0a21ed57fd1..04a73475e74c82 100644
--- a/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.reduce.fsub.ll
+++ b/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.reduce.fsub.ll
@@ -1509,19 +1509,20 @@ define void @divergent_value_float_dpp(ptr addrspace(1) %out, float %id.x) #0 {
; GFX1164DAGISEL-NEXT: scratch_store_b32 off, v4, s32 offset:4
; GFX1164DAGISEL-NEXT: scratch_store_b32 off, v5, s32 offset:8
; GFX1164DAGISEL-NEXT: s_mov_b64 exec, s[0:1]
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1164DAGISEL-NEXT: s_or_saveexec_b64 s[0:1], -1
-; GFX1164DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX1164DAGISEL-NEXT: v_cndmask_b32_e64 v3, 0, v2, s[0:1]
; GFX1164DAGISEL-NEXT: v_mbcnt_lo_u32_b32 v5, -1, 0
-; GFX1164DAGISEL-NEXT: v_add_f32_dpp v3, v3, v3 row_shr:1 row_mask:0xf bank_mask:0xf
; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX1164DAGISEL-NEXT: v_add_f32_dpp v3, v3, v3 row_shr:1 row_mask:0xf bank_mask:0xf
; GFX1164DAGISEL-NEXT: v_mbcnt_hi_u32_b32 v5, -1, v5
-; GFX1164DAGISEL-NEXT: v_add_f32_dpp v3, v3, v3 row_shr:2 row_mask:0xf bank_mask:0xf
; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX1164DAGISEL-NEXT: v_add_f32_dpp v3, v3, v3 row_shr:2 row_mask:0xf bank_mask:0xf
; GFX1164DAGISEL-NEXT: v_add_nc_u32_e32 v5, 32, v5
-; GFX1164DAGISEL-NEXT: v_add_f32_dpp v3, v3, v3 row_shr:4 row_mask:0xf bank_mask:0xf
; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX1164DAGISEL-NEXT: v_add_f32_dpp v3, v3, v3 row_shr:4 row_mask:0xf bank_mask:0xf
; GFX1164DAGISEL-NEXT: v_mul_lo_u32 v5, 4, v5
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX1164DAGISEL-NEXT: v_add_f32_dpp v3, v3, v3 row_shr:8 row_mask:0xf bank_mask:0xf
; GFX1164DAGISEL-NEXT: ds_swizzle_b32 v4, v3 offset:swizzle(BROADCAST,32,15)
; GFX1164DAGISEL-NEXT: s_waitcnt lgkmcnt(0)
@@ -1533,6 +1534,7 @@ define void @divergent_value_float_dpp(ptr addrspace(1) %out, float %id.x) #0 {
; GFX1164DAGISEL-NEXT: v_sub_f32_e32 v3, 0, v3
; GFX1164DAGISEL-NEXT: v_readlane_b32 s2, v3, 63
; GFX1164DAGISEL-NEXT: s_mov_b64 exec, s[0:1]
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v2, s2
; GFX1164DAGISEL-NEXT: global_store_b32 v[0:1], v2, off
; GFX1164DAGISEL-NEXT: s_xor_saveexec_b64 s[0:1], -1
@@ -1554,19 +1556,20 @@ define void @divergent_value_float_dpp(ptr addrspace(1) %out, float %id.x) #0 {
; GFX1164GISEL-NEXT: scratch_store_b32 off, v4, s32 offset:4
; GFX1164GISEL-NEXT: scratch_store_b32 off, v5, s32 offset:8
; GFX1164GISEL-NEXT: s_mov_b64 exec, s[0:1]
+; GFX1164GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1164GISEL-NEXT: s_or_saveexec_b64 s[0:1], -1
-; GFX1164GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX1164GISEL-NEXT: v_cndmask_b32_e64 v3, 0, v2, s[0:1]
; GFX1164GISEL-NEXT: v_mbcnt_lo_u32_b32 v5, -1, 0
-; GFX1164GISEL-NEXT: v_add_f32_dpp v3, v3, v3 row_shr:1 row_mask:0xf bank_mask:0xf
; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX1164GISEL-NEXT: v_add_f32_dpp v3, v3, v3 row_shr:1 row_mask:0xf bank_mask:0xf
; GFX1164GISEL-NEXT: v_mbcnt_hi_u32_b32 v5, -1, v5
-; GFX1164GISEL-NEXT: v_add_f32_dpp v3, v3, v3 row_shr:2 row_mask:0xf bank_mask:0xf
; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX1164GISEL-NEXT: v_add_f32_dpp v3, v3, v3 row_shr:2 row_mask:0xf bank_mask:0xf
; GFX1164GISEL-NEXT: v_add_nc_u32_e32 v5, 32, v5
-; GFX1164GISEL-NEXT: v_add_f32_dpp v3, v3, v3 row_shr:4 row_mask:0xf bank_mask:0xf
; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX1164GISEL-NEXT: v_add_f32_dpp v3, v3, v3 row_shr:4 row_mask:0xf bank_mask:0xf
; GFX1164GISEL-NEXT: v_mul_lo_u32 v5, 4, v5
+; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX1164GISEL-NEXT: v_add_f32_dpp v3, v3, v3 row_shr:8 row_mask:0xf bank_mask:0xf
; GFX1164GISEL-NEXT: ds_swizzle_b32 v4, v3 offset:swizzle(BROADCAST,32,15)
; GFX1164GISEL-NEXT: s_waitcnt lgkmcnt(0)
@@ -1578,6 +1581,7 @@ define void @divergent_value_float_dpp(ptr addrspace(1) %out, float %id.x) #0 {
; GFX1164GISEL-NEXT: v_sub_f32_e32 v3, 0, v3
; GFX1164GISEL-NEXT: v_readlane_b32 s2, v3, 63
; GFX1164GISEL-NEXT: s_mov_b64 exec, s[0:1]
+; GFX1164GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164GISEL-NEXT: v_mov_b32_e32 v2, s2
; GFX1164GISEL-NEXT: global_store_b32 v[0:1], v2, off
; GFX1164GISEL-NEXT: s_xor_saveexec_b64 s[0:1], -1
@@ -1598,22 +1602,23 @@ define void @divergent_value_float_dpp(ptr addrspace(1) %out, float %id.x) #0 {
; GFX1132DAGISEL-NEXT: scratch_store_b32 off, v3, s32
; GFX1132DAGISEL-NEXT: scratch_store_b32 off, v4, s32 offset:4
; GFX1132DAGISEL-NEXT: s_mov_b32 exec_lo, s0
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1132DAGISEL-NEXT: s_or_saveexec_b32 s0, -1
-; GFX1132DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132DAGISEL-NEXT: v_cndmask_b32_e64 v3, 0, v2, s0
-; GFX1132DAGISEL-NEXT: v_add_f32_dpp v3, v3, v3 row_shr:1 row_mask:0xf bank_mask:0xf
; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1132DAGISEL-NEXT: v_add_f32_dpp v3, v3, v3 row_shr:1 row_mask:0xf bank_mask:0xf
; GFX1132DAGISEL-NEXT: v_add_f32_dpp v3, v3, v3 row_shr:2 row_mask:0xf bank_mask:0xf
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132DAGISEL-NEXT: v_add_f32_dpp v3, v3, v3 row_shr:4 row_mask:0xf bank_mask:0xf
-; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_3) | instid1(VALU_DEP_1)
; GFX1132DAGISEL-NEXT: v_add_f32_dpp v3, v3, v3 row_shr:8 row_mask:0xf bank_mask:0xf
; GFX1132DAGISEL-NEXT: ds_swizzle_b32 v4, v3 offset:swizzle(BROADCAST,32,15)
; GFX1132DAGISEL-NEXT: s_waitcnt lgkmcnt(0)
; GFX1132DAGISEL-NEXT: v_sub_f32_e32 v3, v3, v4
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132DAGISEL-NEXT: v_sub_f32_e32 v3, 0, v3
-; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1132DAGISEL-NEXT: v_readlane_b32 s1, v3, 31
; GFX1132DAGISEL-NEXT: s_mov_b32 exec_lo, s0
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1132DAGISEL-NEXT: v_mov_b32_e32 v2, s1
; GFX1132DAGISEL-NEXT: global_store_b32 v[0:1], v2, off
; GFX1132DAGISEL-NEXT: s_xor_saveexec_b32 s0, -1
@@ -1632,22 +1637,23 @@ define void @divergent_value_float_dpp(ptr addrspace(1) %out, float %id.x) #0 {
; GFX1132GISEL-NEXT: scratch_store_b32 off, v3, s32
; GFX1132GISEL-NEXT: scratch_store_b32 off, v4, s32 offset:4
; GFX1132GISEL-NEXT: s_mov_b32 exec_lo, s0
+; GFX1132GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1132GISEL-NEXT: s_or_saveexec_b32 s0, -1
-; GFX1132GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132GISEL-NEXT: v_cndmask_b32_e64 v3, 0, v2, s0
-; GFX1132GISEL-NEXT: v_add_f32_dpp v3, v3, v3 row_shr:1 row_mask:0xf bank_mask:0xf
; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1132GISEL-NEXT: v_add_f32_dpp v3, v3, v3 row_shr:1 row_mask:0xf bank_mask:0xf
; GFX1132GISEL-NEXT: v_add_f32_dpp v3, v3, v3 row_shr:2 row_mask:0xf bank_mask:0xf
+; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132GISEL-NEXT: v_add_f32_dpp v3, v3, v3 row_shr:4 row_mask:0xf bank_mask:0xf
-; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_3) | instid1(VALU_DEP_1)
; GFX1132GISEL-NEXT: v_add_f32_dpp v3, v3, v3 row_shr:8 row_mask:0xf bank_mask:0xf
; GFX1132GISEL-NEXT: ds_swizzle_b32 v4, v3 offset:swizzle(BROADCAST,32,15)
; GFX1132GISEL-NEXT: s_waitcnt lgkmcnt(0)
; GFX1132GISEL-NEXT: v_sub_f32_e32 v3, v3, v4
+; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132GISEL-NEXT: v_sub_f32_e32 v3, 0, v3
-; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1132GISEL-NEXT: v_readlane_b32 s1, v3, 31
; GFX1132GISEL-NEXT: s_mov_b32 exec_lo, s0
+; GFX1132GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1132GISEL-NEXT: v_mov_b32_e32 v2, s1
; GFX1132GISEL-NEXT: global_store_b32 v[0:1], v2, off
; GFX1132GISEL-NEXT: s_xor_saveexec_b32 s0, -1
@@ -1671,20 +1677,21 @@ define void @divergent_value_float_dpp(ptr addrspace(1) %out, float %id.x) #0 {
; GFX12DAGISEL-NEXT: scratch_store_b32 off, v4, s32 offset:4
; GFX12DAGISEL-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12DAGISEL-NEXT: s_mov_b32 exec_lo, s0
+; GFX12DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(VALU_DEP_1)
; GFX12DAGISEL-NEXT: s_or_saveexec_b32 s0, -1
; GFX12DAGISEL-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12DAGISEL-NEXT: v_cndmask_b32_e64 v3, 0, v2, s0
-; GFX12DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX12DAGISEL-NEXT: v_add_f32_dpp v3, v3, v3 row_shr:1 row_mask:0xf bank_mask:0xf
-; GFX12DAGISEL-NEXT: v_add_f32_dpp v3, v3, v3 row_shr:2 row_mask:0xf bank_mask:0xf
; GFX12DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX12DAGISEL-NEXT: v_add_f32_dpp v3, v3, v3 row_shr:2 row_mask:0xf bank_mask:0xf
; GFX12DAGISEL-NEXT: v_add_f32_dpp v3, v3, v3 row_shr:4 row_mask:0xf bank_mask:0xf
+; GFX12DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_3) | instid1(VALU_DEP_1)
; GFX12DAGISEL-NEXT: v_add_f32_dpp v3, v3, v3 row_shr:8 row_mask:0xf bank_mask:0xf
; GFX12DAGISEL-NEXT: ds_swizzle_b32 v4, v3 offset:swizzle(BROADCAST,32,15)
; GFX12DAGISEL-NEXT: s_wait_dscnt 0x0
; GFX12DAGISEL-NEXT: v_sub_f32_e32 v3, v3, v4
-; GFX12DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX12DAGISEL-NEXT: v_sub_f32_e32 v3, 0, v3
+; GFX12DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX12DAGISEL-NEXT: v_readlane_b32 s1, v3, 31
; GFX12DAGISEL-NEXT: s_mov_b32 exec_lo, s0
; GFX12DAGISEL-NEXT: v_mov_b32_e32 v2, s1
@@ -2282,39 +2289,39 @@ define void @divergent_value_double_dpp(ptr addrspace(1) %out, double %in) #0 {
; GFX1164DAGISEL-NEXT: scratch_store_b32 off, v6, s32 offset:16
; GFX1164DAGISEL-NEXT: scratch_store_b64 off, v[7:8], s32 offset:20
; GFX1164DAGISEL-NEXT: s_mov_b64 exec, s[0:1]
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1164DAGISEL-NEXT: s_or_saveexec_b64 s[0:1], -1
-; GFX1164DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX1164DAGISEL-NEXT: v_cndmask_b32_e64 v4, 0, v2, s[0:1]
; GFX1164DAGISEL-NEXT: v_cndmask_b32_e64 v5, 0x80000000, v3, s[0:1]
-; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v6, v4
; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v6, v4
; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v7, v5
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1164DAGISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:1 row_mask:0xf bank_mask:0xf
-; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164DAGISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:1 row_mask:0xf bank_mask:0xf
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164DAGISEL-NEXT: v_add_f64 v[4:5], v[4:5], v[6:7]
-; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v6, v4
-; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v7, v5
; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v7, v5
; GFX1164DAGISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:2 row_mask:0xf bank_mask:0xf
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164DAGISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:2 row_mask:0xf bank_mask:0xf
-; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164DAGISEL-NEXT: v_add_f64 v[4:5], v[4:5], v[6:7]
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v6, v4
-; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v7, v5
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1164DAGISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:4 row_mask:0xf bank_mask:0xf
-; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164DAGISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:4 row_mask:0xf bank_mask:0xf
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164DAGISEL-NEXT: v_add_f64 v[4:5], v[4:5], v[6:7]
-; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v6, v4
-; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v7, v5
; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v7, v5
; GFX1164DAGISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:8 row_mask:0xf bank_mask:0xf
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164DAGISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:8 row_mask:0xf bank_mask:0xf
-; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1164DAGISEL-NEXT: v_add_f64 v[4:5], v[4:5], v[6:7]
; GFX1164DAGISEL-NEXT: ds_swizzle_b32 v6, v4 offset:swizzle(BROADCAST,32,15)
; GFX1164DAGISEL-NEXT: ds_swizzle_b32 v7, v5 offset:swizzle(BROADCAST,32,15)
@@ -2335,6 +2342,7 @@ define void @divergent_value_double_dpp(ptr addrspace(1) %out, double %in) #0 {
; GFX1164DAGISEL-NEXT: v_readlane_b32 s2, v4, 63
; GFX1164DAGISEL-NEXT: v_readlane_b32 s3, v5, 63
; GFX1164DAGISEL-NEXT: s_mov_b64 exec, s[0:1]
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v2, s2
; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v3, s3
; GFX1164DAGISEL-NEXT: global_store_b64 v[0:1], v[2:3], off
@@ -2359,39 +2367,39 @@ define void @divergent_value_double_dpp(ptr addrspace(1) %out, double %in) #0 {
; GFX1164GISEL-NEXT: scratch_store_b32 off, v6, s32 offset:16
; GFX1164GISEL-NEXT: scratch_store_b64 off, v[7:8], s32 offset:20
; GFX1164GISEL-NEXT: s_mov_b64 exec, s[0:1]
+; GFX1164GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1164GISEL-NEXT: s_or_saveexec_b64 s[0:1], -1
-; GFX1164GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX1164GISEL-NEXT: v_cndmask_b32_e64 v4, 0, v2, s[0:1]
; GFX1164GISEL-NEXT: v_cndmask_b32_e64 v5, 0x80000000, v3, s[0:1]
-; GFX1164GISEL-NEXT: v_mov_b32_e32 v6, v4
; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX1164GISEL-NEXT: v_mov_b32_e32 v6, v4
; GFX1164GISEL-NEXT: v_mov_b32_e32 v7, v5
+; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1164GISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:1 row_mask:0xf bank_mask:0xf
-; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164GISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:1 row_mask:0xf bank_mask:0xf
+; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164GISEL-NEXT: v_add_f64 v[4:5], v[4:5], v[6:7]
-; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1164GISEL-NEXT: v_mov_b32_e32 v6, v4
-; GFX1164GISEL-NEXT: v_mov_b32_e32 v7, v5
; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX1164GISEL-NEXT: v_mov_b32_e32 v7, v5
; GFX1164GISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:2 row_mask:0xf bank_mask:0xf
+; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164GISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:2 row_mask:0xf bank_mask:0xf
-; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164GISEL-NEXT: v_add_f64 v[4:5], v[4:5], v[6:7]
+; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1164GISEL-NEXT: v_mov_b32_e32 v6, v4
-; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1164GISEL-NEXT: v_mov_b32_e32 v7, v5
+; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1164GISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:4 row_mask:0xf bank_mask:0xf
-; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164GISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:4 row_mask:0xf bank_mask:0xf
+; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164GISEL-NEXT: v_add_f64 v[4:5], v[4:5], v[6:7]
-; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1164GISEL-NEXT: v_mov_b32_e32 v6, v4
-; GFX1164GISEL-NEXT: v_mov_b32_e32 v7, v5
; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX1164GISEL-NEXT: v_mov_b32_e32 v7, v5
; GFX1164GISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:8 row_mask:0xf bank_mask:0xf
+; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164GISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:8 row_mask:0xf bank_mask:0xf
-; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1164GISEL-NEXT: v_add_f64 v[4:5], v[4:5], v[6:7]
; GFX1164GISEL-NEXT: ds_swizzle_b32 v6, v4 offset:swizzle(BROADCAST,32,15)
; GFX1164GISEL-NEXT: ds_swizzle_b32 v7, v5 offset:swizzle(BROADCAST,32,15)
@@ -2412,6 +2420,7 @@ define void @divergent_value_double_dpp(ptr addrspace(1) %out, double %in) #0 {
; GFX1164GISEL-NEXT: v_readlane_b32 s2, v4, 63
; GFX1164GISEL-NEXT: v_readlane_b32 s3, v5, 63
; GFX1164GISEL-NEXT: s_mov_b64 exec, s[0:1]
+; GFX1164GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164GISEL-NEXT: v_mov_b32_e32 v2, s2
; GFX1164GISEL-NEXT: v_mov_b32_e32 v3, s3
; GFX1164GISEL-NEXT: global_store_b64 v[0:1], v[2:3], off
@@ -2434,41 +2443,42 @@ define void @divergent_value_double_dpp(ptr addrspace(1) %out, double %in) #0 {
; GFX1132DAGISEL-NEXT: scratch_store_b64 off, v[4:5], s32
; GFX1132DAGISEL-NEXT: scratch_store_b64 off, v[6:7], s32 offset:8
; GFX1132DAGISEL-NEXT: s_mov_b32 exec_lo, s0
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1132DAGISEL-NEXT: s_or_saveexec_b32 s2, -1
-; GFX1132DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX1132DAGISEL-NEXT: v_cndmask_b32_e64 v4, 0, v2, s2
; GFX1132DAGISEL-NEXT: v_cndmask_b32_e64 v5, 0x80000000, v3, s2
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132DAGISEL-NEXT: v_dual_mov_b32 v6, v4 :: v_dual_mov_b32 v7, v5
-; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1132DAGISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:1 row_mask:0xf bank_mask:0xf
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132DAGISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:1 row_mask:0xf bank_mask:0xf
-; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132DAGISEL-NEXT: v_add_f64 v[4:5], v[4:5], v[6:7]
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132DAGISEL-NEXT: v_dual_mov_b32 v6, v4 :: v_dual_mov_b32 v7, v5
-; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1132DAGISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:2 row_mask:0xf bank_mask:0xf
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132DAGISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:2 row_mask:0xf bank_mask:0xf
-; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132DAGISEL-NEXT: v_add_f64 v[4:5], v[4:5], v[6:7]
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132DAGISEL-NEXT: v_dual_mov_b32 v6, v4 :: v_dual_mov_b32 v7, v5
-; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1132DAGISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:4 row_mask:0xf bank_mask:0xf
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132DAGISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:4 row_mask:0xf bank_mask:0xf
-; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132DAGISEL-NEXT: v_add_f64 v[4:5], v[4:5], v[6:7]
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132DAGISEL-NEXT: v_dual_mov_b32 v6, v4 :: v_dual_mov_b32 v7, v5
-; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1132DAGISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:8 row_mask:0xf bank_mask:0xf
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132DAGISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:8 row_mask:0xf bank_mask:0xf
-; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_4) | instid1(VALU_DEP_1)
; GFX1132DAGISEL-NEXT: v_add_f64 v[4:5], v[4:5], v[6:7]
; GFX1132DAGISEL-NEXT: ds_swizzle_b32 v6, v4 offset:swizzle(BROADCAST,32,15)
; GFX1132DAGISEL-NEXT: ds_swizzle_b32 v7, v5 offset:swizzle(BROADCAST,32,15)
; GFX1132DAGISEL-NEXT: s_waitcnt lgkmcnt(0)
; GFX1132DAGISEL-NEXT: v_add_f64 v[4:5], v[4:5], v[6:7]
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132DAGISEL-NEXT: v_add_f64 v[4:5], 0x80000000, -v[4:5]
-; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1132DAGISEL-NEXT: v_readlane_b32 s0, v4, 31
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX1132DAGISEL-NEXT: v_readlane_b32 s1, v5, 31
; GFX1132DAGISEL-NEXT: s_mov_b32 exec_lo, s2
; GFX1132DAGISEL-NEXT: v_dual_mov_b32 v3, s1 :: v_dual_mov_b32 v2, s0
@@ -2489,41 +2499,42 @@ define void @divergent_value_double_dpp(ptr addrspace(1) %out, double %in) #0 {
; GFX1132GISEL-NEXT: scratch_store_b64 off, v[4:5], s32
; GFX1132GISEL-NEXT: scratch_store_b64 off, v[6:7], s32 offset:8
; GFX1132GISEL-NEXT: s_mov_b32 exec_lo, s0
+; GFX1132GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1132GISEL-NEXT: s_or_saveexec_b32 s2, -1
-; GFX1132GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX1132GISEL-NEXT: v_cndmask_b32_e64 v4, 0, v2, s2
; GFX1132GISEL-NEXT: v_cndmask_b32_e64 v5, 0x80000000, v3, s2
+; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132GISEL-NEXT: v_dual_mov_b32 v6, v4 :: v_dual_mov_b32 v7, v5
-; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1132GISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:1 row_mask:0xf bank_mask:0xf
+; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132GISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:1 row_mask:0xf bank_mask:0xf
-; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132GISEL-NEXT: v_add_f64 v[4:5], v[4:5], v[6:7]
+; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132GISEL-NEXT: v_dual_mov_b32 v6, v4 :: v_dual_mov_b32 v7, v5
-; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1132GISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:2 row_mask:0xf bank_mask:0xf
+; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132GISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:2 row_mask:0xf bank_mask:0xf
-; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132GISEL-NEXT: v_add_f64 v[4:5], v[4:5], v[6:7]
+; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132GISEL-NEXT: v_dual_mov_b32 v6, v4 :: v_dual_mov_b32 v7, v5
-; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1132GISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:4 row_mask:0xf bank_mask:0xf
+; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132GISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:4 row_mask:0xf bank_mask:0xf
-; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132GISEL-NEXT: v_add_f64 v[4:5], v[4:5], v[6:7]
+; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132GISEL-NEXT: v_dual_mov_b32 v6, v4 :: v_dual_mov_b32 v7, v5
-; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1132GISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:8 row_mask:0xf bank_mask:0xf
+; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132GISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:8 row_mask:0xf bank_mask:0xf
-; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_4) | instid1(VALU_DEP_1)
; GFX1132GISEL-NEXT: v_add_f64 v[4:5], v[4:5], v[6:7]
; GFX1132GISEL-NEXT: ds_swizzle_b32 v6, v4 offset:swizzle(BROADCAST,32,15)
; GFX1132GISEL-NEXT: ds_swizzle_b32 v7, v5 offset:swizzle(BROADCAST,32,15)
; GFX1132GISEL-NEXT: s_waitcnt lgkmcnt(0)
; GFX1132GISEL-NEXT: v_add_f64 v[4:5], v[4:5], v[6:7]
+; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132GISEL-NEXT: v_add_f64 v[4:5], 0x80000000, -v[4:5]
-; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1132GISEL-NEXT: v_readlane_b32 s0, v4, 31
+; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX1132GISEL-NEXT: v_readlane_b32 s1, v5, 31
; GFX1132GISEL-NEXT: s_mov_b32 exec_lo, s2
; GFX1132GISEL-NEXT: v_dual_mov_b32 v3, s1 :: v_dual_mov_b32 v2, s0
@@ -2549,44 +2560,45 @@ define void @divergent_value_double_dpp(ptr addrspace(1) %out, double %in) #0 {
; GFX12DAGISEL-NEXT: scratch_store_b64 off, v[6:7], s32 offset:8
; GFX12DAGISEL-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12DAGISEL-NEXT: s_mov_b32 exec_lo, s0
+; GFX12DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_3) | instid1(VALU_DEP_1)
; GFX12DAGISEL-NEXT: s_or_saveexec_b32 s2, -1
; GFX12DAGISEL-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12DAGISEL-NEXT: v_cndmask_b32_e64 v4, 0, v2, s2
; GFX12DAGISEL-NEXT: v_cndmask_b32_e64 v5, 0x80000000, v3, s2
-; GFX12DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX12DAGISEL-NEXT: v_dual_mov_b32 v6, v4 :: v_dual_mov_b32 v7, v5
+; GFX12DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX12DAGISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:1 row_mask:0xf bank_mask:0xf
-; GFX12DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX12DAGISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:1 row_mask:0xf bank_mask:0xf
-; GFX12DAGISEL-NEXT: v_add_f64_e32 v[4:5], v[4:5], v[6:7]
; GFX12DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX12DAGISEL-NEXT: v_add_f64_e32 v[4:5], v[4:5], v[6:7]
; GFX12DAGISEL-NEXT: v_dual_mov_b32 v6, v4 :: v_dual_mov_b32 v7, v5
+; GFX12DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX12DAGISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:2 row_mask:0xf bank_mask:0xf
-; GFX12DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX12DAGISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:2 row_mask:0xf bank_mask:0xf
-; GFX12DAGISEL-NEXT: v_add_f64_e32 v[4:5], v[4:5], v[6:7]
; GFX12DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX12DAGISEL-NEXT: v_add_f64_e32 v[4:5], v[4:5], v[6:7]
; GFX12DAGISEL-NEXT: v_dual_mov_b32 v6, v4 :: v_dual_mov_b32 v7, v5
+; GFX12DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX12DAGISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:4 row_mask:0xf bank_mask:0xf
-; GFX12DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX12DAGISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:4 row_mask:0xf bank_mask:0xf
-; GFX12DAGISEL-NEXT: v_add_f64_e32 v[4:5], v[4:5], v[6:7]
; GFX12DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX12DAGISEL-NEXT: v_add_f64_e32 v[4:5], v[4:5], v[6:7]
; GFX12DAGISEL-NEXT: v_dual_mov_b32 v6, v4 :: v_dual_mov_b32 v7, v5
+; GFX12DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX12DAGISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:8 row_mask:0xf bank_mask:0xf
-; GFX12DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX12DAGISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:8 row_mask:0xf bank_mask:0xf
+; GFX12DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_4) | instid1(VALU_DEP_1)
; GFX12DAGISEL-NEXT: v_add_f64_e32 v[4:5], v[4:5], v[6:7]
; GFX12DAGISEL-NEXT: ds_swizzle_b32 v6, v4 offset:swizzle(BROADCAST,32,15)
; GFX12DAGISEL-NEXT: ds_swizzle_b32 v7, v5 offset:swizzle(BROADCAST,32,15)
; GFX12DAGISEL-NEXT: s_wait_dscnt 0x0
; GFX12DAGISEL-NEXT: v_add_f64_e32 v[4:5], v[4:5], v[6:7]
-; GFX12DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX12DAGISEL-NEXT: v_add_f64_e64 v[4:5], 0x80000000, -v[4:5]
+; GFX12DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX12DAGISEL-NEXT: v_readlane_b32 s0, v4, 31
-; GFX12DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX12DAGISEL-NEXT: v_readlane_b32 s1, v5, 31
; GFX12DAGISEL-NEXT: s_mov_b32 exec_lo, s2
+; GFX12DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12DAGISEL-NEXT: v_dual_mov_b32 v3, s1 :: v_dual_mov_b32 v2, s0
; GFX12DAGISEL-NEXT: global_store_b64 v[0:1], v[2:3], off
; GFX12DAGISEL-NEXT: s_xor_saveexec_b32 s0, -1
@@ -2906,7 +2918,7 @@ define amdgpu_kernel void @divergent_cfg_float(ptr addrspace(1) %out, float %in,
; GFX1164DAGISEL-NEXT: v_and_b32_e32 v0, 0x3ff, v0
; GFX1164DAGISEL-NEXT: s_mov_b64 s[2:3], exec
; GFX1164DAGISEL-NEXT: ; implicit-def: $sgpr6
-; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164DAGISEL-NEXT: v_cmpx_lt_u32_e32 15, v0
; GFX1164DAGISEL-NEXT: s_xor_b64 s[2:3], exec, s[2:3]
; GFX1164DAGISEL-NEXT: s_cbranch_execz .LBB6_2
@@ -2921,7 +2933,7 @@ define amdgpu_kernel void @divergent_cfg_float(ptr addrspace(1) %out, float %in,
; GFX1164DAGISEL-NEXT: v_readfirstlane_b32 s6, v0
; GFX1164DAGISEL-NEXT: .LBB6_2: ; %Flow
; GFX1164DAGISEL-NEXT: s_or_saveexec_b64 s[2:3], s[2:3]
-; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v0, s6
; GFX1164DAGISEL-NEXT: s_xor_b64 exec, exec, s[2:3]
; GFX1164DAGISEL-NEXT: s_cbranch_execz .LBB6_4
@@ -2950,7 +2962,7 @@ define amdgpu_kernel void @divergent_cfg_float(ptr addrspace(1) %out, float %in,
; GFX1164GISEL-NEXT: v_and_b32_e32 v0, 0x3ff, v0
; GFX1164GISEL-NEXT: s_mov_b64 s[2:3], exec
; GFX1164GISEL-NEXT: ; implicit-def: $sgpr6
-; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164GISEL-NEXT: v_cmpx_le_u32_e32 16, v0
; GFX1164GISEL-NEXT: s_xor_b64 s[2:3], exec, s[2:3]
; GFX1164GISEL-NEXT: s_cbranch_execz .LBB6_2
@@ -2965,7 +2977,7 @@ define amdgpu_kernel void @divergent_cfg_float(ptr addrspace(1) %out, float %in,
; GFX1164GISEL-NEXT: v_readfirstlane_b32 s6, v0
; GFX1164GISEL-NEXT: .LBB6_2: ; %Flow
; GFX1164GISEL-NEXT: s_or_saveexec_b64 s[2:3], s[2:3]
-; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX1164GISEL-NEXT: v_mov_b32_e32 v0, s6
; GFX1164GISEL-NEXT: s_xor_b64 exec, exec, s[2:3]
; GFX1164GISEL-NEXT: s_cbranch_execz .LBB6_4
@@ -2994,9 +3006,10 @@ define amdgpu_kernel void @divergent_cfg_float(ptr addrspace(1) %out, float %in,
; GFX1132DAGISEL-NEXT: v_and_b32_e32 v0, 0x3ff, v0
; GFX1132DAGISEL-NEXT: s_mov_b32 s2, exec_lo
; GFX1132DAGISEL-NEXT: ; implicit-def: $sgpr3
-; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132DAGISEL-NEXT: v_cmpx_lt_u32_e32 15, v0
; GFX1132DAGISEL-NEXT: s_xor_b32 s2, exec_lo, s2
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1132DAGISEL-NEXT: s_cbranch_execz .LBB6_2
; GFX1132DAGISEL-NEXT: ; %bb.1: ; %else
; GFX1132DAGISEL-NEXT: s_mov_b32 s3, exec_lo
@@ -3010,7 +3023,7 @@ define amdgpu_kernel void @divergent_cfg_float(ptr addrspace(1) %out, float %in,
; GFX1132DAGISEL-NEXT: .LBB6_2: ; %Flow
; GFX1132DAGISEL-NEXT: s_waitcnt lgkmcnt(0)
; GFX1132DAGISEL-NEXT: s_or_saveexec_b32 s0, s2
-; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX1132DAGISEL-NEXT: v_mov_b32_e32 v0, s3
; GFX1132DAGISEL-NEXT: s_xor_b32 exec_lo, exec_lo, s0
; GFX1132DAGISEL-NEXT: s_cbranch_execz .LBB6_4
@@ -3038,9 +3051,10 @@ define amdgpu_kernel void @divergent_cfg_float(ptr addrspace(1) %out, float %in,
; GFX1132GISEL-NEXT: v_and_b32_e32 v0, 0x3ff, v0
; GFX1132GISEL-NEXT: s_mov_b32 s2, exec_lo
; GFX1132GISEL-NEXT: ; implicit-def: $sgpr3
-; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132GISEL-NEXT: v_cmpx_le_u32_e32 16, v0
; GFX1132GISEL-NEXT: s_xor_b32 s2, exec_lo, s2
+; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1132GISEL-NEXT: s_cbranch_execz .LBB6_2
; GFX1132GISEL-NEXT: ; %bb.1: ; %else
; GFX1132GISEL-NEXT: s_mov_b32 s3, exec_lo
@@ -3054,7 +3068,7 @@ define amdgpu_kernel void @divergent_cfg_float(ptr addrspace(1) %out, float %in,
; GFX1132GISEL-NEXT: .LBB6_2: ; %Flow
; GFX1132GISEL-NEXT: s_waitcnt lgkmcnt(0)
; GFX1132GISEL-NEXT: s_or_saveexec_b32 s0, s2
-; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX1132GISEL-NEXT: v_mov_b32_e32 v0, s3
; GFX1132GISEL-NEXT: s_xor_b32 exec_lo, exec_lo, s0
; GFX1132GISEL-NEXT: s_cbranch_execz .LBB6_4
@@ -3082,9 +3096,10 @@ define amdgpu_kernel void @divergent_cfg_float(ptr addrspace(1) %out, float %in,
; GFX12DAGISEL-NEXT: v_and_b32_e32 v0, 0x3ff, v0
; GFX12DAGISEL-NEXT: s_mov_b32 s2, exec_lo
; GFX12DAGISEL-NEXT: ; implicit-def: $sgpr3
-; GFX12DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX12DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX12DAGISEL-NEXT: v_cmpx_lt_u32_e32 15, v0
; GFX12DAGISEL-NEXT: s_xor_b32 s2, exec_lo, s2
+; GFX12DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12DAGISEL-NEXT: s_cbranch_execz .LBB6_2
; GFX12DAGISEL-NEXT: ; %bb.1: ; %else
; GFX12DAGISEL-NEXT: s_mov_b32 s3, exec_lo
@@ -3098,7 +3113,7 @@ define amdgpu_kernel void @divergent_cfg_float(ptr addrspace(1) %out, float %in,
; GFX12DAGISEL-NEXT: .LBB6_2: ; %Flow
; GFX12DAGISEL-NEXT: s_wait_kmcnt 0x0
; GFX12DAGISEL-NEXT: s_or_saveexec_b32 s0, s2
-; GFX12DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX12DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
; GFX12DAGISEL-NEXT: v_mov_b32_e32 v0, s3
; GFX12DAGISEL-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12DAGISEL-NEXT: s_xor_b32 exec_lo, exec_lo, s0
@@ -4008,7 +4023,7 @@ define amdgpu_kernel void @divergent_cfg_double(ptr addrspace(1) %out, double %i
; GFX1164DAGISEL-NEXT: v_and_b32_e32 v0, 0x3ff, v0
; GFX1164DAGISEL-NEXT: s_mov_b64 s[6:7], exec
; GFX1164DAGISEL-NEXT: ; implicit-def: $sgpr8_sgpr9
-; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164DAGISEL-NEXT: v_cmpx_lt_u32_e32 15, v0
; GFX1164DAGISEL-NEXT: s_xor_b64 s[6:7], exec, s[6:7]
; GFX1164DAGISEL-NEXT: s_cbranch_execz .LBB9_2
@@ -4030,6 +4045,7 @@ define amdgpu_kernel void @divergent_cfg_double(ptr addrspace(1) %out, double %i
; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v0, s8
; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v1, s9
; GFX1164DAGISEL-NEXT: s_xor_b64 exec, exec, s[2:3]
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164DAGISEL-NEXT: s_cbranch_execz .LBB9_4
; GFX1164DAGISEL-NEXT: ; %bb.3: ; %if
; GFX1164DAGISEL-NEXT: s_mov_b64 s[6:7], exec
@@ -4046,6 +4062,7 @@ define amdgpu_kernel void @divergent_cfg_double(ptr addrspace(1) %out, double %i
; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v1, s5
; GFX1164DAGISEL-NEXT: .LBB9_4: ; %endif
; GFX1164DAGISEL-NEXT: s_or_b64 exec, exec, s[2:3]
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v2, 0
; GFX1164DAGISEL-NEXT: global_store_b64 v2, v[0:1], s[0:1]
; GFX1164DAGISEL-NEXT: s_endpgm
@@ -4056,7 +4073,7 @@ define amdgpu_kernel void @divergent_cfg_double(ptr addrspace(1) %out, double %i
; GFX1164GISEL-NEXT: v_and_b32_e32 v0, 0x3ff, v0
; GFX1164GISEL-NEXT: s_mov_b64 s[6:7], exec
; GFX1164GISEL-NEXT: ; implicit-def: $sgpr8_sgpr9
-; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164GISEL-NEXT: v_cmpx_le_u32_e32 16, v0
; GFX1164GISEL-NEXT: s_xor_b64 s[6:7], exec, s[6:7]
; GFX1164GISEL-NEXT: s_cbranch_execz .LBB9_2
@@ -4078,6 +4095,7 @@ define amdgpu_kernel void @divergent_cfg_double(ptr addrspace(1) %out, double %i
; GFX1164GISEL-NEXT: v_mov_b32_e32 v0, s8
; GFX1164GISEL-NEXT: v_mov_b32_e32 v1, s9
; GFX1164GISEL-NEXT: s_xor_b64 exec, exec, s[2:3]
+; GFX1164GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164GISEL-NEXT: s_cbranch_execz .LBB9_4
; GFX1164GISEL-NEXT: ; %bb.3: ; %if
; GFX1164GISEL-NEXT: s_mov_b64 s[6:7], exec
@@ -4095,6 +4113,7 @@ define amdgpu_kernel void @divergent_cfg_double(ptr addrspace(1) %out, double %i
; GFX1164GISEL-NEXT: v_mov_b32_e32 v1, s5
; GFX1164GISEL-NEXT: .LBB9_4: ; %endif
; GFX1164GISEL-NEXT: s_or_b64 exec, exec, s[2:3]
+; GFX1164GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164GISEL-NEXT: v_mov_b32_e32 v2, 0
; GFX1164GISEL-NEXT: global_store_b64 v2, v[0:1], s[0:1]
; GFX1164GISEL-NEXT: s_endpgm
@@ -4107,9 +4126,10 @@ define amdgpu_kernel void @divergent_cfg_double(ptr addrspace(1) %out, double %i
; GFX1132DAGISEL-NEXT: v_and_b32_e32 v0, 0x3ff, v0
; GFX1132DAGISEL-NEXT: s_mov_b32 s8, exec_lo
; GFX1132DAGISEL-NEXT: ; implicit-def: $sgpr6_sgpr7
-; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132DAGISEL-NEXT: v_cmpx_lt_u32_e32 15, v0
; GFX1132DAGISEL-NEXT: s_xor_b32 s8, exec_lo, s8
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1132DAGISEL-NEXT: s_cbranch_execz .LBB9_2
; GFX1132DAGISEL-NEXT: ; %bb.1: ; %else
; GFX1132DAGISEL-NEXT: s_mov_b32 s6, exec_lo
@@ -4125,7 +4145,7 @@ define amdgpu_kernel void @divergent_cfg_double(ptr addrspace(1) %out, double %i
; GFX1132DAGISEL-NEXT: .LBB9_2: ; %Flow
; GFX1132DAGISEL-NEXT: s_waitcnt lgkmcnt(0)
; GFX1132DAGISEL-NEXT: s_or_saveexec_b32 s2, s8
-; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX1132DAGISEL-NEXT: v_dual_mov_b32 v0, s6 :: v_dual_mov_b32 v1, s7
; GFX1132DAGISEL-NEXT: s_xor_b32 exec_lo, exec_lo, s2
; GFX1132DAGISEL-NEXT: s_cbranch_execz .LBB9_4
@@ -4142,6 +4162,7 @@ define amdgpu_kernel void @divergent_cfg_double(ptr addrspace(1) %out, double %i
; GFX1132DAGISEL-NEXT: v_dual_mov_b32 v0, s4 :: v_dual_mov_b32 v1, s5
; GFX1132DAGISEL-NEXT: .LBB9_4: ; %endif
; GFX1132DAGISEL-NEXT: s_or_b32 exec_lo, exec_lo, s2
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1132DAGISEL-NEXT: v_mov_b32_e32 v2, 0
; GFX1132DAGISEL-NEXT: global_store_b64 v2, v[0:1], s[0:1]
; GFX1132DAGISEL-NEXT: s_endpgm
@@ -4152,9 +4173,10 @@ define amdgpu_kernel void @divergent_cfg_double(ptr addrspace(1) %out, double %i
; GFX1132GISEL-NEXT: v_and_b32_e32 v0, 0x3ff, v0
; GFX1132GISEL-NEXT: s_mov_b32 s8, exec_lo
; GFX1132GISEL-NEXT: ; implicit-def: $sgpr6_sgpr7
-; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132GISEL-NEXT: v_cmpx_le_u32_e32 16, v0
; GFX1132GISEL-NEXT: s_xor_b32 s8, exec_lo, s8
+; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1132GISEL-NEXT: s_cbranch_execz .LBB9_2
; GFX1132GISEL-NEXT: ; %bb.1: ; %else
; GFX1132GISEL-NEXT: s_mov_b32 s6, exec_lo
@@ -4170,7 +4192,7 @@ define amdgpu_kernel void @divergent_cfg_double(ptr addrspace(1) %out, double %i
; GFX1132GISEL-NEXT: .LBB9_2: ; %Flow
; GFX1132GISEL-NEXT: s_waitcnt lgkmcnt(0)
; GFX1132GISEL-NEXT: s_or_saveexec_b32 s2, s8
-; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX1132GISEL-NEXT: v_dual_mov_b32 v0, s6 :: v_dual_mov_b32 v1, s7
; GFX1132GISEL-NEXT: s_xor_b32 exec_lo, exec_lo, s2
; GFX1132GISEL-NEXT: s_cbranch_execz .LBB9_4
@@ -4189,6 +4211,7 @@ define amdgpu_kernel void @divergent_cfg_double(ptr addrspace(1) %out, double %i
; GFX1132GISEL-NEXT: v_dual_mov_b32 v0, s4 :: v_dual_mov_b32 v1, s5
; GFX1132GISEL-NEXT: .LBB9_4: ; %endif
; GFX1132GISEL-NEXT: s_or_b32 exec_lo, exec_lo, s2
+; GFX1132GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1132GISEL-NEXT: v_mov_b32_e32 v2, 0
; GFX1132GISEL-NEXT: global_store_b64 v2, v[0:1], s[0:1]
; GFX1132GISEL-NEXT: s_endpgm
@@ -4201,9 +4224,10 @@ define amdgpu_kernel void @divergent_cfg_double(ptr addrspace(1) %out, double %i
; GFX12DAGISEL-NEXT: v_and_b32_e32 v0, 0x3ff, v0
; GFX12DAGISEL-NEXT: s_mov_b32 s8, exec_lo
; GFX12DAGISEL-NEXT: ; implicit-def: $sgpr6_sgpr7
-; GFX12DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX12DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX12DAGISEL-NEXT: v_cmpx_lt_u32_e32 15, v0
; GFX12DAGISEL-NEXT: s_xor_b32 s8, exec_lo, s8
+; GFX12DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12DAGISEL-NEXT: s_cbranch_execz .LBB9_2
; GFX12DAGISEL-NEXT: ; %bb.1: ; %else
; GFX12DAGISEL-NEXT: s_mov_b32 s6, exec_lo
@@ -4220,7 +4244,7 @@ define amdgpu_kernel void @divergent_cfg_double(ptr addrspace(1) %out, double %i
; GFX12DAGISEL-NEXT: s_wait_kmcnt 0x0
; GFX12DAGISEL-NEXT: s_or_saveexec_b32 s2, s8
; GFX12DAGISEL-NEXT: s_wait_alu depctr_va_sdst(0)
-; GFX12DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX12DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
; GFX12DAGISEL-NEXT: v_dual_mov_b32 v0, s6 :: v_dual_mov_b32 v1, s7
; GFX12DAGISEL-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12DAGISEL-NEXT: s_xor_b32 exec_lo, exec_lo, s2
@@ -4240,6 +4264,7 @@ define amdgpu_kernel void @divergent_cfg_double(ptr addrspace(1) %out, double %i
; GFX12DAGISEL-NEXT: v_dual_mov_b32 v0, s4 :: v_dual_mov_b32 v1, s5
; GFX12DAGISEL-NEXT: .LBB9_4: ; %endif
; GFX12DAGISEL-NEXT: s_or_b32 exec_lo, exec_lo, s2
+; GFX12DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12DAGISEL-NEXT: v_mov_b32_e32 v2, 0
; GFX12DAGISEL-NEXT: global_store_b64 v2, v[0:1], s[0:1]
; GFX12DAGISEL-NEXT: s_endpgm
diff --git a/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.reduce.max.ll b/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.reduce.max.ll
index 7bc7f987bb2d40..8a9c04b02e02f7 100644
--- a/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.reduce.max.ll
+++ b/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.reduce.max.ll
@@ -403,6 +403,7 @@ define void @divergent_value_i16(ptr addrspace(1) %out, i16 %in) {
; GFX1164DAGISEL-NEXT: s_bitset0_b64 s[0:1], s3
; GFX1164DAGISEL-NEXT: s_max_i32 s2, s2, s4
; GFX1164DAGISEL-NEXT: s_cmp_lg_u64 s[0:1], 0
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164DAGISEL-NEXT: s_cbranch_scc1 .LBB1_1
; GFX1164DAGISEL-NEXT: ; %bb.2:
; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v2, s2
@@ -422,6 +423,7 @@ define void @divergent_value_i16(ptr addrspace(1) %out, i16 %in) {
; GFX1164GISEL-NEXT: s_bitset0_b64 s[0:1], s3
; GFX1164GISEL-NEXT: s_max_i32 s2, s2, s4
; GFX1164GISEL-NEXT: s_cmp_lg_u64 s[0:1], 0
+; GFX1164GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164GISEL-NEXT: s_cbranch_scc1 .LBB1_1
; GFX1164GISEL-NEXT: ; %bb.2:
; GFX1164GISEL-NEXT: v_mov_b32_e32 v2, s2
@@ -441,6 +443,7 @@ define void @divergent_value_i16(ptr addrspace(1) %out, i16 %in) {
; GFX1132DAGISEL-NEXT: s_bitset0_b32 s1, s2
; GFX1132DAGISEL-NEXT: s_max_i32 s0, s0, s3
; GFX1132DAGISEL-NEXT: s_cmp_lg_u32 s1, 0
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1132DAGISEL-NEXT: s_cbranch_scc1 .LBB1_1
; GFX1132DAGISEL-NEXT: ; %bb.2:
; GFX1132DAGISEL-NEXT: v_mov_b32_e32 v2, s0
@@ -460,6 +463,7 @@ define void @divergent_value_i16(ptr addrspace(1) %out, i16 %in) {
; GFX1132GISEL-NEXT: s_bitset0_b32 s1, s2
; GFX1132GISEL-NEXT: s_max_i32 s0, s0, s3
; GFX1132GISEL-NEXT: s_cmp_lg_u32 s1, 0
+; GFX1132GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1132GISEL-NEXT: s_cbranch_scc1 .LBB1_1
; GFX1132GISEL-NEXT: ; %bb.2:
; GFX1132GISEL-NEXT: v_mov_b32_e32 v2, s0
@@ -814,6 +818,7 @@ define amdgpu_kernel void @divergent_value(ptr addrspace(1) %out) #0 {
; GFX1164DAGISEL-NEXT: s_bitset0_b64 s[2:3], s5
; GFX1164DAGISEL-NEXT: s_max_i32 s4, s4, s6
; GFX1164DAGISEL-NEXT: s_cmp_lg_u64 s[2:3], 0
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164DAGISEL-NEXT: s_cbranch_scc1 .LBB3_1
; GFX1164DAGISEL-NEXT: ; %bb.2:
; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v0, s4
@@ -834,6 +839,7 @@ define amdgpu_kernel void @divergent_value(ptr addrspace(1) %out) #0 {
; GFX1164GISEL-NEXT: s_bitset0_b64 s[2:3], s5
; GFX1164GISEL-NEXT: s_max_i32 s4, s4, s6
; GFX1164GISEL-NEXT: s_cmp_lg_u64 s[2:3], 0
+; GFX1164GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164GISEL-NEXT: s_cbranch_scc1 .LBB3_1
; GFX1164GISEL-NEXT: ; %bb.2:
; GFX1164GISEL-NEXT: v_mov_b32_e32 v0, s4
@@ -855,6 +861,7 @@ define amdgpu_kernel void @divergent_value(ptr addrspace(1) %out) #0 {
; GFX1132DAGISEL-NEXT: s_bitset0_b32 s3, s4
; GFX1132DAGISEL-NEXT: s_max_i32 s2, s2, s5
; GFX1132DAGISEL-NEXT: s_cmp_lg_u32 s3, 0
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1132DAGISEL-NEXT: s_cbranch_scc1 .LBB3_1
; GFX1132DAGISEL-NEXT: ; %bb.2:
; GFX1132DAGISEL-NEXT: v_mov_b32_e32 v0, s2
@@ -875,6 +882,7 @@ define amdgpu_kernel void @divergent_value(ptr addrspace(1) %out) #0 {
; GFX1132GISEL-NEXT: s_bitset0_b32 s3, s4
; GFX1132GISEL-NEXT: s_max_i32 s2, s2, s5
; GFX1132GISEL-NEXT: s_cmp_lg_u32 s3, 0
+; GFX1132GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1132GISEL-NEXT: s_cbranch_scc1 .LBB3_1
; GFX1132GISEL-NEXT: ; %bb.2:
; GFX1132GISEL-NEXT: v_dual_mov_b32 v0, s2 :: v_dual_mov_b32 v1, 0
@@ -1298,7 +1306,7 @@ define amdgpu_kernel void @divergent_value_dpp(ptr addrspace(1) %out) #0 {
; GFX1164DAGISEL-NEXT: s_or_saveexec_b64 s[2:3], -1
; GFX1164DAGISEL-NEXT: s_waitcnt lgkmcnt(0)
; GFX1164DAGISEL-NEXT: v_max_i32_e32 v1, v1, v2
-; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX1164DAGISEL-NEXT: v_readlane_b32 s4, v1, 63
; GFX1164DAGISEL-NEXT: s_mov_b64 exec, s[2:3]
; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v0, 0
@@ -1333,7 +1341,7 @@ define amdgpu_kernel void @divergent_value_dpp(ptr addrspace(1) %out) #0 {
; GFX1164GISEL-NEXT: s_mov_b64 exec, s[0:1]
; GFX1164GISEL-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
; GFX1164GISEL-NEXT: s_or_saveexec_b64 s[2:3], -1
-; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX1164GISEL-NEXT: v_readlane_b32 s4, v1, 63
; GFX1164GISEL-NEXT: s_mov_b64 exec, s[2:3]
; GFX1164GISEL-NEXT: v_mov_b32_e32 v0, s4
@@ -1360,7 +1368,7 @@ define amdgpu_kernel void @divergent_value_dpp(ptr addrspace(1) %out) #0 {
; GFX1132DAGISEL-NEXT: s_or_saveexec_b32 s2, -1
; GFX1132DAGISEL-NEXT: s_waitcnt lgkmcnt(0)
; GFX1132DAGISEL-NEXT: v_max_i32_e32 v1, v1, v2
-; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX1132DAGISEL-NEXT: v_readlane_b32 s3, v1, 31
; GFX1132DAGISEL-NEXT: s_mov_b32 exec_lo, s2
; GFX1132DAGISEL-NEXT: v_dual_mov_b32 v0, 0 :: v_dual_mov_b32 v3, s3
@@ -1385,7 +1393,7 @@ define amdgpu_kernel void @divergent_value_dpp(ptr addrspace(1) %out) #0 {
; GFX1132GISEL-NEXT: s_mov_b32 exec_lo, s0
; GFX1132GISEL-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
; GFX1132GISEL-NEXT: s_or_saveexec_b32 s2, -1
-; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX1132GISEL-NEXT: v_readlane_b32 s3, v1, 31
; GFX1132GISEL-NEXT: s_mov_b32 exec_lo, s2
; GFX1132GISEL-NEXT: v_dual_mov_b32 v0, s3 :: v_dual_mov_b32 v3, 0
@@ -2125,52 +2133,53 @@ define void @divergent_value_dpp_i64(ptr addrspace(1) %out, i64 %in) #0 {
; GFX1164DAGISEL-NEXT: scratch_store_b32 off, v4, s32 offset:20
; GFX1164DAGISEL-NEXT: scratch_store_b32 off, v5, s32 offset:24
; GFX1164DAGISEL-NEXT: s_mov_b64 exec, s[0:1]
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1164DAGISEL-NEXT: s_or_saveexec_b64 s[0:1], -1
-; GFX1164DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(VALU_DEP_3)
; GFX1164DAGISEL-NEXT: v_cndmask_b32_e64 v4, 0, v2, s[0:1]
; GFX1164DAGISEL-NEXT: v_cndmask_b32_e64 v5, 0x80000000, v3, s[0:1]
; GFX1164DAGISEL-NEXT: v_mbcnt_lo_u32_b32 v8, -1, 0
-; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v6, v4
; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v6, v4
; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v7, v5
-; GFX1164DAGISEL-NEXT: v_mbcnt_hi_u32_b32 v8, -1, v8
; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX1164DAGISEL-NEXT: v_mbcnt_hi_u32_b32 v8, -1, v8
; GFX1164DAGISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:1 row_mask:0xf bank_mask:0xf
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_3)
; GFX1164DAGISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:1 row_mask:0xf bank_mask:0xf
-; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1164DAGISEL-NEXT: v_add_nc_u32_e32 v8, 32, v8
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1164DAGISEL-NEXT: v_cmp_gt_i64_e32 vcc, v[4:5], v[6:7]
-; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_2) | instid1(VALU_DEP_2)
; GFX1164DAGISEL-NEXT: v_mul_lo_u32 v8, 4, v8
; GFX1164DAGISEL-NEXT: v_cndmask_b32_e32 v4, v6, v4, vcc
; GFX1164DAGISEL-NEXT: v_cndmask_b32_e32 v5, v7, v5, vcc
-; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v6, v4
; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v6, v4
; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v7, v5
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1164DAGISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:2 row_mask:0xf bank_mask:0xf
-; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164DAGISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:2 row_mask:0xf bank_mask:0xf
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_3) | instid1(VALU_DEP_2)
; GFX1164DAGISEL-NEXT: v_cmp_gt_i64_e32 vcc, v[4:5], v[6:7]
; GFX1164DAGISEL-NEXT: s_waitcnt_depctr depctr_va_vcc(0)
; GFX1164DAGISEL-NEXT: v_cndmask_b32_e32 v4, v6, v4, vcc
; GFX1164DAGISEL-NEXT: v_cndmask_b32_e32 v5, v7, v5, vcc
-; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v6, v4
-; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v7, v5
; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v7, v5
; GFX1164DAGISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:4 row_mask:0xf bank_mask:0xf
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164DAGISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:4 row_mask:0xf bank_mask:0xf
-; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_3) | instid1(VALU_DEP_2)
; GFX1164DAGISEL-NEXT: v_cmp_gt_i64_e32 vcc, v[4:5], v[6:7]
; GFX1164DAGISEL-NEXT: s_waitcnt_depctr depctr_va_vcc(0)
; GFX1164DAGISEL-NEXT: v_cndmask_b32_e32 v4, v6, v4, vcc
; GFX1164DAGISEL-NEXT: v_cndmask_b32_e32 v5, v7, v5, vcc
-; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v6, v4
; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v6, v4
; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v7, v5
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1164DAGISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:8 row_mask:0xf bank_mask:0xf
-; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164DAGISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:8 row_mask:0xf bank_mask:0xf
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1164DAGISEL-NEXT: v_cmp_gt_i64_e32 vcc, v[4:5], v[6:7]
; GFX1164DAGISEL-NEXT: s_waitcnt_depctr depctr_va_vcc(0)
; GFX1164DAGISEL-NEXT: v_cndmask_b32_e32 v4, v6, v4, vcc
@@ -2193,6 +2202,7 @@ define void @divergent_value_dpp_i64(ptr addrspace(1) %out, i64 %in) #0 {
; GFX1164DAGISEL-NEXT: v_readlane_b32 s2, v4, 63
; GFX1164DAGISEL-NEXT: v_readlane_b32 s3, v5, 63
; GFX1164DAGISEL-NEXT: s_mov_b64 exec, s[0:1]
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v2, s2
; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v3, s3
; GFX1164DAGISEL-NEXT: global_store_b64 v[0:1], v[2:3], off
@@ -2219,52 +2229,53 @@ define void @divergent_value_dpp_i64(ptr addrspace(1) %out, i64 %in) #0 {
; GFX1164GISEL-NEXT: scratch_store_b32 off, v4, s32 offset:20
; GFX1164GISEL-NEXT: scratch_store_b32 off, v5, s32 offset:24
; GFX1164GISEL-NEXT: s_mov_b64 exec, s[0:1]
+; GFX1164GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1164GISEL-NEXT: s_or_saveexec_b64 s[0:1], -1
-; GFX1164GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(VALU_DEP_3)
; GFX1164GISEL-NEXT: v_cndmask_b32_e64 v4, 0, v2, s[0:1]
; GFX1164GISEL-NEXT: v_cndmask_b32_e64 v5, 0x80000000, v3, s[0:1]
; GFX1164GISEL-NEXT: v_mbcnt_lo_u32_b32 v8, -1, 0
-; GFX1164GISEL-NEXT: v_mov_b32_e32 v6, v4
; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX1164GISEL-NEXT: v_mov_b32_e32 v6, v4
; GFX1164GISEL-NEXT: v_mov_b32_e32 v7, v5
-; GFX1164GISEL-NEXT: v_mbcnt_hi_u32_b32 v8, -1, v8
; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX1164GISEL-NEXT: v_mbcnt_hi_u32_b32 v8, -1, v8
; GFX1164GISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:1 row_mask:0xf bank_mask:0xf
+; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_3)
; GFX1164GISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:1 row_mask:0xf bank_mask:0xf
-; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1164GISEL-NEXT: v_add_nc_u32_e32 v8, 32, v8
+; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1164GISEL-NEXT: v_cmp_gt_i64_e32 vcc, v[4:5], v[6:7]
-; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_2) | instid1(VALU_DEP_2)
; GFX1164GISEL-NEXT: v_mul_lo_u32 v8, 4, v8
; GFX1164GISEL-NEXT: v_cndmask_b32_e32 v4, v6, v4, vcc
; GFX1164GISEL-NEXT: v_cndmask_b32_e32 v5, v7, v5, vcc
-; GFX1164GISEL-NEXT: v_mov_b32_e32 v6, v4
; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX1164GISEL-NEXT: v_mov_b32_e32 v6, v4
; GFX1164GISEL-NEXT: v_mov_b32_e32 v7, v5
+; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1164GISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:2 row_mask:0xf bank_mask:0xf
-; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164GISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:2 row_mask:0xf bank_mask:0xf
+; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_3) | instid1(VALU_DEP_2)
; GFX1164GISEL-NEXT: v_cmp_gt_i64_e32 vcc, v[4:5], v[6:7]
; GFX1164GISEL-NEXT: s_waitcnt_depctr depctr_va_vcc(0)
; GFX1164GISEL-NEXT: v_cndmask_b32_e32 v4, v6, v4, vcc
; GFX1164GISEL-NEXT: v_cndmask_b32_e32 v5, v7, v5, vcc
-; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1164GISEL-NEXT: v_mov_b32_e32 v6, v4
-; GFX1164GISEL-NEXT: v_mov_b32_e32 v7, v5
; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX1164GISEL-NEXT: v_mov_b32_e32 v7, v5
; GFX1164GISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:4 row_mask:0xf bank_mask:0xf
+; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164GISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:4 row_mask:0xf bank_mask:0xf
-; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_3) | instid1(VALU_DEP_2)
; GFX1164GISEL-NEXT: v_cmp_gt_i64_e32 vcc, v[4:5], v[6:7]
; GFX1164GISEL-NEXT: s_waitcnt_depctr depctr_va_vcc(0)
; GFX1164GISEL-NEXT: v_cndmask_b32_e32 v4, v6, v4, vcc
; GFX1164GISEL-NEXT: v_cndmask_b32_e32 v5, v7, v5, vcc
-; GFX1164GISEL-NEXT: v_mov_b32_e32 v6, v4
; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX1164GISEL-NEXT: v_mov_b32_e32 v6, v4
; GFX1164GISEL-NEXT: v_mov_b32_e32 v7, v5
+; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1164GISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:8 row_mask:0xf bank_mask:0xf
-; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164GISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:8 row_mask:0xf bank_mask:0xf
+; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1164GISEL-NEXT: v_cmp_gt_i64_e32 vcc, v[4:5], v[6:7]
; GFX1164GISEL-NEXT: s_waitcnt_depctr depctr_va_vcc(0)
; GFX1164GISEL-NEXT: v_cndmask_b32_e32 v4, v6, v4, vcc
@@ -2287,6 +2298,7 @@ define void @divergent_value_dpp_i64(ptr addrspace(1) %out, i64 %in) #0 {
; GFX1164GISEL-NEXT: v_readlane_b32 s2, v4, 63
; GFX1164GISEL-NEXT: v_readlane_b32 s3, v5, 63
; GFX1164GISEL-NEXT: s_mov_b64 exec, s[0:1]
+; GFX1164GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164GISEL-NEXT: v_mov_b32_e32 v2, s2
; GFX1164GISEL-NEXT: v_mov_b32_e32 v3, s3
; GFX1164GISEL-NEXT: global_store_b64 v[0:1], v[2:3], off
@@ -2312,36 +2324,36 @@ define void @divergent_value_dpp_i64(ptr addrspace(1) %out, i64 %in) #0 {
; GFX1132DAGISEL-NEXT: scratch_store_b32 off, v4, s32 offset:16
; GFX1132DAGISEL-NEXT: scratch_store_b32 off, v5, s32 offset:20
; GFX1132DAGISEL-NEXT: s_mov_b32 exec_lo, s0
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1132DAGISEL-NEXT: s_or_saveexec_b32 s2, -1
-; GFX1132DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX1132DAGISEL-NEXT: v_cndmask_b32_e64 v4, 0, v2, s2
; GFX1132DAGISEL-NEXT: v_cndmask_b32_e64 v5, 0x80000000, v3, s2
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132DAGISEL-NEXT: v_dual_mov_b32 v6, v4 :: v_dual_mov_b32 v7, v5
-; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1132DAGISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:1 row_mask:0xf bank_mask:0xf
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132DAGISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:1 row_mask:0xf bank_mask:0xf
-; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX1132DAGISEL-NEXT: v_cmp_gt_i64_e32 vcc_lo, v[4:5], v[6:7]
; GFX1132DAGISEL-NEXT: v_dual_cndmask_b32 v4, v6, v4 :: v_dual_cndmask_b32 v5, v7, v5
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132DAGISEL-NEXT: v_dual_mov_b32 v6, v4 :: v_dual_mov_b32 v7, v5
-; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1132DAGISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:2 row_mask:0xf bank_mask:0xf
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132DAGISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:2 row_mask:0xf bank_mask:0xf
-; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX1132DAGISEL-NEXT: v_cmp_gt_i64_e32 vcc_lo, v[4:5], v[6:7]
; GFX1132DAGISEL-NEXT: v_dual_cndmask_b32 v4, v6, v4 :: v_dual_cndmask_b32 v5, v7, v5
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132DAGISEL-NEXT: v_dual_mov_b32 v6, v4 :: v_dual_mov_b32 v7, v5
-; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1132DAGISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:4 row_mask:0xf bank_mask:0xf
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132DAGISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:4 row_mask:0xf bank_mask:0xf
-; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX1132DAGISEL-NEXT: v_cmp_gt_i64_e32 vcc_lo, v[4:5], v[6:7]
; GFX1132DAGISEL-NEXT: v_dual_cndmask_b32 v4, v6, v4 :: v_dual_cndmask_b32 v5, v7, v5
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132DAGISEL-NEXT: v_dual_mov_b32 v6, v4 :: v_dual_mov_b32 v7, v5
-; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1132DAGISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:8 row_mask:0xf bank_mask:0xf
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132DAGISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:8 row_mask:0xf bank_mask:0xf
-; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1132DAGISEL-NEXT: v_cmp_gt_i64_e32 vcc_lo, v[4:5], v[6:7]
; GFX1132DAGISEL-NEXT: v_dual_cndmask_b32 v4, v6, v4 :: v_dual_cndmask_b32 v5, v7, v5
; GFX1132DAGISEL-NEXT: ds_swizzle_b32 v6, v4 offset:swizzle(BROADCAST,32,15)
@@ -2353,6 +2365,7 @@ define void @divergent_value_dpp_i64(ptr addrspace(1) %out, i64 %in) #0 {
; GFX1132DAGISEL-NEXT: v_readlane_b32 s0, v4, 31
; GFX1132DAGISEL-NEXT: v_readlane_b32 s1, v5, 31
; GFX1132DAGISEL-NEXT: s_mov_b32 exec_lo, s2
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1132DAGISEL-NEXT: v_dual_mov_b32 v3, s1 :: v_dual_mov_b32 v2, s0
; GFX1132DAGISEL-NEXT: global_store_b64 v[0:1], v[2:3], off
; GFX1132DAGISEL-NEXT: s_xor_saveexec_b32 s0, -1
@@ -2375,36 +2388,36 @@ define void @divergent_value_dpp_i64(ptr addrspace(1) %out, i64 %in) #0 {
; GFX1132GISEL-NEXT: scratch_store_b32 off, v4, s32 offset:16
; GFX1132GISEL-NEXT: scratch_store_b32 off, v5, s32 offset:20
; GFX1132GISEL-NEXT: s_mov_b32 exec_lo, s0
+; GFX1132GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1132GISEL-NEXT: s_or_saveexec_b32 s2, -1
-; GFX1132GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX1132GISEL-NEXT: v_cndmask_b32_e64 v4, 0, v2, s2
; GFX1132GISEL-NEXT: v_cndmask_b32_e64 v5, 0x80000000, v3, s2
+; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132GISEL-NEXT: v_dual_mov_b32 v6, v4 :: v_dual_mov_b32 v7, v5
-; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1132GISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:1 row_mask:0xf bank_mask:0xf
+; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132GISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:1 row_mask:0xf bank_mask:0xf
-; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX1132GISEL-NEXT: v_cmp_gt_i64_e32 vcc_lo, v[4:5], v[6:7]
; GFX1132GISEL-NEXT: v_dual_cndmask_b32 v4, v6, v4 :: v_dual_cndmask_b32 v5, v7, v5
+; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132GISEL-NEXT: v_dual_mov_b32 v6, v4 :: v_dual_mov_b32 v7, v5
-; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1132GISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:2 row_mask:0xf bank_mask:0xf
+; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132GISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:2 row_mask:0xf bank_mask:0xf
-; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX1132GISEL-NEXT: v_cmp_gt_i64_e32 vcc_lo, v[4:5], v[6:7]
; GFX1132GISEL-NEXT: v_dual_cndmask_b32 v4, v6, v4 :: v_dual_cndmask_b32 v5, v7, v5
+; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132GISEL-NEXT: v_dual_mov_b32 v6, v4 :: v_dual_mov_b32 v7, v5
-; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1132GISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:4 row_mask:0xf bank_mask:0xf
+; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132GISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:4 row_mask:0xf bank_mask:0xf
-; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX1132GISEL-NEXT: v_cmp_gt_i64_e32 vcc_lo, v[4:5], v[6:7]
; GFX1132GISEL-NEXT: v_dual_cndmask_b32 v4, v6, v4 :: v_dual_cndmask_b32 v5, v7, v5
+; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132GISEL-NEXT: v_dual_mov_b32 v6, v4 :: v_dual_mov_b32 v7, v5
-; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1132GISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:8 row_mask:0xf bank_mask:0xf
+; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132GISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:8 row_mask:0xf bank_mask:0xf
-; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1132GISEL-NEXT: v_cmp_gt_i64_e32 vcc_lo, v[4:5], v[6:7]
; GFX1132GISEL-NEXT: v_dual_cndmask_b32 v4, v6, v4 :: v_dual_cndmask_b32 v5, v7, v5
; GFX1132GISEL-NEXT: ds_swizzle_b32 v6, v4 offset:swizzle(BROADCAST,32,15)
@@ -2416,6 +2429,7 @@ define void @divergent_value_dpp_i64(ptr addrspace(1) %out, i64 %in) #0 {
; GFX1132GISEL-NEXT: v_readlane_b32 s0, v4, 31
; GFX1132GISEL-NEXT: v_readlane_b32 s1, v5, 31
; GFX1132GISEL-NEXT: s_mov_b32 exec_lo, s2
+; GFX1132GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1132GISEL-NEXT: v_dual_mov_b32 v3, s1 :: v_dual_mov_b32 v2, s0
; GFX1132GISEL-NEXT: global_store_b64 v[0:1], v[2:3], off
; GFX1132GISEL-NEXT: s_xor_saveexec_b32 s0, -1
@@ -2709,7 +2723,7 @@ define amdgpu_kernel void @default_stratergy(ptr addrspace(1) %out) #0 {
; GFX1164DAGISEL-NEXT: s_or_saveexec_b64 s[2:3], -1
; GFX1164DAGISEL-NEXT: s_waitcnt lgkmcnt(0)
; GFX1164DAGISEL-NEXT: v_max_i32_e32 v1, v1, v2
-; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX1164DAGISEL-NEXT: v_readlane_b32 s4, v1, 63
; GFX1164DAGISEL-NEXT: s_mov_b64 exec, s[2:3]
; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v0, 0
@@ -2744,7 +2758,7 @@ define amdgpu_kernel void @default_stratergy(ptr addrspace(1) %out) #0 {
; GFX1164GISEL-NEXT: s_mov_b64 exec, s[0:1]
; GFX1164GISEL-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
; GFX1164GISEL-NEXT: s_or_saveexec_b64 s[2:3], -1
-; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX1164GISEL-NEXT: v_readlane_b32 s4, v1, 63
; GFX1164GISEL-NEXT: s_mov_b64 exec, s[2:3]
; GFX1164GISEL-NEXT: v_mov_b32_e32 v0, s4
@@ -2771,7 +2785,7 @@ define amdgpu_kernel void @default_stratergy(ptr addrspace(1) %out) #0 {
; GFX1132DAGISEL-NEXT: s_or_saveexec_b32 s2, -1
; GFX1132DAGISEL-NEXT: s_waitcnt lgkmcnt(0)
; GFX1132DAGISEL-NEXT: v_max_i32_e32 v1, v1, v2
-; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX1132DAGISEL-NEXT: v_readlane_b32 s3, v1, 31
; GFX1132DAGISEL-NEXT: s_mov_b32 exec_lo, s2
; GFX1132DAGISEL-NEXT: v_dual_mov_b32 v0, 0 :: v_dual_mov_b32 v3, s3
@@ -2796,7 +2810,7 @@ define amdgpu_kernel void @default_stratergy(ptr addrspace(1) %out) #0 {
; GFX1132GISEL-NEXT: s_mov_b32 exec_lo, s0
; GFX1132GISEL-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
; GFX1132GISEL-NEXT: s_or_saveexec_b32 s2, -1
-; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX1132GISEL-NEXT: v_readlane_b32 s3, v1, 31
; GFX1132GISEL-NEXT: s_mov_b32 exec_lo, s2
; GFX1132GISEL-NEXT: v_dual_mov_b32 v0, s3 :: v_dual_mov_b32 v3, 0
@@ -3178,7 +3192,7 @@ define amdgpu_kernel void @divergent_cfg(ptr addrspace(1) %out, i32 %in) #0 {
; GFX1164DAGISEL-NEXT: v_and_b32_e32 v0, 0x3ff, v0
; GFX1164DAGISEL-NEXT: s_mov_b64 s[0:1], exec
; GFX1164DAGISEL-NEXT: ; implicit-def: $sgpr2
-; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164DAGISEL-NEXT: v_cmpx_lt_u32_e32 15, v0
; GFX1164DAGISEL-NEXT: s_xor_b64 s[0:1], exec, s[0:1]
; GFX1164DAGISEL-NEXT: ; %bb.1: ; %else
@@ -3189,13 +3203,14 @@ define amdgpu_kernel void @divergent_cfg(ptr addrspace(1) %out, i32 %in) #0 {
; GFX1164DAGISEL-NEXT: s_waitcnt lgkmcnt(0)
; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v1, s2
; GFX1164DAGISEL-NEXT: s_xor_b64 exec, exec, s[0:1]
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164DAGISEL-NEXT: s_cbranch_execz .LBB8_6
; GFX1164DAGISEL-NEXT: ; %bb.3: ; %if
; GFX1164DAGISEL-NEXT: s_mov_b64 s[2:3], exec
; GFX1164DAGISEL-NEXT: s_brev_b32 s6, 1
; GFX1164DAGISEL-NEXT: .LBB8_4: ; =>This Inner Loop Header: Depth=1
; GFX1164DAGISEL-NEXT: s_ctz_i32_b64 s7, s[2:3]
-; GFX1164DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_3) | instid1(SALU_CYCLE_1)
; GFX1164DAGISEL-NEXT: v_readlane_b32 s8, v0, s7
; GFX1164DAGISEL-NEXT: s_bitset0_b64 s[2:3], s7
; GFX1164DAGISEL-NEXT: s_max_i32 s6, s6, s8
@@ -3216,7 +3231,7 @@ define amdgpu_kernel void @divergent_cfg(ptr addrspace(1) %out, i32 %in) #0 {
; GFX1164GISEL-NEXT: v_and_b32_e32 v0, 0x3ff, v0
; GFX1164GISEL-NEXT: s_mov_b64 s[0:1], exec
; GFX1164GISEL-NEXT: ; implicit-def: $sgpr2
-; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164GISEL-NEXT: v_cmpx_le_u32_e32 16, v0
; GFX1164GISEL-NEXT: s_xor_b64 s[0:1], exec, s[0:1]
; GFX1164GISEL-NEXT: s_cbranch_execz .LBB8_2
@@ -3229,13 +3244,14 @@ define amdgpu_kernel void @divergent_cfg(ptr addrspace(1) %out, i32 %in) #0 {
; GFX1164GISEL-NEXT: s_or_saveexec_b64 s[0:1], s[0:1]
; GFX1164GISEL-NEXT: v_mov_b32_e32 v1, s2
; GFX1164GISEL-NEXT: s_xor_b64 exec, exec, s[0:1]
+; GFX1164GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164GISEL-NEXT: s_cbranch_execz .LBB8_6
; GFX1164GISEL-NEXT: ; %bb.3: ; %if
; GFX1164GISEL-NEXT: s_mov_b64 s[2:3], exec
; GFX1164GISEL-NEXT: s_brev_b32 s6, 1
; GFX1164GISEL-NEXT: .LBB8_4: ; =>This Inner Loop Header: Depth=1
; GFX1164GISEL-NEXT: s_ctz_i32_b64 s7, s[2:3]
-; GFX1164GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1164GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_3) | instid1(SALU_CYCLE_1)
; GFX1164GISEL-NEXT: v_readlane_b32 s8, v0, s7
; GFX1164GISEL-NEXT: s_bitset0_b64 s[2:3], s7
; GFX1164GISEL-NEXT: s_max_i32 s6, s6, s8
@@ -3256,13 +3272,14 @@ define amdgpu_kernel void @divergent_cfg(ptr addrspace(1) %out, i32 %in) #0 {
; GFX1132DAGISEL-NEXT: v_and_b32_e32 v0, 0x3ff, v0
; GFX1132DAGISEL-NEXT: s_mov_b32 s0, exec_lo
; GFX1132DAGISEL-NEXT: ; implicit-def: $sgpr1
-; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132DAGISEL-NEXT: v_cmpx_lt_u32_e32 15, v0
; GFX1132DAGISEL-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX1132DAGISEL-NEXT: ; %bb.1: ; %else
; GFX1132DAGISEL-NEXT: s_load_b32 s1, s[4:5], 0x2c
; GFX1132DAGISEL-NEXT: ; implicit-def: $vgpr0
; GFX1132DAGISEL-NEXT: ; %bb.2: ; %Flow
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_3) | instid1(SALU_CYCLE_1)
; GFX1132DAGISEL-NEXT: s_or_saveexec_b32 s0, s0
; GFX1132DAGISEL-NEXT: s_waitcnt lgkmcnt(0)
; GFX1132DAGISEL-NEXT: v_mov_b32_e32 v1, s1
@@ -3273,7 +3290,7 @@ define amdgpu_kernel void @divergent_cfg(ptr addrspace(1) %out, i32 %in) #0 {
; GFX1132DAGISEL-NEXT: s_brev_b32 s1, 1
; GFX1132DAGISEL-NEXT: .LBB8_4: ; =>This Inner Loop Header: Depth=1
; GFX1132DAGISEL-NEXT: s_ctz_i32_b32 s3, s2
-; GFX1132DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_3) | instid1(SALU_CYCLE_1)
; GFX1132DAGISEL-NEXT: v_readlane_b32 s6, v0, s3
; GFX1132DAGISEL-NEXT: s_bitset0_b32 s2, s3
; GFX1132DAGISEL-NEXT: s_max_i32 s1, s1, s6
@@ -3294,9 +3311,10 @@ define amdgpu_kernel void @divergent_cfg(ptr addrspace(1) %out, i32 %in) #0 {
; GFX1132GISEL-NEXT: v_and_b32_e32 v0, 0x3ff, v0
; GFX1132GISEL-NEXT: s_mov_b32 s0, exec_lo
; GFX1132GISEL-NEXT: ; implicit-def: $sgpr1
-; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132GISEL-NEXT: v_cmpx_le_u32_e32 16, v0
; GFX1132GISEL-NEXT: s_xor_b32 s0, exec_lo, s0
+; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1132GISEL-NEXT: s_cbranch_execz .LBB8_2
; GFX1132GISEL-NEXT: ; %bb.1: ; %else
; GFX1132GISEL-NEXT: s_load_b32 s1, s[4:5], 0x2c
@@ -3307,13 +3325,14 @@ define amdgpu_kernel void @divergent_cfg(ptr addrspace(1) %out, i32 %in) #0 {
; GFX1132GISEL-NEXT: s_or_saveexec_b32 s0, s0
; GFX1132GISEL-NEXT: v_mov_b32_e32 v1, s1
; GFX1132GISEL-NEXT: s_xor_b32 exec_lo, exec_lo, s0
+; GFX1132GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1132GISEL-NEXT: s_cbranch_execz .LBB8_6
; GFX1132GISEL-NEXT: ; %bb.3: ; %if
; GFX1132GISEL-NEXT: s_mov_b32 s2, exec_lo
; GFX1132GISEL-NEXT: s_brev_b32 s1, 1
; GFX1132GISEL-NEXT: .LBB8_4: ; =>This Inner Loop Header: Depth=1
; GFX1132GISEL-NEXT: s_ctz_i32_b32 s3, s2
-; GFX1132GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1132GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_3) | instid1(SALU_CYCLE_1)
; GFX1132GISEL-NEXT: v_readlane_b32 s6, v0, s3
; GFX1132GISEL-NEXT: s_bitset0_b32 s2, s3
; GFX1132GISEL-NEXT: s_max_i32 s1, s1, s6
@@ -3748,6 +3767,7 @@ define void @divergent_value_i64(ptr addrspace(1) %out, i64 %id.x) #0 {
; GFX1164DAGISEL-NEXT: v_readlane_b32 s4, v2, s8
; GFX1164DAGISEL-NEXT: v_readlane_b32 s5, v3, s8
; GFX1164DAGISEL-NEXT: v_cmp_gt_i64_e32 vcc, s[4:5], v[4:5]
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_3) | instid1(SALU_CYCLE_1)
; GFX1164DAGISEL-NEXT: s_and_b64 s[6:7], vcc, s[2:3]
; GFX1164DAGISEL-NEXT: s_bitset0_b64 s[2:3], s8
; GFX1164DAGISEL-NEXT: s_cselect_b64 s[0:1], s[4:5], s[0:1]
@@ -3773,6 +3793,7 @@ define void @divergent_value_i64(ptr addrspace(1) %out, i64 %id.x) #0 {
; GFX1164GISEL-NEXT: v_readlane_b32 s4, v2, s8
; GFX1164GISEL-NEXT: v_readlane_b32 s5, v3, s8
; GFX1164GISEL-NEXT: v_cmp_gt_i64_e32 vcc, s[4:5], v[4:5]
+; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_3) | instid1(SALU_CYCLE_1)
; GFX1164GISEL-NEXT: s_and_b64 s[6:7], vcc, s[2:3]
; GFX1164GISEL-NEXT: s_bitset0_b64 s[2:3], s8
; GFX1164GISEL-NEXT: s_cselect_b64 s[0:1], s[4:5], s[0:1]
@@ -3801,6 +3822,7 @@ define void @divergent_value_i64(ptr addrspace(1) %out, i64 %id.x) #0 {
; GFX1132DAGISEL-NEXT: s_bitset0_b32 s2, s3
; GFX1132DAGISEL-NEXT: s_cselect_b64 s[0:1], s[4:5], s[0:1]
; GFX1132DAGISEL-NEXT: s_cmp_lg_u32 s2, 0
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1132DAGISEL-NEXT: s_cbranch_scc1 .LBB10_1
; GFX1132DAGISEL-NEXT: ; %bb.2:
; GFX1132DAGISEL-NEXT: v_dual_mov_b32 v3, s1 :: v_dual_mov_b32 v2, s0
@@ -3824,6 +3846,7 @@ define void @divergent_value_i64(ptr addrspace(1) %out, i64 %id.x) #0 {
; GFX1132GISEL-NEXT: s_bitset0_b32 s2, s3
; GFX1132GISEL-NEXT: s_cselect_b64 s[0:1], s[4:5], s[0:1]
; GFX1132GISEL-NEXT: s_cmp_lg_u32 s2, 0
+; GFX1132GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1132GISEL-NEXT: s_cbranch_scc1 .LBB10_1
; GFX1132GISEL-NEXT: ; %bb.2:
; GFX1132GISEL-NEXT: v_dual_mov_b32 v3, s1 :: v_dual_mov_b32 v2, s0
@@ -4095,19 +4118,22 @@ define amdgpu_kernel void @divergent_cfg_i64(ptr addrspace(1) %out, i64 %in, i64
; GFX1164DAGISEL-NEXT: s_load_b64 s[4:5], s[4:5], 0x34
; GFX1164DAGISEL-NEXT: v_and_b32_e32 v0, 0x3ff, v0
; GFX1164DAGISEL-NEXT: s_mov_b64 s[6:7], exec
-; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164DAGISEL-NEXT: v_cmpx_lt_u32_e32 15, v0
; GFX1164DAGISEL-NEXT: s_xor_b64 s[6:7], exec, s[6:7]
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164DAGISEL-NEXT: s_or_saveexec_b64 s[6:7], s[6:7]
; GFX1164DAGISEL-NEXT: s_waitcnt lgkmcnt(0)
; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v0, s2
; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v1, s3
; GFX1164DAGISEL-NEXT: s_xor_b64 exec, exec, s[6:7]
; GFX1164DAGISEL-NEXT: ; %bb.1: ; %if
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v0, s4
; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v1, s5
; GFX1164DAGISEL-NEXT: ; %bb.2: ; %endif
; GFX1164DAGISEL-NEXT: s_or_b64 exec, exec, s[6:7]
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v2, 0
; GFX1164DAGISEL-NEXT: global_store_b64 v2, v[0:1], s[0:1]
; GFX1164DAGISEL-NEXT: s_endpgm
@@ -4118,7 +4144,7 @@ define amdgpu_kernel void @divergent_cfg_i64(ptr addrspace(1) %out, i64 %in, i64
; GFX1164GISEL-NEXT: v_and_b32_e32 v0, 0x3ff, v0
; GFX1164GISEL-NEXT: s_mov_b64 s[8:9], exec
; GFX1164GISEL-NEXT: ; implicit-def: $sgpr6_sgpr7
-; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164GISEL-NEXT: v_cmpx_le_u32_e32 16, v0
; GFX1164GISEL-NEXT: s_xor_b64 s[8:9], exec, s[8:9]
; GFX1164GISEL-NEXT: s_cbranch_execz .LBB11_2
@@ -4131,6 +4157,7 @@ define amdgpu_kernel void @divergent_cfg_i64(ptr addrspace(1) %out, i64 %in, i64
; GFX1164GISEL-NEXT: v_mov_b32_e32 v0, s6
; GFX1164GISEL-NEXT: v_mov_b32_e32 v1, s7
; GFX1164GISEL-NEXT: s_xor_b64 exec, exec, s[2:3]
+; GFX1164GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164GISEL-NEXT: s_cbranch_execz .LBB11_4
; GFX1164GISEL-NEXT: ; %bb.3: ; %if
; GFX1164GISEL-NEXT: s_load_b64 s[4:5], s[4:5], 0x34
@@ -4141,6 +4168,7 @@ define amdgpu_kernel void @divergent_cfg_i64(ptr addrspace(1) %out, i64 %in, i64
; GFX1164GISEL-NEXT: v_mov_b32_e32 v1, s5
; GFX1164GISEL-NEXT: .LBB11_4: ; %endif
; GFX1164GISEL-NEXT: s_or_b64 exec, exec, s[2:3]
+; GFX1164GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164GISEL-NEXT: v_mov_b32_e32 v2, 0
; GFX1164GISEL-NEXT: global_store_b64 v2, v[0:1], s[0:1]
; GFX1164GISEL-NEXT: s_endpgm
@@ -4152,17 +4180,20 @@ define amdgpu_kernel void @divergent_cfg_i64(ptr addrspace(1) %out, i64 %in, i64
; GFX1132DAGISEL-NEXT: s_load_b64 s[4:5], s[4:5], 0x34
; GFX1132DAGISEL-NEXT: v_and_b32_e32 v0, 0x3ff, v0
; GFX1132DAGISEL-NEXT: s_mov_b32 s6, exec_lo
-; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132DAGISEL-NEXT: v_cmpx_lt_u32_e32 15, v0
; GFX1132DAGISEL-NEXT: s_xor_b32 s6, exec_lo, s6
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX1132DAGISEL-NEXT: s_or_saveexec_b32 s6, s6
; GFX1132DAGISEL-NEXT: s_waitcnt lgkmcnt(0)
; GFX1132DAGISEL-NEXT: v_dual_mov_b32 v0, s2 :: v_dual_mov_b32 v1, s3
; GFX1132DAGISEL-NEXT: s_xor_b32 exec_lo, exec_lo, s6
; GFX1132DAGISEL-NEXT: ; %bb.1: ; %if
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1132DAGISEL-NEXT: v_dual_mov_b32 v0, s4 :: v_dual_mov_b32 v1, s5
; GFX1132DAGISEL-NEXT: ; %bb.2: ; %endif
; GFX1132DAGISEL-NEXT: s_or_b32 exec_lo, exec_lo, s6
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1132DAGISEL-NEXT: v_mov_b32_e32 v2, 0
; GFX1132DAGISEL-NEXT: global_store_b64 v2, v[0:1], s[0:1]
; GFX1132DAGISEL-NEXT: s_endpgm
@@ -4173,9 +4204,10 @@ define amdgpu_kernel void @divergent_cfg_i64(ptr addrspace(1) %out, i64 %in, i64
; GFX1132GISEL-NEXT: v_and_b32_e32 v0, 0x3ff, v0
; GFX1132GISEL-NEXT: s_mov_b32 s8, exec_lo
; GFX1132GISEL-NEXT: ; implicit-def: $sgpr6_sgpr7
-; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132GISEL-NEXT: v_cmpx_le_u32_e32 16, v0
; GFX1132GISEL-NEXT: s_xor_b32 s8, exec_lo, s8
+; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1132GISEL-NEXT: s_cbranch_execz .LBB11_2
; GFX1132GISEL-NEXT: ; %bb.1: ; %else
; GFX1132GISEL-NEXT: s_waitcnt lgkmcnt(0)
@@ -4185,6 +4217,7 @@ define amdgpu_kernel void @divergent_cfg_i64(ptr addrspace(1) %out, i64 %in, i64
; GFX1132GISEL-NEXT: s_or_saveexec_b32 s2, s8
; GFX1132GISEL-NEXT: v_dual_mov_b32 v0, s6 :: v_dual_mov_b32 v1, s7
; GFX1132GISEL-NEXT: s_xor_b32 exec_lo, exec_lo, s2
+; GFX1132GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1132GISEL-NEXT: s_cbranch_execz .LBB11_4
; GFX1132GISEL-NEXT: ; %bb.3: ; %if
; GFX1132GISEL-NEXT: s_load_b64 s[4:5], s[4:5], 0x34
@@ -4194,6 +4227,7 @@ define amdgpu_kernel void @divergent_cfg_i64(ptr addrspace(1) %out, i64 %in, i64
; GFX1132GISEL-NEXT: v_dual_mov_b32 v0, s4 :: v_dual_mov_b32 v1, s5
; GFX1132GISEL-NEXT: .LBB11_4: ; %endif
; GFX1132GISEL-NEXT: s_or_b32 exec_lo, exec_lo, s2
+; GFX1132GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1132GISEL-NEXT: v_mov_b32_e32 v2, 0
; GFX1132GISEL-NEXT: global_store_b64 v2, v[0:1], s[0:1]
; GFX1132GISEL-NEXT: s_endpgm
diff --git a/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.reduce.min.ll b/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.reduce.min.ll
index 6f472a6e185843..5aca0141aa195e 100644
--- a/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.reduce.min.ll
+++ b/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.reduce.min.ll
@@ -403,6 +403,7 @@ define void @divergent_value_i16(ptr addrspace(1) %out, i16 %in) {
; GFX1164DAGISEL-NEXT: s_bitset0_b64 s[0:1], s3
; GFX1164DAGISEL-NEXT: s_min_i32 s2, s2, s4
; GFX1164DAGISEL-NEXT: s_cmp_lg_u64 s[0:1], 0
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164DAGISEL-NEXT: s_cbranch_scc1 .LBB1_1
; GFX1164DAGISEL-NEXT: ; %bb.2:
; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v2, s2
@@ -422,6 +423,7 @@ define void @divergent_value_i16(ptr addrspace(1) %out, i16 %in) {
; GFX1164GISEL-NEXT: s_bitset0_b64 s[0:1], s3
; GFX1164GISEL-NEXT: s_min_i32 s2, s2, s4
; GFX1164GISEL-NEXT: s_cmp_lg_u64 s[0:1], 0
+; GFX1164GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164GISEL-NEXT: s_cbranch_scc1 .LBB1_1
; GFX1164GISEL-NEXT: ; %bb.2:
; GFX1164GISEL-NEXT: v_mov_b32_e32 v2, s2
@@ -441,6 +443,7 @@ define void @divergent_value_i16(ptr addrspace(1) %out, i16 %in) {
; GFX1132DAGISEL-NEXT: s_bitset0_b32 s1, s2
; GFX1132DAGISEL-NEXT: s_min_i32 s0, s0, s3
; GFX1132DAGISEL-NEXT: s_cmp_lg_u32 s1, 0
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1132DAGISEL-NEXT: s_cbranch_scc1 .LBB1_1
; GFX1132DAGISEL-NEXT: ; %bb.2:
; GFX1132DAGISEL-NEXT: v_mov_b32_e32 v2, s0
@@ -460,6 +463,7 @@ define void @divergent_value_i16(ptr addrspace(1) %out, i16 %in) {
; GFX1132GISEL-NEXT: s_bitset0_b32 s1, s2
; GFX1132GISEL-NEXT: s_min_i32 s0, s0, s3
; GFX1132GISEL-NEXT: s_cmp_lg_u32 s1, 0
+; GFX1132GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1132GISEL-NEXT: s_cbranch_scc1 .LBB1_1
; GFX1132GISEL-NEXT: ; %bb.2:
; GFX1132GISEL-NEXT: v_mov_b32_e32 v2, s0
@@ -814,6 +818,7 @@ define amdgpu_kernel void @divergent_value(ptr addrspace(1) %out) #0 {
; GFX1164DAGISEL-NEXT: s_bitset0_b64 s[2:3], s5
; GFX1164DAGISEL-NEXT: s_min_i32 s4, s4, s6
; GFX1164DAGISEL-NEXT: s_cmp_lg_u64 s[2:3], 0
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164DAGISEL-NEXT: s_cbranch_scc1 .LBB3_1
; GFX1164DAGISEL-NEXT: ; %bb.2:
; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v0, s4
@@ -834,6 +839,7 @@ define amdgpu_kernel void @divergent_value(ptr addrspace(1) %out) #0 {
; GFX1164GISEL-NEXT: s_bitset0_b64 s[2:3], s5
; GFX1164GISEL-NEXT: s_min_i32 s4, s4, s6
; GFX1164GISEL-NEXT: s_cmp_lg_u64 s[2:3], 0
+; GFX1164GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164GISEL-NEXT: s_cbranch_scc1 .LBB3_1
; GFX1164GISEL-NEXT: ; %bb.2:
; GFX1164GISEL-NEXT: v_mov_b32_e32 v0, s4
@@ -855,6 +861,7 @@ define amdgpu_kernel void @divergent_value(ptr addrspace(1) %out) #0 {
; GFX1132DAGISEL-NEXT: s_bitset0_b32 s3, s4
; GFX1132DAGISEL-NEXT: s_min_i32 s2, s2, s5
; GFX1132DAGISEL-NEXT: s_cmp_lg_u32 s3, 0
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1132DAGISEL-NEXT: s_cbranch_scc1 .LBB3_1
; GFX1132DAGISEL-NEXT: ; %bb.2:
; GFX1132DAGISEL-NEXT: v_mov_b32_e32 v0, s2
@@ -875,6 +882,7 @@ define amdgpu_kernel void @divergent_value(ptr addrspace(1) %out) #0 {
; GFX1132GISEL-NEXT: s_bitset0_b32 s3, s4
; GFX1132GISEL-NEXT: s_min_i32 s2, s2, s5
; GFX1132GISEL-NEXT: s_cmp_lg_u32 s3, 0
+; GFX1132GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1132GISEL-NEXT: s_cbranch_scc1 .LBB3_1
; GFX1132GISEL-NEXT: ; %bb.2:
; GFX1132GISEL-NEXT: v_dual_mov_b32 v0, s2 :: v_dual_mov_b32 v1, 0
@@ -1298,7 +1306,7 @@ define amdgpu_kernel void @divergent_value_dpp(ptr addrspace(1) %out) #0 {
; GFX1164DAGISEL-NEXT: s_or_saveexec_b64 s[2:3], -1
; GFX1164DAGISEL-NEXT: s_waitcnt lgkmcnt(0)
; GFX1164DAGISEL-NEXT: v_min_i32_e32 v1, v1, v2
-; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX1164DAGISEL-NEXT: v_readlane_b32 s4, v1, 63
; GFX1164DAGISEL-NEXT: s_mov_b64 exec, s[2:3]
; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v0, 0
@@ -1333,7 +1341,7 @@ define amdgpu_kernel void @divergent_value_dpp(ptr addrspace(1) %out) #0 {
; GFX1164GISEL-NEXT: s_mov_b64 exec, s[0:1]
; GFX1164GISEL-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
; GFX1164GISEL-NEXT: s_or_saveexec_b64 s[2:3], -1
-; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX1164GISEL-NEXT: v_readlane_b32 s4, v1, 63
; GFX1164GISEL-NEXT: s_mov_b64 exec, s[2:3]
; GFX1164GISEL-NEXT: v_mov_b32_e32 v0, s4
@@ -1360,7 +1368,7 @@ define amdgpu_kernel void @divergent_value_dpp(ptr addrspace(1) %out) #0 {
; GFX1132DAGISEL-NEXT: s_or_saveexec_b32 s2, -1
; GFX1132DAGISEL-NEXT: s_waitcnt lgkmcnt(0)
; GFX1132DAGISEL-NEXT: v_min_i32_e32 v1, v1, v2
-; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX1132DAGISEL-NEXT: v_readlane_b32 s3, v1, 31
; GFX1132DAGISEL-NEXT: s_mov_b32 exec_lo, s2
; GFX1132DAGISEL-NEXT: v_dual_mov_b32 v0, 0 :: v_dual_mov_b32 v3, s3
@@ -1385,7 +1393,7 @@ define amdgpu_kernel void @divergent_value_dpp(ptr addrspace(1) %out) #0 {
; GFX1132GISEL-NEXT: s_mov_b32 exec_lo, s0
; GFX1132GISEL-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
; GFX1132GISEL-NEXT: s_or_saveexec_b32 s2, -1
-; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX1132GISEL-NEXT: v_readlane_b32 s3, v1, 31
; GFX1132GISEL-NEXT: s_mov_b32 exec_lo, s2
; GFX1132GISEL-NEXT: v_dual_mov_b32 v0, s3 :: v_dual_mov_b32 v3, 0
@@ -2125,52 +2133,53 @@ define void @divergent_value_dpp_i64(ptr addrspace(1) %out, i64 %in) #0 {
; GFX1164DAGISEL-NEXT: scratch_store_b32 off, v4, s32 offset:20
; GFX1164DAGISEL-NEXT: scratch_store_b32 off, v5, s32 offset:24
; GFX1164DAGISEL-NEXT: s_mov_b64 exec, s[0:1]
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1164DAGISEL-NEXT: s_or_saveexec_b64 s[0:1], -1
-; GFX1164DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(VALU_DEP_3)
; GFX1164DAGISEL-NEXT: v_cndmask_b32_e64 v4, -1, v2, s[0:1]
; GFX1164DAGISEL-NEXT: v_cndmask_b32_e64 v5, 0x7fffffff, v3, s[0:1]
; GFX1164DAGISEL-NEXT: v_mbcnt_lo_u32_b32 v8, -1, 0
-; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v6, v4
; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v6, v4
; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v7, v5
-; GFX1164DAGISEL-NEXT: v_mbcnt_hi_u32_b32 v8, -1, v8
; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX1164DAGISEL-NEXT: v_mbcnt_hi_u32_b32 v8, -1, v8
; GFX1164DAGISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:1 row_mask:0xf bank_mask:0xf
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_3)
; GFX1164DAGISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:1 row_mask:0xf bank_mask:0xf
-; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1164DAGISEL-NEXT: v_add_nc_u32_e32 v8, 32, v8
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1164DAGISEL-NEXT: v_cmp_lt_i64_e32 vcc, v[4:5], v[6:7]
-; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_2) | instid1(VALU_DEP_2)
; GFX1164DAGISEL-NEXT: v_mul_lo_u32 v8, 4, v8
; GFX1164DAGISEL-NEXT: v_cndmask_b32_e32 v4, v6, v4, vcc
; GFX1164DAGISEL-NEXT: v_cndmask_b32_e32 v5, v7, v5, vcc
-; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v6, v4
; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v6, v4
; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v7, v5
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1164DAGISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:2 row_mask:0xf bank_mask:0xf
-; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164DAGISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:2 row_mask:0xf bank_mask:0xf
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_3) | instid1(VALU_DEP_2)
; GFX1164DAGISEL-NEXT: v_cmp_lt_i64_e32 vcc, v[4:5], v[6:7]
; GFX1164DAGISEL-NEXT: s_waitcnt_depctr depctr_va_vcc(0)
; GFX1164DAGISEL-NEXT: v_cndmask_b32_e32 v4, v6, v4, vcc
; GFX1164DAGISEL-NEXT: v_cndmask_b32_e32 v5, v7, v5, vcc
-; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v6, v4
-; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v7, v5
; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v7, v5
; GFX1164DAGISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:4 row_mask:0xf bank_mask:0xf
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164DAGISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:4 row_mask:0xf bank_mask:0xf
-; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_3) | instid1(VALU_DEP_2)
; GFX1164DAGISEL-NEXT: v_cmp_lt_i64_e32 vcc, v[4:5], v[6:7]
; GFX1164DAGISEL-NEXT: s_waitcnt_depctr depctr_va_vcc(0)
; GFX1164DAGISEL-NEXT: v_cndmask_b32_e32 v4, v6, v4, vcc
; GFX1164DAGISEL-NEXT: v_cndmask_b32_e32 v5, v7, v5, vcc
-; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v6, v4
; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v6, v4
; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v7, v5
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1164DAGISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:8 row_mask:0xf bank_mask:0xf
-; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164DAGISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:8 row_mask:0xf bank_mask:0xf
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1164DAGISEL-NEXT: v_cmp_lt_i64_e32 vcc, v[4:5], v[6:7]
; GFX1164DAGISEL-NEXT: s_waitcnt_depctr depctr_va_vcc(0)
; GFX1164DAGISEL-NEXT: v_cndmask_b32_e32 v4, v6, v4, vcc
@@ -2193,6 +2202,7 @@ define void @divergent_value_dpp_i64(ptr addrspace(1) %out, i64 %in) #0 {
; GFX1164DAGISEL-NEXT: v_readlane_b32 s2, v4, 63
; GFX1164DAGISEL-NEXT: v_readlane_b32 s3, v5, 63
; GFX1164DAGISEL-NEXT: s_mov_b64 exec, s[0:1]
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v2, s2
; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v3, s3
; GFX1164DAGISEL-NEXT: global_store_b64 v[0:1], v[2:3], off
@@ -2219,52 +2229,53 @@ define void @divergent_value_dpp_i64(ptr addrspace(1) %out, i64 %in) #0 {
; GFX1164GISEL-NEXT: scratch_store_b32 off, v4, s32 offset:20
; GFX1164GISEL-NEXT: scratch_store_b32 off, v5, s32 offset:24
; GFX1164GISEL-NEXT: s_mov_b64 exec, s[0:1]
+; GFX1164GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1164GISEL-NEXT: s_or_saveexec_b64 s[0:1], -1
-; GFX1164GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(VALU_DEP_3)
; GFX1164GISEL-NEXT: v_cndmask_b32_e64 v4, -1, v2, s[0:1]
; GFX1164GISEL-NEXT: v_cndmask_b32_e64 v5, 0x7fffffff, v3, s[0:1]
; GFX1164GISEL-NEXT: v_mbcnt_lo_u32_b32 v8, -1, 0
-; GFX1164GISEL-NEXT: v_mov_b32_e32 v6, v4
; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX1164GISEL-NEXT: v_mov_b32_e32 v6, v4
; GFX1164GISEL-NEXT: v_mov_b32_e32 v7, v5
-; GFX1164GISEL-NEXT: v_mbcnt_hi_u32_b32 v8, -1, v8
; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX1164GISEL-NEXT: v_mbcnt_hi_u32_b32 v8, -1, v8
; GFX1164GISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:1 row_mask:0xf bank_mask:0xf
+; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_3)
; GFX1164GISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:1 row_mask:0xf bank_mask:0xf
-; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1164GISEL-NEXT: v_add_nc_u32_e32 v8, 32, v8
+; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1164GISEL-NEXT: v_cmp_lt_i64_e32 vcc, v[4:5], v[6:7]
-; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_2) | instid1(VALU_DEP_2)
; GFX1164GISEL-NEXT: v_mul_lo_u32 v8, 4, v8
; GFX1164GISEL-NEXT: v_cndmask_b32_e32 v4, v6, v4, vcc
; GFX1164GISEL-NEXT: v_cndmask_b32_e32 v5, v7, v5, vcc
-; GFX1164GISEL-NEXT: v_mov_b32_e32 v6, v4
; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX1164GISEL-NEXT: v_mov_b32_e32 v6, v4
; GFX1164GISEL-NEXT: v_mov_b32_e32 v7, v5
+; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1164GISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:2 row_mask:0xf bank_mask:0xf
-; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164GISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:2 row_mask:0xf bank_mask:0xf
+; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_3) | instid1(VALU_DEP_2)
; GFX1164GISEL-NEXT: v_cmp_lt_i64_e32 vcc, v[4:5], v[6:7]
; GFX1164GISEL-NEXT: s_waitcnt_depctr depctr_va_vcc(0)
; GFX1164GISEL-NEXT: v_cndmask_b32_e32 v4, v6, v4, vcc
; GFX1164GISEL-NEXT: v_cndmask_b32_e32 v5, v7, v5, vcc
-; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1164GISEL-NEXT: v_mov_b32_e32 v6, v4
-; GFX1164GISEL-NEXT: v_mov_b32_e32 v7, v5
; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX1164GISEL-NEXT: v_mov_b32_e32 v7, v5
; GFX1164GISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:4 row_mask:0xf bank_mask:0xf
+; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164GISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:4 row_mask:0xf bank_mask:0xf
-; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_3) | instid1(VALU_DEP_2)
; GFX1164GISEL-NEXT: v_cmp_lt_i64_e32 vcc, v[4:5], v[6:7]
; GFX1164GISEL-NEXT: s_waitcnt_depctr depctr_va_vcc(0)
; GFX1164GISEL-NEXT: v_cndmask_b32_e32 v4, v6, v4, vcc
; GFX1164GISEL-NEXT: v_cndmask_b32_e32 v5, v7, v5, vcc
-; GFX1164GISEL-NEXT: v_mov_b32_e32 v6, v4
; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX1164GISEL-NEXT: v_mov_b32_e32 v6, v4
; GFX1164GISEL-NEXT: v_mov_b32_e32 v7, v5
+; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1164GISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:8 row_mask:0xf bank_mask:0xf
-; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164GISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:8 row_mask:0xf bank_mask:0xf
+; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1164GISEL-NEXT: v_cmp_lt_i64_e32 vcc, v[4:5], v[6:7]
; GFX1164GISEL-NEXT: s_waitcnt_depctr depctr_va_vcc(0)
; GFX1164GISEL-NEXT: v_cndmask_b32_e32 v4, v6, v4, vcc
@@ -2287,6 +2298,7 @@ define void @divergent_value_dpp_i64(ptr addrspace(1) %out, i64 %in) #0 {
; GFX1164GISEL-NEXT: v_readlane_b32 s2, v4, 63
; GFX1164GISEL-NEXT: v_readlane_b32 s3, v5, 63
; GFX1164GISEL-NEXT: s_mov_b64 exec, s[0:1]
+; GFX1164GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164GISEL-NEXT: v_mov_b32_e32 v2, s2
; GFX1164GISEL-NEXT: v_mov_b32_e32 v3, s3
; GFX1164GISEL-NEXT: global_store_b64 v[0:1], v[2:3], off
@@ -2312,36 +2324,36 @@ define void @divergent_value_dpp_i64(ptr addrspace(1) %out, i64 %in) #0 {
; GFX1132DAGISEL-NEXT: scratch_store_b32 off, v4, s32 offset:16
; GFX1132DAGISEL-NEXT: scratch_store_b32 off, v5, s32 offset:20
; GFX1132DAGISEL-NEXT: s_mov_b32 exec_lo, s0
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1132DAGISEL-NEXT: s_or_saveexec_b32 s2, -1
-; GFX1132DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX1132DAGISEL-NEXT: v_cndmask_b32_e64 v4, -1, v2, s2
; GFX1132DAGISEL-NEXT: v_cndmask_b32_e64 v5, 0x7fffffff, v3, s2
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132DAGISEL-NEXT: v_dual_mov_b32 v6, v4 :: v_dual_mov_b32 v7, v5
-; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1132DAGISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:1 row_mask:0xf bank_mask:0xf
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132DAGISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:1 row_mask:0xf bank_mask:0xf
-; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX1132DAGISEL-NEXT: v_cmp_lt_i64_e32 vcc_lo, v[4:5], v[6:7]
; GFX1132DAGISEL-NEXT: v_dual_cndmask_b32 v4, v6, v4 :: v_dual_cndmask_b32 v5, v7, v5
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132DAGISEL-NEXT: v_dual_mov_b32 v6, v4 :: v_dual_mov_b32 v7, v5
-; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1132DAGISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:2 row_mask:0xf bank_mask:0xf
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132DAGISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:2 row_mask:0xf bank_mask:0xf
-; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX1132DAGISEL-NEXT: v_cmp_lt_i64_e32 vcc_lo, v[4:5], v[6:7]
; GFX1132DAGISEL-NEXT: v_dual_cndmask_b32 v4, v6, v4 :: v_dual_cndmask_b32 v5, v7, v5
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132DAGISEL-NEXT: v_dual_mov_b32 v6, v4 :: v_dual_mov_b32 v7, v5
-; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1132DAGISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:4 row_mask:0xf bank_mask:0xf
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132DAGISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:4 row_mask:0xf bank_mask:0xf
-; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX1132DAGISEL-NEXT: v_cmp_lt_i64_e32 vcc_lo, v[4:5], v[6:7]
; GFX1132DAGISEL-NEXT: v_dual_cndmask_b32 v4, v6, v4 :: v_dual_cndmask_b32 v5, v7, v5
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132DAGISEL-NEXT: v_dual_mov_b32 v6, v4 :: v_dual_mov_b32 v7, v5
-; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1132DAGISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:8 row_mask:0xf bank_mask:0xf
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132DAGISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:8 row_mask:0xf bank_mask:0xf
-; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1132DAGISEL-NEXT: v_cmp_lt_i64_e32 vcc_lo, v[4:5], v[6:7]
; GFX1132DAGISEL-NEXT: v_dual_cndmask_b32 v4, v6, v4 :: v_dual_cndmask_b32 v5, v7, v5
; GFX1132DAGISEL-NEXT: ds_swizzle_b32 v6, v4 offset:swizzle(BROADCAST,32,15)
@@ -2353,6 +2365,7 @@ define void @divergent_value_dpp_i64(ptr addrspace(1) %out, i64 %in) #0 {
; GFX1132DAGISEL-NEXT: v_readlane_b32 s0, v4, 31
; GFX1132DAGISEL-NEXT: v_readlane_b32 s1, v5, 31
; GFX1132DAGISEL-NEXT: s_mov_b32 exec_lo, s2
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1132DAGISEL-NEXT: v_dual_mov_b32 v3, s1 :: v_dual_mov_b32 v2, s0
; GFX1132DAGISEL-NEXT: global_store_b64 v[0:1], v[2:3], off
; GFX1132DAGISEL-NEXT: s_xor_saveexec_b32 s0, -1
@@ -2375,36 +2388,36 @@ define void @divergent_value_dpp_i64(ptr addrspace(1) %out, i64 %in) #0 {
; GFX1132GISEL-NEXT: scratch_store_b32 off, v4, s32 offset:16
; GFX1132GISEL-NEXT: scratch_store_b32 off, v5, s32 offset:20
; GFX1132GISEL-NEXT: s_mov_b32 exec_lo, s0
+; GFX1132GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1132GISEL-NEXT: s_or_saveexec_b32 s2, -1
-; GFX1132GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX1132GISEL-NEXT: v_cndmask_b32_e64 v4, -1, v2, s2
; GFX1132GISEL-NEXT: v_cndmask_b32_e64 v5, 0x7fffffff, v3, s2
+; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132GISEL-NEXT: v_dual_mov_b32 v6, v4 :: v_dual_mov_b32 v7, v5
-; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1132GISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:1 row_mask:0xf bank_mask:0xf
+; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132GISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:1 row_mask:0xf bank_mask:0xf
-; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX1132GISEL-NEXT: v_cmp_lt_i64_e32 vcc_lo, v[4:5], v[6:7]
; GFX1132GISEL-NEXT: v_dual_cndmask_b32 v4, v6, v4 :: v_dual_cndmask_b32 v5, v7, v5
+; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132GISEL-NEXT: v_dual_mov_b32 v6, v4 :: v_dual_mov_b32 v7, v5
-; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1132GISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:2 row_mask:0xf bank_mask:0xf
+; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132GISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:2 row_mask:0xf bank_mask:0xf
-; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX1132GISEL-NEXT: v_cmp_lt_i64_e32 vcc_lo, v[4:5], v[6:7]
; GFX1132GISEL-NEXT: v_dual_cndmask_b32 v4, v6, v4 :: v_dual_cndmask_b32 v5, v7, v5
+; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132GISEL-NEXT: v_dual_mov_b32 v6, v4 :: v_dual_mov_b32 v7, v5
-; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1132GISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:4 row_mask:0xf bank_mask:0xf
+; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132GISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:4 row_mask:0xf bank_mask:0xf
-; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX1132GISEL-NEXT: v_cmp_lt_i64_e32 vcc_lo, v[4:5], v[6:7]
; GFX1132GISEL-NEXT: v_dual_cndmask_b32 v4, v6, v4 :: v_dual_cndmask_b32 v5, v7, v5
+; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132GISEL-NEXT: v_dual_mov_b32 v6, v4 :: v_dual_mov_b32 v7, v5
-; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1132GISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:8 row_mask:0xf bank_mask:0xf
+; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132GISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:8 row_mask:0xf bank_mask:0xf
-; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1132GISEL-NEXT: v_cmp_lt_i64_e32 vcc_lo, v[4:5], v[6:7]
; GFX1132GISEL-NEXT: v_dual_cndmask_b32 v4, v6, v4 :: v_dual_cndmask_b32 v5, v7, v5
; GFX1132GISEL-NEXT: ds_swizzle_b32 v6, v4 offset:swizzle(BROADCAST,32,15)
@@ -2416,6 +2429,7 @@ define void @divergent_value_dpp_i64(ptr addrspace(1) %out, i64 %in) #0 {
; GFX1132GISEL-NEXT: v_readlane_b32 s0, v4, 31
; GFX1132GISEL-NEXT: v_readlane_b32 s1, v5, 31
; GFX1132GISEL-NEXT: s_mov_b32 exec_lo, s2
+; GFX1132GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1132GISEL-NEXT: v_dual_mov_b32 v3, s1 :: v_dual_mov_b32 v2, s0
; GFX1132GISEL-NEXT: global_store_b64 v[0:1], v[2:3], off
; GFX1132GISEL-NEXT: s_xor_saveexec_b32 s0, -1
@@ -2709,7 +2723,7 @@ define amdgpu_kernel void @default_stratergy(ptr addrspace(1) %out) #0 {
; GFX1164DAGISEL-NEXT: s_or_saveexec_b64 s[2:3], -1
; GFX1164DAGISEL-NEXT: s_waitcnt lgkmcnt(0)
; GFX1164DAGISEL-NEXT: v_min_i32_e32 v1, v1, v2
-; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX1164DAGISEL-NEXT: v_readlane_b32 s4, v1, 63
; GFX1164DAGISEL-NEXT: s_mov_b64 exec, s[2:3]
; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v0, 0
@@ -2744,7 +2758,7 @@ define amdgpu_kernel void @default_stratergy(ptr addrspace(1) %out) #0 {
; GFX1164GISEL-NEXT: s_mov_b64 exec, s[0:1]
; GFX1164GISEL-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
; GFX1164GISEL-NEXT: s_or_saveexec_b64 s[2:3], -1
-; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX1164GISEL-NEXT: v_readlane_b32 s4, v1, 63
; GFX1164GISEL-NEXT: s_mov_b64 exec, s[2:3]
; GFX1164GISEL-NEXT: v_mov_b32_e32 v0, s4
@@ -2771,7 +2785,7 @@ define amdgpu_kernel void @default_stratergy(ptr addrspace(1) %out) #0 {
; GFX1132DAGISEL-NEXT: s_or_saveexec_b32 s2, -1
; GFX1132DAGISEL-NEXT: s_waitcnt lgkmcnt(0)
; GFX1132DAGISEL-NEXT: v_min_i32_e32 v1, v1, v2
-; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX1132DAGISEL-NEXT: v_readlane_b32 s3, v1, 31
; GFX1132DAGISEL-NEXT: s_mov_b32 exec_lo, s2
; GFX1132DAGISEL-NEXT: v_dual_mov_b32 v0, 0 :: v_dual_mov_b32 v3, s3
@@ -2796,7 +2810,7 @@ define amdgpu_kernel void @default_stratergy(ptr addrspace(1) %out) #0 {
; GFX1132GISEL-NEXT: s_mov_b32 exec_lo, s0
; GFX1132GISEL-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
; GFX1132GISEL-NEXT: s_or_saveexec_b32 s2, -1
-; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX1132GISEL-NEXT: v_readlane_b32 s3, v1, 31
; GFX1132GISEL-NEXT: s_mov_b32 exec_lo, s2
; GFX1132GISEL-NEXT: v_dual_mov_b32 v0, s3 :: v_dual_mov_b32 v3, 0
@@ -3178,7 +3192,7 @@ define amdgpu_kernel void @divergent_cfg(ptr addrspace(1) %out, i32 %in) #0 {
; GFX1164DAGISEL-NEXT: v_and_b32_e32 v0, 0x3ff, v0
; GFX1164DAGISEL-NEXT: s_mov_b64 s[0:1], exec
; GFX1164DAGISEL-NEXT: ; implicit-def: $sgpr2
-; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164DAGISEL-NEXT: v_cmpx_lt_u32_e32 15, v0
; GFX1164DAGISEL-NEXT: s_xor_b64 s[0:1], exec, s[0:1]
; GFX1164DAGISEL-NEXT: ; %bb.1: ; %else
@@ -3189,13 +3203,14 @@ define amdgpu_kernel void @divergent_cfg(ptr addrspace(1) %out, i32 %in) #0 {
; GFX1164DAGISEL-NEXT: s_waitcnt lgkmcnt(0)
; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v1, s2
; GFX1164DAGISEL-NEXT: s_xor_b64 exec, exec, s[0:1]
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164DAGISEL-NEXT: s_cbranch_execz .LBB8_6
; GFX1164DAGISEL-NEXT: ; %bb.3: ; %if
; GFX1164DAGISEL-NEXT: s_mov_b64 s[2:3], exec
; GFX1164DAGISEL-NEXT: s_brev_b32 s6, -2
; GFX1164DAGISEL-NEXT: .LBB8_4: ; =>This Inner Loop Header: Depth=1
; GFX1164DAGISEL-NEXT: s_ctz_i32_b64 s7, s[2:3]
-; GFX1164DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_3) | instid1(SALU_CYCLE_1)
; GFX1164DAGISEL-NEXT: v_readlane_b32 s8, v0, s7
; GFX1164DAGISEL-NEXT: s_bitset0_b64 s[2:3], s7
; GFX1164DAGISEL-NEXT: s_min_i32 s6, s6, s8
@@ -3216,7 +3231,7 @@ define amdgpu_kernel void @divergent_cfg(ptr addrspace(1) %out, i32 %in) #0 {
; GFX1164GISEL-NEXT: v_and_b32_e32 v0, 0x3ff, v0
; GFX1164GISEL-NEXT: s_mov_b64 s[0:1], exec
; GFX1164GISEL-NEXT: ; implicit-def: $sgpr2
-; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164GISEL-NEXT: v_cmpx_le_u32_e32 16, v0
; GFX1164GISEL-NEXT: s_xor_b64 s[0:1], exec, s[0:1]
; GFX1164GISEL-NEXT: s_cbranch_execz .LBB8_2
@@ -3229,13 +3244,14 @@ define amdgpu_kernel void @divergent_cfg(ptr addrspace(1) %out, i32 %in) #0 {
; GFX1164GISEL-NEXT: s_or_saveexec_b64 s[0:1], s[0:1]
; GFX1164GISEL-NEXT: v_mov_b32_e32 v1, s2
; GFX1164GISEL-NEXT: s_xor_b64 exec, exec, s[0:1]
+; GFX1164GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164GISEL-NEXT: s_cbranch_execz .LBB8_6
; GFX1164GISEL-NEXT: ; %bb.3: ; %if
; GFX1164GISEL-NEXT: s_mov_b64 s[2:3], exec
; GFX1164GISEL-NEXT: s_brev_b32 s6, -2
; GFX1164GISEL-NEXT: .LBB8_4: ; =>This Inner Loop Header: Depth=1
; GFX1164GISEL-NEXT: s_ctz_i32_b64 s7, s[2:3]
-; GFX1164GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1164GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_3) | instid1(SALU_CYCLE_1)
; GFX1164GISEL-NEXT: v_readlane_b32 s8, v0, s7
; GFX1164GISEL-NEXT: s_bitset0_b64 s[2:3], s7
; GFX1164GISEL-NEXT: s_min_i32 s6, s6, s8
@@ -3256,13 +3272,14 @@ define amdgpu_kernel void @divergent_cfg(ptr addrspace(1) %out, i32 %in) #0 {
; GFX1132DAGISEL-NEXT: v_and_b32_e32 v0, 0x3ff, v0
; GFX1132DAGISEL-NEXT: s_mov_b32 s0, exec_lo
; GFX1132DAGISEL-NEXT: ; implicit-def: $sgpr1
-; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132DAGISEL-NEXT: v_cmpx_lt_u32_e32 15, v0
; GFX1132DAGISEL-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX1132DAGISEL-NEXT: ; %bb.1: ; %else
; GFX1132DAGISEL-NEXT: s_load_b32 s1, s[4:5], 0x2c
; GFX1132DAGISEL-NEXT: ; implicit-def: $vgpr0
; GFX1132DAGISEL-NEXT: ; %bb.2: ; %Flow
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_3) | instid1(SALU_CYCLE_1)
; GFX1132DAGISEL-NEXT: s_or_saveexec_b32 s0, s0
; GFX1132DAGISEL-NEXT: s_waitcnt lgkmcnt(0)
; GFX1132DAGISEL-NEXT: v_mov_b32_e32 v1, s1
@@ -3273,7 +3290,7 @@ define amdgpu_kernel void @divergent_cfg(ptr addrspace(1) %out, i32 %in) #0 {
; GFX1132DAGISEL-NEXT: s_brev_b32 s1, -2
; GFX1132DAGISEL-NEXT: .LBB8_4: ; =>This Inner Loop Header: Depth=1
; GFX1132DAGISEL-NEXT: s_ctz_i32_b32 s3, s2
-; GFX1132DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_3) | instid1(SALU_CYCLE_1)
; GFX1132DAGISEL-NEXT: v_readlane_b32 s6, v0, s3
; GFX1132DAGISEL-NEXT: s_bitset0_b32 s2, s3
; GFX1132DAGISEL-NEXT: s_min_i32 s1, s1, s6
@@ -3294,9 +3311,10 @@ define amdgpu_kernel void @divergent_cfg(ptr addrspace(1) %out, i32 %in) #0 {
; GFX1132GISEL-NEXT: v_and_b32_e32 v0, 0x3ff, v0
; GFX1132GISEL-NEXT: s_mov_b32 s0, exec_lo
; GFX1132GISEL-NEXT: ; implicit-def: $sgpr1
-; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132GISEL-NEXT: v_cmpx_le_u32_e32 16, v0
; GFX1132GISEL-NEXT: s_xor_b32 s0, exec_lo, s0
+; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1132GISEL-NEXT: s_cbranch_execz .LBB8_2
; GFX1132GISEL-NEXT: ; %bb.1: ; %else
; GFX1132GISEL-NEXT: s_load_b32 s1, s[4:5], 0x2c
@@ -3307,13 +3325,14 @@ define amdgpu_kernel void @divergent_cfg(ptr addrspace(1) %out, i32 %in) #0 {
; GFX1132GISEL-NEXT: s_or_saveexec_b32 s0, s0
; GFX1132GISEL-NEXT: v_mov_b32_e32 v1, s1
; GFX1132GISEL-NEXT: s_xor_b32 exec_lo, exec_lo, s0
+; GFX1132GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1132GISEL-NEXT: s_cbranch_execz .LBB8_6
; GFX1132GISEL-NEXT: ; %bb.3: ; %if
; GFX1132GISEL-NEXT: s_mov_b32 s2, exec_lo
; GFX1132GISEL-NEXT: s_brev_b32 s1, -2
; GFX1132GISEL-NEXT: .LBB8_4: ; =>This Inner Loop Header: Depth=1
; GFX1132GISEL-NEXT: s_ctz_i32_b32 s3, s2
-; GFX1132GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1132GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_3) | instid1(SALU_CYCLE_1)
; GFX1132GISEL-NEXT: v_readlane_b32 s6, v0, s3
; GFX1132GISEL-NEXT: s_bitset0_b32 s2, s3
; GFX1132GISEL-NEXT: s_min_i32 s1, s1, s6
@@ -3748,6 +3767,7 @@ define void @divergent_value_i64(ptr addrspace(1) %out, i64 %id.x) #0 {
; GFX1164DAGISEL-NEXT: v_readlane_b32 s4, v2, s8
; GFX1164DAGISEL-NEXT: v_readlane_b32 s5, v3, s8
; GFX1164DAGISEL-NEXT: v_cmp_lt_i64_e32 vcc, s[4:5], v[4:5]
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_3) | instid1(SALU_CYCLE_1)
; GFX1164DAGISEL-NEXT: s_and_b64 s[6:7], vcc, s[2:3]
; GFX1164DAGISEL-NEXT: s_bitset0_b64 s[2:3], s8
; GFX1164DAGISEL-NEXT: s_cselect_b64 s[0:1], s[4:5], s[0:1]
@@ -3773,6 +3793,7 @@ define void @divergent_value_i64(ptr addrspace(1) %out, i64 %id.x) #0 {
; GFX1164GISEL-NEXT: v_readlane_b32 s4, v2, s8
; GFX1164GISEL-NEXT: v_readlane_b32 s5, v3, s8
; GFX1164GISEL-NEXT: v_cmp_lt_i64_e32 vcc, s[4:5], v[4:5]
+; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_3) | instid1(SALU_CYCLE_1)
; GFX1164GISEL-NEXT: s_and_b64 s[6:7], vcc, s[2:3]
; GFX1164GISEL-NEXT: s_bitset0_b64 s[2:3], s8
; GFX1164GISEL-NEXT: s_cselect_b64 s[0:1], s[4:5], s[0:1]
@@ -3801,6 +3822,7 @@ define void @divergent_value_i64(ptr addrspace(1) %out, i64 %id.x) #0 {
; GFX1132DAGISEL-NEXT: s_bitset0_b32 s2, s3
; GFX1132DAGISEL-NEXT: s_cselect_b64 s[0:1], s[4:5], s[0:1]
; GFX1132DAGISEL-NEXT: s_cmp_lg_u32 s2, 0
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1132DAGISEL-NEXT: s_cbranch_scc1 .LBB10_1
; GFX1132DAGISEL-NEXT: ; %bb.2:
; GFX1132DAGISEL-NEXT: v_dual_mov_b32 v3, s1 :: v_dual_mov_b32 v2, s0
@@ -3824,6 +3846,7 @@ define void @divergent_value_i64(ptr addrspace(1) %out, i64 %id.x) #0 {
; GFX1132GISEL-NEXT: s_bitset0_b32 s2, s3
; GFX1132GISEL-NEXT: s_cselect_b64 s[0:1], s[4:5], s[0:1]
; GFX1132GISEL-NEXT: s_cmp_lg_u32 s2, 0
+; GFX1132GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1132GISEL-NEXT: s_cbranch_scc1 .LBB10_1
; GFX1132GISEL-NEXT: ; %bb.2:
; GFX1132GISEL-NEXT: v_dual_mov_b32 v3, s1 :: v_dual_mov_b32 v2, s0
@@ -4095,19 +4118,22 @@ define amdgpu_kernel void @divergent_cfg_i64(ptr addrspace(1) %out, i64 %in, i64
; GFX1164DAGISEL-NEXT: s_load_b64 s[4:5], s[4:5], 0x34
; GFX1164DAGISEL-NEXT: v_and_b32_e32 v0, 0x3ff, v0
; GFX1164DAGISEL-NEXT: s_mov_b64 s[6:7], exec
-; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164DAGISEL-NEXT: v_cmpx_lt_u32_e32 15, v0
; GFX1164DAGISEL-NEXT: s_xor_b64 s[6:7], exec, s[6:7]
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164DAGISEL-NEXT: s_or_saveexec_b64 s[6:7], s[6:7]
; GFX1164DAGISEL-NEXT: s_waitcnt lgkmcnt(0)
; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v0, s2
; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v1, s3
; GFX1164DAGISEL-NEXT: s_xor_b64 exec, exec, s[6:7]
; GFX1164DAGISEL-NEXT: ; %bb.1: ; %if
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v0, s4
; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v1, s5
; GFX1164DAGISEL-NEXT: ; %bb.2: ; %endif
; GFX1164DAGISEL-NEXT: s_or_b64 exec, exec, s[6:7]
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v2, 0
; GFX1164DAGISEL-NEXT: global_store_b64 v2, v[0:1], s[0:1]
; GFX1164DAGISEL-NEXT: s_endpgm
@@ -4118,7 +4144,7 @@ define amdgpu_kernel void @divergent_cfg_i64(ptr addrspace(1) %out, i64 %in, i64
; GFX1164GISEL-NEXT: v_and_b32_e32 v0, 0x3ff, v0
; GFX1164GISEL-NEXT: s_mov_b64 s[8:9], exec
; GFX1164GISEL-NEXT: ; implicit-def: $sgpr6_sgpr7
-; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164GISEL-NEXT: v_cmpx_le_u32_e32 16, v0
; GFX1164GISEL-NEXT: s_xor_b64 s[8:9], exec, s[8:9]
; GFX1164GISEL-NEXT: s_cbranch_execz .LBB11_2
@@ -4131,6 +4157,7 @@ define amdgpu_kernel void @divergent_cfg_i64(ptr addrspace(1) %out, i64 %in, i64
; GFX1164GISEL-NEXT: v_mov_b32_e32 v0, s6
; GFX1164GISEL-NEXT: v_mov_b32_e32 v1, s7
; GFX1164GISEL-NEXT: s_xor_b64 exec, exec, s[2:3]
+; GFX1164GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164GISEL-NEXT: s_cbranch_execz .LBB11_4
; GFX1164GISEL-NEXT: ; %bb.3: ; %if
; GFX1164GISEL-NEXT: s_load_b64 s[4:5], s[4:5], 0x34
@@ -4141,6 +4168,7 @@ define amdgpu_kernel void @divergent_cfg_i64(ptr addrspace(1) %out, i64 %in, i64
; GFX1164GISEL-NEXT: v_mov_b32_e32 v1, s5
; GFX1164GISEL-NEXT: .LBB11_4: ; %endif
; GFX1164GISEL-NEXT: s_or_b64 exec, exec, s[2:3]
+; GFX1164GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164GISEL-NEXT: v_mov_b32_e32 v2, 0
; GFX1164GISEL-NEXT: global_store_b64 v2, v[0:1], s[0:1]
; GFX1164GISEL-NEXT: s_endpgm
@@ -4152,17 +4180,20 @@ define amdgpu_kernel void @divergent_cfg_i64(ptr addrspace(1) %out, i64 %in, i64
; GFX1132DAGISEL-NEXT: s_load_b64 s[4:5], s[4:5], 0x34
; GFX1132DAGISEL-NEXT: v_and_b32_e32 v0, 0x3ff, v0
; GFX1132DAGISEL-NEXT: s_mov_b32 s6, exec_lo
-; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132DAGISEL-NEXT: v_cmpx_lt_u32_e32 15, v0
; GFX1132DAGISEL-NEXT: s_xor_b32 s6, exec_lo, s6
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX1132DAGISEL-NEXT: s_or_saveexec_b32 s6, s6
; GFX1132DAGISEL-NEXT: s_waitcnt lgkmcnt(0)
; GFX1132DAGISEL-NEXT: v_dual_mov_b32 v0, s2 :: v_dual_mov_b32 v1, s3
; GFX1132DAGISEL-NEXT: s_xor_b32 exec_lo, exec_lo, s6
; GFX1132DAGISEL-NEXT: ; %bb.1: ; %if
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1132DAGISEL-NEXT: v_dual_mov_b32 v0, s4 :: v_dual_mov_b32 v1, s5
; GFX1132DAGISEL-NEXT: ; %bb.2: ; %endif
; GFX1132DAGISEL-NEXT: s_or_b32 exec_lo, exec_lo, s6
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1132DAGISEL-NEXT: v_mov_b32_e32 v2, 0
; GFX1132DAGISEL-NEXT: global_store_b64 v2, v[0:1], s[0:1]
; GFX1132DAGISEL-NEXT: s_endpgm
@@ -4173,9 +4204,10 @@ define amdgpu_kernel void @divergent_cfg_i64(ptr addrspace(1) %out, i64 %in, i64
; GFX1132GISEL-NEXT: v_and_b32_e32 v0, 0x3ff, v0
; GFX1132GISEL-NEXT: s_mov_b32 s8, exec_lo
; GFX1132GISEL-NEXT: ; implicit-def: $sgpr6_sgpr7
-; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132GISEL-NEXT: v_cmpx_le_u32_e32 16, v0
; GFX1132GISEL-NEXT: s_xor_b32 s8, exec_lo, s8
+; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1132GISEL-NEXT: s_cbranch_execz .LBB11_2
; GFX1132GISEL-NEXT: ; %bb.1: ; %else
; GFX1132GISEL-NEXT: s_waitcnt lgkmcnt(0)
@@ -4185,6 +4217,7 @@ define amdgpu_kernel void @divergent_cfg_i64(ptr addrspace(1) %out, i64 %in, i64
; GFX1132GISEL-NEXT: s_or_saveexec_b32 s2, s8
; GFX1132GISEL-NEXT: v_dual_mov_b32 v0, s6 :: v_dual_mov_b32 v1, s7
; GFX1132GISEL-NEXT: s_xor_b32 exec_lo, exec_lo, s2
+; GFX1132GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1132GISEL-NEXT: s_cbranch_execz .LBB11_4
; GFX1132GISEL-NEXT: ; %bb.3: ; %if
; GFX1132GISEL-NEXT: s_load_b64 s[4:5], s[4:5], 0x34
@@ -4194,6 +4227,7 @@ define amdgpu_kernel void @divergent_cfg_i64(ptr addrspace(1) %out, i64 %in, i64
; GFX1132GISEL-NEXT: v_dual_mov_b32 v0, s4 :: v_dual_mov_b32 v1, s5
; GFX1132GISEL-NEXT: .LBB11_4: ; %endif
; GFX1132GISEL-NEXT: s_or_b32 exec_lo, exec_lo, s2
+; GFX1132GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1132GISEL-NEXT: v_mov_b32_e32 v2, 0
; GFX1132GISEL-NEXT: global_store_b64 v2, v[0:1], s[0:1]
; GFX1132GISEL-NEXT: s_endpgm
diff --git a/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.reduce.or.ll b/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.reduce.or.ll
index 1f3d19668f86bb..93face2fb40e92 100644
--- a/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.reduce.or.ll
+++ b/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.reduce.or.ll
@@ -403,6 +403,7 @@ define void @divergent_value_i16(ptr addrspace(1) %out, i16 %in) {
; GFX1164DAGISEL-FAKE16-NEXT: s_bitset0_b64 s[0:1], s3
; GFX1164DAGISEL-FAKE16-NEXT: s_or_b32 s2, s2, s4
; GFX1164DAGISEL-FAKE16-NEXT: s_cmp_lg_u64 s[0:1], 0
+; GFX1164DAGISEL-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164DAGISEL-FAKE16-NEXT: s_cbranch_scc1 .LBB1_1
; GFX1164DAGISEL-FAKE16-NEXT: ; %bb.2:
; GFX1164DAGISEL-FAKE16-NEXT: v_mov_b32_e32 v2, s2
@@ -422,6 +423,7 @@ define void @divergent_value_i16(ptr addrspace(1) %out, i16 %in) {
; GFX1164GISEL-NEXT: s_bitset0_b64 s[0:1], s3
; GFX1164GISEL-NEXT: s_or_b32 s2, s2, s4
; GFX1164GISEL-NEXT: s_cmp_lg_u64 s[0:1], 0
+; GFX1164GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164GISEL-NEXT: s_cbranch_scc1 .LBB1_1
; GFX1164GISEL-NEXT: ; %bb.2:
; GFX1164GISEL-NEXT: v_mov_b32_e32 v2, s2
@@ -441,6 +443,7 @@ define void @divergent_value_i16(ptr addrspace(1) %out, i16 %in) {
; GFX1132DAGISEL-FAKE16-NEXT: s_bitset0_b32 s1, s2
; GFX1132DAGISEL-FAKE16-NEXT: s_or_b32 s0, s0, s3
; GFX1132DAGISEL-FAKE16-NEXT: s_cmp_lg_u32 s1, 0
+; GFX1132DAGISEL-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1132DAGISEL-FAKE16-NEXT: s_cbranch_scc1 .LBB1_1
; GFX1132DAGISEL-FAKE16-NEXT: ; %bb.2:
; GFX1132DAGISEL-FAKE16-NEXT: v_mov_b32_e32 v2, s0
@@ -460,6 +463,7 @@ define void @divergent_value_i16(ptr addrspace(1) %out, i16 %in) {
; GFX1132GISEL-NEXT: s_bitset0_b32 s1, s2
; GFX1132GISEL-NEXT: s_or_b32 s0, s0, s3
; GFX1132GISEL-NEXT: s_cmp_lg_u32 s1, 0
+; GFX1132GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1132GISEL-NEXT: s_cbranch_scc1 .LBB1_1
; GFX1132GISEL-NEXT: ; %bb.2:
; GFX1132GISEL-NEXT: v_mov_b32_e32 v2, s0
@@ -479,6 +483,7 @@ define void @divergent_value_i16(ptr addrspace(1) %out, i16 %in) {
; GFX1164DAGISEL-TRUE16-NEXT: s_bitset0_b64 s[0:1], s3
; GFX1164DAGISEL-TRUE16-NEXT: s_or_b32 s2, s2, s4
; GFX1164DAGISEL-TRUE16-NEXT: s_cmp_lg_u64 s[0:1], 0
+; GFX1164DAGISEL-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164DAGISEL-TRUE16-NEXT: s_cbranch_scc1 .LBB1_1
; GFX1164DAGISEL-TRUE16-NEXT: ; %bb.2:
; GFX1164DAGISEL-TRUE16-NEXT: v_mov_b32_e32 v2, s2
@@ -498,6 +503,7 @@ define void @divergent_value_i16(ptr addrspace(1) %out, i16 %in) {
; GFX1132DAGISEL-TRUE16-NEXT: s_bitset0_b32 s1, s2
; GFX1132DAGISEL-TRUE16-NEXT: s_or_b32 s0, s0, s3
; GFX1132DAGISEL-TRUE16-NEXT: s_cmp_lg_u32 s1, 0
+; GFX1132DAGISEL-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1132DAGISEL-TRUE16-NEXT: s_cbranch_scc1 .LBB1_1
; GFX1132DAGISEL-TRUE16-NEXT: ; %bb.2:
; GFX1132DAGISEL-TRUE16-NEXT: v_mov_b32_e32 v2, s0
@@ -852,6 +858,7 @@ define amdgpu_kernel void @divergent_value(ptr addrspace(1) %out) #0 {
; GFX1164DAGISEL-NEXT: s_bitset0_b64 s[2:3], s5
; GFX1164DAGISEL-NEXT: s_or_b32 s4, s4, s6
; GFX1164DAGISEL-NEXT: s_cmp_lg_u64 s[2:3], 0
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164DAGISEL-NEXT: s_cbranch_scc1 .LBB3_1
; GFX1164DAGISEL-NEXT: ; %bb.2:
; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v0, s4
@@ -872,6 +879,7 @@ define amdgpu_kernel void @divergent_value(ptr addrspace(1) %out) #0 {
; GFX1164GISEL-NEXT: s_bitset0_b64 s[2:3], s5
; GFX1164GISEL-NEXT: s_or_b32 s4, s4, s6
; GFX1164GISEL-NEXT: s_cmp_lg_u64 s[2:3], 0
+; GFX1164GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164GISEL-NEXT: s_cbranch_scc1 .LBB3_1
; GFX1164GISEL-NEXT: ; %bb.2:
; GFX1164GISEL-NEXT: v_mov_b32_e32 v0, s4
@@ -893,6 +901,7 @@ define amdgpu_kernel void @divergent_value(ptr addrspace(1) %out) #0 {
; GFX1132DAGISEL-NEXT: s_bitset0_b32 s3, s4
; GFX1132DAGISEL-NEXT: s_or_b32 s2, s2, s5
; GFX1132DAGISEL-NEXT: s_cmp_lg_u32 s3, 0
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1132DAGISEL-NEXT: s_cbranch_scc1 .LBB3_1
; GFX1132DAGISEL-NEXT: ; %bb.2:
; GFX1132DAGISEL-NEXT: v_mov_b32_e32 v0, s2
@@ -913,6 +922,7 @@ define amdgpu_kernel void @divergent_value(ptr addrspace(1) %out) #0 {
; GFX1132GISEL-NEXT: s_bitset0_b32 s3, s4
; GFX1132GISEL-NEXT: s_or_b32 s2, s2, s5
; GFX1132GISEL-NEXT: s_cmp_lg_u32 s3, 0
+; GFX1132GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1132GISEL-NEXT: s_cbranch_scc1 .LBB3_1
; GFX1132GISEL-NEXT: ; %bb.2:
; GFX1132GISEL-NEXT: v_dual_mov_b32 v0, s2 :: v_dual_mov_b32 v1, 0
@@ -1332,7 +1342,7 @@ define amdgpu_kernel void @divergent_value_dpp(ptr addrspace(1) %out) #0 {
; GFX1164DAGISEL-NEXT: s_or_saveexec_b64 s[2:3], -1
; GFX1164DAGISEL-NEXT: s_waitcnt lgkmcnt(0)
; GFX1164DAGISEL-NEXT: v_or_b32_e32 v1, v1, v2
-; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX1164DAGISEL-NEXT: v_readlane_b32 s4, v1, 63
; GFX1164DAGISEL-NEXT: s_mov_b64 exec, s[2:3]
; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v0, 0
@@ -1367,7 +1377,7 @@ define amdgpu_kernel void @divergent_value_dpp(ptr addrspace(1) %out) #0 {
; GFX1164GISEL-NEXT: s_mov_b64 exec, s[0:1]
; GFX1164GISEL-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
; GFX1164GISEL-NEXT: s_or_saveexec_b64 s[2:3], -1
-; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX1164GISEL-NEXT: v_readlane_b32 s4, v1, 63
; GFX1164GISEL-NEXT: s_mov_b64 exec, s[2:3]
; GFX1164GISEL-NEXT: v_mov_b32_e32 v0, s4
@@ -1394,7 +1404,7 @@ define amdgpu_kernel void @divergent_value_dpp(ptr addrspace(1) %out) #0 {
; GFX1132DAGISEL-NEXT: s_or_saveexec_b32 s2, -1
; GFX1132DAGISEL-NEXT: s_waitcnt lgkmcnt(0)
; GFX1132DAGISEL-NEXT: v_or_b32_e32 v1, v1, v2
-; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX1132DAGISEL-NEXT: v_readlane_b32 s3, v1, 31
; GFX1132DAGISEL-NEXT: s_mov_b32 exec_lo, s2
; GFX1132DAGISEL-NEXT: v_dual_mov_b32 v0, 0 :: v_dual_mov_b32 v3, s3
@@ -1419,7 +1429,7 @@ define amdgpu_kernel void @divergent_value_dpp(ptr addrspace(1) %out) #0 {
; GFX1132GISEL-NEXT: s_mov_b32 exec_lo, s0
; GFX1132GISEL-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
; GFX1132GISEL-NEXT: s_or_saveexec_b32 s2, -1
-; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX1132GISEL-NEXT: v_readlane_b32 s3, v1, 31
; GFX1132GISEL-NEXT: s_mov_b32 exec_lo, s2
; GFX1132GISEL-NEXT: v_dual_mov_b32 v0, s3 :: v_dual_mov_b32 v3, 0
@@ -2033,50 +2043,51 @@ define void @divergent_value_dpp_i64(ptr addrspace(1) %out, i64 %in) #0 {
; GFX1164DAGISEL-NEXT: scratch_store_b32 off, v7, s32 offset:12
; GFX1164DAGISEL-NEXT: scratch_store_b32 off, v8, s32 offset:16
; GFX1164DAGISEL-NEXT: s_mov_b64 exec, s[0:1]
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1164DAGISEL-NEXT: s_or_saveexec_b64 s[0:1], -1
-; GFX1164DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(VALU_DEP_3)
; GFX1164DAGISEL-NEXT: v_cndmask_b32_e64 v4, 0, v2, s[0:1]
; GFX1164DAGISEL-NEXT: v_cndmask_b32_e64 v5, 0, v3, s[0:1]
; GFX1164DAGISEL-NEXT: v_mbcnt_lo_u32_b32 v8, -1, 0
-; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v6, v4
; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v6, v4
; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v7, v5
-; GFX1164DAGISEL-NEXT: v_mbcnt_hi_u32_b32 v8, -1, v8
; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX1164DAGISEL-NEXT: v_mbcnt_hi_u32_b32 v8, -1, v8
; GFX1164DAGISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:1 row_mask:0xf bank_mask:0xf
-; GFX1164DAGISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:1 row_mask:0xf bank_mask:0xf
; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX1164DAGISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:1 row_mask:0xf bank_mask:0xf
; GFX1164DAGISEL-NEXT: v_add_nc_u32_e32 v8, 32, v8
-; GFX1164DAGISEL-NEXT: v_or_b32_e32 v4, v4, v6
; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX1164DAGISEL-NEXT: v_or_b32_e32 v4, v4, v6
; GFX1164DAGISEL-NEXT: v_or_b32_e32 v5, v5, v7
-; GFX1164DAGISEL-NEXT: v_mul_lo_u32 v8, 4, v8
; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX1164DAGISEL-NEXT: v_mul_lo_u32 v8, 4, v8
; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v6, v4
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v7, v5
-; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1164DAGISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:2 row_mask:0xf bank_mask:0xf
-; GFX1164DAGISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:2 row_mask:0xf bank_mask:0xf
; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX1164DAGISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:2 row_mask:0xf bank_mask:0xf
; GFX1164DAGISEL-NEXT: v_or_b32_e32 v4, v4, v6
-; GFX1164DAGISEL-NEXT: v_or_b32_e32 v5, v5, v7
; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX1164DAGISEL-NEXT: v_or_b32_e32 v5, v5, v7
; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v6, v4
-; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v7, v5
; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v7, v5
; GFX1164DAGISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:4 row_mask:0xf bank_mask:0xf
-; GFX1164DAGISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:4 row_mask:0xf bank_mask:0xf
; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX1164DAGISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:4 row_mask:0xf bank_mask:0xf
; GFX1164DAGISEL-NEXT: v_or_b32_e32 v4, v4, v6
-; GFX1164DAGISEL-NEXT: v_or_b32_e32 v5, v5, v7
; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX1164DAGISEL-NEXT: v_or_b32_e32 v5, v5, v7
; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v6, v4
-; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v7, v5
; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v7, v5
; GFX1164DAGISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:8 row_mask:0xf bank_mask:0xf
-; GFX1164DAGISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:8 row_mask:0xf bank_mask:0xf
; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX1164DAGISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:8 row_mask:0xf bank_mask:0xf
; GFX1164DAGISEL-NEXT: v_or_b32_e32 v4, v4, v6
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX1164DAGISEL-NEXT: v_or_b32_e32 v5, v5, v7
; GFX1164DAGISEL-NEXT: ds_swizzle_b32 v6, v4 offset:swizzle(BROADCAST,32,15)
; GFX1164DAGISEL-NEXT: ds_swizzle_b32 v7, v5 offset:swizzle(BROADCAST,32,15)
@@ -2094,6 +2105,7 @@ define void @divergent_value_dpp_i64(ptr addrspace(1) %out, i64 %in) #0 {
; GFX1164DAGISEL-NEXT: v_readlane_b32 s2, v4, 63
; GFX1164DAGISEL-NEXT: v_readlane_b32 s3, v5, 63
; GFX1164DAGISEL-NEXT: s_mov_b64 exec, s[0:1]
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v2, s2
; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v3, s3
; GFX1164DAGISEL-NEXT: global_store_b64 v[0:1], v[2:3], off
@@ -2120,50 +2132,51 @@ define void @divergent_value_dpp_i64(ptr addrspace(1) %out, i64 %in) #0 {
; GFX1164GISEL-NEXT: scratch_store_b32 off, v7, s32 offset:12
; GFX1164GISEL-NEXT: scratch_store_b32 off, v8, s32 offset:16
; GFX1164GISEL-NEXT: s_mov_b64 exec, s[0:1]
+; GFX1164GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1164GISEL-NEXT: s_or_saveexec_b64 s[0:1], -1
-; GFX1164GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(VALU_DEP_3)
; GFX1164GISEL-NEXT: v_cndmask_b32_e64 v4, 0, v2, s[0:1]
; GFX1164GISEL-NEXT: v_cndmask_b32_e64 v5, 0, v3, s[0:1]
; GFX1164GISEL-NEXT: v_mbcnt_lo_u32_b32 v8, -1, 0
-; GFX1164GISEL-NEXT: v_mov_b32_e32 v6, v4
; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX1164GISEL-NEXT: v_mov_b32_e32 v6, v4
; GFX1164GISEL-NEXT: v_mov_b32_e32 v7, v5
-; GFX1164GISEL-NEXT: v_mbcnt_hi_u32_b32 v8, -1, v8
; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX1164GISEL-NEXT: v_mbcnt_hi_u32_b32 v8, -1, v8
; GFX1164GISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:1 row_mask:0xf bank_mask:0xf
-; GFX1164GISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:1 row_mask:0xf bank_mask:0xf
; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX1164GISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:1 row_mask:0xf bank_mask:0xf
; GFX1164GISEL-NEXT: v_add_nc_u32_e32 v8, 32, v8
-; GFX1164GISEL-NEXT: v_or_b32_e32 v4, v4, v6
; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX1164GISEL-NEXT: v_or_b32_e32 v4, v4, v6
; GFX1164GISEL-NEXT: v_or_b32_e32 v5, v5, v7
-; GFX1164GISEL-NEXT: v_mul_lo_u32 v8, 4, v8
; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX1164GISEL-NEXT: v_mul_lo_u32 v8, 4, v8
; GFX1164GISEL-NEXT: v_mov_b32_e32 v6, v4
+; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1164GISEL-NEXT: v_mov_b32_e32 v7, v5
-; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1164GISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:2 row_mask:0xf bank_mask:0xf
-; GFX1164GISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:2 row_mask:0xf bank_mask:0xf
; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX1164GISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:2 row_mask:0xf bank_mask:0xf
; GFX1164GISEL-NEXT: v_or_b32_e32 v4, v4, v6
-; GFX1164GISEL-NEXT: v_or_b32_e32 v5, v5, v7
; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX1164GISEL-NEXT: v_or_b32_e32 v5, v5, v7
; GFX1164GISEL-NEXT: v_mov_b32_e32 v6, v4
-; GFX1164GISEL-NEXT: v_mov_b32_e32 v7, v5
; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX1164GISEL-NEXT: v_mov_b32_e32 v7, v5
; GFX1164GISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:4 row_mask:0xf bank_mask:0xf
-; GFX1164GISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:4 row_mask:0xf bank_mask:0xf
; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX1164GISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:4 row_mask:0xf bank_mask:0xf
; GFX1164GISEL-NEXT: v_or_b32_e32 v4, v4, v6
-; GFX1164GISEL-NEXT: v_or_b32_e32 v5, v5, v7
; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX1164GISEL-NEXT: v_or_b32_e32 v5, v5, v7
; GFX1164GISEL-NEXT: v_mov_b32_e32 v6, v4
-; GFX1164GISEL-NEXT: v_mov_b32_e32 v7, v5
; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX1164GISEL-NEXT: v_mov_b32_e32 v7, v5
; GFX1164GISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:8 row_mask:0xf bank_mask:0xf
-; GFX1164GISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:8 row_mask:0xf bank_mask:0xf
; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX1164GISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:8 row_mask:0xf bank_mask:0xf
; GFX1164GISEL-NEXT: v_or_b32_e32 v4, v4, v6
+; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX1164GISEL-NEXT: v_or_b32_e32 v5, v5, v7
; GFX1164GISEL-NEXT: ds_swizzle_b32 v6, v4 offset:swizzle(BROADCAST,32,15)
; GFX1164GISEL-NEXT: ds_swizzle_b32 v7, v5 offset:swizzle(BROADCAST,32,15)
@@ -2181,6 +2194,7 @@ define void @divergent_value_dpp_i64(ptr addrspace(1) %out, i64 %in) #0 {
; GFX1164GISEL-NEXT: v_readlane_b32 s2, v4, 63
; GFX1164GISEL-NEXT: v_readlane_b32 s3, v5, 63
; GFX1164GISEL-NEXT: s_mov_b64 exec, s[0:1]
+; GFX1164GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164GISEL-NEXT: v_mov_b32_e32 v2, s2
; GFX1164GISEL-NEXT: v_mov_b32_e32 v3, s3
; GFX1164GISEL-NEXT: global_store_b64 v[0:1], v[2:3], off
@@ -2206,39 +2220,39 @@ define void @divergent_value_dpp_i64(ptr addrspace(1) %out, i64 %in) #0 {
; GFX1132DAGISEL-NEXT: scratch_store_b32 off, v6, s32 offset:8
; GFX1132DAGISEL-NEXT: scratch_store_b32 off, v7, s32 offset:12
; GFX1132DAGISEL-NEXT: s_mov_b32 exec_lo, s0
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1132DAGISEL-NEXT: s_or_saveexec_b32 s2, -1
-; GFX1132DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX1132DAGISEL-NEXT: v_cndmask_b32_e64 v4, 0, v2, s2
; GFX1132DAGISEL-NEXT: v_cndmask_b32_e64 v5, 0, v3, s2
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132DAGISEL-NEXT: v_dual_mov_b32 v6, v4 :: v_dual_mov_b32 v7, v5
-; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1132DAGISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:1 row_mask:0xf bank_mask:0xf
-; GFX1132DAGISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:1 row_mask:0xf bank_mask:0xf
; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX1132DAGISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:1 row_mask:0xf bank_mask:0xf
; GFX1132DAGISEL-NEXT: v_or_b32_e32 v4, v4, v6
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132DAGISEL-NEXT: v_or_b32_e32 v5, v5, v7
-; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132DAGISEL-NEXT: v_dual_mov_b32 v6, v4 :: v_dual_mov_b32 v7, v5
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1132DAGISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:2 row_mask:0xf bank_mask:0xf
-; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1132DAGISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:2 row_mask:0xf bank_mask:0xf
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1132DAGISEL-NEXT: v_or_b32_e32 v4, v4, v6
-; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132DAGISEL-NEXT: v_or_b32_e32 v5, v5, v7
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132DAGISEL-NEXT: v_dual_mov_b32 v6, v4 :: v_dual_mov_b32 v7, v5
-; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1132DAGISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:4 row_mask:0xf bank_mask:0xf
-; GFX1132DAGISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:4 row_mask:0xf bank_mask:0xf
; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX1132DAGISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:4 row_mask:0xf bank_mask:0xf
; GFX1132DAGISEL-NEXT: v_or_b32_e32 v4, v4, v6
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132DAGISEL-NEXT: v_or_b32_e32 v5, v5, v7
-; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132DAGISEL-NEXT: v_dual_mov_b32 v6, v4 :: v_dual_mov_b32 v7, v5
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1132DAGISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:8 row_mask:0xf bank_mask:0xf
-; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1132DAGISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:8 row_mask:0xf bank_mask:0xf
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1132DAGISEL-NEXT: v_or_b32_e32 v4, v4, v6
-; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX1132DAGISEL-NEXT: v_or_b32_e32 v5, v5, v7
; GFX1132DAGISEL-NEXT: ds_swizzle_b32 v6, v4 offset:swizzle(BROADCAST,32,15)
; GFX1132DAGISEL-NEXT: ds_swizzle_b32 v7, v5 offset:swizzle(BROADCAST,32,15)
@@ -2250,6 +2264,7 @@ define void @divergent_value_dpp_i64(ptr addrspace(1) %out, i64 %in) #0 {
; GFX1132DAGISEL-NEXT: v_readlane_b32 s0, v4, 31
; GFX1132DAGISEL-NEXT: v_readlane_b32 s1, v5, 31
; GFX1132DAGISEL-NEXT: s_mov_b32 exec_lo, s2
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1132DAGISEL-NEXT: v_dual_mov_b32 v3, s1 :: v_dual_mov_b32 v2, s0
; GFX1132DAGISEL-NEXT: global_store_b64 v[0:1], v[2:3], off
; GFX1132DAGISEL-NEXT: s_xor_saveexec_b32 s0, -1
@@ -2272,39 +2287,39 @@ define void @divergent_value_dpp_i64(ptr addrspace(1) %out, i64 %in) #0 {
; GFX1132GISEL-NEXT: scratch_store_b32 off, v6, s32 offset:8
; GFX1132GISEL-NEXT: scratch_store_b32 off, v7, s32 offset:12
; GFX1132GISEL-NEXT: s_mov_b32 exec_lo, s0
+; GFX1132GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1132GISEL-NEXT: s_or_saveexec_b32 s2, -1
-; GFX1132GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX1132GISEL-NEXT: v_cndmask_b32_e64 v4, 0, v2, s2
; GFX1132GISEL-NEXT: v_cndmask_b32_e64 v5, 0, v3, s2
+; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132GISEL-NEXT: v_dual_mov_b32 v6, v4 :: v_dual_mov_b32 v7, v5
-; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1132GISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:1 row_mask:0xf bank_mask:0xf
-; GFX1132GISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:1 row_mask:0xf bank_mask:0xf
; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX1132GISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:1 row_mask:0xf bank_mask:0xf
; GFX1132GISEL-NEXT: v_or_b32_e32 v4, v4, v6
+; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132GISEL-NEXT: v_or_b32_e32 v5, v5, v7
-; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132GISEL-NEXT: v_dual_mov_b32 v6, v4 :: v_dual_mov_b32 v7, v5
+; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1132GISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:2 row_mask:0xf bank_mask:0xf
-; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1132GISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:2 row_mask:0xf bank_mask:0xf
+; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1132GISEL-NEXT: v_or_b32_e32 v4, v4, v6
-; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132GISEL-NEXT: v_or_b32_e32 v5, v5, v7
+; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132GISEL-NEXT: v_dual_mov_b32 v6, v4 :: v_dual_mov_b32 v7, v5
-; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1132GISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:4 row_mask:0xf bank_mask:0xf
-; GFX1132GISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:4 row_mask:0xf bank_mask:0xf
; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX1132GISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:4 row_mask:0xf bank_mask:0xf
; GFX1132GISEL-NEXT: v_or_b32_e32 v4, v4, v6
+; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132GISEL-NEXT: v_or_b32_e32 v5, v5, v7
-; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132GISEL-NEXT: v_dual_mov_b32 v6, v4 :: v_dual_mov_b32 v7, v5
+; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1132GISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:8 row_mask:0xf bank_mask:0xf
-; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1132GISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:8 row_mask:0xf bank_mask:0xf
+; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1132GISEL-NEXT: v_or_b32_e32 v4, v4, v6
-; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX1132GISEL-NEXT: v_or_b32_e32 v5, v5, v7
; GFX1132GISEL-NEXT: ds_swizzle_b32 v6, v4 offset:swizzle(BROADCAST,32,15)
; GFX1132GISEL-NEXT: ds_swizzle_b32 v7, v5 offset:swizzle(BROADCAST,32,15)
@@ -2316,6 +2331,7 @@ define void @divergent_value_dpp_i64(ptr addrspace(1) %out, i64 %in) #0 {
; GFX1132GISEL-NEXT: v_readlane_b32 s0, v4, 31
; GFX1132GISEL-NEXT: v_readlane_b32 s1, v5, 31
; GFX1132GISEL-NEXT: s_mov_b32 exec_lo, s2
+; GFX1132GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1132GISEL-NEXT: v_dual_mov_b32 v3, s1 :: v_dual_mov_b32 v2, s0
; GFX1132GISEL-NEXT: global_store_b64 v[0:1], v[2:3], off
; GFX1132GISEL-NEXT: s_xor_saveexec_b32 s0, -1
@@ -2605,7 +2621,7 @@ define amdgpu_kernel void @default_stratergy(ptr addrspace(1) %out) #0 {
; GFX1164DAGISEL-NEXT: s_or_saveexec_b64 s[2:3], -1
; GFX1164DAGISEL-NEXT: s_waitcnt lgkmcnt(0)
; GFX1164DAGISEL-NEXT: v_or_b32_e32 v1, v1, v2
-; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX1164DAGISEL-NEXT: v_readlane_b32 s4, v1, 63
; GFX1164DAGISEL-NEXT: s_mov_b64 exec, s[2:3]
; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v0, 0
@@ -2640,7 +2656,7 @@ define amdgpu_kernel void @default_stratergy(ptr addrspace(1) %out) #0 {
; GFX1164GISEL-NEXT: s_mov_b64 exec, s[0:1]
; GFX1164GISEL-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
; GFX1164GISEL-NEXT: s_or_saveexec_b64 s[2:3], -1
-; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX1164GISEL-NEXT: v_readlane_b32 s4, v1, 63
; GFX1164GISEL-NEXT: s_mov_b64 exec, s[2:3]
; GFX1164GISEL-NEXT: v_mov_b32_e32 v0, s4
@@ -2667,7 +2683,7 @@ define amdgpu_kernel void @default_stratergy(ptr addrspace(1) %out) #0 {
; GFX1132DAGISEL-NEXT: s_or_saveexec_b32 s2, -1
; GFX1132DAGISEL-NEXT: s_waitcnt lgkmcnt(0)
; GFX1132DAGISEL-NEXT: v_or_b32_e32 v1, v1, v2
-; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX1132DAGISEL-NEXT: v_readlane_b32 s3, v1, 31
; GFX1132DAGISEL-NEXT: s_mov_b32 exec_lo, s2
; GFX1132DAGISEL-NEXT: v_dual_mov_b32 v0, 0 :: v_dual_mov_b32 v3, s3
@@ -2692,7 +2708,7 @@ define amdgpu_kernel void @default_stratergy(ptr addrspace(1) %out) #0 {
; GFX1132GISEL-NEXT: s_mov_b32 exec_lo, s0
; GFX1132GISEL-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
; GFX1132GISEL-NEXT: s_or_saveexec_b32 s2, -1
-; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX1132GISEL-NEXT: v_readlane_b32 s3, v1, 31
; GFX1132GISEL-NEXT: s_mov_b32 exec_lo, s2
; GFX1132GISEL-NEXT: v_dual_mov_b32 v0, s3 :: v_dual_mov_b32 v3, 0
@@ -3074,7 +3090,7 @@ define amdgpu_kernel void @divergent_cfg(ptr addrspace(1) %out, i32 %in) #0 {
; GFX1164DAGISEL-NEXT: v_and_b32_e32 v0, 0x3ff, v0
; GFX1164DAGISEL-NEXT: s_mov_b64 s[0:1], exec
; GFX1164DAGISEL-NEXT: ; implicit-def: $sgpr2
-; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164DAGISEL-NEXT: v_cmpx_lt_u32_e32 15, v0
; GFX1164DAGISEL-NEXT: s_xor_b64 s[0:1], exec, s[0:1]
; GFX1164DAGISEL-NEXT: ; %bb.1: ; %else
@@ -3085,13 +3101,14 @@ define amdgpu_kernel void @divergent_cfg(ptr addrspace(1) %out, i32 %in) #0 {
; GFX1164DAGISEL-NEXT: s_waitcnt lgkmcnt(0)
; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v1, s2
; GFX1164DAGISEL-NEXT: s_xor_b64 exec, exec, s[0:1]
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164DAGISEL-NEXT: s_cbranch_execz .LBB8_6
; GFX1164DAGISEL-NEXT: ; %bb.3: ; %if
; GFX1164DAGISEL-NEXT: s_mov_b64 s[2:3], exec
; GFX1164DAGISEL-NEXT: s_mov_b32 s6, 0
; GFX1164DAGISEL-NEXT: .LBB8_4: ; =>This Inner Loop Header: Depth=1
; GFX1164DAGISEL-NEXT: s_ctz_i32_b64 s7, s[2:3]
-; GFX1164DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_3) | instid1(SALU_CYCLE_1)
; GFX1164DAGISEL-NEXT: v_readlane_b32 s8, v0, s7
; GFX1164DAGISEL-NEXT: s_bitset0_b64 s[2:3], s7
; GFX1164DAGISEL-NEXT: s_or_b32 s6, s6, s8
@@ -3112,7 +3129,7 @@ define amdgpu_kernel void @divergent_cfg(ptr addrspace(1) %out, i32 %in) #0 {
; GFX1164GISEL-NEXT: v_and_b32_e32 v0, 0x3ff, v0
; GFX1164GISEL-NEXT: s_mov_b64 s[0:1], exec
; GFX1164GISEL-NEXT: ; implicit-def: $sgpr2
-; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164GISEL-NEXT: v_cmpx_le_u32_e32 16, v0
; GFX1164GISEL-NEXT: s_xor_b64 s[0:1], exec, s[0:1]
; GFX1164GISEL-NEXT: s_cbranch_execz .LBB8_2
@@ -3125,13 +3142,14 @@ define amdgpu_kernel void @divergent_cfg(ptr addrspace(1) %out, i32 %in) #0 {
; GFX1164GISEL-NEXT: s_or_saveexec_b64 s[0:1], s[0:1]
; GFX1164GISEL-NEXT: v_mov_b32_e32 v1, s2
; GFX1164GISEL-NEXT: s_xor_b64 exec, exec, s[0:1]
+; GFX1164GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164GISEL-NEXT: s_cbranch_execz .LBB8_6
; GFX1164GISEL-NEXT: ; %bb.3: ; %if
; GFX1164GISEL-NEXT: s_mov_b64 s[2:3], exec
; GFX1164GISEL-NEXT: s_mov_b32 s6, 0
; GFX1164GISEL-NEXT: .LBB8_4: ; =>This Inner Loop Header: Depth=1
; GFX1164GISEL-NEXT: s_ctz_i32_b64 s7, s[2:3]
-; GFX1164GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1164GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_3) | instid1(SALU_CYCLE_1)
; GFX1164GISEL-NEXT: v_readlane_b32 s8, v0, s7
; GFX1164GISEL-NEXT: s_bitset0_b64 s[2:3], s7
; GFX1164GISEL-NEXT: s_or_b32 s6, s6, s8
@@ -3152,13 +3170,14 @@ define amdgpu_kernel void @divergent_cfg(ptr addrspace(1) %out, i32 %in) #0 {
; GFX1132DAGISEL-NEXT: v_and_b32_e32 v0, 0x3ff, v0
; GFX1132DAGISEL-NEXT: s_mov_b32 s0, exec_lo
; GFX1132DAGISEL-NEXT: ; implicit-def: $sgpr1
-; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132DAGISEL-NEXT: v_cmpx_lt_u32_e32 15, v0
; GFX1132DAGISEL-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX1132DAGISEL-NEXT: ; %bb.1: ; %else
; GFX1132DAGISEL-NEXT: s_load_b32 s1, s[4:5], 0x2c
; GFX1132DAGISEL-NEXT: ; implicit-def: $vgpr0
; GFX1132DAGISEL-NEXT: ; %bb.2: ; %Flow
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_3) | instid1(SALU_CYCLE_1)
; GFX1132DAGISEL-NEXT: s_or_saveexec_b32 s0, s0
; GFX1132DAGISEL-NEXT: s_waitcnt lgkmcnt(0)
; GFX1132DAGISEL-NEXT: v_mov_b32_e32 v1, s1
@@ -3169,7 +3188,7 @@ define amdgpu_kernel void @divergent_cfg(ptr addrspace(1) %out, i32 %in) #0 {
; GFX1132DAGISEL-NEXT: s_mov_b32 s1, 0
; GFX1132DAGISEL-NEXT: .LBB8_4: ; =>This Inner Loop Header: Depth=1
; GFX1132DAGISEL-NEXT: s_ctz_i32_b32 s3, s2
-; GFX1132DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_3) | instid1(SALU_CYCLE_1)
; GFX1132DAGISEL-NEXT: v_readlane_b32 s6, v0, s3
; GFX1132DAGISEL-NEXT: s_bitset0_b32 s2, s3
; GFX1132DAGISEL-NEXT: s_or_b32 s1, s1, s6
@@ -3190,9 +3209,10 @@ define amdgpu_kernel void @divergent_cfg(ptr addrspace(1) %out, i32 %in) #0 {
; GFX1132GISEL-NEXT: v_and_b32_e32 v0, 0x3ff, v0
; GFX1132GISEL-NEXT: s_mov_b32 s0, exec_lo
; GFX1132GISEL-NEXT: ; implicit-def: $sgpr1
-; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132GISEL-NEXT: v_cmpx_le_u32_e32 16, v0
; GFX1132GISEL-NEXT: s_xor_b32 s0, exec_lo, s0
+; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1132GISEL-NEXT: s_cbranch_execz .LBB8_2
; GFX1132GISEL-NEXT: ; %bb.1: ; %else
; GFX1132GISEL-NEXT: s_load_b32 s1, s[4:5], 0x2c
@@ -3203,13 +3223,14 @@ define amdgpu_kernel void @divergent_cfg(ptr addrspace(1) %out, i32 %in) #0 {
; GFX1132GISEL-NEXT: s_or_saveexec_b32 s0, s0
; GFX1132GISEL-NEXT: v_mov_b32_e32 v1, s1
; GFX1132GISEL-NEXT: s_xor_b32 exec_lo, exec_lo, s0
+; GFX1132GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1132GISEL-NEXT: s_cbranch_execz .LBB8_6
; GFX1132GISEL-NEXT: ; %bb.3: ; %if
; GFX1132GISEL-NEXT: s_mov_b32 s2, exec_lo
; GFX1132GISEL-NEXT: s_mov_b32 s1, 0
; GFX1132GISEL-NEXT: .LBB8_4: ; =>This Inner Loop Header: Depth=1
; GFX1132GISEL-NEXT: s_ctz_i32_b32 s3, s2
-; GFX1132GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1132GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_3) | instid1(SALU_CYCLE_1)
; GFX1132GISEL-NEXT: v_readlane_b32 s6, v0, s3
; GFX1132GISEL-NEXT: s_bitset0_b32 s2, s3
; GFX1132GISEL-NEXT: s_or_b32 s1, s1, s6
@@ -3593,6 +3614,7 @@ define void @divergent_value_i64(ptr addrspace(1) %out, i64 %id.x) #0 {
; GFX1164DAGISEL-NEXT: s_bitset0_b64 s[2:3], s6
; GFX1164DAGISEL-NEXT: s_or_b64 s[0:1], s[0:1], s[4:5]
; GFX1164DAGISEL-NEXT: s_cmp_lg_u64 s[2:3], 0
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164DAGISEL-NEXT: s_cbranch_scc1 .LBB10_1
; GFX1164DAGISEL-NEXT: ; %bb.2:
; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v3, s1
@@ -3613,6 +3635,7 @@ define void @divergent_value_i64(ptr addrspace(1) %out, i64 %id.x) #0 {
; GFX1164GISEL-NEXT: s_bitset0_b64 s[2:3], s6
; GFX1164GISEL-NEXT: s_or_b64 s[0:1], s[0:1], s[4:5]
; GFX1164GISEL-NEXT: s_cmp_lg_u64 s[2:3], 0
+; GFX1164GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164GISEL-NEXT: s_cbranch_scc1 .LBB10_1
; GFX1164GISEL-NEXT: ; %bb.2:
; GFX1164GISEL-NEXT: v_mov_b32_e32 v3, s1
@@ -3633,6 +3656,7 @@ define void @divergent_value_i64(ptr addrspace(1) %out, i64 %id.x) #0 {
; GFX1132DAGISEL-NEXT: s_bitset0_b32 s2, s3
; GFX1132DAGISEL-NEXT: s_or_b64 s[0:1], s[0:1], s[4:5]
; GFX1132DAGISEL-NEXT: s_cmp_lg_u32 s2, 0
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1132DAGISEL-NEXT: s_cbranch_scc1 .LBB10_1
; GFX1132DAGISEL-NEXT: ; %bb.2:
; GFX1132DAGISEL-NEXT: v_dual_mov_b32 v3, s1 :: v_dual_mov_b32 v2, s0
@@ -3652,6 +3676,7 @@ define void @divergent_value_i64(ptr addrspace(1) %out, i64 %id.x) #0 {
; GFX1132GISEL-NEXT: s_bitset0_b32 s2, s3
; GFX1132GISEL-NEXT: s_or_b64 s[0:1], s[0:1], s[4:5]
; GFX1132GISEL-NEXT: s_cmp_lg_u32 s2, 0
+; GFX1132GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1132GISEL-NEXT: s_cbranch_scc1 .LBB10_1
; GFX1132GISEL-NEXT: ; %bb.2:
; GFX1132GISEL-NEXT: v_dual_mov_b32 v3, s1 :: v_dual_mov_b32 v2, s0
@@ -3924,19 +3949,22 @@ define amdgpu_kernel void @divergent_cfg_i64(ptr addrspace(1) %out, i64 %in, i64
; GFX1164DAGISEL-NEXT: s_load_b64 s[4:5], s[4:5], 0x34
; GFX1164DAGISEL-NEXT: v_and_b32_e32 v0, 0x3ff, v0
; GFX1164DAGISEL-NEXT: s_mov_b64 s[6:7], exec
-; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164DAGISEL-NEXT: v_cmpx_lt_u32_e32 15, v0
; GFX1164DAGISEL-NEXT: s_xor_b64 s[6:7], exec, s[6:7]
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164DAGISEL-NEXT: s_or_saveexec_b64 s[6:7], s[6:7]
; GFX1164DAGISEL-NEXT: s_waitcnt lgkmcnt(0)
; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v0, s2
; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v1, s3
; GFX1164DAGISEL-NEXT: s_xor_b64 exec, exec, s[6:7]
; GFX1164DAGISEL-NEXT: ; %bb.1: ; %if
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v0, s4
; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v1, s5
; GFX1164DAGISEL-NEXT: ; %bb.2: ; %endif
; GFX1164DAGISEL-NEXT: s_or_b64 exec, exec, s[6:7]
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v2, 0
; GFX1164DAGISEL-NEXT: global_store_b64 v2, v[0:1], s[0:1]
; GFX1164DAGISEL-NEXT: s_endpgm
@@ -3947,7 +3975,7 @@ define amdgpu_kernel void @divergent_cfg_i64(ptr addrspace(1) %out, i64 %in, i64
; GFX1164GISEL-NEXT: v_and_b32_e32 v0, 0x3ff, v0
; GFX1164GISEL-NEXT: s_mov_b64 s[8:9], exec
; GFX1164GISEL-NEXT: ; implicit-def: $sgpr6_sgpr7
-; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164GISEL-NEXT: v_cmpx_le_u32_e32 16, v0
; GFX1164GISEL-NEXT: s_xor_b64 s[8:9], exec, s[8:9]
; GFX1164GISEL-NEXT: s_cbranch_execz .LBB11_2
@@ -3960,6 +3988,7 @@ define amdgpu_kernel void @divergent_cfg_i64(ptr addrspace(1) %out, i64 %in, i64
; GFX1164GISEL-NEXT: v_mov_b32_e32 v0, s6
; GFX1164GISEL-NEXT: v_mov_b32_e32 v1, s7
; GFX1164GISEL-NEXT: s_xor_b64 exec, exec, s[2:3]
+; GFX1164GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164GISEL-NEXT: s_cbranch_execz .LBB11_4
; GFX1164GISEL-NEXT: ; %bb.3: ; %if
; GFX1164GISEL-NEXT: s_load_b64 s[4:5], s[4:5], 0x34
@@ -3970,6 +3999,7 @@ define amdgpu_kernel void @divergent_cfg_i64(ptr addrspace(1) %out, i64 %in, i64
; GFX1164GISEL-NEXT: v_mov_b32_e32 v1, s5
; GFX1164GISEL-NEXT: .LBB11_4: ; %endif
; GFX1164GISEL-NEXT: s_or_b64 exec, exec, s[2:3]
+; GFX1164GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164GISEL-NEXT: v_mov_b32_e32 v2, 0
; GFX1164GISEL-NEXT: global_store_b64 v2, v[0:1], s[0:1]
; GFX1164GISEL-NEXT: s_endpgm
@@ -3981,17 +4011,20 @@ define amdgpu_kernel void @divergent_cfg_i64(ptr addrspace(1) %out, i64 %in, i64
; GFX1132DAGISEL-NEXT: s_load_b64 s[4:5], s[4:5], 0x34
; GFX1132DAGISEL-NEXT: v_and_b32_e32 v0, 0x3ff, v0
; GFX1132DAGISEL-NEXT: s_mov_b32 s6, exec_lo
-; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132DAGISEL-NEXT: v_cmpx_lt_u32_e32 15, v0
; GFX1132DAGISEL-NEXT: s_xor_b32 s6, exec_lo, s6
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX1132DAGISEL-NEXT: s_or_saveexec_b32 s6, s6
; GFX1132DAGISEL-NEXT: s_waitcnt lgkmcnt(0)
; GFX1132DAGISEL-NEXT: v_dual_mov_b32 v0, s2 :: v_dual_mov_b32 v1, s3
; GFX1132DAGISEL-NEXT: s_xor_b32 exec_lo, exec_lo, s6
; GFX1132DAGISEL-NEXT: ; %bb.1: ; %if
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1132DAGISEL-NEXT: v_dual_mov_b32 v0, s4 :: v_dual_mov_b32 v1, s5
; GFX1132DAGISEL-NEXT: ; %bb.2: ; %endif
; GFX1132DAGISEL-NEXT: s_or_b32 exec_lo, exec_lo, s6
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1132DAGISEL-NEXT: v_mov_b32_e32 v2, 0
; GFX1132DAGISEL-NEXT: global_store_b64 v2, v[0:1], s[0:1]
; GFX1132DAGISEL-NEXT: s_endpgm
@@ -4002,9 +4035,10 @@ define amdgpu_kernel void @divergent_cfg_i64(ptr addrspace(1) %out, i64 %in, i64
; GFX1132GISEL-NEXT: v_and_b32_e32 v0, 0x3ff, v0
; GFX1132GISEL-NEXT: s_mov_b32 s8, exec_lo
; GFX1132GISEL-NEXT: ; implicit-def: $sgpr6_sgpr7
-; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132GISEL-NEXT: v_cmpx_le_u32_e32 16, v0
; GFX1132GISEL-NEXT: s_xor_b32 s8, exec_lo, s8
+; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1132GISEL-NEXT: s_cbranch_execz .LBB11_2
; GFX1132GISEL-NEXT: ; %bb.1: ; %else
; GFX1132GISEL-NEXT: s_waitcnt lgkmcnt(0)
@@ -4014,6 +4048,7 @@ define amdgpu_kernel void @divergent_cfg_i64(ptr addrspace(1) %out, i64 %in, i64
; GFX1132GISEL-NEXT: s_or_saveexec_b32 s2, s8
; GFX1132GISEL-NEXT: v_dual_mov_b32 v0, s6 :: v_dual_mov_b32 v1, s7
; GFX1132GISEL-NEXT: s_xor_b32 exec_lo, exec_lo, s2
+; GFX1132GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1132GISEL-NEXT: s_cbranch_execz .LBB11_4
; GFX1132GISEL-NEXT: ; %bb.3: ; %if
; GFX1132GISEL-NEXT: s_load_b64 s[4:5], s[4:5], 0x34
@@ -4023,6 +4058,7 @@ define amdgpu_kernel void @divergent_cfg_i64(ptr addrspace(1) %out, i64 %in, i64
; GFX1132GISEL-NEXT: v_dual_mov_b32 v0, s4 :: v_dual_mov_b32 v1, s5
; GFX1132GISEL-NEXT: .LBB11_4: ; %endif
; GFX1132GISEL-NEXT: s_or_b32 exec_lo, exec_lo, s2
+; GFX1132GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1132GISEL-NEXT: v_mov_b32_e32 v2, 0
; GFX1132GISEL-NEXT: global_store_b64 v2, v[0:1], s[0:1]
; GFX1132GISEL-NEXT: s_endpgm
diff --git a/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.reduce.sub.ll b/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.reduce.sub.ll
index ae2dcfac5a4d02..c493dc7c069a98 100644
--- a/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.reduce.sub.ll
+++ b/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.reduce.sub.ll
@@ -511,6 +511,7 @@ define void @divergent_value_i16(ptr addrspace(1) %out, i16 %in) {
; GFX1164DAGISEL-NEXT: s_bitset0_b64 s[0:1], s3
; GFX1164DAGISEL-NEXT: s_sub_i32 s2, s2, s4
; GFX1164DAGISEL-NEXT: s_cmp_lg_u64 s[0:1], 0
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164DAGISEL-NEXT: s_cbranch_scc1 .LBB1_1
; GFX1164DAGISEL-NEXT: ; %bb.2:
; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v2, s2
@@ -530,6 +531,7 @@ define void @divergent_value_i16(ptr addrspace(1) %out, i16 %in) {
; GFX1164GISEL-NEXT: s_bitset0_b64 s[0:1], s3
; GFX1164GISEL-NEXT: s_sub_i32 s2, s2, s4
; GFX1164GISEL-NEXT: s_cmp_lg_u64 s[0:1], 0
+; GFX1164GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164GISEL-NEXT: s_cbranch_scc1 .LBB1_1
; GFX1164GISEL-NEXT: ; %bb.2:
; GFX1164GISEL-NEXT: v_mov_b32_e32 v2, s2
@@ -549,6 +551,7 @@ define void @divergent_value_i16(ptr addrspace(1) %out, i16 %in) {
; GFX1132DAGISEL-NEXT: s_bitset0_b32 s1, s2
; GFX1132DAGISEL-NEXT: s_sub_i32 s0, s0, s3
; GFX1132DAGISEL-NEXT: s_cmp_lg_u32 s1, 0
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1132DAGISEL-NEXT: s_cbranch_scc1 .LBB1_1
; GFX1132DAGISEL-NEXT: ; %bb.2:
; GFX1132DAGISEL-NEXT: v_mov_b32_e32 v2, s0
@@ -568,6 +571,7 @@ define void @divergent_value_i16(ptr addrspace(1) %out, i16 %in) {
; GFX1132GISEL-NEXT: s_bitset0_b32 s1, s2
; GFX1132GISEL-NEXT: s_sub_i32 s0, s0, s3
; GFX1132GISEL-NEXT: s_cmp_lg_u32 s1, 0
+; GFX1132GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1132GISEL-NEXT: s_cbranch_scc1 .LBB1_1
; GFX1132GISEL-NEXT: ; %bb.2:
; GFX1132GISEL-NEXT: v_mov_b32_e32 v2, s0
@@ -1043,6 +1047,7 @@ define amdgpu_kernel void @divergent_value(ptr addrspace(1) %out) #0 {
; GFX1164DAGISEL-NEXT: s_bitset0_b64 s[2:3], s5
; GFX1164DAGISEL-NEXT: s_sub_i32 s4, s4, s6
; GFX1164DAGISEL-NEXT: s_cmp_lg_u64 s[2:3], 0
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164DAGISEL-NEXT: s_cbranch_scc1 .LBB3_1
; GFX1164DAGISEL-NEXT: ; %bb.2:
; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v0, s4
@@ -1063,6 +1068,7 @@ define amdgpu_kernel void @divergent_value(ptr addrspace(1) %out) #0 {
; GFX1164GISEL-NEXT: s_bitset0_b64 s[2:3], s5
; GFX1164GISEL-NEXT: s_sub_i32 s4, s4, s6
; GFX1164GISEL-NEXT: s_cmp_lg_u64 s[2:3], 0
+; GFX1164GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164GISEL-NEXT: s_cbranch_scc1 .LBB3_1
; GFX1164GISEL-NEXT: ; %bb.2:
; GFX1164GISEL-NEXT: v_mov_b32_e32 v0, s4
@@ -1084,6 +1090,7 @@ define amdgpu_kernel void @divergent_value(ptr addrspace(1) %out) #0 {
; GFX1132DAGISEL-NEXT: s_bitset0_b32 s3, s4
; GFX1132DAGISEL-NEXT: s_sub_i32 s2, s2, s5
; GFX1132DAGISEL-NEXT: s_cmp_lg_u32 s3, 0
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1132DAGISEL-NEXT: s_cbranch_scc1 .LBB3_1
; GFX1132DAGISEL-NEXT: ; %bb.2:
; GFX1132DAGISEL-NEXT: v_mov_b32_e32 v0, s2
@@ -1104,6 +1111,7 @@ define amdgpu_kernel void @divergent_value(ptr addrspace(1) %out) #0 {
; GFX1132GISEL-NEXT: s_bitset0_b32 s3, s4
; GFX1132GISEL-NEXT: s_sub_i32 s2, s2, s5
; GFX1132GISEL-NEXT: s_cmp_lg_u32 s3, 0
+; GFX1132GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1132GISEL-NEXT: s_cbranch_scc1 .LBB3_1
; GFX1132GISEL-NEXT: ; %bb.2:
; GFX1132GISEL-NEXT: v_dual_mov_b32 v0, s2 :: v_dual_mov_b32 v1, 0
@@ -2398,50 +2406,51 @@ define void @divergent_value_dpp_i64(ptr addrspace(1) %out, i64 %in) #0 {
; GFX1164DAGISEL-NEXT: scratch_store_b32 off, v7, s32 offset:12
; GFX1164DAGISEL-NEXT: scratch_store_b32 off, v8, s32 offset:16
; GFX1164DAGISEL-NEXT: s_mov_b64 exec, s[0:1]
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1164DAGISEL-NEXT: s_or_saveexec_b64 s[0:1], -1
-; GFX1164DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(VALU_DEP_3)
; GFX1164DAGISEL-NEXT: v_cndmask_b32_e64 v4, 0, v2, s[0:1]
; GFX1164DAGISEL-NEXT: v_cndmask_b32_e64 v5, 0, v3, s[0:1]
; GFX1164DAGISEL-NEXT: v_mbcnt_lo_u32_b32 v8, -1, 0
-; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v6, v4
; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v6, v4
; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v7, v5
-; GFX1164DAGISEL-NEXT: v_mbcnt_hi_u32_b32 v8, -1, v8
; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX1164DAGISEL-NEXT: v_mbcnt_hi_u32_b32 v8, -1, v8
; GFX1164DAGISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:1 row_mask:0xf bank_mask:0xf
-; GFX1164DAGISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:1 row_mask:0xf bank_mask:0xf
; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX1164DAGISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:1 row_mask:0xf bank_mask:0xf
; GFX1164DAGISEL-NEXT: v_add_nc_u32_e32 v8, 32, v8
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1164DAGISEL-NEXT: v_add_co_u32 v4, vcc, v4, v6
-; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1164DAGISEL-NEXT: v_mul_lo_u32 v8, 4, v8
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v6, v4
; GFX1164DAGISEL-NEXT: v_add_co_ci_u32_e64 v5, null, v5, v7, vcc
-; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1164DAGISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:2 row_mask:0xf bank_mask:0xf
-; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v7, v5
; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v7, v5
; GFX1164DAGISEL-NEXT: v_add_co_u32 v4, vcc, v4, v6
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1164DAGISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:2 row_mask:0xf bank_mask:0xf
-; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v6, v4
; GFX1164DAGISEL-NEXT: s_waitcnt_depctr depctr_va_vcc(0)
-; GFX1164DAGISEL-NEXT: v_add_co_ci_u32_e64 v5, null, v5, v7, vcc
; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX1164DAGISEL-NEXT: v_add_co_ci_u32_e64 v5, null, v5, v7, vcc
; GFX1164DAGISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:4 row_mask:0xf bank_mask:0xf
-; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v7, v5
; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v7, v5
; GFX1164DAGISEL-NEXT: v_add_co_u32 v4, vcc, v4, v6
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1164DAGISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:4 row_mask:0xf bank_mask:0xf
-; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v6, v4
; GFX1164DAGISEL-NEXT: s_waitcnt_depctr depctr_va_vcc(0)
-; GFX1164DAGISEL-NEXT: v_add_co_ci_u32_e64 v5, null, v5, v7, vcc
; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX1164DAGISEL-NEXT: v_add_co_ci_u32_e64 v5, null, v5, v7, vcc
; GFX1164DAGISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:8 row_mask:0xf bank_mask:0xf
-; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v7, v5
; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v7, v5
; GFX1164DAGISEL-NEXT: v_add_co_u32 v4, vcc, v4, v6
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX1164DAGISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:8 row_mask:0xf bank_mask:0xf
; GFX1164DAGISEL-NEXT: ds_swizzle_b32 v6, v4 offset:swizzle(BROADCAST,32,15)
; GFX1164DAGISEL-NEXT: s_waitcnt_depctr depctr_va_vcc(0)
@@ -2491,50 +2500,51 @@ define void @divergent_value_dpp_i64(ptr addrspace(1) %out, i64 %in) #0 {
; GFX1164GISEL-NEXT: scratch_store_b32 off, v7, s32 offset:12
; GFX1164GISEL-NEXT: scratch_store_b32 off, v8, s32 offset:16
; GFX1164GISEL-NEXT: s_mov_b64 exec, s[0:1]
+; GFX1164GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1164GISEL-NEXT: s_or_saveexec_b64 s[0:1], -1
-; GFX1164GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(VALU_DEP_3)
; GFX1164GISEL-NEXT: v_cndmask_b32_e64 v4, 0, v2, s[0:1]
; GFX1164GISEL-NEXT: v_cndmask_b32_e64 v5, 0, v3, s[0:1]
; GFX1164GISEL-NEXT: v_mbcnt_lo_u32_b32 v8, -1, 0
-; GFX1164GISEL-NEXT: v_mov_b32_e32 v6, v4
; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX1164GISEL-NEXT: v_mov_b32_e32 v6, v4
; GFX1164GISEL-NEXT: v_mov_b32_e32 v7, v5
-; GFX1164GISEL-NEXT: v_mbcnt_hi_u32_b32 v8, -1, v8
; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX1164GISEL-NEXT: v_mbcnt_hi_u32_b32 v8, -1, v8
; GFX1164GISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:1 row_mask:0xf bank_mask:0xf
-; GFX1164GISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:1 row_mask:0xf bank_mask:0xf
; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX1164GISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:1 row_mask:0xf bank_mask:0xf
; GFX1164GISEL-NEXT: v_add_nc_u32_e32 v8, 32, v8
+; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1164GISEL-NEXT: v_add_co_u32 v4, vcc, v4, v6
-; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1164GISEL-NEXT: v_mul_lo_u32 v8, 4, v8
+; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX1164GISEL-NEXT: v_mov_b32_e32 v6, v4
; GFX1164GISEL-NEXT: v_add_co_ci_u32_e64 v5, null, v5, v7, vcc
-; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1164GISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:2 row_mask:0xf bank_mask:0xf
-; GFX1164GISEL-NEXT: v_mov_b32_e32 v7, v5
; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX1164GISEL-NEXT: v_mov_b32_e32 v7, v5
; GFX1164GISEL-NEXT: v_add_co_u32 v4, vcc, v4, v6
+; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1164GISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:2 row_mask:0xf bank_mask:0xf
-; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX1164GISEL-NEXT: v_mov_b32_e32 v6, v4
; GFX1164GISEL-NEXT: s_waitcnt_depctr depctr_va_vcc(0)
-; GFX1164GISEL-NEXT: v_add_co_ci_u32_e64 v5, null, v5, v7, vcc
; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX1164GISEL-NEXT: v_add_co_ci_u32_e64 v5, null, v5, v7, vcc
; GFX1164GISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:4 row_mask:0xf bank_mask:0xf
-; GFX1164GISEL-NEXT: v_mov_b32_e32 v7, v5
; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX1164GISEL-NEXT: v_mov_b32_e32 v7, v5
; GFX1164GISEL-NEXT: v_add_co_u32 v4, vcc, v4, v6
+; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1164GISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:4 row_mask:0xf bank_mask:0xf
-; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX1164GISEL-NEXT: v_mov_b32_e32 v6, v4
; GFX1164GISEL-NEXT: s_waitcnt_depctr depctr_va_vcc(0)
-; GFX1164GISEL-NEXT: v_add_co_ci_u32_e64 v5, null, v5, v7, vcc
; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX1164GISEL-NEXT: v_add_co_ci_u32_e64 v5, null, v5, v7, vcc
; GFX1164GISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:8 row_mask:0xf bank_mask:0xf
-; GFX1164GISEL-NEXT: v_mov_b32_e32 v7, v5
; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX1164GISEL-NEXT: v_mov_b32_e32 v7, v5
; GFX1164GISEL-NEXT: v_add_co_u32 v4, vcc, v4, v6
+; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX1164GISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:8 row_mask:0xf bank_mask:0xf
; GFX1164GISEL-NEXT: ds_swizzle_b32 v6, v4 offset:swizzle(BROADCAST,32,15)
; GFX1164GISEL-NEXT: s_waitcnt_depctr depctr_va_vcc(0)
@@ -2583,39 +2593,39 @@ define void @divergent_value_dpp_i64(ptr addrspace(1) %out, i64 %in) #0 {
; GFX1132DAGISEL-NEXT: scratch_store_b32 off, v6, s32 offset:8
; GFX1132DAGISEL-NEXT: scratch_store_b32 off, v7, s32 offset:12
; GFX1132DAGISEL-NEXT: s_mov_b32 exec_lo, s0
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1132DAGISEL-NEXT: s_or_saveexec_b32 s0, -1
-; GFX1132DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX1132DAGISEL-NEXT: v_cndmask_b32_e64 v4, 0, v2, s0
; GFX1132DAGISEL-NEXT: v_cndmask_b32_e64 v5, 0, v3, s0
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132DAGISEL-NEXT: v_dual_mov_b32 v6, v4 :: v_dual_mov_b32 v7, v5
-; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1132DAGISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:1 row_mask:0xf bank_mask:0xf
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1132DAGISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:1 row_mask:0xf bank_mask:0xf
-; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132DAGISEL-NEXT: v_add_co_u32 v4, vcc_lo, v4, v6
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132DAGISEL-NEXT: v_add_co_ci_u32_e64 v5, null, v5, v7, vcc_lo
-; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132DAGISEL-NEXT: v_dual_mov_b32 v6, v4 :: v_dual_mov_b32 v7, v5
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1132DAGISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:2 row_mask:0xf bank_mask:0xf
-; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1132DAGISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:2 row_mask:0xf bank_mask:0xf
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1132DAGISEL-NEXT: v_add_co_u32 v4, vcc_lo, v4, v6
-; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132DAGISEL-NEXT: v_add_co_ci_u32_e64 v5, null, v5, v7, vcc_lo
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132DAGISEL-NEXT: v_dual_mov_b32 v6, v4 :: v_dual_mov_b32 v7, v5
-; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1132DAGISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:4 row_mask:0xf bank_mask:0xf
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1132DAGISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:4 row_mask:0xf bank_mask:0xf
-; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132DAGISEL-NEXT: v_add_co_u32 v4, vcc_lo, v4, v6
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132DAGISEL-NEXT: v_add_co_ci_u32_e64 v5, null, v5, v7, vcc_lo
-; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132DAGISEL-NEXT: v_dual_mov_b32 v6, v4 :: v_dual_mov_b32 v7, v5
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1132DAGISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:8 row_mask:0xf bank_mask:0xf
-; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1132DAGISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:8 row_mask:0xf bank_mask:0xf
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1132DAGISEL-NEXT: v_add_co_u32 v4, vcc_lo, v4, v6
-; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1132DAGISEL-NEXT: v_add_co_ci_u32_e64 v5, null, v5, v7, vcc_lo
; GFX1132DAGISEL-NEXT: ds_swizzle_b32 v6, v4 offset:swizzle(BROADCAST,32,15)
; GFX1132DAGISEL-NEXT: ds_swizzle_b32 v7, v5 offset:swizzle(BROADCAST,32,15)
@@ -2652,39 +2662,39 @@ define void @divergent_value_dpp_i64(ptr addrspace(1) %out, i64 %in) #0 {
; GFX1132GISEL-NEXT: scratch_store_b32 off, v6, s32 offset:8
; GFX1132GISEL-NEXT: scratch_store_b32 off, v7, s32 offset:12
; GFX1132GISEL-NEXT: s_mov_b32 exec_lo, s0
+; GFX1132GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1132GISEL-NEXT: s_or_saveexec_b32 s0, -1
-; GFX1132GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX1132GISEL-NEXT: v_cndmask_b32_e64 v4, 0, v2, s0
; GFX1132GISEL-NEXT: v_cndmask_b32_e64 v5, 0, v3, s0
+; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132GISEL-NEXT: v_dual_mov_b32 v6, v4 :: v_dual_mov_b32 v7, v5
-; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1132GISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:1 row_mask:0xf bank_mask:0xf
+; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1132GISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:1 row_mask:0xf bank_mask:0xf
-; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132GISEL-NEXT: v_add_co_u32 v4, vcc_lo, v4, v6
+; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132GISEL-NEXT: v_add_co_ci_u32_e64 v5, null, v5, v7, vcc_lo
-; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132GISEL-NEXT: v_dual_mov_b32 v6, v4 :: v_dual_mov_b32 v7, v5
+; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1132GISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:2 row_mask:0xf bank_mask:0xf
-; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1132GISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:2 row_mask:0xf bank_mask:0xf
+; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1132GISEL-NEXT: v_add_co_u32 v4, vcc_lo, v4, v6
-; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132GISEL-NEXT: v_add_co_ci_u32_e64 v5, null, v5, v7, vcc_lo
+; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132GISEL-NEXT: v_dual_mov_b32 v6, v4 :: v_dual_mov_b32 v7, v5
-; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1132GISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:4 row_mask:0xf bank_mask:0xf
+; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1132GISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:4 row_mask:0xf bank_mask:0xf
-; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132GISEL-NEXT: v_add_co_u32 v4, vcc_lo, v4, v6
+; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132GISEL-NEXT: v_add_co_ci_u32_e64 v5, null, v5, v7, vcc_lo
-; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132GISEL-NEXT: v_dual_mov_b32 v6, v4 :: v_dual_mov_b32 v7, v5
+; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1132GISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:8 row_mask:0xf bank_mask:0xf
-; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1132GISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:8 row_mask:0xf bank_mask:0xf
+; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1132GISEL-NEXT: v_add_co_u32 v4, vcc_lo, v4, v6
-; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1132GISEL-NEXT: v_add_co_ci_u32_e64 v5, null, v5, v7, vcc_lo
; GFX1132GISEL-NEXT: ds_swizzle_b32 v6, v4 offset:swizzle(BROADCAST,32,15)
; GFX1132GISEL-NEXT: ds_swizzle_b32 v7, v5 offset:swizzle(BROADCAST,32,15)
@@ -3617,7 +3627,7 @@ define amdgpu_kernel void @divergent_cfg(ptr addrspace(1) %out, i32 %in) #0 {
; GFX1164DAGISEL-NEXT: v_and_b32_e32 v0, 0x3ff, v0
; GFX1164DAGISEL-NEXT: s_mov_b64 s[0:1], exec
; GFX1164DAGISEL-NEXT: ; implicit-def: $sgpr2
-; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164DAGISEL-NEXT: v_cmpx_lt_u32_e32 15, v0
; GFX1164DAGISEL-NEXT: s_xor_b64 s[0:1], exec, s[0:1]
; GFX1164DAGISEL-NEXT: s_cbranch_execz .LBB8_2
@@ -3634,13 +3644,14 @@ define amdgpu_kernel void @divergent_cfg(ptr addrspace(1) %out, i32 %in) #0 {
; GFX1164DAGISEL-NEXT: s_or_saveexec_b64 s[0:1], s[0:1]
; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v1, s2
; GFX1164DAGISEL-NEXT: s_xor_b64 exec, exec, s[0:1]
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164DAGISEL-NEXT: s_cbranch_execz .LBB8_6
; GFX1164DAGISEL-NEXT: ; %bb.3: ; %if
; GFX1164DAGISEL-NEXT: s_mov_b64 s[2:3], exec
; GFX1164DAGISEL-NEXT: s_mov_b32 s6, 0
; GFX1164DAGISEL-NEXT: .LBB8_4: ; =>This Inner Loop Header: Depth=1
; GFX1164DAGISEL-NEXT: s_ctz_i32_b64 s7, s[2:3]
-; GFX1164DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_3) | instid1(SALU_CYCLE_1)
; GFX1164DAGISEL-NEXT: v_readlane_b32 s8, v0, s7
; GFX1164DAGISEL-NEXT: s_bitset0_b64 s[2:3], s7
; GFX1164DAGISEL-NEXT: s_sub_i32 s6, s6, s8
@@ -3661,7 +3672,7 @@ define amdgpu_kernel void @divergent_cfg(ptr addrspace(1) %out, i32 %in) #0 {
; GFX1164GISEL-NEXT: v_and_b32_e32 v0, 0x3ff, v0
; GFX1164GISEL-NEXT: s_mov_b64 s[0:1], exec
; GFX1164GISEL-NEXT: ; implicit-def: $sgpr2
-; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164GISEL-NEXT: v_cmpx_le_u32_e32 16, v0
; GFX1164GISEL-NEXT: s_xor_b64 s[0:1], exec, s[0:1]
; GFX1164GISEL-NEXT: s_cbranch_execz .LBB8_2
@@ -3678,13 +3689,14 @@ define amdgpu_kernel void @divergent_cfg(ptr addrspace(1) %out, i32 %in) #0 {
; GFX1164GISEL-NEXT: s_or_saveexec_b64 s[0:1], s[0:1]
; GFX1164GISEL-NEXT: v_mov_b32_e32 v1, s2
; GFX1164GISEL-NEXT: s_xor_b64 exec, exec, s[0:1]
+; GFX1164GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164GISEL-NEXT: s_cbranch_execz .LBB8_6
; GFX1164GISEL-NEXT: ; %bb.3: ; %if
; GFX1164GISEL-NEXT: s_mov_b64 s[2:3], exec
; GFX1164GISEL-NEXT: s_mov_b32 s6, 0
; GFX1164GISEL-NEXT: .LBB8_4: ; =>This Inner Loop Header: Depth=1
; GFX1164GISEL-NEXT: s_ctz_i32_b64 s7, s[2:3]
-; GFX1164GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1164GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_3) | instid1(SALU_CYCLE_1)
; GFX1164GISEL-NEXT: v_readlane_b32 s8, v0, s7
; GFX1164GISEL-NEXT: s_bitset0_b64 s[2:3], s7
; GFX1164GISEL-NEXT: s_sub_i32 s6, s6, s8
@@ -3705,9 +3717,10 @@ define amdgpu_kernel void @divergent_cfg(ptr addrspace(1) %out, i32 %in) #0 {
; GFX1132DAGISEL-NEXT: v_and_b32_e32 v0, 0x3ff, v0
; GFX1132DAGISEL-NEXT: s_mov_b32 s0, exec_lo
; GFX1132DAGISEL-NEXT: ; implicit-def: $sgpr1
-; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132DAGISEL-NEXT: v_cmpx_lt_u32_e32 15, v0
; GFX1132DAGISEL-NEXT: s_xor_b32 s0, exec_lo, s0
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1132DAGISEL-NEXT: s_cbranch_execz .LBB8_2
; GFX1132DAGISEL-NEXT: ; %bb.1: ; %else
; GFX1132DAGISEL-NEXT: s_load_b32 s1, s[4:5], 0x2c
@@ -3722,13 +3735,14 @@ define amdgpu_kernel void @divergent_cfg(ptr addrspace(1) %out, i32 %in) #0 {
; GFX1132DAGISEL-NEXT: s_or_saveexec_b32 s0, s0
; GFX1132DAGISEL-NEXT: v_mov_b32_e32 v1, s1
; GFX1132DAGISEL-NEXT: s_xor_b32 exec_lo, exec_lo, s0
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1132DAGISEL-NEXT: s_cbranch_execz .LBB8_6
; GFX1132DAGISEL-NEXT: ; %bb.3: ; %if
; GFX1132DAGISEL-NEXT: s_mov_b32 s2, exec_lo
; GFX1132DAGISEL-NEXT: s_mov_b32 s1, 0
; GFX1132DAGISEL-NEXT: .LBB8_4: ; =>This Inner Loop Header: Depth=1
; GFX1132DAGISEL-NEXT: s_ctz_i32_b32 s3, s2
-; GFX1132DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_3) | instid1(SALU_CYCLE_1)
; GFX1132DAGISEL-NEXT: v_readlane_b32 s6, v0, s3
; GFX1132DAGISEL-NEXT: s_bitset0_b32 s2, s3
; GFX1132DAGISEL-NEXT: s_sub_i32 s1, s1, s6
@@ -3749,9 +3763,10 @@ define amdgpu_kernel void @divergent_cfg(ptr addrspace(1) %out, i32 %in) #0 {
; GFX1132GISEL-NEXT: v_and_b32_e32 v0, 0x3ff, v0
; GFX1132GISEL-NEXT: s_mov_b32 s0, exec_lo
; GFX1132GISEL-NEXT: ; implicit-def: $sgpr1
-; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132GISEL-NEXT: v_cmpx_le_u32_e32 16, v0
; GFX1132GISEL-NEXT: s_xor_b32 s0, exec_lo, s0
+; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1132GISEL-NEXT: s_cbranch_execz .LBB8_2
; GFX1132GISEL-NEXT: ; %bb.1: ; %else
; GFX1132GISEL-NEXT: s_load_b32 s1, s[4:5], 0x2c
@@ -3766,13 +3781,14 @@ define amdgpu_kernel void @divergent_cfg(ptr addrspace(1) %out, i32 %in) #0 {
; GFX1132GISEL-NEXT: s_or_saveexec_b32 s0, s0
; GFX1132GISEL-NEXT: v_mov_b32_e32 v1, s1
; GFX1132GISEL-NEXT: s_xor_b32 exec_lo, exec_lo, s0
+; GFX1132GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1132GISEL-NEXT: s_cbranch_execz .LBB8_6
; GFX1132GISEL-NEXT: ; %bb.3: ; %if
; GFX1132GISEL-NEXT: s_mov_b32 s2, exec_lo
; GFX1132GISEL-NEXT: s_mov_b32 s1, 0
; GFX1132GISEL-NEXT: .LBB8_4: ; =>This Inner Loop Header: Depth=1
; GFX1132GISEL-NEXT: s_ctz_i32_b32 s3, s2
-; GFX1132GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1132GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_3) | instid1(SALU_CYCLE_1)
; GFX1132GISEL-NEXT: v_readlane_b32 s6, v0, s3
; GFX1132GISEL-NEXT: s_bitset0_b32 s2, s3
; GFX1132GISEL-NEXT: s_sub_i32 s1, s1, s6
@@ -4407,6 +4423,7 @@ define void @divergent_value_i64(ptr addrspace(1) %out, i64 %id.x) #0 {
; GFX1164DAGISEL-NEXT: s_sub_u32 s0, s0, s5
; GFX1164DAGISEL-NEXT: s_subb_u32 s1, s1, s6
; GFX1164DAGISEL-NEXT: s_cmp_lg_u64 s[2:3], 0
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164DAGISEL-NEXT: s_cbranch_scc1 .LBB10_1
; GFX1164DAGISEL-NEXT: ; %bb.2:
; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v3, s1
@@ -4428,6 +4445,7 @@ define void @divergent_value_i64(ptr addrspace(1) %out, i64 %id.x) #0 {
; GFX1164GISEL-NEXT: s_sub_u32 s0, s0, s5
; GFX1164GISEL-NEXT: s_subb_u32 s1, s1, s6
; GFX1164GISEL-NEXT: s_cmp_lg_u64 s[2:3], 0
+; GFX1164GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164GISEL-NEXT: s_cbranch_scc1 .LBB10_1
; GFX1164GISEL-NEXT: ; %bb.2:
; GFX1164GISEL-NEXT: v_mov_b32_e32 v3, s1
@@ -4449,6 +4467,7 @@ define void @divergent_value_i64(ptr addrspace(1) %out, i64 %id.x) #0 {
; GFX1132DAGISEL-NEXT: s_sub_u32 s0, s0, s4
; GFX1132DAGISEL-NEXT: s_subb_u32 s1, s1, s5
; GFX1132DAGISEL-NEXT: s_cmp_lg_u32 s2, 0
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1132DAGISEL-NEXT: s_cbranch_scc1 .LBB10_1
; GFX1132DAGISEL-NEXT: ; %bb.2:
; GFX1132DAGISEL-NEXT: v_dual_mov_b32 v3, s1 :: v_dual_mov_b32 v2, s0
@@ -4469,6 +4488,7 @@ define void @divergent_value_i64(ptr addrspace(1) %out, i64 %id.x) #0 {
; GFX1132GISEL-NEXT: s_sub_u32 s0, s0, s4
; GFX1132GISEL-NEXT: s_subb_u32 s1, s1, s5
; GFX1132GISEL-NEXT: s_cmp_lg_u32 s2, 0
+; GFX1132GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1132GISEL-NEXT: s_cbranch_scc1 .LBB10_1
; GFX1132GISEL-NEXT: ; %bb.2:
; GFX1132GISEL-NEXT: v_dual_mov_b32 v3, s1 :: v_dual_mov_b32 v2, s0
@@ -5008,7 +5028,7 @@ define amdgpu_kernel void @divergent_cfg_i64(ptr addrspace(1) %out, i64 %in, i64
; GFX1164DAGISEL-NEXT: v_and_b32_e32 v0, 0x3ff, v0
; GFX1164DAGISEL-NEXT: s_mov_b64 s[6:7], exec
; GFX1164DAGISEL-NEXT: ; implicit-def: $sgpr8_sgpr9
-; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164DAGISEL-NEXT: v_cmpx_lt_u32_e32 15, v0
; GFX1164DAGISEL-NEXT: s_xor_b64 s[6:7], exec, s[6:7]
; GFX1164DAGISEL-NEXT: s_cbranch_execz .LBB11_2
@@ -5049,6 +5069,7 @@ define amdgpu_kernel void @divergent_cfg_i64(ptr addrspace(1) %out, i64 %in, i64
; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v1, s5
; GFX1164DAGISEL-NEXT: ; %bb.4: ; %endif
; GFX1164DAGISEL-NEXT: s_or_b64 exec, exec, s[2:3]
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v2, 0
; GFX1164DAGISEL-NEXT: global_store_b64 v2, v[0:1], s[0:1]
; GFX1164DAGISEL-NEXT: s_endpgm
@@ -5059,7 +5080,7 @@ define amdgpu_kernel void @divergent_cfg_i64(ptr addrspace(1) %out, i64 %in, i64
; GFX1164GISEL-NEXT: v_and_b32_e32 v0, 0x3ff, v0
; GFX1164GISEL-NEXT: s_mov_b64 s[6:7], exec
; GFX1164GISEL-NEXT: ; implicit-def: $sgpr8_sgpr9
-; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164GISEL-NEXT: v_cmpx_le_u32_e32 16, v0
; GFX1164GISEL-NEXT: s_xor_b64 s[6:7], exec, s[6:7]
; GFX1164GISEL-NEXT: s_cbranch_execz .LBB11_2
@@ -5083,6 +5104,7 @@ define amdgpu_kernel void @divergent_cfg_i64(ptr addrspace(1) %out, i64 %in, i64
; GFX1164GISEL-NEXT: v_mov_b32_e32 v0, s8
; GFX1164GISEL-NEXT: v_mov_b32_e32 v1, s9
; GFX1164GISEL-NEXT: s_xor_b64 exec, exec, s[2:3]
+; GFX1164GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164GISEL-NEXT: s_cbranch_execz .LBB11_4
; GFX1164GISEL-NEXT: ; %bb.3: ; %if
; GFX1164GISEL-NEXT: s_load_b64 s[4:5], s[4:5], 0x34
@@ -5103,6 +5125,7 @@ define amdgpu_kernel void @divergent_cfg_i64(ptr addrspace(1) %out, i64 %in, i64
; GFX1164GISEL-NEXT: v_mov_b32_e32 v1, s5
; GFX1164GISEL-NEXT: .LBB11_4: ; %endif
; GFX1164GISEL-NEXT: s_or_b64 exec, exec, s[2:3]
+; GFX1164GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164GISEL-NEXT: v_mov_b32_e32 v2, 0
; GFX1164GISEL-NEXT: global_store_b64 v2, v[0:1], s[0:1]
; GFX1164GISEL-NEXT: s_endpgm
@@ -5115,9 +5138,10 @@ define amdgpu_kernel void @divergent_cfg_i64(ptr addrspace(1) %out, i64 %in, i64
; GFX1132DAGISEL-NEXT: v_and_b32_e32 v0, 0x3ff, v0
; GFX1132DAGISEL-NEXT: s_mov_b32 s8, exec_lo
; GFX1132DAGISEL-NEXT: ; implicit-def: $sgpr6_sgpr7
-; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132DAGISEL-NEXT: v_cmpx_lt_u32_e32 15, v0
; GFX1132DAGISEL-NEXT: s_xor_b32 s8, exec_lo, s8
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1132DAGISEL-NEXT: s_cbranch_execz .LBB11_2
; GFX1132DAGISEL-NEXT: ; %bb.1: ; %else
; GFX1132DAGISEL-NEXT: s_mov_b32 s6, exec_lo
@@ -5155,6 +5179,7 @@ define amdgpu_kernel void @divergent_cfg_i64(ptr addrspace(1) %out, i64 %in, i64
; GFX1132DAGISEL-NEXT: v_dual_mov_b32 v0, s4 :: v_dual_mov_b32 v1, s5
; GFX1132DAGISEL-NEXT: ; %bb.4: ; %endif
; GFX1132DAGISEL-NEXT: s_or_b32 exec_lo, exec_lo, s2
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1132DAGISEL-NEXT: v_mov_b32_e32 v2, 0
; GFX1132DAGISEL-NEXT: global_store_b64 v2, v[0:1], s[0:1]
; GFX1132DAGISEL-NEXT: s_endpgm
@@ -5165,9 +5190,10 @@ define amdgpu_kernel void @divergent_cfg_i64(ptr addrspace(1) %out, i64 %in, i64
; GFX1132GISEL-NEXT: v_and_b32_e32 v0, 0x3ff, v0
; GFX1132GISEL-NEXT: s_mov_b32 s8, exec_lo
; GFX1132GISEL-NEXT: ; implicit-def: $sgpr6_sgpr7
-; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132GISEL-NEXT: v_cmpx_le_u32_e32 16, v0
; GFX1132GISEL-NEXT: s_xor_b32 s8, exec_lo, s8
+; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1132GISEL-NEXT: s_cbranch_execz .LBB11_2
; GFX1132GISEL-NEXT: ; %bb.1: ; %else
; GFX1132GISEL-NEXT: s_mov_b32 s6, exec_lo
@@ -5188,6 +5214,7 @@ define amdgpu_kernel void @divergent_cfg_i64(ptr addrspace(1) %out, i64 %in, i64
; GFX1132GISEL-NEXT: s_or_saveexec_b32 s2, s8
; GFX1132GISEL-NEXT: v_dual_mov_b32 v0, s6 :: v_dual_mov_b32 v1, s7
; GFX1132GISEL-NEXT: s_xor_b32 exec_lo, exec_lo, s2
+; GFX1132GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1132GISEL-NEXT: s_cbranch_execz .LBB11_4
; GFX1132GISEL-NEXT: ; %bb.3: ; %if
; GFX1132GISEL-NEXT: s_load_b64 s[4:5], s[4:5], 0x34
@@ -5208,6 +5235,7 @@ define amdgpu_kernel void @divergent_cfg_i64(ptr addrspace(1) %out, i64 %in, i64
; GFX1132GISEL-NEXT: v_dual_mov_b32 v0, s4 :: v_dual_mov_b32 v1, s5
; GFX1132GISEL-NEXT: .LBB11_4: ; %endif
; GFX1132GISEL-NEXT: s_or_b32 exec_lo, exec_lo, s2
+; GFX1132GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1132GISEL-NEXT: v_mov_b32_e32 v2, 0
; GFX1132GISEL-NEXT: global_store_b64 v2, v[0:1], s[0:1]
; GFX1132GISEL-NEXT: s_endpgm
diff --git a/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.reduce.umax.ll b/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.reduce.umax.ll
index 69b24c6f1b17a0..283e7f3b16b6cf 100644
--- a/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.reduce.umax.ll
+++ b/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.reduce.umax.ll
@@ -415,6 +415,7 @@ define void @divergent_value_i16(ptr addrspace(1) %out, i16 %in) {
; GFX1164DAGISEL-FAKE16-NEXT: s_bitset0_b64 s[0:1], s3
; GFX1164DAGISEL-FAKE16-NEXT: s_max_u32 s2, s2, s4
; GFX1164DAGISEL-FAKE16-NEXT: s_cmp_lg_u64 s[0:1], 0
+; GFX1164DAGISEL-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164DAGISEL-FAKE16-NEXT: s_cbranch_scc1 .LBB1_1
; GFX1164DAGISEL-FAKE16-NEXT: ; %bb.2:
; GFX1164DAGISEL-FAKE16-NEXT: v_mov_b32_e32 v2, s2
@@ -434,6 +435,7 @@ define void @divergent_value_i16(ptr addrspace(1) %out, i16 %in) {
; GFX1164GISEL-NEXT: s_bitset0_b64 s[0:1], s3
; GFX1164GISEL-NEXT: s_max_u32 s2, s2, s4
; GFX1164GISEL-NEXT: s_cmp_lg_u64 s[0:1], 0
+; GFX1164GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164GISEL-NEXT: s_cbranch_scc1 .LBB1_1
; GFX1164GISEL-NEXT: ; %bb.2:
; GFX1164GISEL-NEXT: v_mov_b32_e32 v2, s2
@@ -453,6 +455,7 @@ define void @divergent_value_i16(ptr addrspace(1) %out, i16 %in) {
; GFX1132DAGISEL-FAKE16-NEXT: s_bitset0_b32 s1, s2
; GFX1132DAGISEL-FAKE16-NEXT: s_max_u32 s0, s0, s3
; GFX1132DAGISEL-FAKE16-NEXT: s_cmp_lg_u32 s1, 0
+; GFX1132DAGISEL-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1132DAGISEL-FAKE16-NEXT: s_cbranch_scc1 .LBB1_1
; GFX1132DAGISEL-FAKE16-NEXT: ; %bb.2:
; GFX1132DAGISEL-FAKE16-NEXT: v_mov_b32_e32 v2, s0
@@ -472,6 +475,7 @@ define void @divergent_value_i16(ptr addrspace(1) %out, i16 %in) {
; GFX1132GISEL-NEXT: s_bitset0_b32 s1, s2
; GFX1132GISEL-NEXT: s_max_u32 s0, s0, s3
; GFX1132GISEL-NEXT: s_cmp_lg_u32 s1, 0
+; GFX1132GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1132GISEL-NEXT: s_cbranch_scc1 .LBB1_1
; GFX1132GISEL-NEXT: ; %bb.2:
; GFX1132GISEL-NEXT: v_mov_b32_e32 v2, s0
@@ -491,6 +495,7 @@ define void @divergent_value_i16(ptr addrspace(1) %out, i16 %in) {
; GFX1164DAGISEL-TRUE16-NEXT: s_bitset0_b64 s[0:1], s3
; GFX1164DAGISEL-TRUE16-NEXT: s_max_u32 s2, s2, s4
; GFX1164DAGISEL-TRUE16-NEXT: s_cmp_lg_u64 s[0:1], 0
+; GFX1164DAGISEL-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164DAGISEL-TRUE16-NEXT: s_cbranch_scc1 .LBB1_1
; GFX1164DAGISEL-TRUE16-NEXT: ; %bb.2:
; GFX1164DAGISEL-TRUE16-NEXT: v_mov_b32_e32 v2, s2
@@ -510,6 +515,7 @@ define void @divergent_value_i16(ptr addrspace(1) %out, i16 %in) {
; GFX1132DAGISEL-TRUE16-NEXT: s_bitset0_b32 s1, s2
; GFX1132DAGISEL-TRUE16-NEXT: s_max_u32 s0, s0, s3
; GFX1132DAGISEL-TRUE16-NEXT: s_cmp_lg_u32 s1, 0
+; GFX1132DAGISEL-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1132DAGISEL-TRUE16-NEXT: s_cbranch_scc1 .LBB1_1
; GFX1132DAGISEL-TRUE16-NEXT: ; %bb.2:
; GFX1132DAGISEL-TRUE16-NEXT: v_mov_b32_e32 v2, s0
@@ -864,6 +870,7 @@ define amdgpu_kernel void @divergent_value(ptr addrspace(1) %out, i32 %in) #0 {
; GFX1164DAGISEL-NEXT: s_bitset0_b64 s[2:3], s5
; GFX1164DAGISEL-NEXT: s_max_u32 s4, s4, s6
; GFX1164DAGISEL-NEXT: s_cmp_lg_u64 s[2:3], 0
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164DAGISEL-NEXT: s_cbranch_scc1 .LBB3_1
; GFX1164DAGISEL-NEXT: ; %bb.2:
; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v0, s4
@@ -884,6 +891,7 @@ define amdgpu_kernel void @divergent_value(ptr addrspace(1) %out, i32 %in) #0 {
; GFX1164GISEL-NEXT: s_bitset0_b64 s[2:3], s5
; GFX1164GISEL-NEXT: s_max_u32 s4, s4, s6
; GFX1164GISEL-NEXT: s_cmp_lg_u64 s[2:3], 0
+; GFX1164GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164GISEL-NEXT: s_cbranch_scc1 .LBB3_1
; GFX1164GISEL-NEXT: ; %bb.2:
; GFX1164GISEL-NEXT: v_mov_b32_e32 v0, s4
@@ -905,6 +913,7 @@ define amdgpu_kernel void @divergent_value(ptr addrspace(1) %out, i32 %in) #0 {
; GFX1132DAGISEL-NEXT: s_bitset0_b32 s3, s4
; GFX1132DAGISEL-NEXT: s_max_u32 s2, s2, s5
; GFX1132DAGISEL-NEXT: s_cmp_lg_u32 s3, 0
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1132DAGISEL-NEXT: s_cbranch_scc1 .LBB3_1
; GFX1132DAGISEL-NEXT: ; %bb.2:
; GFX1132DAGISEL-NEXT: v_mov_b32_e32 v0, s2
@@ -925,6 +934,7 @@ define amdgpu_kernel void @divergent_value(ptr addrspace(1) %out, i32 %in) #0 {
; GFX1132GISEL-NEXT: s_bitset0_b32 s3, s4
; GFX1132GISEL-NEXT: s_max_u32 s2, s2, s5
; GFX1132GISEL-NEXT: s_cmp_lg_u32 s3, 0
+; GFX1132GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1132GISEL-NEXT: s_cbranch_scc1 .LBB3_1
; GFX1132GISEL-NEXT: ; %bb.2:
; GFX1132GISEL-NEXT: v_dual_mov_b32 v0, s2 :: v_dual_mov_b32 v1, 0
@@ -1344,7 +1354,7 @@ define amdgpu_kernel void @divergent_value_dpp(ptr addrspace(1) %out, i32 %in) #
; GFX1164DAGISEL-NEXT: s_or_saveexec_b64 s[2:3], -1
; GFX1164DAGISEL-NEXT: s_waitcnt lgkmcnt(0)
; GFX1164DAGISEL-NEXT: v_max_u32_e32 v1, v1, v2
-; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX1164DAGISEL-NEXT: v_readlane_b32 s4, v1, 63
; GFX1164DAGISEL-NEXT: s_mov_b64 exec, s[2:3]
; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v0, 0
@@ -1379,7 +1389,7 @@ define amdgpu_kernel void @divergent_value_dpp(ptr addrspace(1) %out, i32 %in) #
; GFX1164GISEL-NEXT: s_mov_b64 exec, s[0:1]
; GFX1164GISEL-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
; GFX1164GISEL-NEXT: s_or_saveexec_b64 s[2:3], -1
-; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX1164GISEL-NEXT: v_readlane_b32 s4, v1, 63
; GFX1164GISEL-NEXT: s_mov_b64 exec, s[2:3]
; GFX1164GISEL-NEXT: v_mov_b32_e32 v0, s4
@@ -1406,7 +1416,7 @@ define amdgpu_kernel void @divergent_value_dpp(ptr addrspace(1) %out, i32 %in) #
; GFX1132DAGISEL-NEXT: s_or_saveexec_b32 s2, -1
; GFX1132DAGISEL-NEXT: s_waitcnt lgkmcnt(0)
; GFX1132DAGISEL-NEXT: v_max_u32_e32 v1, v1, v2
-; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX1132DAGISEL-NEXT: v_readlane_b32 s3, v1, 31
; GFX1132DAGISEL-NEXT: s_mov_b32 exec_lo, s2
; GFX1132DAGISEL-NEXT: v_dual_mov_b32 v0, 0 :: v_dual_mov_b32 v3, s3
@@ -1431,7 +1441,7 @@ define amdgpu_kernel void @divergent_value_dpp(ptr addrspace(1) %out, i32 %in) #
; GFX1132GISEL-NEXT: s_mov_b32 exec_lo, s0
; GFX1132GISEL-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
; GFX1132GISEL-NEXT: s_or_saveexec_b32 s2, -1
-; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX1132GISEL-NEXT: v_readlane_b32 s3, v1, 31
; GFX1132GISEL-NEXT: s_mov_b32 exec_lo, s2
; GFX1132GISEL-NEXT: v_dual_mov_b32 v0, s3 :: v_dual_mov_b32 v3, 0
@@ -2129,52 +2139,53 @@ define void @divergent_value_dpp_i64(ptr addrspace(1) %out, i64 %in) #0 {
; GFX1164DAGISEL-NEXT: scratch_store_b32 off, v4, s32 offset:20
; GFX1164DAGISEL-NEXT: scratch_store_b32 off, v5, s32 offset:24
; GFX1164DAGISEL-NEXT: s_mov_b64 exec, s[0:1]
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1164DAGISEL-NEXT: s_or_saveexec_b64 s[0:1], -1
-; GFX1164DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(VALU_DEP_3)
; GFX1164DAGISEL-NEXT: v_cndmask_b32_e64 v4, 0, v2, s[0:1]
; GFX1164DAGISEL-NEXT: v_cndmask_b32_e64 v5, 0, v3, s[0:1]
; GFX1164DAGISEL-NEXT: v_mbcnt_lo_u32_b32 v8, -1, 0
-; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v6, v4
; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v6, v4
; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v7, v5
-; GFX1164DAGISEL-NEXT: v_mbcnt_hi_u32_b32 v8, -1, v8
; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX1164DAGISEL-NEXT: v_mbcnt_hi_u32_b32 v8, -1, v8
; GFX1164DAGISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:1 row_mask:0xf bank_mask:0xf
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_3)
; GFX1164DAGISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:1 row_mask:0xf bank_mask:0xf
-; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1164DAGISEL-NEXT: v_add_nc_u32_e32 v8, 32, v8
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1164DAGISEL-NEXT: v_cmp_gt_u64_e32 vcc, v[4:5], v[6:7]
-; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_2) | instid1(VALU_DEP_2)
; GFX1164DAGISEL-NEXT: v_mul_lo_u32 v8, 4, v8
; GFX1164DAGISEL-NEXT: v_cndmask_b32_e32 v4, v6, v4, vcc
; GFX1164DAGISEL-NEXT: v_cndmask_b32_e32 v5, v7, v5, vcc
-; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v6, v4
; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v6, v4
; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v7, v5
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1164DAGISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:2 row_mask:0xf bank_mask:0xf
-; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164DAGISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:2 row_mask:0xf bank_mask:0xf
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_3) | instid1(VALU_DEP_2)
; GFX1164DAGISEL-NEXT: v_cmp_gt_u64_e32 vcc, v[4:5], v[6:7]
; GFX1164DAGISEL-NEXT: s_waitcnt_depctr depctr_va_vcc(0)
; GFX1164DAGISEL-NEXT: v_cndmask_b32_e32 v4, v6, v4, vcc
; GFX1164DAGISEL-NEXT: v_cndmask_b32_e32 v5, v7, v5, vcc
-; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v6, v4
-; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v7, v5
; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v7, v5
; GFX1164DAGISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:4 row_mask:0xf bank_mask:0xf
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164DAGISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:4 row_mask:0xf bank_mask:0xf
-; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_3) | instid1(VALU_DEP_2)
; GFX1164DAGISEL-NEXT: v_cmp_gt_u64_e32 vcc, v[4:5], v[6:7]
; GFX1164DAGISEL-NEXT: s_waitcnt_depctr depctr_va_vcc(0)
; GFX1164DAGISEL-NEXT: v_cndmask_b32_e32 v4, v6, v4, vcc
; GFX1164DAGISEL-NEXT: v_cndmask_b32_e32 v5, v7, v5, vcc
-; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v6, v4
; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v6, v4
; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v7, v5
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1164DAGISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:8 row_mask:0xf bank_mask:0xf
-; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164DAGISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:8 row_mask:0xf bank_mask:0xf
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1164DAGISEL-NEXT: v_cmp_gt_u64_e32 vcc, v[4:5], v[6:7]
; GFX1164DAGISEL-NEXT: s_waitcnt_depctr depctr_va_vcc(0)
; GFX1164DAGISEL-NEXT: v_cndmask_b32_e32 v4, v6, v4, vcc
@@ -2197,6 +2208,7 @@ define void @divergent_value_dpp_i64(ptr addrspace(1) %out, i64 %in) #0 {
; GFX1164DAGISEL-NEXT: v_readlane_b32 s2, v4, 63
; GFX1164DAGISEL-NEXT: v_readlane_b32 s3, v5, 63
; GFX1164DAGISEL-NEXT: s_mov_b64 exec, s[0:1]
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v2, s2
; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v3, s3
; GFX1164DAGISEL-NEXT: global_store_b64 v[0:1], v[2:3], off
@@ -2223,52 +2235,53 @@ define void @divergent_value_dpp_i64(ptr addrspace(1) %out, i64 %in) #0 {
; GFX1164GISEL-NEXT: scratch_store_b32 off, v4, s32 offset:20
; GFX1164GISEL-NEXT: scratch_store_b32 off, v5, s32 offset:24
; GFX1164GISEL-NEXT: s_mov_b64 exec, s[0:1]
+; GFX1164GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1164GISEL-NEXT: s_or_saveexec_b64 s[0:1], -1
-; GFX1164GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(VALU_DEP_3)
; GFX1164GISEL-NEXT: v_cndmask_b32_e64 v4, 0, v2, s[0:1]
; GFX1164GISEL-NEXT: v_cndmask_b32_e64 v5, 0, v3, s[0:1]
; GFX1164GISEL-NEXT: v_mbcnt_lo_u32_b32 v8, -1, 0
-; GFX1164GISEL-NEXT: v_mov_b32_e32 v6, v4
; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX1164GISEL-NEXT: v_mov_b32_e32 v6, v4
; GFX1164GISEL-NEXT: v_mov_b32_e32 v7, v5
-; GFX1164GISEL-NEXT: v_mbcnt_hi_u32_b32 v8, -1, v8
; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX1164GISEL-NEXT: v_mbcnt_hi_u32_b32 v8, -1, v8
; GFX1164GISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:1 row_mask:0xf bank_mask:0xf
+; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_3)
; GFX1164GISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:1 row_mask:0xf bank_mask:0xf
-; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1164GISEL-NEXT: v_add_nc_u32_e32 v8, 32, v8
+; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1164GISEL-NEXT: v_cmp_gt_u64_e32 vcc, v[4:5], v[6:7]
-; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_2) | instid1(VALU_DEP_2)
; GFX1164GISEL-NEXT: v_mul_lo_u32 v8, 4, v8
; GFX1164GISEL-NEXT: v_cndmask_b32_e32 v4, v6, v4, vcc
; GFX1164GISEL-NEXT: v_cndmask_b32_e32 v5, v7, v5, vcc
-; GFX1164GISEL-NEXT: v_mov_b32_e32 v6, v4
; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX1164GISEL-NEXT: v_mov_b32_e32 v6, v4
; GFX1164GISEL-NEXT: v_mov_b32_e32 v7, v5
+; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1164GISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:2 row_mask:0xf bank_mask:0xf
-; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164GISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:2 row_mask:0xf bank_mask:0xf
+; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_3) | instid1(VALU_DEP_2)
; GFX1164GISEL-NEXT: v_cmp_gt_u64_e32 vcc, v[4:5], v[6:7]
; GFX1164GISEL-NEXT: s_waitcnt_depctr depctr_va_vcc(0)
; GFX1164GISEL-NEXT: v_cndmask_b32_e32 v4, v6, v4, vcc
; GFX1164GISEL-NEXT: v_cndmask_b32_e32 v5, v7, v5, vcc
-; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1164GISEL-NEXT: v_mov_b32_e32 v6, v4
-; GFX1164GISEL-NEXT: v_mov_b32_e32 v7, v5
; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX1164GISEL-NEXT: v_mov_b32_e32 v7, v5
; GFX1164GISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:4 row_mask:0xf bank_mask:0xf
+; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164GISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:4 row_mask:0xf bank_mask:0xf
-; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_3) | instid1(VALU_DEP_2)
; GFX1164GISEL-NEXT: v_cmp_gt_u64_e32 vcc, v[4:5], v[6:7]
; GFX1164GISEL-NEXT: s_waitcnt_depctr depctr_va_vcc(0)
; GFX1164GISEL-NEXT: v_cndmask_b32_e32 v4, v6, v4, vcc
; GFX1164GISEL-NEXT: v_cndmask_b32_e32 v5, v7, v5, vcc
-; GFX1164GISEL-NEXT: v_mov_b32_e32 v6, v4
; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX1164GISEL-NEXT: v_mov_b32_e32 v6, v4
; GFX1164GISEL-NEXT: v_mov_b32_e32 v7, v5
+; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1164GISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:8 row_mask:0xf bank_mask:0xf
-; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164GISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:8 row_mask:0xf bank_mask:0xf
+; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1164GISEL-NEXT: v_cmp_gt_u64_e32 vcc, v[4:5], v[6:7]
; GFX1164GISEL-NEXT: s_waitcnt_depctr depctr_va_vcc(0)
; GFX1164GISEL-NEXT: v_cndmask_b32_e32 v4, v6, v4, vcc
@@ -2291,6 +2304,7 @@ define void @divergent_value_dpp_i64(ptr addrspace(1) %out, i64 %in) #0 {
; GFX1164GISEL-NEXT: v_readlane_b32 s2, v4, 63
; GFX1164GISEL-NEXT: v_readlane_b32 s3, v5, 63
; GFX1164GISEL-NEXT: s_mov_b64 exec, s[0:1]
+; GFX1164GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164GISEL-NEXT: v_mov_b32_e32 v2, s2
; GFX1164GISEL-NEXT: v_mov_b32_e32 v3, s3
; GFX1164GISEL-NEXT: global_store_b64 v[0:1], v[2:3], off
@@ -2316,36 +2330,36 @@ define void @divergent_value_dpp_i64(ptr addrspace(1) %out, i64 %in) #0 {
; GFX1132DAGISEL-NEXT: scratch_store_b32 off, v4, s32 offset:16
; GFX1132DAGISEL-NEXT: scratch_store_b32 off, v5, s32 offset:20
; GFX1132DAGISEL-NEXT: s_mov_b32 exec_lo, s0
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1132DAGISEL-NEXT: s_or_saveexec_b32 s2, -1
-; GFX1132DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX1132DAGISEL-NEXT: v_cndmask_b32_e64 v4, 0, v2, s2
; GFX1132DAGISEL-NEXT: v_cndmask_b32_e64 v5, 0, v3, s2
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132DAGISEL-NEXT: v_dual_mov_b32 v6, v4 :: v_dual_mov_b32 v7, v5
-; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1132DAGISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:1 row_mask:0xf bank_mask:0xf
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132DAGISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:1 row_mask:0xf bank_mask:0xf
-; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX1132DAGISEL-NEXT: v_cmp_gt_u64_e32 vcc_lo, v[4:5], v[6:7]
; GFX1132DAGISEL-NEXT: v_dual_cndmask_b32 v4, v6, v4 :: v_dual_cndmask_b32 v5, v7, v5
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132DAGISEL-NEXT: v_dual_mov_b32 v6, v4 :: v_dual_mov_b32 v7, v5
-; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1132DAGISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:2 row_mask:0xf bank_mask:0xf
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132DAGISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:2 row_mask:0xf bank_mask:0xf
-; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX1132DAGISEL-NEXT: v_cmp_gt_u64_e32 vcc_lo, v[4:5], v[6:7]
; GFX1132DAGISEL-NEXT: v_dual_cndmask_b32 v4, v6, v4 :: v_dual_cndmask_b32 v5, v7, v5
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132DAGISEL-NEXT: v_dual_mov_b32 v6, v4 :: v_dual_mov_b32 v7, v5
-; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1132DAGISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:4 row_mask:0xf bank_mask:0xf
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132DAGISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:4 row_mask:0xf bank_mask:0xf
-; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX1132DAGISEL-NEXT: v_cmp_gt_u64_e32 vcc_lo, v[4:5], v[6:7]
; GFX1132DAGISEL-NEXT: v_dual_cndmask_b32 v4, v6, v4 :: v_dual_cndmask_b32 v5, v7, v5
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132DAGISEL-NEXT: v_dual_mov_b32 v6, v4 :: v_dual_mov_b32 v7, v5
-; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1132DAGISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:8 row_mask:0xf bank_mask:0xf
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132DAGISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:8 row_mask:0xf bank_mask:0xf
-; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1132DAGISEL-NEXT: v_cmp_gt_u64_e32 vcc_lo, v[4:5], v[6:7]
; GFX1132DAGISEL-NEXT: v_dual_cndmask_b32 v4, v6, v4 :: v_dual_cndmask_b32 v5, v7, v5
; GFX1132DAGISEL-NEXT: ds_swizzle_b32 v6, v4 offset:swizzle(BROADCAST,32,15)
@@ -2357,6 +2371,7 @@ define void @divergent_value_dpp_i64(ptr addrspace(1) %out, i64 %in) #0 {
; GFX1132DAGISEL-NEXT: v_readlane_b32 s0, v4, 31
; GFX1132DAGISEL-NEXT: v_readlane_b32 s1, v5, 31
; GFX1132DAGISEL-NEXT: s_mov_b32 exec_lo, s2
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1132DAGISEL-NEXT: v_dual_mov_b32 v3, s1 :: v_dual_mov_b32 v2, s0
; GFX1132DAGISEL-NEXT: global_store_b64 v[0:1], v[2:3], off
; GFX1132DAGISEL-NEXT: s_xor_saveexec_b32 s0, -1
@@ -2379,36 +2394,36 @@ define void @divergent_value_dpp_i64(ptr addrspace(1) %out, i64 %in) #0 {
; GFX1132GISEL-NEXT: scratch_store_b32 off, v4, s32 offset:16
; GFX1132GISEL-NEXT: scratch_store_b32 off, v5, s32 offset:20
; GFX1132GISEL-NEXT: s_mov_b32 exec_lo, s0
+; GFX1132GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1132GISEL-NEXT: s_or_saveexec_b32 s2, -1
-; GFX1132GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX1132GISEL-NEXT: v_cndmask_b32_e64 v4, 0, v2, s2
; GFX1132GISEL-NEXT: v_cndmask_b32_e64 v5, 0, v3, s2
+; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132GISEL-NEXT: v_dual_mov_b32 v6, v4 :: v_dual_mov_b32 v7, v5
-; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1132GISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:1 row_mask:0xf bank_mask:0xf
+; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132GISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:1 row_mask:0xf bank_mask:0xf
-; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX1132GISEL-NEXT: v_cmp_gt_u64_e32 vcc_lo, v[4:5], v[6:7]
; GFX1132GISEL-NEXT: v_dual_cndmask_b32 v4, v6, v4 :: v_dual_cndmask_b32 v5, v7, v5
+; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132GISEL-NEXT: v_dual_mov_b32 v6, v4 :: v_dual_mov_b32 v7, v5
-; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1132GISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:2 row_mask:0xf bank_mask:0xf
+; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132GISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:2 row_mask:0xf bank_mask:0xf
-; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX1132GISEL-NEXT: v_cmp_gt_u64_e32 vcc_lo, v[4:5], v[6:7]
; GFX1132GISEL-NEXT: v_dual_cndmask_b32 v4, v6, v4 :: v_dual_cndmask_b32 v5, v7, v5
+; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132GISEL-NEXT: v_dual_mov_b32 v6, v4 :: v_dual_mov_b32 v7, v5
-; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1132GISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:4 row_mask:0xf bank_mask:0xf
+; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132GISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:4 row_mask:0xf bank_mask:0xf
-; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX1132GISEL-NEXT: v_cmp_gt_u64_e32 vcc_lo, v[4:5], v[6:7]
; GFX1132GISEL-NEXT: v_dual_cndmask_b32 v4, v6, v4 :: v_dual_cndmask_b32 v5, v7, v5
+; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132GISEL-NEXT: v_dual_mov_b32 v6, v4 :: v_dual_mov_b32 v7, v5
-; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1132GISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:8 row_mask:0xf bank_mask:0xf
+; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132GISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:8 row_mask:0xf bank_mask:0xf
-; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1132GISEL-NEXT: v_cmp_gt_u64_e32 vcc_lo, v[4:5], v[6:7]
; GFX1132GISEL-NEXT: v_dual_cndmask_b32 v4, v6, v4 :: v_dual_cndmask_b32 v5, v7, v5
; GFX1132GISEL-NEXT: ds_swizzle_b32 v6, v4 offset:swizzle(BROADCAST,32,15)
@@ -2420,6 +2435,7 @@ define void @divergent_value_dpp_i64(ptr addrspace(1) %out, i64 %in) #0 {
; GFX1132GISEL-NEXT: v_readlane_b32 s0, v4, 31
; GFX1132GISEL-NEXT: v_readlane_b32 s1, v5, 31
; GFX1132GISEL-NEXT: s_mov_b32 exec_lo, s2
+; GFX1132GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1132GISEL-NEXT: v_dual_mov_b32 v3, s1 :: v_dual_mov_b32 v2, s0
; GFX1132GISEL-NEXT: global_store_b64 v[0:1], v[2:3], off
; GFX1132GISEL-NEXT: s_xor_saveexec_b32 s0, -1
@@ -2709,7 +2725,7 @@ define amdgpu_kernel void @default_stratergy(ptr addrspace(1) %out) #0 {
; GFX1164DAGISEL-NEXT: s_or_saveexec_b64 s[2:3], -1
; GFX1164DAGISEL-NEXT: s_waitcnt lgkmcnt(0)
; GFX1164DAGISEL-NEXT: v_max_u32_e32 v1, v1, v2
-; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX1164DAGISEL-NEXT: v_readlane_b32 s4, v1, 63
; GFX1164DAGISEL-NEXT: s_mov_b64 exec, s[2:3]
; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v0, 0
@@ -2744,7 +2760,7 @@ define amdgpu_kernel void @default_stratergy(ptr addrspace(1) %out) #0 {
; GFX1164GISEL-NEXT: s_mov_b64 exec, s[0:1]
; GFX1164GISEL-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
; GFX1164GISEL-NEXT: s_or_saveexec_b64 s[2:3], -1
-; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX1164GISEL-NEXT: v_readlane_b32 s4, v1, 63
; GFX1164GISEL-NEXT: s_mov_b64 exec, s[2:3]
; GFX1164GISEL-NEXT: v_mov_b32_e32 v0, s4
@@ -2771,7 +2787,7 @@ define amdgpu_kernel void @default_stratergy(ptr addrspace(1) %out) #0 {
; GFX1132DAGISEL-NEXT: s_or_saveexec_b32 s2, -1
; GFX1132DAGISEL-NEXT: s_waitcnt lgkmcnt(0)
; GFX1132DAGISEL-NEXT: v_max_u32_e32 v1, v1, v2
-; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX1132DAGISEL-NEXT: v_readlane_b32 s3, v1, 31
; GFX1132DAGISEL-NEXT: s_mov_b32 exec_lo, s2
; GFX1132DAGISEL-NEXT: v_dual_mov_b32 v0, 0 :: v_dual_mov_b32 v3, s3
@@ -2796,7 +2812,7 @@ define amdgpu_kernel void @default_stratergy(ptr addrspace(1) %out) #0 {
; GFX1132GISEL-NEXT: s_mov_b32 exec_lo, s0
; GFX1132GISEL-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
; GFX1132GISEL-NEXT: s_or_saveexec_b32 s2, -1
-; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX1132GISEL-NEXT: v_readlane_b32 s3, v1, 31
; GFX1132GISEL-NEXT: s_mov_b32 exec_lo, s2
; GFX1132GISEL-NEXT: v_dual_mov_b32 v0, s3 :: v_dual_mov_b32 v3, 0
@@ -3178,7 +3194,7 @@ define amdgpu_kernel void @divergent_cfg(ptr addrspace(1) %out, i32 %in) #0 {
; GFX1164DAGISEL-NEXT: v_and_b32_e32 v0, 0x3ff, v0
; GFX1164DAGISEL-NEXT: s_mov_b64 s[0:1], exec
; GFX1164DAGISEL-NEXT: ; implicit-def: $sgpr2
-; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164DAGISEL-NEXT: v_cmpx_lt_u32_e32 15, v0
; GFX1164DAGISEL-NEXT: s_xor_b64 s[0:1], exec, s[0:1]
; GFX1164DAGISEL-NEXT: ; %bb.1: ; %else
@@ -3189,13 +3205,14 @@ define amdgpu_kernel void @divergent_cfg(ptr addrspace(1) %out, i32 %in) #0 {
; GFX1164DAGISEL-NEXT: s_waitcnt lgkmcnt(0)
; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v1, s2
; GFX1164DAGISEL-NEXT: s_xor_b64 exec, exec, s[0:1]
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164DAGISEL-NEXT: s_cbranch_execz .LBB8_6
; GFX1164DAGISEL-NEXT: ; %bb.3: ; %if
; GFX1164DAGISEL-NEXT: s_mov_b64 s[2:3], exec
; GFX1164DAGISEL-NEXT: s_mov_b32 s6, 0
; GFX1164DAGISEL-NEXT: .LBB8_4: ; =>This Inner Loop Header: Depth=1
; GFX1164DAGISEL-NEXT: s_ctz_i32_b64 s7, s[2:3]
-; GFX1164DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_3) | instid1(SALU_CYCLE_1)
; GFX1164DAGISEL-NEXT: v_readlane_b32 s8, v0, s7
; GFX1164DAGISEL-NEXT: s_bitset0_b64 s[2:3], s7
; GFX1164DAGISEL-NEXT: s_max_u32 s6, s6, s8
@@ -3216,7 +3233,7 @@ define amdgpu_kernel void @divergent_cfg(ptr addrspace(1) %out, i32 %in) #0 {
; GFX1164GISEL-NEXT: v_and_b32_e32 v0, 0x3ff, v0
; GFX1164GISEL-NEXT: s_mov_b64 s[0:1], exec
; GFX1164GISEL-NEXT: ; implicit-def: $sgpr2
-; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164GISEL-NEXT: v_cmpx_le_u32_e32 16, v0
; GFX1164GISEL-NEXT: s_xor_b64 s[0:1], exec, s[0:1]
; GFX1164GISEL-NEXT: s_cbranch_execz .LBB8_2
@@ -3229,13 +3246,14 @@ define amdgpu_kernel void @divergent_cfg(ptr addrspace(1) %out, i32 %in) #0 {
; GFX1164GISEL-NEXT: s_or_saveexec_b64 s[0:1], s[0:1]
; GFX1164GISEL-NEXT: v_mov_b32_e32 v1, s2
; GFX1164GISEL-NEXT: s_xor_b64 exec, exec, s[0:1]
+; GFX1164GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164GISEL-NEXT: s_cbranch_execz .LBB8_6
; GFX1164GISEL-NEXT: ; %bb.3: ; %if
; GFX1164GISEL-NEXT: s_mov_b64 s[2:3], exec
; GFX1164GISEL-NEXT: s_mov_b32 s6, 0
; GFX1164GISEL-NEXT: .LBB8_4: ; =>This Inner Loop Header: Depth=1
; GFX1164GISEL-NEXT: s_ctz_i32_b64 s7, s[2:3]
-; GFX1164GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1164GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_3) | instid1(SALU_CYCLE_1)
; GFX1164GISEL-NEXT: v_readlane_b32 s8, v0, s7
; GFX1164GISEL-NEXT: s_bitset0_b64 s[2:3], s7
; GFX1164GISEL-NEXT: s_max_u32 s6, s6, s8
@@ -3256,13 +3274,14 @@ define amdgpu_kernel void @divergent_cfg(ptr addrspace(1) %out, i32 %in) #0 {
; GFX1132DAGISEL-NEXT: v_and_b32_e32 v0, 0x3ff, v0
; GFX1132DAGISEL-NEXT: s_mov_b32 s0, exec_lo
; GFX1132DAGISEL-NEXT: ; implicit-def: $sgpr1
-; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132DAGISEL-NEXT: v_cmpx_lt_u32_e32 15, v0
; GFX1132DAGISEL-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX1132DAGISEL-NEXT: ; %bb.1: ; %else
; GFX1132DAGISEL-NEXT: s_load_b32 s1, s[4:5], 0x2c
; GFX1132DAGISEL-NEXT: ; implicit-def: $vgpr0
; GFX1132DAGISEL-NEXT: ; %bb.2: ; %Flow
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_3) | instid1(SALU_CYCLE_1)
; GFX1132DAGISEL-NEXT: s_or_saveexec_b32 s0, s0
; GFX1132DAGISEL-NEXT: s_waitcnt lgkmcnt(0)
; GFX1132DAGISEL-NEXT: v_mov_b32_e32 v1, s1
@@ -3273,7 +3292,7 @@ define amdgpu_kernel void @divergent_cfg(ptr addrspace(1) %out, i32 %in) #0 {
; GFX1132DAGISEL-NEXT: s_mov_b32 s1, 0
; GFX1132DAGISEL-NEXT: .LBB8_4: ; =>This Inner Loop Header: Depth=1
; GFX1132DAGISEL-NEXT: s_ctz_i32_b32 s3, s2
-; GFX1132DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_3) | instid1(SALU_CYCLE_1)
; GFX1132DAGISEL-NEXT: v_readlane_b32 s6, v0, s3
; GFX1132DAGISEL-NEXT: s_bitset0_b32 s2, s3
; GFX1132DAGISEL-NEXT: s_max_u32 s1, s1, s6
@@ -3294,9 +3313,10 @@ define amdgpu_kernel void @divergent_cfg(ptr addrspace(1) %out, i32 %in) #0 {
; GFX1132GISEL-NEXT: v_and_b32_e32 v0, 0x3ff, v0
; GFX1132GISEL-NEXT: s_mov_b32 s0, exec_lo
; GFX1132GISEL-NEXT: ; implicit-def: $sgpr1
-; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132GISEL-NEXT: v_cmpx_le_u32_e32 16, v0
; GFX1132GISEL-NEXT: s_xor_b32 s0, exec_lo, s0
+; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1132GISEL-NEXT: s_cbranch_execz .LBB8_2
; GFX1132GISEL-NEXT: ; %bb.1: ; %else
; GFX1132GISEL-NEXT: s_load_b32 s1, s[4:5], 0x2c
@@ -3307,13 +3327,14 @@ define amdgpu_kernel void @divergent_cfg(ptr addrspace(1) %out, i32 %in) #0 {
; GFX1132GISEL-NEXT: s_or_saveexec_b32 s0, s0
; GFX1132GISEL-NEXT: v_mov_b32_e32 v1, s1
; GFX1132GISEL-NEXT: s_xor_b32 exec_lo, exec_lo, s0
+; GFX1132GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1132GISEL-NEXT: s_cbranch_execz .LBB8_6
; GFX1132GISEL-NEXT: ; %bb.3: ; %if
; GFX1132GISEL-NEXT: s_mov_b32 s2, exec_lo
; GFX1132GISEL-NEXT: s_mov_b32 s1, 0
; GFX1132GISEL-NEXT: .LBB8_4: ; =>This Inner Loop Header: Depth=1
; GFX1132GISEL-NEXT: s_ctz_i32_b32 s3, s2
-; GFX1132GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1132GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_3) | instid1(SALU_CYCLE_1)
; GFX1132GISEL-NEXT: v_readlane_b32 s6, v0, s3
; GFX1132GISEL-NEXT: s_bitset0_b32 s2, s3
; GFX1132GISEL-NEXT: s_max_u32 s1, s1, s6
@@ -3921,6 +3942,7 @@ define void @divergent_value_i64(ptr addrspace(1) %out, i64 %id.x) #0 {
; GFX1164DAGISEL-NEXT: v_readlane_b32 s4, v2, s8
; GFX1164DAGISEL-NEXT: v_readlane_b32 s5, v3, s8
; GFX1164DAGISEL-NEXT: v_cmp_gt_u64_e32 vcc, s[4:5], v[4:5]
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_3) | instid1(SALU_CYCLE_1)
; GFX1164DAGISEL-NEXT: s_and_b64 s[6:7], vcc, s[2:3]
; GFX1164DAGISEL-NEXT: s_bitset0_b64 s[2:3], s8
; GFX1164DAGISEL-NEXT: s_cselect_b64 s[0:1], s[4:5], s[0:1]
@@ -3945,6 +3967,7 @@ define void @divergent_value_i64(ptr addrspace(1) %out, i64 %id.x) #0 {
; GFX1164GISEL-NEXT: v_readlane_b32 s4, v2, s8
; GFX1164GISEL-NEXT: v_readlane_b32 s5, v3, s8
; GFX1164GISEL-NEXT: v_cmp_gt_u64_e32 vcc, s[4:5], v[4:5]
+; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_3) | instid1(SALU_CYCLE_1)
; GFX1164GISEL-NEXT: s_and_b64 s[6:7], vcc, s[2:3]
; GFX1164GISEL-NEXT: s_bitset0_b64 s[2:3], s8
; GFX1164GISEL-NEXT: s_cselect_b64 s[0:1], s[4:5], s[0:1]
@@ -3972,6 +3995,7 @@ define void @divergent_value_i64(ptr addrspace(1) %out, i64 %id.x) #0 {
; GFX1132DAGISEL-NEXT: s_bitset0_b32 s2, s3
; GFX1132DAGISEL-NEXT: s_cselect_b64 s[0:1], s[4:5], s[0:1]
; GFX1132DAGISEL-NEXT: s_cmp_lg_u32 s2, 0
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1132DAGISEL-NEXT: s_cbranch_scc1 .LBB12_1
; GFX1132DAGISEL-NEXT: ; %bb.2:
; GFX1132DAGISEL-NEXT: v_dual_mov_b32 v3, s1 :: v_dual_mov_b32 v2, s0
@@ -3994,6 +4018,7 @@ define void @divergent_value_i64(ptr addrspace(1) %out, i64 %id.x) #0 {
; GFX1132GISEL-NEXT: s_bitset0_b32 s2, s3
; GFX1132GISEL-NEXT: s_cselect_b64 s[0:1], s[4:5], s[0:1]
; GFX1132GISEL-NEXT: s_cmp_lg_u32 s2, 0
+; GFX1132GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1132GISEL-NEXT: s_cbranch_scc1 .LBB12_1
; GFX1132GISEL-NEXT: ; %bb.2:
; GFX1132GISEL-NEXT: v_dual_mov_b32 v3, s1 :: v_dual_mov_b32 v2, s0
@@ -4265,19 +4290,22 @@ define amdgpu_kernel void @divergent_cfg_i64(ptr addrspace(1) %out, i64 %in, i64
; GFX1164DAGISEL-NEXT: s_load_b64 s[4:5], s[4:5], 0x34
; GFX1164DAGISEL-NEXT: v_and_b32_e32 v0, 0x3ff, v0
; GFX1164DAGISEL-NEXT: s_mov_b64 s[6:7], exec
-; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164DAGISEL-NEXT: v_cmpx_lt_u32_e32 15, v0
; GFX1164DAGISEL-NEXT: s_xor_b64 s[6:7], exec, s[6:7]
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164DAGISEL-NEXT: s_or_saveexec_b64 s[6:7], s[6:7]
; GFX1164DAGISEL-NEXT: s_waitcnt lgkmcnt(0)
; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v0, s2
; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v1, s3
; GFX1164DAGISEL-NEXT: s_xor_b64 exec, exec, s[6:7]
; GFX1164DAGISEL-NEXT: ; %bb.1: ; %if
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v0, s4
; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v1, s5
; GFX1164DAGISEL-NEXT: ; %bb.2: ; %endif
; GFX1164DAGISEL-NEXT: s_or_b64 exec, exec, s[6:7]
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v2, 0
; GFX1164DAGISEL-NEXT: global_store_b64 v2, v[0:1], s[0:1]
; GFX1164DAGISEL-NEXT: s_endpgm
@@ -4288,7 +4316,7 @@ define amdgpu_kernel void @divergent_cfg_i64(ptr addrspace(1) %out, i64 %in, i64
; GFX1164GISEL-NEXT: v_and_b32_e32 v0, 0x3ff, v0
; GFX1164GISEL-NEXT: s_mov_b64 s[8:9], exec
; GFX1164GISEL-NEXT: ; implicit-def: $sgpr6_sgpr7
-; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164GISEL-NEXT: v_cmpx_le_u32_e32 16, v0
; GFX1164GISEL-NEXT: s_xor_b64 s[8:9], exec, s[8:9]
; GFX1164GISEL-NEXT: s_cbranch_execz .LBB13_2
@@ -4301,6 +4329,7 @@ define amdgpu_kernel void @divergent_cfg_i64(ptr addrspace(1) %out, i64 %in, i64
; GFX1164GISEL-NEXT: v_mov_b32_e32 v0, s6
; GFX1164GISEL-NEXT: v_mov_b32_e32 v1, s7
; GFX1164GISEL-NEXT: s_xor_b64 exec, exec, s[2:3]
+; GFX1164GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164GISEL-NEXT: s_cbranch_execz .LBB13_4
; GFX1164GISEL-NEXT: ; %bb.3: ; %if
; GFX1164GISEL-NEXT: s_load_b64 s[4:5], s[4:5], 0x34
@@ -4311,6 +4340,7 @@ define amdgpu_kernel void @divergent_cfg_i64(ptr addrspace(1) %out, i64 %in, i64
; GFX1164GISEL-NEXT: v_mov_b32_e32 v1, s5
; GFX1164GISEL-NEXT: .LBB13_4: ; %endif
; GFX1164GISEL-NEXT: s_or_b64 exec, exec, s[2:3]
+; GFX1164GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164GISEL-NEXT: v_mov_b32_e32 v2, 0
; GFX1164GISEL-NEXT: global_store_b64 v2, v[0:1], s[0:1]
; GFX1164GISEL-NEXT: s_endpgm
@@ -4322,17 +4352,20 @@ define amdgpu_kernel void @divergent_cfg_i64(ptr addrspace(1) %out, i64 %in, i64
; GFX1132DAGISEL-NEXT: s_load_b64 s[4:5], s[4:5], 0x34
; GFX1132DAGISEL-NEXT: v_and_b32_e32 v0, 0x3ff, v0
; GFX1132DAGISEL-NEXT: s_mov_b32 s6, exec_lo
-; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132DAGISEL-NEXT: v_cmpx_lt_u32_e32 15, v0
; GFX1132DAGISEL-NEXT: s_xor_b32 s6, exec_lo, s6
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX1132DAGISEL-NEXT: s_or_saveexec_b32 s6, s6
; GFX1132DAGISEL-NEXT: s_waitcnt lgkmcnt(0)
; GFX1132DAGISEL-NEXT: v_dual_mov_b32 v0, s2 :: v_dual_mov_b32 v1, s3
; GFX1132DAGISEL-NEXT: s_xor_b32 exec_lo, exec_lo, s6
; GFX1132DAGISEL-NEXT: ; %bb.1: ; %if
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1132DAGISEL-NEXT: v_dual_mov_b32 v0, s4 :: v_dual_mov_b32 v1, s5
; GFX1132DAGISEL-NEXT: ; %bb.2: ; %endif
; GFX1132DAGISEL-NEXT: s_or_b32 exec_lo, exec_lo, s6
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1132DAGISEL-NEXT: v_mov_b32_e32 v2, 0
; GFX1132DAGISEL-NEXT: global_store_b64 v2, v[0:1], s[0:1]
; GFX1132DAGISEL-NEXT: s_endpgm
@@ -4343,9 +4376,10 @@ define amdgpu_kernel void @divergent_cfg_i64(ptr addrspace(1) %out, i64 %in, i64
; GFX1132GISEL-NEXT: v_and_b32_e32 v0, 0x3ff, v0
; GFX1132GISEL-NEXT: s_mov_b32 s8, exec_lo
; GFX1132GISEL-NEXT: ; implicit-def: $sgpr6_sgpr7
-; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132GISEL-NEXT: v_cmpx_le_u32_e32 16, v0
; GFX1132GISEL-NEXT: s_xor_b32 s8, exec_lo, s8
+; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1132GISEL-NEXT: s_cbranch_execz .LBB13_2
; GFX1132GISEL-NEXT: ; %bb.1: ; %else
; GFX1132GISEL-NEXT: s_waitcnt lgkmcnt(0)
@@ -4355,6 +4389,7 @@ define amdgpu_kernel void @divergent_cfg_i64(ptr addrspace(1) %out, i64 %in, i64
; GFX1132GISEL-NEXT: s_or_saveexec_b32 s2, s8
; GFX1132GISEL-NEXT: v_dual_mov_b32 v0, s6 :: v_dual_mov_b32 v1, s7
; GFX1132GISEL-NEXT: s_xor_b32 exec_lo, exec_lo, s2
+; GFX1132GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1132GISEL-NEXT: s_cbranch_execz .LBB13_4
; GFX1132GISEL-NEXT: ; %bb.3: ; %if
; GFX1132GISEL-NEXT: s_load_b64 s[4:5], s[4:5], 0x34
@@ -4364,6 +4399,7 @@ define amdgpu_kernel void @divergent_cfg_i64(ptr addrspace(1) %out, i64 %in, i64
; GFX1132GISEL-NEXT: v_dual_mov_b32 v0, s4 :: v_dual_mov_b32 v1, s5
; GFX1132GISEL-NEXT: .LBB13_4: ; %endif
; GFX1132GISEL-NEXT: s_or_b32 exec_lo, exec_lo, s2
+; GFX1132GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1132GISEL-NEXT: v_mov_b32_e32 v2, 0
; GFX1132GISEL-NEXT: global_store_b64 v2, v[0:1], s[0:1]
; GFX1132GISEL-NEXT: s_endpgm
diff --git a/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.reduce.umin.ll b/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.reduce.umin.ll
index 7a3a9398910896..4f2f188c7b4f6e 100644
--- a/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.reduce.umin.ll
+++ b/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.reduce.umin.ll
@@ -403,6 +403,7 @@ define void @divergent_value_i16(ptr addrspace(1) %out, i16 %in) {
; GFX1164DAGISEL-FAKE16-NEXT: s_bitset0_b64 s[0:1], s3
; GFX1164DAGISEL-FAKE16-NEXT: s_min_u32 s2, s2, s4
; GFX1164DAGISEL-FAKE16-NEXT: s_cmp_lg_u64 s[0:1], 0
+; GFX1164DAGISEL-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164DAGISEL-FAKE16-NEXT: s_cbranch_scc1 .LBB1_1
; GFX1164DAGISEL-FAKE16-NEXT: ; %bb.2:
; GFX1164DAGISEL-FAKE16-NEXT: v_mov_b32_e32 v2, s2
@@ -422,6 +423,7 @@ define void @divergent_value_i16(ptr addrspace(1) %out, i16 %in) {
; GFX1164GISEL-NEXT: s_bitset0_b64 s[0:1], s3
; GFX1164GISEL-NEXT: s_min_u32 s2, s2, s4
; GFX1164GISEL-NEXT: s_cmp_lg_u64 s[0:1], 0
+; GFX1164GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164GISEL-NEXT: s_cbranch_scc1 .LBB1_1
; GFX1164GISEL-NEXT: ; %bb.2:
; GFX1164GISEL-NEXT: v_mov_b32_e32 v2, s2
@@ -441,6 +443,7 @@ define void @divergent_value_i16(ptr addrspace(1) %out, i16 %in) {
; GFX1132DAGISEL-FAKE16-NEXT: s_bitset0_b32 s1, s2
; GFX1132DAGISEL-FAKE16-NEXT: s_min_u32 s0, s0, s3
; GFX1132DAGISEL-FAKE16-NEXT: s_cmp_lg_u32 s1, 0
+; GFX1132DAGISEL-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1132DAGISEL-FAKE16-NEXT: s_cbranch_scc1 .LBB1_1
; GFX1132DAGISEL-FAKE16-NEXT: ; %bb.2:
; GFX1132DAGISEL-FAKE16-NEXT: v_mov_b32_e32 v2, s0
@@ -460,6 +463,7 @@ define void @divergent_value_i16(ptr addrspace(1) %out, i16 %in) {
; GFX1132GISEL-NEXT: s_bitset0_b32 s1, s2
; GFX1132GISEL-NEXT: s_min_u32 s0, s0, s3
; GFX1132GISEL-NEXT: s_cmp_lg_u32 s1, 0
+; GFX1132GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1132GISEL-NEXT: s_cbranch_scc1 .LBB1_1
; GFX1132GISEL-NEXT: ; %bb.2:
; GFX1132GISEL-NEXT: v_mov_b32_e32 v2, s0
@@ -479,6 +483,7 @@ define void @divergent_value_i16(ptr addrspace(1) %out, i16 %in) {
; GFX1164DAGISEL-TRUE16-NEXT: s_bitset0_b64 s[0:1], s3
; GFX1164DAGISEL-TRUE16-NEXT: s_min_u32 s2, s2, s4
; GFX1164DAGISEL-TRUE16-NEXT: s_cmp_lg_u64 s[0:1], 0
+; GFX1164DAGISEL-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164DAGISEL-TRUE16-NEXT: s_cbranch_scc1 .LBB1_1
; GFX1164DAGISEL-TRUE16-NEXT: ; %bb.2:
; GFX1164DAGISEL-TRUE16-NEXT: v_mov_b32_e32 v2, s2
@@ -498,6 +503,7 @@ define void @divergent_value_i16(ptr addrspace(1) %out, i16 %in) {
; GFX1132DAGISEL-TRUE16-NEXT: s_bitset0_b32 s1, s2
; GFX1132DAGISEL-TRUE16-NEXT: s_min_u32 s0, s0, s3
; GFX1132DAGISEL-TRUE16-NEXT: s_cmp_lg_u32 s1, 0
+; GFX1132DAGISEL-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1132DAGISEL-TRUE16-NEXT: s_cbranch_scc1 .LBB1_1
; GFX1132DAGISEL-TRUE16-NEXT: ; %bb.2:
; GFX1132DAGISEL-TRUE16-NEXT: v_mov_b32_e32 v2, s0
@@ -852,6 +858,7 @@ define amdgpu_kernel void @divergent_value(ptr addrspace(1) %out) #0 {
; GFX1164DAGISEL-NEXT: s_bitset0_b64 s[2:3], s5
; GFX1164DAGISEL-NEXT: s_min_u32 s4, s4, s6
; GFX1164DAGISEL-NEXT: s_cmp_lg_u64 s[2:3], 0
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164DAGISEL-NEXT: s_cbranch_scc1 .LBB3_1
; GFX1164DAGISEL-NEXT: ; %bb.2:
; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v0, s4
@@ -872,6 +879,7 @@ define amdgpu_kernel void @divergent_value(ptr addrspace(1) %out) #0 {
; GFX1164GISEL-NEXT: s_bitset0_b64 s[2:3], s5
; GFX1164GISEL-NEXT: s_min_u32 s4, s4, s6
; GFX1164GISEL-NEXT: s_cmp_lg_u64 s[2:3], 0
+; GFX1164GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164GISEL-NEXT: s_cbranch_scc1 .LBB3_1
; GFX1164GISEL-NEXT: ; %bb.2:
; GFX1164GISEL-NEXT: v_mov_b32_e32 v0, s4
@@ -893,6 +901,7 @@ define amdgpu_kernel void @divergent_value(ptr addrspace(1) %out) #0 {
; GFX1132DAGISEL-NEXT: s_bitset0_b32 s3, s4
; GFX1132DAGISEL-NEXT: s_min_u32 s2, s2, s5
; GFX1132DAGISEL-NEXT: s_cmp_lg_u32 s3, 0
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1132DAGISEL-NEXT: s_cbranch_scc1 .LBB3_1
; GFX1132DAGISEL-NEXT: ; %bb.2:
; GFX1132DAGISEL-NEXT: v_mov_b32_e32 v0, s2
@@ -913,6 +922,7 @@ define amdgpu_kernel void @divergent_value(ptr addrspace(1) %out) #0 {
; GFX1132GISEL-NEXT: s_bitset0_b32 s3, s4
; GFX1132GISEL-NEXT: s_min_u32 s2, s2, s5
; GFX1132GISEL-NEXT: s_cmp_lg_u32 s3, 0
+; GFX1132GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1132GISEL-NEXT: s_cbranch_scc1 .LBB3_1
; GFX1132GISEL-NEXT: ; %bb.2:
; GFX1132GISEL-NEXT: v_dual_mov_b32 v0, s2 :: v_dual_mov_b32 v1, 0
@@ -1332,7 +1342,7 @@ define amdgpu_kernel void @divergent_value_dpp(ptr addrspace(1) %out) #0 {
; GFX1164DAGISEL-NEXT: s_or_saveexec_b64 s[2:3], -1
; GFX1164DAGISEL-NEXT: s_waitcnt lgkmcnt(0)
; GFX1164DAGISEL-NEXT: v_min_u32_e32 v1, v1, v2
-; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX1164DAGISEL-NEXT: v_readlane_b32 s4, v1, 63
; GFX1164DAGISEL-NEXT: s_mov_b64 exec, s[2:3]
; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v0, 0
@@ -1367,7 +1377,7 @@ define amdgpu_kernel void @divergent_value_dpp(ptr addrspace(1) %out) #0 {
; GFX1164GISEL-NEXT: s_mov_b64 exec, s[0:1]
; GFX1164GISEL-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
; GFX1164GISEL-NEXT: s_or_saveexec_b64 s[2:3], -1
-; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX1164GISEL-NEXT: v_readlane_b32 s4, v1, 63
; GFX1164GISEL-NEXT: s_mov_b64 exec, s[2:3]
; GFX1164GISEL-NEXT: v_mov_b32_e32 v0, s4
@@ -1394,7 +1404,7 @@ define amdgpu_kernel void @divergent_value_dpp(ptr addrspace(1) %out) #0 {
; GFX1132DAGISEL-NEXT: s_or_saveexec_b32 s2, -1
; GFX1132DAGISEL-NEXT: s_waitcnt lgkmcnt(0)
; GFX1132DAGISEL-NEXT: v_min_u32_e32 v1, v1, v2
-; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX1132DAGISEL-NEXT: v_readlane_b32 s3, v1, 31
; GFX1132DAGISEL-NEXT: s_mov_b32 exec_lo, s2
; GFX1132DAGISEL-NEXT: v_dual_mov_b32 v0, 0 :: v_dual_mov_b32 v3, s3
@@ -1419,7 +1429,7 @@ define amdgpu_kernel void @divergent_value_dpp(ptr addrspace(1) %out) #0 {
; GFX1132GISEL-NEXT: s_mov_b32 exec_lo, s0
; GFX1132GISEL-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
; GFX1132GISEL-NEXT: s_or_saveexec_b32 s2, -1
-; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX1132GISEL-NEXT: v_readlane_b32 s3, v1, 31
; GFX1132GISEL-NEXT: s_mov_b32 exec_lo, s2
; GFX1132GISEL-NEXT: v_dual_mov_b32 v0, s3 :: v_dual_mov_b32 v3, 0
@@ -2117,52 +2127,53 @@ define void @divergent_value_dpp_i64(ptr addrspace(1) %out, i64 %in) #0 {
; GFX1164DAGISEL-NEXT: scratch_store_b32 off, v4, s32 offset:20
; GFX1164DAGISEL-NEXT: scratch_store_b32 off, v5, s32 offset:24
; GFX1164DAGISEL-NEXT: s_mov_b64 exec, s[0:1]
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1164DAGISEL-NEXT: s_or_saveexec_b64 s[0:1], -1
-; GFX1164DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(VALU_DEP_3)
; GFX1164DAGISEL-NEXT: v_cndmask_b32_e64 v4, -1, v2, s[0:1]
; GFX1164DAGISEL-NEXT: v_cndmask_b32_e64 v5, -1, v3, s[0:1]
; GFX1164DAGISEL-NEXT: v_mbcnt_lo_u32_b32 v8, -1, 0
-; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v6, v4
; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v6, v4
; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v7, v5
-; GFX1164DAGISEL-NEXT: v_mbcnt_hi_u32_b32 v8, -1, v8
; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX1164DAGISEL-NEXT: v_mbcnt_hi_u32_b32 v8, -1, v8
; GFX1164DAGISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:1 row_mask:0xf bank_mask:0xf
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_3)
; GFX1164DAGISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:1 row_mask:0xf bank_mask:0xf
-; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1164DAGISEL-NEXT: v_add_nc_u32_e32 v8, 32, v8
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1164DAGISEL-NEXT: v_cmp_lt_u64_e32 vcc, v[4:5], v[6:7]
-; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_2) | instid1(VALU_DEP_2)
; GFX1164DAGISEL-NEXT: v_mul_lo_u32 v8, 4, v8
; GFX1164DAGISEL-NEXT: v_cndmask_b32_e32 v4, v6, v4, vcc
; GFX1164DAGISEL-NEXT: v_cndmask_b32_e32 v5, v7, v5, vcc
-; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v6, v4
; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v6, v4
; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v7, v5
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1164DAGISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:2 row_mask:0xf bank_mask:0xf
-; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164DAGISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:2 row_mask:0xf bank_mask:0xf
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_3) | instid1(VALU_DEP_2)
; GFX1164DAGISEL-NEXT: v_cmp_lt_u64_e32 vcc, v[4:5], v[6:7]
; GFX1164DAGISEL-NEXT: s_waitcnt_depctr depctr_va_vcc(0)
; GFX1164DAGISEL-NEXT: v_cndmask_b32_e32 v4, v6, v4, vcc
; GFX1164DAGISEL-NEXT: v_cndmask_b32_e32 v5, v7, v5, vcc
-; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v6, v4
-; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v7, v5
; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v7, v5
; GFX1164DAGISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:4 row_mask:0xf bank_mask:0xf
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164DAGISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:4 row_mask:0xf bank_mask:0xf
-; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_3) | instid1(VALU_DEP_2)
; GFX1164DAGISEL-NEXT: v_cmp_lt_u64_e32 vcc, v[4:5], v[6:7]
; GFX1164DAGISEL-NEXT: s_waitcnt_depctr depctr_va_vcc(0)
; GFX1164DAGISEL-NEXT: v_cndmask_b32_e32 v4, v6, v4, vcc
; GFX1164DAGISEL-NEXT: v_cndmask_b32_e32 v5, v7, v5, vcc
-; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v6, v4
; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v6, v4
; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v7, v5
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1164DAGISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:8 row_mask:0xf bank_mask:0xf
-; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164DAGISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:8 row_mask:0xf bank_mask:0xf
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1164DAGISEL-NEXT: v_cmp_lt_u64_e32 vcc, v[4:5], v[6:7]
; GFX1164DAGISEL-NEXT: s_waitcnt_depctr depctr_va_vcc(0)
; GFX1164DAGISEL-NEXT: v_cndmask_b32_e32 v4, v6, v4, vcc
@@ -2185,6 +2196,7 @@ define void @divergent_value_dpp_i64(ptr addrspace(1) %out, i64 %in) #0 {
; GFX1164DAGISEL-NEXT: v_readlane_b32 s2, v4, 63
; GFX1164DAGISEL-NEXT: v_readlane_b32 s3, v5, 63
; GFX1164DAGISEL-NEXT: s_mov_b64 exec, s[0:1]
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v2, s2
; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v3, s3
; GFX1164DAGISEL-NEXT: global_store_b64 v[0:1], v[2:3], off
@@ -2211,52 +2223,53 @@ define void @divergent_value_dpp_i64(ptr addrspace(1) %out, i64 %in) #0 {
; GFX1164GISEL-NEXT: scratch_store_b32 off, v4, s32 offset:20
; GFX1164GISEL-NEXT: scratch_store_b32 off, v5, s32 offset:24
; GFX1164GISEL-NEXT: s_mov_b64 exec, s[0:1]
+; GFX1164GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1164GISEL-NEXT: s_or_saveexec_b64 s[0:1], -1
-; GFX1164GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(VALU_DEP_3)
; GFX1164GISEL-NEXT: v_cndmask_b32_e64 v4, -1, v2, s[0:1]
; GFX1164GISEL-NEXT: v_cndmask_b32_e64 v5, -1, v3, s[0:1]
; GFX1164GISEL-NEXT: v_mbcnt_lo_u32_b32 v8, -1, 0
-; GFX1164GISEL-NEXT: v_mov_b32_e32 v6, v4
; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX1164GISEL-NEXT: v_mov_b32_e32 v6, v4
; GFX1164GISEL-NEXT: v_mov_b32_e32 v7, v5
-; GFX1164GISEL-NEXT: v_mbcnt_hi_u32_b32 v8, -1, v8
; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX1164GISEL-NEXT: v_mbcnt_hi_u32_b32 v8, -1, v8
; GFX1164GISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:1 row_mask:0xf bank_mask:0xf
+; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_3)
; GFX1164GISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:1 row_mask:0xf bank_mask:0xf
-; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1164GISEL-NEXT: v_add_nc_u32_e32 v8, 32, v8
+; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1164GISEL-NEXT: v_cmp_lt_u64_e32 vcc, v[4:5], v[6:7]
-; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_2) | instid1(VALU_DEP_2)
; GFX1164GISEL-NEXT: v_mul_lo_u32 v8, 4, v8
; GFX1164GISEL-NEXT: v_cndmask_b32_e32 v4, v6, v4, vcc
; GFX1164GISEL-NEXT: v_cndmask_b32_e32 v5, v7, v5, vcc
-; GFX1164GISEL-NEXT: v_mov_b32_e32 v6, v4
; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX1164GISEL-NEXT: v_mov_b32_e32 v6, v4
; GFX1164GISEL-NEXT: v_mov_b32_e32 v7, v5
+; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1164GISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:2 row_mask:0xf bank_mask:0xf
-; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164GISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:2 row_mask:0xf bank_mask:0xf
+; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_3) | instid1(VALU_DEP_2)
; GFX1164GISEL-NEXT: v_cmp_lt_u64_e32 vcc, v[4:5], v[6:7]
; GFX1164GISEL-NEXT: s_waitcnt_depctr depctr_va_vcc(0)
; GFX1164GISEL-NEXT: v_cndmask_b32_e32 v4, v6, v4, vcc
; GFX1164GISEL-NEXT: v_cndmask_b32_e32 v5, v7, v5, vcc
-; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1164GISEL-NEXT: v_mov_b32_e32 v6, v4
-; GFX1164GISEL-NEXT: v_mov_b32_e32 v7, v5
; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX1164GISEL-NEXT: v_mov_b32_e32 v7, v5
; GFX1164GISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:4 row_mask:0xf bank_mask:0xf
+; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164GISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:4 row_mask:0xf bank_mask:0xf
-; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_3) | instid1(VALU_DEP_2)
; GFX1164GISEL-NEXT: v_cmp_lt_u64_e32 vcc, v[4:5], v[6:7]
; GFX1164GISEL-NEXT: s_waitcnt_depctr depctr_va_vcc(0)
; GFX1164GISEL-NEXT: v_cndmask_b32_e32 v4, v6, v4, vcc
; GFX1164GISEL-NEXT: v_cndmask_b32_e32 v5, v7, v5, vcc
-; GFX1164GISEL-NEXT: v_mov_b32_e32 v6, v4
; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX1164GISEL-NEXT: v_mov_b32_e32 v6, v4
; GFX1164GISEL-NEXT: v_mov_b32_e32 v7, v5
+; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1164GISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:8 row_mask:0xf bank_mask:0xf
-; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164GISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:8 row_mask:0xf bank_mask:0xf
+; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1164GISEL-NEXT: v_cmp_lt_u64_e32 vcc, v[4:5], v[6:7]
; GFX1164GISEL-NEXT: s_waitcnt_depctr depctr_va_vcc(0)
; GFX1164GISEL-NEXT: v_cndmask_b32_e32 v4, v6, v4, vcc
@@ -2279,6 +2292,7 @@ define void @divergent_value_dpp_i64(ptr addrspace(1) %out, i64 %in) #0 {
; GFX1164GISEL-NEXT: v_readlane_b32 s2, v4, 63
; GFX1164GISEL-NEXT: v_readlane_b32 s3, v5, 63
; GFX1164GISEL-NEXT: s_mov_b64 exec, s[0:1]
+; GFX1164GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164GISEL-NEXT: v_mov_b32_e32 v2, s2
; GFX1164GISEL-NEXT: v_mov_b32_e32 v3, s3
; GFX1164GISEL-NEXT: global_store_b64 v[0:1], v[2:3], off
@@ -2304,36 +2318,36 @@ define void @divergent_value_dpp_i64(ptr addrspace(1) %out, i64 %in) #0 {
; GFX1132DAGISEL-NEXT: scratch_store_b32 off, v4, s32 offset:16
; GFX1132DAGISEL-NEXT: scratch_store_b32 off, v5, s32 offset:20
; GFX1132DAGISEL-NEXT: s_mov_b32 exec_lo, s0
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1132DAGISEL-NEXT: s_or_saveexec_b32 s2, -1
-; GFX1132DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX1132DAGISEL-NEXT: v_cndmask_b32_e64 v4, -1, v2, s2
; GFX1132DAGISEL-NEXT: v_cndmask_b32_e64 v5, -1, v3, s2
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132DAGISEL-NEXT: v_dual_mov_b32 v6, v4 :: v_dual_mov_b32 v7, v5
-; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1132DAGISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:1 row_mask:0xf bank_mask:0xf
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132DAGISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:1 row_mask:0xf bank_mask:0xf
-; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX1132DAGISEL-NEXT: v_cmp_lt_u64_e32 vcc_lo, v[4:5], v[6:7]
; GFX1132DAGISEL-NEXT: v_dual_cndmask_b32 v4, v6, v4 :: v_dual_cndmask_b32 v5, v7, v5
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132DAGISEL-NEXT: v_dual_mov_b32 v6, v4 :: v_dual_mov_b32 v7, v5
-; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1132DAGISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:2 row_mask:0xf bank_mask:0xf
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132DAGISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:2 row_mask:0xf bank_mask:0xf
-; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX1132DAGISEL-NEXT: v_cmp_lt_u64_e32 vcc_lo, v[4:5], v[6:7]
; GFX1132DAGISEL-NEXT: v_dual_cndmask_b32 v4, v6, v4 :: v_dual_cndmask_b32 v5, v7, v5
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132DAGISEL-NEXT: v_dual_mov_b32 v6, v4 :: v_dual_mov_b32 v7, v5
-; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1132DAGISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:4 row_mask:0xf bank_mask:0xf
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132DAGISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:4 row_mask:0xf bank_mask:0xf
-; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX1132DAGISEL-NEXT: v_cmp_lt_u64_e32 vcc_lo, v[4:5], v[6:7]
; GFX1132DAGISEL-NEXT: v_dual_cndmask_b32 v4, v6, v4 :: v_dual_cndmask_b32 v5, v7, v5
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132DAGISEL-NEXT: v_dual_mov_b32 v6, v4 :: v_dual_mov_b32 v7, v5
-; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1132DAGISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:8 row_mask:0xf bank_mask:0xf
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132DAGISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:8 row_mask:0xf bank_mask:0xf
-; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1132DAGISEL-NEXT: v_cmp_lt_u64_e32 vcc_lo, v[4:5], v[6:7]
; GFX1132DAGISEL-NEXT: v_dual_cndmask_b32 v4, v6, v4 :: v_dual_cndmask_b32 v5, v7, v5
; GFX1132DAGISEL-NEXT: ds_swizzle_b32 v6, v4 offset:swizzle(BROADCAST,32,15)
@@ -2345,6 +2359,7 @@ define void @divergent_value_dpp_i64(ptr addrspace(1) %out, i64 %in) #0 {
; GFX1132DAGISEL-NEXT: v_readlane_b32 s0, v4, 31
; GFX1132DAGISEL-NEXT: v_readlane_b32 s1, v5, 31
; GFX1132DAGISEL-NEXT: s_mov_b32 exec_lo, s2
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1132DAGISEL-NEXT: v_dual_mov_b32 v3, s1 :: v_dual_mov_b32 v2, s0
; GFX1132DAGISEL-NEXT: global_store_b64 v[0:1], v[2:3], off
; GFX1132DAGISEL-NEXT: s_xor_saveexec_b32 s0, -1
@@ -2367,36 +2382,36 @@ define void @divergent_value_dpp_i64(ptr addrspace(1) %out, i64 %in) #0 {
; GFX1132GISEL-NEXT: scratch_store_b32 off, v4, s32 offset:16
; GFX1132GISEL-NEXT: scratch_store_b32 off, v5, s32 offset:20
; GFX1132GISEL-NEXT: s_mov_b32 exec_lo, s0
+; GFX1132GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1132GISEL-NEXT: s_or_saveexec_b32 s2, -1
-; GFX1132GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX1132GISEL-NEXT: v_cndmask_b32_e64 v4, -1, v2, s2
; GFX1132GISEL-NEXT: v_cndmask_b32_e64 v5, -1, v3, s2
+; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132GISEL-NEXT: v_dual_mov_b32 v6, v4 :: v_dual_mov_b32 v7, v5
-; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1132GISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:1 row_mask:0xf bank_mask:0xf
+; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132GISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:1 row_mask:0xf bank_mask:0xf
-; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX1132GISEL-NEXT: v_cmp_lt_u64_e32 vcc_lo, v[4:5], v[6:7]
; GFX1132GISEL-NEXT: v_dual_cndmask_b32 v4, v6, v4 :: v_dual_cndmask_b32 v5, v7, v5
+; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132GISEL-NEXT: v_dual_mov_b32 v6, v4 :: v_dual_mov_b32 v7, v5
-; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1132GISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:2 row_mask:0xf bank_mask:0xf
+; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132GISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:2 row_mask:0xf bank_mask:0xf
-; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX1132GISEL-NEXT: v_cmp_lt_u64_e32 vcc_lo, v[4:5], v[6:7]
; GFX1132GISEL-NEXT: v_dual_cndmask_b32 v4, v6, v4 :: v_dual_cndmask_b32 v5, v7, v5
+; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132GISEL-NEXT: v_dual_mov_b32 v6, v4 :: v_dual_mov_b32 v7, v5
-; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1132GISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:4 row_mask:0xf bank_mask:0xf
+; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132GISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:4 row_mask:0xf bank_mask:0xf
-; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX1132GISEL-NEXT: v_cmp_lt_u64_e32 vcc_lo, v[4:5], v[6:7]
; GFX1132GISEL-NEXT: v_dual_cndmask_b32 v4, v6, v4 :: v_dual_cndmask_b32 v5, v7, v5
+; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132GISEL-NEXT: v_dual_mov_b32 v6, v4 :: v_dual_mov_b32 v7, v5
-; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1132GISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:8 row_mask:0xf bank_mask:0xf
+; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132GISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:8 row_mask:0xf bank_mask:0xf
-; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1132GISEL-NEXT: v_cmp_lt_u64_e32 vcc_lo, v[4:5], v[6:7]
; GFX1132GISEL-NEXT: v_dual_cndmask_b32 v4, v6, v4 :: v_dual_cndmask_b32 v5, v7, v5
; GFX1132GISEL-NEXT: ds_swizzle_b32 v6, v4 offset:swizzle(BROADCAST,32,15)
@@ -2408,6 +2423,7 @@ define void @divergent_value_dpp_i64(ptr addrspace(1) %out, i64 %in) #0 {
; GFX1132GISEL-NEXT: v_readlane_b32 s0, v4, 31
; GFX1132GISEL-NEXT: v_readlane_b32 s1, v5, 31
; GFX1132GISEL-NEXT: s_mov_b32 exec_lo, s2
+; GFX1132GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1132GISEL-NEXT: v_dual_mov_b32 v3, s1 :: v_dual_mov_b32 v2, s0
; GFX1132GISEL-NEXT: global_store_b64 v[0:1], v[2:3], off
; GFX1132GISEL-NEXT: s_xor_saveexec_b32 s0, -1
@@ -2697,7 +2713,7 @@ define amdgpu_kernel void @default_stratergy(ptr addrspace(1) %out) #0 {
; GFX1164DAGISEL-NEXT: s_or_saveexec_b64 s[2:3], -1
; GFX1164DAGISEL-NEXT: s_waitcnt lgkmcnt(0)
; GFX1164DAGISEL-NEXT: v_min_u32_e32 v1, v1, v2
-; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX1164DAGISEL-NEXT: v_readlane_b32 s4, v1, 63
; GFX1164DAGISEL-NEXT: s_mov_b64 exec, s[2:3]
; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v0, 0
@@ -2732,7 +2748,7 @@ define amdgpu_kernel void @default_stratergy(ptr addrspace(1) %out) #0 {
; GFX1164GISEL-NEXT: s_mov_b64 exec, s[0:1]
; GFX1164GISEL-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
; GFX1164GISEL-NEXT: s_or_saveexec_b64 s[2:3], -1
-; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX1164GISEL-NEXT: v_readlane_b32 s4, v1, 63
; GFX1164GISEL-NEXT: s_mov_b64 exec, s[2:3]
; GFX1164GISEL-NEXT: v_mov_b32_e32 v0, s4
@@ -2759,7 +2775,7 @@ define amdgpu_kernel void @default_stratergy(ptr addrspace(1) %out) #0 {
; GFX1132DAGISEL-NEXT: s_or_saveexec_b32 s2, -1
; GFX1132DAGISEL-NEXT: s_waitcnt lgkmcnt(0)
; GFX1132DAGISEL-NEXT: v_min_u32_e32 v1, v1, v2
-; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX1132DAGISEL-NEXT: v_readlane_b32 s3, v1, 31
; GFX1132DAGISEL-NEXT: s_mov_b32 exec_lo, s2
; GFX1132DAGISEL-NEXT: v_dual_mov_b32 v0, 0 :: v_dual_mov_b32 v3, s3
@@ -2784,7 +2800,7 @@ define amdgpu_kernel void @default_stratergy(ptr addrspace(1) %out) #0 {
; GFX1132GISEL-NEXT: s_mov_b32 exec_lo, s0
; GFX1132GISEL-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
; GFX1132GISEL-NEXT: s_or_saveexec_b32 s2, -1
-; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX1132GISEL-NEXT: v_readlane_b32 s3, v1, 31
; GFX1132GISEL-NEXT: s_mov_b32 exec_lo, s2
; GFX1132GISEL-NEXT: v_dual_mov_b32 v0, s3 :: v_dual_mov_b32 v3, 0
@@ -3166,7 +3182,7 @@ define amdgpu_kernel void @divergent_cfg(ptr addrspace(1) %out, i32 %in) #0 {
; GFX1164DAGISEL-NEXT: v_and_b32_e32 v0, 0x3ff, v0
; GFX1164DAGISEL-NEXT: s_mov_b64 s[0:1], exec
; GFX1164DAGISEL-NEXT: ; implicit-def: $sgpr2
-; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164DAGISEL-NEXT: v_cmpx_lt_u32_e32 15, v0
; GFX1164DAGISEL-NEXT: s_xor_b64 s[0:1], exec, s[0:1]
; GFX1164DAGISEL-NEXT: ; %bb.1: ; %else
@@ -3177,13 +3193,14 @@ define amdgpu_kernel void @divergent_cfg(ptr addrspace(1) %out, i32 %in) #0 {
; GFX1164DAGISEL-NEXT: s_waitcnt lgkmcnt(0)
; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v1, s2
; GFX1164DAGISEL-NEXT: s_xor_b64 exec, exec, s[0:1]
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164DAGISEL-NEXT: s_cbranch_execz .LBB8_6
; GFX1164DAGISEL-NEXT: ; %bb.3: ; %if
; GFX1164DAGISEL-NEXT: s_mov_b64 s[2:3], exec
; GFX1164DAGISEL-NEXT: s_mov_b32 s6, -1
; GFX1164DAGISEL-NEXT: .LBB8_4: ; =>This Inner Loop Header: Depth=1
; GFX1164DAGISEL-NEXT: s_ctz_i32_b64 s7, s[2:3]
-; GFX1164DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_3) | instid1(SALU_CYCLE_1)
; GFX1164DAGISEL-NEXT: v_readlane_b32 s8, v0, s7
; GFX1164DAGISEL-NEXT: s_bitset0_b64 s[2:3], s7
; GFX1164DAGISEL-NEXT: s_min_u32 s6, s6, s8
@@ -3204,7 +3221,7 @@ define amdgpu_kernel void @divergent_cfg(ptr addrspace(1) %out, i32 %in) #0 {
; GFX1164GISEL-NEXT: v_and_b32_e32 v0, 0x3ff, v0
; GFX1164GISEL-NEXT: s_mov_b64 s[0:1], exec
; GFX1164GISEL-NEXT: ; implicit-def: $sgpr2
-; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164GISEL-NEXT: v_cmpx_le_u32_e32 16, v0
; GFX1164GISEL-NEXT: s_xor_b64 s[0:1], exec, s[0:1]
; GFX1164GISEL-NEXT: s_cbranch_execz .LBB8_2
@@ -3217,13 +3234,14 @@ define amdgpu_kernel void @divergent_cfg(ptr addrspace(1) %out, i32 %in) #0 {
; GFX1164GISEL-NEXT: s_or_saveexec_b64 s[0:1], s[0:1]
; GFX1164GISEL-NEXT: v_mov_b32_e32 v1, s2
; GFX1164GISEL-NEXT: s_xor_b64 exec, exec, s[0:1]
+; GFX1164GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164GISEL-NEXT: s_cbranch_execz .LBB8_6
; GFX1164GISEL-NEXT: ; %bb.3: ; %if
; GFX1164GISEL-NEXT: s_mov_b64 s[2:3], exec
; GFX1164GISEL-NEXT: s_mov_b32 s6, -1
; GFX1164GISEL-NEXT: .LBB8_4: ; =>This Inner Loop Header: Depth=1
; GFX1164GISEL-NEXT: s_ctz_i32_b64 s7, s[2:3]
-; GFX1164GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1164GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_3) | instid1(SALU_CYCLE_1)
; GFX1164GISEL-NEXT: v_readlane_b32 s8, v0, s7
; GFX1164GISEL-NEXT: s_bitset0_b64 s[2:3], s7
; GFX1164GISEL-NEXT: s_min_u32 s6, s6, s8
@@ -3244,13 +3262,14 @@ define amdgpu_kernel void @divergent_cfg(ptr addrspace(1) %out, i32 %in) #0 {
; GFX1132DAGISEL-NEXT: v_and_b32_e32 v0, 0x3ff, v0
; GFX1132DAGISEL-NEXT: s_mov_b32 s0, exec_lo
; GFX1132DAGISEL-NEXT: ; implicit-def: $sgpr1
-; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132DAGISEL-NEXT: v_cmpx_lt_u32_e32 15, v0
; GFX1132DAGISEL-NEXT: s_xor_b32 s0, exec_lo, s0
; GFX1132DAGISEL-NEXT: ; %bb.1: ; %else
; GFX1132DAGISEL-NEXT: s_load_b32 s1, s[4:5], 0x2c
; GFX1132DAGISEL-NEXT: ; implicit-def: $vgpr0
; GFX1132DAGISEL-NEXT: ; %bb.2: ; %Flow
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_3) | instid1(SALU_CYCLE_1)
; GFX1132DAGISEL-NEXT: s_or_saveexec_b32 s0, s0
; GFX1132DAGISEL-NEXT: s_waitcnt lgkmcnt(0)
; GFX1132DAGISEL-NEXT: v_mov_b32_e32 v1, s1
@@ -3261,7 +3280,7 @@ define amdgpu_kernel void @divergent_cfg(ptr addrspace(1) %out, i32 %in) #0 {
; GFX1132DAGISEL-NEXT: s_mov_b32 s1, -1
; GFX1132DAGISEL-NEXT: .LBB8_4: ; =>This Inner Loop Header: Depth=1
; GFX1132DAGISEL-NEXT: s_ctz_i32_b32 s3, s2
-; GFX1132DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_3) | instid1(SALU_CYCLE_1)
; GFX1132DAGISEL-NEXT: v_readlane_b32 s6, v0, s3
; GFX1132DAGISEL-NEXT: s_bitset0_b32 s2, s3
; GFX1132DAGISEL-NEXT: s_min_u32 s1, s1, s6
@@ -3282,9 +3301,10 @@ define amdgpu_kernel void @divergent_cfg(ptr addrspace(1) %out, i32 %in) #0 {
; GFX1132GISEL-NEXT: v_and_b32_e32 v0, 0x3ff, v0
; GFX1132GISEL-NEXT: s_mov_b32 s0, exec_lo
; GFX1132GISEL-NEXT: ; implicit-def: $sgpr1
-; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132GISEL-NEXT: v_cmpx_le_u32_e32 16, v0
; GFX1132GISEL-NEXT: s_xor_b32 s0, exec_lo, s0
+; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1132GISEL-NEXT: s_cbranch_execz .LBB8_2
; GFX1132GISEL-NEXT: ; %bb.1: ; %else
; GFX1132GISEL-NEXT: s_load_b32 s1, s[4:5], 0x2c
@@ -3295,13 +3315,14 @@ define amdgpu_kernel void @divergent_cfg(ptr addrspace(1) %out, i32 %in) #0 {
; GFX1132GISEL-NEXT: s_or_saveexec_b32 s0, s0
; GFX1132GISEL-NEXT: v_mov_b32_e32 v1, s1
; GFX1132GISEL-NEXT: s_xor_b32 exec_lo, exec_lo, s0
+; GFX1132GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1132GISEL-NEXT: s_cbranch_execz .LBB8_6
; GFX1132GISEL-NEXT: ; %bb.3: ; %if
; GFX1132GISEL-NEXT: s_mov_b32 s2, exec_lo
; GFX1132GISEL-NEXT: s_mov_b32 s1, -1
; GFX1132GISEL-NEXT: .LBB8_4: ; =>This Inner Loop Header: Depth=1
; GFX1132GISEL-NEXT: s_ctz_i32_b32 s3, s2
-; GFX1132GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1132GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_3) | instid1(SALU_CYCLE_1)
; GFX1132GISEL-NEXT: v_readlane_b32 s6, v0, s3
; GFX1132GISEL-NEXT: s_bitset0_b32 s2, s3
; GFX1132GISEL-NEXT: s_min_u32 s1, s1, s6
@@ -3725,6 +3746,7 @@ define void @divergent_value_i64(ptr addrspace(1) %out, i64 %id.x) #0 {
; GFX1164DAGISEL-NEXT: v_readlane_b32 s4, v2, s8
; GFX1164DAGISEL-NEXT: v_readlane_b32 s5, v3, s8
; GFX1164DAGISEL-NEXT: v_cmp_lt_u64_e32 vcc, s[4:5], v[4:5]
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_3) | instid1(SALU_CYCLE_1)
; GFX1164DAGISEL-NEXT: s_and_b64 s[6:7], vcc, s[2:3]
; GFX1164DAGISEL-NEXT: s_bitset0_b64 s[2:3], s8
; GFX1164DAGISEL-NEXT: s_cselect_b64 s[0:1], s[4:5], s[0:1]
@@ -3749,6 +3771,7 @@ define void @divergent_value_i64(ptr addrspace(1) %out, i64 %id.x) #0 {
; GFX1164GISEL-NEXT: v_readlane_b32 s4, v2, s8
; GFX1164GISEL-NEXT: v_readlane_b32 s5, v3, s8
; GFX1164GISEL-NEXT: v_cmp_lt_u64_e32 vcc, s[4:5], v[4:5]
+; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_3) | instid1(SALU_CYCLE_1)
; GFX1164GISEL-NEXT: s_and_b64 s[6:7], vcc, s[2:3]
; GFX1164GISEL-NEXT: s_bitset0_b64 s[2:3], s8
; GFX1164GISEL-NEXT: s_cselect_b64 s[0:1], s[4:5], s[0:1]
@@ -3776,6 +3799,7 @@ define void @divergent_value_i64(ptr addrspace(1) %out, i64 %id.x) #0 {
; GFX1132DAGISEL-NEXT: s_bitset0_b32 s2, s3
; GFX1132DAGISEL-NEXT: s_cselect_b64 s[0:1], s[4:5], s[0:1]
; GFX1132DAGISEL-NEXT: s_cmp_lg_u32 s2, 0
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1132DAGISEL-NEXT: s_cbranch_scc1 .LBB10_1
; GFX1132DAGISEL-NEXT: ; %bb.2:
; GFX1132DAGISEL-NEXT: v_dual_mov_b32 v3, s1 :: v_dual_mov_b32 v2, s0
@@ -3798,6 +3822,7 @@ define void @divergent_value_i64(ptr addrspace(1) %out, i64 %id.x) #0 {
; GFX1132GISEL-NEXT: s_bitset0_b32 s2, s3
; GFX1132GISEL-NEXT: s_cselect_b64 s[0:1], s[4:5], s[0:1]
; GFX1132GISEL-NEXT: s_cmp_lg_u32 s2, 0
+; GFX1132GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1132GISEL-NEXT: s_cbranch_scc1 .LBB10_1
; GFX1132GISEL-NEXT: ; %bb.2:
; GFX1132GISEL-NEXT: v_dual_mov_b32 v3, s1 :: v_dual_mov_b32 v2, s0
@@ -4069,19 +4094,22 @@ define amdgpu_kernel void @divergent_cfg_i64(ptr addrspace(1) %out, i64 %in, i64
; GFX1164DAGISEL-NEXT: s_load_b64 s[4:5], s[4:5], 0x34
; GFX1164DAGISEL-NEXT: v_and_b32_e32 v0, 0x3ff, v0
; GFX1164DAGISEL-NEXT: s_mov_b64 s[6:7], exec
-; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164DAGISEL-NEXT: v_cmpx_lt_u32_e32 15, v0
; GFX1164DAGISEL-NEXT: s_xor_b64 s[6:7], exec, s[6:7]
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164DAGISEL-NEXT: s_or_saveexec_b64 s[6:7], s[6:7]
; GFX1164DAGISEL-NEXT: s_waitcnt lgkmcnt(0)
; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v0, s2
; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v1, s3
; GFX1164DAGISEL-NEXT: s_xor_b64 exec, exec, s[6:7]
; GFX1164DAGISEL-NEXT: ; %bb.1: ; %if
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v0, s4
; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v1, s5
; GFX1164DAGISEL-NEXT: ; %bb.2: ; %endif
; GFX1164DAGISEL-NEXT: s_or_b64 exec, exec, s[6:7]
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v2, 0
; GFX1164DAGISEL-NEXT: global_store_b64 v2, v[0:1], s[0:1]
; GFX1164DAGISEL-NEXT: s_endpgm
@@ -4092,7 +4120,7 @@ define amdgpu_kernel void @divergent_cfg_i64(ptr addrspace(1) %out, i64 %in, i64
; GFX1164GISEL-NEXT: v_and_b32_e32 v0, 0x3ff, v0
; GFX1164GISEL-NEXT: s_mov_b64 s[8:9], exec
; GFX1164GISEL-NEXT: ; implicit-def: $sgpr6_sgpr7
-; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164GISEL-NEXT: v_cmpx_le_u32_e32 16, v0
; GFX1164GISEL-NEXT: s_xor_b64 s[8:9], exec, s[8:9]
; GFX1164GISEL-NEXT: s_cbranch_execz .LBB11_2
@@ -4105,6 +4133,7 @@ define amdgpu_kernel void @divergent_cfg_i64(ptr addrspace(1) %out, i64 %in, i64
; GFX1164GISEL-NEXT: v_mov_b32_e32 v0, s6
; GFX1164GISEL-NEXT: v_mov_b32_e32 v1, s7
; GFX1164GISEL-NEXT: s_xor_b64 exec, exec, s[2:3]
+; GFX1164GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164GISEL-NEXT: s_cbranch_execz .LBB11_4
; GFX1164GISEL-NEXT: ; %bb.3: ; %if
; GFX1164GISEL-NEXT: s_load_b64 s[4:5], s[4:5], 0x34
@@ -4115,6 +4144,7 @@ define amdgpu_kernel void @divergent_cfg_i64(ptr addrspace(1) %out, i64 %in, i64
; GFX1164GISEL-NEXT: v_mov_b32_e32 v1, s5
; GFX1164GISEL-NEXT: .LBB11_4: ; %endif
; GFX1164GISEL-NEXT: s_or_b64 exec, exec, s[2:3]
+; GFX1164GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164GISEL-NEXT: v_mov_b32_e32 v2, 0
; GFX1164GISEL-NEXT: global_store_b64 v2, v[0:1], s[0:1]
; GFX1164GISEL-NEXT: s_endpgm
@@ -4126,17 +4156,20 @@ define amdgpu_kernel void @divergent_cfg_i64(ptr addrspace(1) %out, i64 %in, i64
; GFX1132DAGISEL-NEXT: s_load_b64 s[4:5], s[4:5], 0x34
; GFX1132DAGISEL-NEXT: v_and_b32_e32 v0, 0x3ff, v0
; GFX1132DAGISEL-NEXT: s_mov_b32 s6, exec_lo
-; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132DAGISEL-NEXT: v_cmpx_lt_u32_e32 15, v0
; GFX1132DAGISEL-NEXT: s_xor_b32 s6, exec_lo, s6
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX1132DAGISEL-NEXT: s_or_saveexec_b32 s6, s6
; GFX1132DAGISEL-NEXT: s_waitcnt lgkmcnt(0)
; GFX1132DAGISEL-NEXT: v_dual_mov_b32 v0, s2 :: v_dual_mov_b32 v1, s3
; GFX1132DAGISEL-NEXT: s_xor_b32 exec_lo, exec_lo, s6
; GFX1132DAGISEL-NEXT: ; %bb.1: ; %if
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1132DAGISEL-NEXT: v_dual_mov_b32 v0, s4 :: v_dual_mov_b32 v1, s5
; GFX1132DAGISEL-NEXT: ; %bb.2: ; %endif
; GFX1132DAGISEL-NEXT: s_or_b32 exec_lo, exec_lo, s6
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1132DAGISEL-NEXT: v_mov_b32_e32 v2, 0
; GFX1132DAGISEL-NEXT: global_store_b64 v2, v[0:1], s[0:1]
; GFX1132DAGISEL-NEXT: s_endpgm
@@ -4147,9 +4180,10 @@ define amdgpu_kernel void @divergent_cfg_i64(ptr addrspace(1) %out, i64 %in, i64
; GFX1132GISEL-NEXT: v_and_b32_e32 v0, 0x3ff, v0
; GFX1132GISEL-NEXT: s_mov_b32 s8, exec_lo
; GFX1132GISEL-NEXT: ; implicit-def: $sgpr6_sgpr7
-; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132GISEL-NEXT: v_cmpx_le_u32_e32 16, v0
; GFX1132GISEL-NEXT: s_xor_b32 s8, exec_lo, s8
+; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1132GISEL-NEXT: s_cbranch_execz .LBB11_2
; GFX1132GISEL-NEXT: ; %bb.1: ; %else
; GFX1132GISEL-NEXT: s_waitcnt lgkmcnt(0)
@@ -4159,6 +4193,7 @@ define amdgpu_kernel void @divergent_cfg_i64(ptr addrspace(1) %out, i64 %in, i64
; GFX1132GISEL-NEXT: s_or_saveexec_b32 s2, s8
; GFX1132GISEL-NEXT: v_dual_mov_b32 v0, s6 :: v_dual_mov_b32 v1, s7
; GFX1132GISEL-NEXT: s_xor_b32 exec_lo, exec_lo, s2
+; GFX1132GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1132GISEL-NEXT: s_cbranch_execz .LBB11_4
; GFX1132GISEL-NEXT: ; %bb.3: ; %if
; GFX1132GISEL-NEXT: s_load_b64 s[4:5], s[4:5], 0x34
@@ -4168,6 +4203,7 @@ define amdgpu_kernel void @divergent_cfg_i64(ptr addrspace(1) %out, i64 %in, i64
; GFX1132GISEL-NEXT: v_dual_mov_b32 v0, s4 :: v_dual_mov_b32 v1, s5
; GFX1132GISEL-NEXT: .LBB11_4: ; %endif
; GFX1132GISEL-NEXT: s_or_b32 exec_lo, exec_lo, s2
+; GFX1132GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1132GISEL-NEXT: v_mov_b32_e32 v2, 0
; GFX1132GISEL-NEXT: global_store_b64 v2, v[0:1], s[0:1]
; GFX1132GISEL-NEXT: s_endpgm
diff --git a/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.reduce.xor.ll b/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.reduce.xor.ll
index 7d6c296515733c..76b917f9f09372 100644
--- a/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.reduce.xor.ll
+++ b/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.reduce.xor.ll
@@ -497,6 +497,7 @@ define void @divergent_value_i16(ptr addrspace(1) %out, i16 %in) {
; GFX1164DAGISEL-FAKE16-NEXT: s_bitset0_b64 s[0:1], s3
; GFX1164DAGISEL-FAKE16-NEXT: s_xor_b32 s2, s2, s4
; GFX1164DAGISEL-FAKE16-NEXT: s_cmp_lg_u64 s[0:1], 0
+; GFX1164DAGISEL-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164DAGISEL-FAKE16-NEXT: s_cbranch_scc1 .LBB1_1
; GFX1164DAGISEL-FAKE16-NEXT: ; %bb.2:
; GFX1164DAGISEL-FAKE16-NEXT: v_mov_b32_e32 v2, s2
@@ -516,6 +517,7 @@ define void @divergent_value_i16(ptr addrspace(1) %out, i16 %in) {
; GFX1164GISEL-NEXT: s_bitset0_b64 s[0:1], s3
; GFX1164GISEL-NEXT: s_xor_b32 s2, s2, s4
; GFX1164GISEL-NEXT: s_cmp_lg_u64 s[0:1], 0
+; GFX1164GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164GISEL-NEXT: s_cbranch_scc1 .LBB1_1
; GFX1164GISEL-NEXT: ; %bb.2:
; GFX1164GISEL-NEXT: v_mov_b32_e32 v2, s2
@@ -535,6 +537,7 @@ define void @divergent_value_i16(ptr addrspace(1) %out, i16 %in) {
; GFX1132DAGISEL-FAKE16-NEXT: s_bitset0_b32 s1, s2
; GFX1132DAGISEL-FAKE16-NEXT: s_xor_b32 s0, s0, s3
; GFX1132DAGISEL-FAKE16-NEXT: s_cmp_lg_u32 s1, 0
+; GFX1132DAGISEL-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1132DAGISEL-FAKE16-NEXT: s_cbranch_scc1 .LBB1_1
; GFX1132DAGISEL-FAKE16-NEXT: ; %bb.2:
; GFX1132DAGISEL-FAKE16-NEXT: v_mov_b32_e32 v2, s0
@@ -554,6 +557,7 @@ define void @divergent_value_i16(ptr addrspace(1) %out, i16 %in) {
; GFX1132GISEL-NEXT: s_bitset0_b32 s1, s2
; GFX1132GISEL-NEXT: s_xor_b32 s0, s0, s3
; GFX1132GISEL-NEXT: s_cmp_lg_u32 s1, 0
+; GFX1132GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1132GISEL-NEXT: s_cbranch_scc1 .LBB1_1
; GFX1132GISEL-NEXT: ; %bb.2:
; GFX1132GISEL-NEXT: v_mov_b32_e32 v2, s0
@@ -573,6 +577,7 @@ define void @divergent_value_i16(ptr addrspace(1) %out, i16 %in) {
; GFX1164DAGISEL-TRUE16-NEXT: s_bitset0_b64 s[0:1], s3
; GFX1164DAGISEL-TRUE16-NEXT: s_xor_b32 s2, s2, s4
; GFX1164DAGISEL-TRUE16-NEXT: s_cmp_lg_u64 s[0:1], 0
+; GFX1164DAGISEL-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164DAGISEL-TRUE16-NEXT: s_cbranch_scc1 .LBB1_1
; GFX1164DAGISEL-TRUE16-NEXT: ; %bb.2:
; GFX1164DAGISEL-TRUE16-NEXT: v_mov_b32_e32 v2, s2
@@ -592,6 +597,7 @@ define void @divergent_value_i16(ptr addrspace(1) %out, i16 %in) {
; GFX1132DAGISEL-TRUE16-NEXT: s_bitset0_b32 s1, s2
; GFX1132DAGISEL-TRUE16-NEXT: s_xor_b32 s0, s0, s3
; GFX1132DAGISEL-TRUE16-NEXT: s_cmp_lg_u32 s1, 0
+; GFX1132DAGISEL-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1132DAGISEL-TRUE16-NEXT: s_cbranch_scc1 .LBB1_1
; GFX1132DAGISEL-TRUE16-NEXT: ; %bb.2:
; GFX1132DAGISEL-TRUE16-NEXT: v_mov_b32_e32 v2, s0
@@ -1030,6 +1036,7 @@ define amdgpu_kernel void @divergent_value(ptr addrspace(1) %out) #0 {
; GFX1164DAGISEL-NEXT: s_bitset0_b64 s[2:3], s5
; GFX1164DAGISEL-NEXT: s_xor_b32 s4, s4, s6
; GFX1164DAGISEL-NEXT: s_cmp_lg_u64 s[2:3], 0
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164DAGISEL-NEXT: s_cbranch_scc1 .LBB3_1
; GFX1164DAGISEL-NEXT: ; %bb.2:
; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v0, s4
@@ -1050,6 +1057,7 @@ define amdgpu_kernel void @divergent_value(ptr addrspace(1) %out) #0 {
; GFX1164GISEL-NEXT: s_bitset0_b64 s[2:3], s5
; GFX1164GISEL-NEXT: s_xor_b32 s4, s4, s6
; GFX1164GISEL-NEXT: s_cmp_lg_u64 s[2:3], 0
+; GFX1164GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164GISEL-NEXT: s_cbranch_scc1 .LBB3_1
; GFX1164GISEL-NEXT: ; %bb.2:
; GFX1164GISEL-NEXT: v_mov_b32_e32 v0, s4
@@ -1071,6 +1079,7 @@ define amdgpu_kernel void @divergent_value(ptr addrspace(1) %out) #0 {
; GFX1132DAGISEL-NEXT: s_bitset0_b32 s3, s4
; GFX1132DAGISEL-NEXT: s_xor_b32 s2, s2, s5
; GFX1132DAGISEL-NEXT: s_cmp_lg_u32 s3, 0
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1132DAGISEL-NEXT: s_cbranch_scc1 .LBB3_1
; GFX1132DAGISEL-NEXT: ; %bb.2:
; GFX1132DAGISEL-NEXT: v_mov_b32_e32 v0, s2
@@ -1091,6 +1100,7 @@ define amdgpu_kernel void @divergent_value(ptr addrspace(1) %out) #0 {
; GFX1132GISEL-NEXT: s_bitset0_b32 s3, s4
; GFX1132GISEL-NEXT: s_xor_b32 s2, s2, s5
; GFX1132GISEL-NEXT: s_cmp_lg_u32 s3, 0
+; GFX1132GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1132GISEL-NEXT: s_cbranch_scc1 .LBB3_1
; GFX1132GISEL-NEXT: ; %bb.2:
; GFX1132GISEL-NEXT: v_dual_mov_b32 v0, s2 :: v_dual_mov_b32 v1, 0
@@ -1594,7 +1604,7 @@ define amdgpu_kernel void @divergent_value_dpp(ptr addrspace(1) %out) #0 {
; GFX1164DAGISEL-NEXT: s_or_saveexec_b64 s[2:3], -1
; GFX1164DAGISEL-NEXT: s_waitcnt lgkmcnt(0)
; GFX1164DAGISEL-NEXT: v_xor_b32_e32 v1, v1, v2
-; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX1164DAGISEL-NEXT: v_readlane_b32 s4, v1, 63
; GFX1164DAGISEL-NEXT: s_mov_b64 exec, s[2:3]
; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v0, 0
@@ -1629,7 +1639,7 @@ define amdgpu_kernel void @divergent_value_dpp(ptr addrspace(1) %out) #0 {
; GFX1164GISEL-NEXT: s_mov_b64 exec, s[0:1]
; GFX1164GISEL-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
; GFX1164GISEL-NEXT: s_or_saveexec_b64 s[2:3], -1
-; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX1164GISEL-NEXT: v_readlane_b32 s4, v1, 63
; GFX1164GISEL-NEXT: s_mov_b64 exec, s[2:3]
; GFX1164GISEL-NEXT: v_mov_b32_e32 v0, s4
@@ -1656,7 +1666,7 @@ define amdgpu_kernel void @divergent_value_dpp(ptr addrspace(1) %out) #0 {
; GFX1132DAGISEL-NEXT: s_or_saveexec_b32 s2, -1
; GFX1132DAGISEL-NEXT: s_waitcnt lgkmcnt(0)
; GFX1132DAGISEL-NEXT: v_xor_b32_e32 v1, v1, v2
-; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX1132DAGISEL-NEXT: v_readlane_b32 s3, v1, 31
; GFX1132DAGISEL-NEXT: s_mov_b32 exec_lo, s2
; GFX1132DAGISEL-NEXT: v_dual_mov_b32 v0, 0 :: v_dual_mov_b32 v3, s3
@@ -1681,7 +1691,7 @@ define amdgpu_kernel void @divergent_value_dpp(ptr addrspace(1) %out) #0 {
; GFX1132GISEL-NEXT: s_mov_b32 exec_lo, s0
; GFX1132GISEL-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
; GFX1132GISEL-NEXT: s_or_saveexec_b32 s2, -1
-; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX1132GISEL-NEXT: v_readlane_b32 s3, v1, 31
; GFX1132GISEL-NEXT: s_mov_b32 exec_lo, s2
; GFX1132GISEL-NEXT: v_dual_mov_b32 v0, s3 :: v_dual_mov_b32 v3, 0
@@ -2295,50 +2305,51 @@ define void @divergent_value_dpp_i64(ptr addrspace(1) %out, i64 %in) #0 {
; GFX1164DAGISEL-NEXT: scratch_store_b32 off, v7, s32 offset:12
; GFX1164DAGISEL-NEXT: scratch_store_b32 off, v8, s32 offset:16
; GFX1164DAGISEL-NEXT: s_mov_b64 exec, s[0:1]
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1164DAGISEL-NEXT: s_or_saveexec_b64 s[0:1], -1
-; GFX1164DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(VALU_DEP_3)
; GFX1164DAGISEL-NEXT: v_cndmask_b32_e64 v4, 0, v2, s[0:1]
; GFX1164DAGISEL-NEXT: v_cndmask_b32_e64 v5, 0, v3, s[0:1]
; GFX1164DAGISEL-NEXT: v_mbcnt_lo_u32_b32 v8, -1, 0
-; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v6, v4
; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v6, v4
; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v7, v5
-; GFX1164DAGISEL-NEXT: v_mbcnt_hi_u32_b32 v8, -1, v8
; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX1164DAGISEL-NEXT: v_mbcnt_hi_u32_b32 v8, -1, v8
; GFX1164DAGISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:1 row_mask:0xf bank_mask:0xf
-; GFX1164DAGISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:1 row_mask:0xf bank_mask:0xf
; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX1164DAGISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:1 row_mask:0xf bank_mask:0xf
; GFX1164DAGISEL-NEXT: v_add_nc_u32_e32 v8, 32, v8
-; GFX1164DAGISEL-NEXT: v_xor_b32_e32 v4, v4, v6
; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX1164DAGISEL-NEXT: v_xor_b32_e32 v4, v4, v6
; GFX1164DAGISEL-NEXT: v_xor_b32_e32 v5, v5, v7
-; GFX1164DAGISEL-NEXT: v_mul_lo_u32 v8, 4, v8
; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX1164DAGISEL-NEXT: v_mul_lo_u32 v8, 4, v8
; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v6, v4
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v7, v5
-; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1164DAGISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:2 row_mask:0xf bank_mask:0xf
-; GFX1164DAGISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:2 row_mask:0xf bank_mask:0xf
; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX1164DAGISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:2 row_mask:0xf bank_mask:0xf
; GFX1164DAGISEL-NEXT: v_xor_b32_e32 v4, v4, v6
-; GFX1164DAGISEL-NEXT: v_xor_b32_e32 v5, v5, v7
; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX1164DAGISEL-NEXT: v_xor_b32_e32 v5, v5, v7
; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v6, v4
-; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v7, v5
; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v7, v5
; GFX1164DAGISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:4 row_mask:0xf bank_mask:0xf
-; GFX1164DAGISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:4 row_mask:0xf bank_mask:0xf
; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX1164DAGISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:4 row_mask:0xf bank_mask:0xf
; GFX1164DAGISEL-NEXT: v_xor_b32_e32 v4, v4, v6
-; GFX1164DAGISEL-NEXT: v_xor_b32_e32 v5, v5, v7
; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX1164DAGISEL-NEXT: v_xor_b32_e32 v5, v5, v7
; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v6, v4
-; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v7, v5
; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v7, v5
; GFX1164DAGISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:8 row_mask:0xf bank_mask:0xf
-; GFX1164DAGISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:8 row_mask:0xf bank_mask:0xf
; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX1164DAGISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:8 row_mask:0xf bank_mask:0xf
; GFX1164DAGISEL-NEXT: v_xor_b32_e32 v4, v4, v6
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX1164DAGISEL-NEXT: v_xor_b32_e32 v5, v5, v7
; GFX1164DAGISEL-NEXT: ds_swizzle_b32 v6, v4 offset:swizzle(BROADCAST,32,15)
; GFX1164DAGISEL-NEXT: ds_swizzle_b32 v7, v5 offset:swizzle(BROADCAST,32,15)
@@ -2356,6 +2367,7 @@ define void @divergent_value_dpp_i64(ptr addrspace(1) %out, i64 %in) #0 {
; GFX1164DAGISEL-NEXT: v_readlane_b32 s2, v4, 63
; GFX1164DAGISEL-NEXT: v_readlane_b32 s3, v5, 63
; GFX1164DAGISEL-NEXT: s_mov_b64 exec, s[0:1]
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v2, s2
; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v3, s3
; GFX1164DAGISEL-NEXT: global_store_b64 v[0:1], v[2:3], off
@@ -2382,50 +2394,51 @@ define void @divergent_value_dpp_i64(ptr addrspace(1) %out, i64 %in) #0 {
; GFX1164GISEL-NEXT: scratch_store_b32 off, v7, s32 offset:12
; GFX1164GISEL-NEXT: scratch_store_b32 off, v8, s32 offset:16
; GFX1164GISEL-NEXT: s_mov_b64 exec, s[0:1]
+; GFX1164GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1164GISEL-NEXT: s_or_saveexec_b64 s[0:1], -1
-; GFX1164GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(VALU_DEP_3)
; GFX1164GISEL-NEXT: v_cndmask_b32_e64 v4, 0, v2, s[0:1]
; GFX1164GISEL-NEXT: v_cndmask_b32_e64 v5, 0, v3, s[0:1]
; GFX1164GISEL-NEXT: v_mbcnt_lo_u32_b32 v8, -1, 0
-; GFX1164GISEL-NEXT: v_mov_b32_e32 v6, v4
; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX1164GISEL-NEXT: v_mov_b32_e32 v6, v4
; GFX1164GISEL-NEXT: v_mov_b32_e32 v7, v5
-; GFX1164GISEL-NEXT: v_mbcnt_hi_u32_b32 v8, -1, v8
; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX1164GISEL-NEXT: v_mbcnt_hi_u32_b32 v8, -1, v8
; GFX1164GISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:1 row_mask:0xf bank_mask:0xf
-; GFX1164GISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:1 row_mask:0xf bank_mask:0xf
; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX1164GISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:1 row_mask:0xf bank_mask:0xf
; GFX1164GISEL-NEXT: v_add_nc_u32_e32 v8, 32, v8
-; GFX1164GISEL-NEXT: v_xor_b32_e32 v4, v4, v6
; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX1164GISEL-NEXT: v_xor_b32_e32 v4, v4, v6
; GFX1164GISEL-NEXT: v_xor_b32_e32 v5, v5, v7
-; GFX1164GISEL-NEXT: v_mul_lo_u32 v8, 4, v8
; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX1164GISEL-NEXT: v_mul_lo_u32 v8, 4, v8
; GFX1164GISEL-NEXT: v_mov_b32_e32 v6, v4
+; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1164GISEL-NEXT: v_mov_b32_e32 v7, v5
-; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1164GISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:2 row_mask:0xf bank_mask:0xf
-; GFX1164GISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:2 row_mask:0xf bank_mask:0xf
; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX1164GISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:2 row_mask:0xf bank_mask:0xf
; GFX1164GISEL-NEXT: v_xor_b32_e32 v4, v4, v6
-; GFX1164GISEL-NEXT: v_xor_b32_e32 v5, v5, v7
; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX1164GISEL-NEXT: v_xor_b32_e32 v5, v5, v7
; GFX1164GISEL-NEXT: v_mov_b32_e32 v6, v4
-; GFX1164GISEL-NEXT: v_mov_b32_e32 v7, v5
; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX1164GISEL-NEXT: v_mov_b32_e32 v7, v5
; GFX1164GISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:4 row_mask:0xf bank_mask:0xf
-; GFX1164GISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:4 row_mask:0xf bank_mask:0xf
; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX1164GISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:4 row_mask:0xf bank_mask:0xf
; GFX1164GISEL-NEXT: v_xor_b32_e32 v4, v4, v6
-; GFX1164GISEL-NEXT: v_xor_b32_e32 v5, v5, v7
; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX1164GISEL-NEXT: v_xor_b32_e32 v5, v5, v7
; GFX1164GISEL-NEXT: v_mov_b32_e32 v6, v4
-; GFX1164GISEL-NEXT: v_mov_b32_e32 v7, v5
; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX1164GISEL-NEXT: v_mov_b32_e32 v7, v5
; GFX1164GISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:8 row_mask:0xf bank_mask:0xf
-; GFX1164GISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:8 row_mask:0xf bank_mask:0xf
; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX1164GISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:8 row_mask:0xf bank_mask:0xf
; GFX1164GISEL-NEXT: v_xor_b32_e32 v4, v4, v6
+; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX1164GISEL-NEXT: v_xor_b32_e32 v5, v5, v7
; GFX1164GISEL-NEXT: ds_swizzle_b32 v6, v4 offset:swizzle(BROADCAST,32,15)
; GFX1164GISEL-NEXT: ds_swizzle_b32 v7, v5 offset:swizzle(BROADCAST,32,15)
@@ -2443,6 +2456,7 @@ define void @divergent_value_dpp_i64(ptr addrspace(1) %out, i64 %in) #0 {
; GFX1164GISEL-NEXT: v_readlane_b32 s2, v4, 63
; GFX1164GISEL-NEXT: v_readlane_b32 s3, v5, 63
; GFX1164GISEL-NEXT: s_mov_b64 exec, s[0:1]
+; GFX1164GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164GISEL-NEXT: v_mov_b32_e32 v2, s2
; GFX1164GISEL-NEXT: v_mov_b32_e32 v3, s3
; GFX1164GISEL-NEXT: global_store_b64 v[0:1], v[2:3], off
@@ -2468,39 +2482,39 @@ define void @divergent_value_dpp_i64(ptr addrspace(1) %out, i64 %in) #0 {
; GFX1132DAGISEL-NEXT: scratch_store_b32 off, v6, s32 offset:8
; GFX1132DAGISEL-NEXT: scratch_store_b32 off, v7, s32 offset:12
; GFX1132DAGISEL-NEXT: s_mov_b32 exec_lo, s0
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1132DAGISEL-NEXT: s_or_saveexec_b32 s2, -1
-; GFX1132DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX1132DAGISEL-NEXT: v_cndmask_b32_e64 v4, 0, v2, s2
; GFX1132DAGISEL-NEXT: v_cndmask_b32_e64 v5, 0, v3, s2
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132DAGISEL-NEXT: v_dual_mov_b32 v6, v4 :: v_dual_mov_b32 v7, v5
-; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1132DAGISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:1 row_mask:0xf bank_mask:0xf
-; GFX1132DAGISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:1 row_mask:0xf bank_mask:0xf
; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX1132DAGISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:1 row_mask:0xf bank_mask:0xf
; GFX1132DAGISEL-NEXT: v_xor_b32_e32 v4, v4, v6
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132DAGISEL-NEXT: v_xor_b32_e32 v5, v5, v7
-; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132DAGISEL-NEXT: v_dual_mov_b32 v6, v4 :: v_dual_mov_b32 v7, v5
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1132DAGISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:2 row_mask:0xf bank_mask:0xf
-; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1132DAGISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:2 row_mask:0xf bank_mask:0xf
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1132DAGISEL-NEXT: v_xor_b32_e32 v4, v4, v6
-; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132DAGISEL-NEXT: v_xor_b32_e32 v5, v5, v7
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132DAGISEL-NEXT: v_dual_mov_b32 v6, v4 :: v_dual_mov_b32 v7, v5
-; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1132DAGISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:4 row_mask:0xf bank_mask:0xf
-; GFX1132DAGISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:4 row_mask:0xf bank_mask:0xf
; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX1132DAGISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:4 row_mask:0xf bank_mask:0xf
; GFX1132DAGISEL-NEXT: v_xor_b32_e32 v4, v4, v6
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132DAGISEL-NEXT: v_xor_b32_e32 v5, v5, v7
-; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132DAGISEL-NEXT: v_dual_mov_b32 v6, v4 :: v_dual_mov_b32 v7, v5
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1132DAGISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:8 row_mask:0xf bank_mask:0xf
-; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1132DAGISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:8 row_mask:0xf bank_mask:0xf
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1132DAGISEL-NEXT: v_xor_b32_e32 v4, v4, v6
-; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX1132DAGISEL-NEXT: v_xor_b32_e32 v5, v5, v7
; GFX1132DAGISEL-NEXT: ds_swizzle_b32 v6, v4 offset:swizzle(BROADCAST,32,15)
; GFX1132DAGISEL-NEXT: ds_swizzle_b32 v7, v5 offset:swizzle(BROADCAST,32,15)
@@ -2512,6 +2526,7 @@ define void @divergent_value_dpp_i64(ptr addrspace(1) %out, i64 %in) #0 {
; GFX1132DAGISEL-NEXT: v_readlane_b32 s0, v4, 31
; GFX1132DAGISEL-NEXT: v_readlane_b32 s1, v5, 31
; GFX1132DAGISEL-NEXT: s_mov_b32 exec_lo, s2
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1132DAGISEL-NEXT: v_dual_mov_b32 v3, s1 :: v_dual_mov_b32 v2, s0
; GFX1132DAGISEL-NEXT: global_store_b64 v[0:1], v[2:3], off
; GFX1132DAGISEL-NEXT: s_xor_saveexec_b32 s0, -1
@@ -2534,39 +2549,39 @@ define void @divergent_value_dpp_i64(ptr addrspace(1) %out, i64 %in) #0 {
; GFX1132GISEL-NEXT: scratch_store_b32 off, v6, s32 offset:8
; GFX1132GISEL-NEXT: scratch_store_b32 off, v7, s32 offset:12
; GFX1132GISEL-NEXT: s_mov_b32 exec_lo, s0
+; GFX1132GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1132GISEL-NEXT: s_or_saveexec_b32 s2, -1
-; GFX1132GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX1132GISEL-NEXT: v_cndmask_b32_e64 v4, 0, v2, s2
; GFX1132GISEL-NEXT: v_cndmask_b32_e64 v5, 0, v3, s2
+; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132GISEL-NEXT: v_dual_mov_b32 v6, v4 :: v_dual_mov_b32 v7, v5
-; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1132GISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:1 row_mask:0xf bank_mask:0xf
-; GFX1132GISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:1 row_mask:0xf bank_mask:0xf
; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX1132GISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:1 row_mask:0xf bank_mask:0xf
; GFX1132GISEL-NEXT: v_xor_b32_e32 v4, v4, v6
+; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132GISEL-NEXT: v_xor_b32_e32 v5, v5, v7
-; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132GISEL-NEXT: v_dual_mov_b32 v6, v4 :: v_dual_mov_b32 v7, v5
+; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1132GISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:2 row_mask:0xf bank_mask:0xf
-; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1132GISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:2 row_mask:0xf bank_mask:0xf
+; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1132GISEL-NEXT: v_xor_b32_e32 v4, v4, v6
-; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132GISEL-NEXT: v_xor_b32_e32 v5, v5, v7
+; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132GISEL-NEXT: v_dual_mov_b32 v6, v4 :: v_dual_mov_b32 v7, v5
-; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1132GISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:4 row_mask:0xf bank_mask:0xf
-; GFX1132GISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:4 row_mask:0xf bank_mask:0xf
; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX1132GISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:4 row_mask:0xf bank_mask:0xf
; GFX1132GISEL-NEXT: v_xor_b32_e32 v4, v4, v6
+; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132GISEL-NEXT: v_xor_b32_e32 v5, v5, v7
-; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132GISEL-NEXT: v_dual_mov_b32 v6, v4 :: v_dual_mov_b32 v7, v5
+; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1132GISEL-NEXT: v_mov_b32_dpp v6, v6 row_shr:8 row_mask:0xf bank_mask:0xf
-; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1132GISEL-NEXT: v_mov_b32_dpp v7, v7 row_shr:8 row_mask:0xf bank_mask:0xf
+; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1132GISEL-NEXT: v_xor_b32_e32 v4, v4, v6
-; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX1132GISEL-NEXT: v_xor_b32_e32 v5, v5, v7
; GFX1132GISEL-NEXT: ds_swizzle_b32 v6, v4 offset:swizzle(BROADCAST,32,15)
; GFX1132GISEL-NEXT: ds_swizzle_b32 v7, v5 offset:swizzle(BROADCAST,32,15)
@@ -2578,6 +2593,7 @@ define void @divergent_value_dpp_i64(ptr addrspace(1) %out, i64 %in) #0 {
; GFX1132GISEL-NEXT: v_readlane_b32 s0, v4, 31
; GFX1132GISEL-NEXT: v_readlane_b32 s1, v5, 31
; GFX1132GISEL-NEXT: s_mov_b32 exec_lo, s2
+; GFX1132GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1132GISEL-NEXT: v_dual_mov_b32 v3, s1 :: v_dual_mov_b32 v2, s0
; GFX1132GISEL-NEXT: global_store_b64 v[0:1], v[2:3], off
; GFX1132GISEL-NEXT: s_xor_saveexec_b32 s0, -1
@@ -2867,7 +2883,7 @@ define amdgpu_kernel void @default_stratergy(ptr addrspace(1) %out) #0 {
; GFX1164DAGISEL-NEXT: s_or_saveexec_b64 s[2:3], -1
; GFX1164DAGISEL-NEXT: s_waitcnt lgkmcnt(0)
; GFX1164DAGISEL-NEXT: v_xor_b32_e32 v1, v1, v2
-; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX1164DAGISEL-NEXT: v_readlane_b32 s4, v1, 63
; GFX1164DAGISEL-NEXT: s_mov_b64 exec, s[2:3]
; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v0, 0
@@ -2902,7 +2918,7 @@ define amdgpu_kernel void @default_stratergy(ptr addrspace(1) %out) #0 {
; GFX1164GISEL-NEXT: s_mov_b64 exec, s[0:1]
; GFX1164GISEL-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
; GFX1164GISEL-NEXT: s_or_saveexec_b64 s[2:3], -1
-; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX1164GISEL-NEXT: v_readlane_b32 s4, v1, 63
; GFX1164GISEL-NEXT: s_mov_b64 exec, s[2:3]
; GFX1164GISEL-NEXT: v_mov_b32_e32 v0, s4
@@ -2929,7 +2945,7 @@ define amdgpu_kernel void @default_stratergy(ptr addrspace(1) %out) #0 {
; GFX1132DAGISEL-NEXT: s_or_saveexec_b32 s2, -1
; GFX1132DAGISEL-NEXT: s_waitcnt lgkmcnt(0)
; GFX1132DAGISEL-NEXT: v_xor_b32_e32 v1, v1, v2
-; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX1132DAGISEL-NEXT: v_readlane_b32 s3, v1, 31
; GFX1132DAGISEL-NEXT: s_mov_b32 exec_lo, s2
; GFX1132DAGISEL-NEXT: v_dual_mov_b32 v0, 0 :: v_dual_mov_b32 v3, s3
@@ -2954,7 +2970,7 @@ define amdgpu_kernel void @default_stratergy(ptr addrspace(1) %out) #0 {
; GFX1132GISEL-NEXT: s_mov_b32 exec_lo, s0
; GFX1132GISEL-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
; GFX1132GISEL-NEXT: s_or_saveexec_b32 s2, -1
-; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX1132GISEL-NEXT: v_readlane_b32 s3, v1, 31
; GFX1132GISEL-NEXT: s_mov_b32 exec_lo, s2
; GFX1132GISEL-NEXT: v_dual_mov_b32 v0, s3 :: v_dual_mov_b32 v3, 0
@@ -3378,7 +3394,7 @@ define amdgpu_kernel void @divergent_cfg(ptr addrspace(1) %out, i32 %in) #0 {
; GFX1164DAGISEL-NEXT: v_and_b32_e32 v0, 0x3ff, v0
; GFX1164DAGISEL-NEXT: s_mov_b64 s[0:1], exec
; GFX1164DAGISEL-NEXT: ; implicit-def: $sgpr2
-; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164DAGISEL-NEXT: v_cmpx_lt_u32_e32 15, v0
; GFX1164DAGISEL-NEXT: s_xor_b64 s[0:1], exec, s[0:1]
; GFX1164DAGISEL-NEXT: s_cbranch_execz .LBB8_2
@@ -3395,13 +3411,14 @@ define amdgpu_kernel void @divergent_cfg(ptr addrspace(1) %out, i32 %in) #0 {
; GFX1164DAGISEL-NEXT: s_or_saveexec_b64 s[0:1], s[0:1]
; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v1, s2
; GFX1164DAGISEL-NEXT: s_xor_b64 exec, exec, s[0:1]
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164DAGISEL-NEXT: s_cbranch_execz .LBB8_6
; GFX1164DAGISEL-NEXT: ; %bb.3: ; %if
; GFX1164DAGISEL-NEXT: s_mov_b64 s[2:3], exec
; GFX1164DAGISEL-NEXT: s_mov_b32 s6, 0
; GFX1164DAGISEL-NEXT: .LBB8_4: ; =>This Inner Loop Header: Depth=1
; GFX1164DAGISEL-NEXT: s_ctz_i32_b64 s7, s[2:3]
-; GFX1164DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_3) | instid1(SALU_CYCLE_1)
; GFX1164DAGISEL-NEXT: v_readlane_b32 s8, v0, s7
; GFX1164DAGISEL-NEXT: s_bitset0_b64 s[2:3], s7
; GFX1164DAGISEL-NEXT: s_xor_b32 s6, s6, s8
@@ -3422,7 +3439,7 @@ define amdgpu_kernel void @divergent_cfg(ptr addrspace(1) %out, i32 %in) #0 {
; GFX1164GISEL-NEXT: v_and_b32_e32 v0, 0x3ff, v0
; GFX1164GISEL-NEXT: s_mov_b64 s[0:1], exec
; GFX1164GISEL-NEXT: ; implicit-def: $sgpr2
-; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164GISEL-NEXT: v_cmpx_le_u32_e32 16, v0
; GFX1164GISEL-NEXT: s_xor_b64 s[0:1], exec, s[0:1]
; GFX1164GISEL-NEXT: s_cbranch_execz .LBB8_2
@@ -3439,13 +3456,14 @@ define amdgpu_kernel void @divergent_cfg(ptr addrspace(1) %out, i32 %in) #0 {
; GFX1164GISEL-NEXT: s_or_saveexec_b64 s[0:1], s[0:1]
; GFX1164GISEL-NEXT: v_mov_b32_e32 v1, s2
; GFX1164GISEL-NEXT: s_xor_b64 exec, exec, s[0:1]
+; GFX1164GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164GISEL-NEXT: s_cbranch_execz .LBB8_6
; GFX1164GISEL-NEXT: ; %bb.3: ; %if
; GFX1164GISEL-NEXT: s_mov_b64 s[2:3], exec
; GFX1164GISEL-NEXT: s_mov_b32 s6, 0
; GFX1164GISEL-NEXT: .LBB8_4: ; =>This Inner Loop Header: Depth=1
; GFX1164GISEL-NEXT: s_ctz_i32_b64 s7, s[2:3]
-; GFX1164GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1164GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_3) | instid1(SALU_CYCLE_1)
; GFX1164GISEL-NEXT: v_readlane_b32 s8, v0, s7
; GFX1164GISEL-NEXT: s_bitset0_b64 s[2:3], s7
; GFX1164GISEL-NEXT: s_xor_b32 s6, s6, s8
@@ -3466,9 +3484,10 @@ define amdgpu_kernel void @divergent_cfg(ptr addrspace(1) %out, i32 %in) #0 {
; GFX1132DAGISEL-NEXT: v_and_b32_e32 v0, 0x3ff, v0
; GFX1132DAGISEL-NEXT: s_mov_b32 s0, exec_lo
; GFX1132DAGISEL-NEXT: ; implicit-def: $sgpr1
-; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132DAGISEL-NEXT: v_cmpx_lt_u32_e32 15, v0
; GFX1132DAGISEL-NEXT: s_xor_b32 s0, exec_lo, s0
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1132DAGISEL-NEXT: s_cbranch_execz .LBB8_2
; GFX1132DAGISEL-NEXT: ; %bb.1: ; %else
; GFX1132DAGISEL-NEXT: s_load_b32 s1, s[4:5], 0x2c
@@ -3483,13 +3502,14 @@ define amdgpu_kernel void @divergent_cfg(ptr addrspace(1) %out, i32 %in) #0 {
; GFX1132DAGISEL-NEXT: s_or_saveexec_b32 s0, s0
; GFX1132DAGISEL-NEXT: v_mov_b32_e32 v1, s1
; GFX1132DAGISEL-NEXT: s_xor_b32 exec_lo, exec_lo, s0
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1132DAGISEL-NEXT: s_cbranch_execz .LBB8_6
; GFX1132DAGISEL-NEXT: ; %bb.3: ; %if
; GFX1132DAGISEL-NEXT: s_mov_b32 s2, exec_lo
; GFX1132DAGISEL-NEXT: s_mov_b32 s1, 0
; GFX1132DAGISEL-NEXT: .LBB8_4: ; =>This Inner Loop Header: Depth=1
; GFX1132DAGISEL-NEXT: s_ctz_i32_b32 s3, s2
-; GFX1132DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_3) | instid1(SALU_CYCLE_1)
; GFX1132DAGISEL-NEXT: v_readlane_b32 s6, v0, s3
; GFX1132DAGISEL-NEXT: s_bitset0_b32 s2, s3
; GFX1132DAGISEL-NEXT: s_xor_b32 s1, s1, s6
@@ -3510,9 +3530,10 @@ define amdgpu_kernel void @divergent_cfg(ptr addrspace(1) %out, i32 %in) #0 {
; GFX1132GISEL-NEXT: v_and_b32_e32 v0, 0x3ff, v0
; GFX1132GISEL-NEXT: s_mov_b32 s0, exec_lo
; GFX1132GISEL-NEXT: ; implicit-def: $sgpr1
-; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132GISEL-NEXT: v_cmpx_le_u32_e32 16, v0
; GFX1132GISEL-NEXT: s_xor_b32 s0, exec_lo, s0
+; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1132GISEL-NEXT: s_cbranch_execz .LBB8_2
; GFX1132GISEL-NEXT: ; %bb.1: ; %else
; GFX1132GISEL-NEXT: s_load_b32 s1, s[4:5], 0x2c
@@ -3527,13 +3548,14 @@ define amdgpu_kernel void @divergent_cfg(ptr addrspace(1) %out, i32 %in) #0 {
; GFX1132GISEL-NEXT: s_or_saveexec_b32 s0, s0
; GFX1132GISEL-NEXT: v_mov_b32_e32 v1, s1
; GFX1132GISEL-NEXT: s_xor_b32 exec_lo, exec_lo, s0
+; GFX1132GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1132GISEL-NEXT: s_cbranch_execz .LBB8_6
; GFX1132GISEL-NEXT: ; %bb.3: ; %if
; GFX1132GISEL-NEXT: s_mov_b32 s2, exec_lo
; GFX1132GISEL-NEXT: s_mov_b32 s1, 0
; GFX1132GISEL-NEXT: .LBB8_4: ; =>This Inner Loop Header: Depth=1
; GFX1132GISEL-NEXT: s_ctz_i32_b32 s3, s2
-; GFX1132GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1132GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_3) | instid1(SALU_CYCLE_1)
; GFX1132GISEL-NEXT: v_readlane_b32 s6, v0, s3
; GFX1132GISEL-NEXT: s_bitset0_b32 s2, s3
; GFX1132GISEL-NEXT: s_xor_b32 s1, s1, s6
@@ -4012,6 +4034,7 @@ define void @divergent_value_i64(ptr addrspace(1) %out, i64 %id.x) #0 {
; GFX1164DAGISEL-NEXT: s_bitset0_b64 s[2:3], s6
; GFX1164DAGISEL-NEXT: s_xor_b64 s[0:1], s[0:1], s[4:5]
; GFX1164DAGISEL-NEXT: s_cmp_lg_u64 s[2:3], 0
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164DAGISEL-NEXT: s_cbranch_scc1 .LBB10_1
; GFX1164DAGISEL-NEXT: ; %bb.2:
; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v3, s1
@@ -4032,6 +4055,7 @@ define void @divergent_value_i64(ptr addrspace(1) %out, i64 %id.x) #0 {
; GFX1164GISEL-NEXT: s_bitset0_b64 s[2:3], s6
; GFX1164GISEL-NEXT: s_xor_b64 s[0:1], s[0:1], s[4:5]
; GFX1164GISEL-NEXT: s_cmp_lg_u64 s[2:3], 0
+; GFX1164GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164GISEL-NEXT: s_cbranch_scc1 .LBB10_1
; GFX1164GISEL-NEXT: ; %bb.2:
; GFX1164GISEL-NEXT: v_mov_b32_e32 v3, s1
@@ -4052,6 +4076,7 @@ define void @divergent_value_i64(ptr addrspace(1) %out, i64 %id.x) #0 {
; GFX1132DAGISEL-NEXT: s_bitset0_b32 s2, s3
; GFX1132DAGISEL-NEXT: s_xor_b64 s[0:1], s[0:1], s[4:5]
; GFX1132DAGISEL-NEXT: s_cmp_lg_u32 s2, 0
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1132DAGISEL-NEXT: s_cbranch_scc1 .LBB10_1
; GFX1132DAGISEL-NEXT: ; %bb.2:
; GFX1132DAGISEL-NEXT: v_dual_mov_b32 v3, s1 :: v_dual_mov_b32 v2, s0
@@ -4071,6 +4096,7 @@ define void @divergent_value_i64(ptr addrspace(1) %out, i64 %id.x) #0 {
; GFX1132GISEL-NEXT: s_bitset0_b32 s2, s3
; GFX1132GISEL-NEXT: s_xor_b64 s[0:1], s[0:1], s[4:5]
; GFX1132GISEL-NEXT: s_cmp_lg_u32 s2, 0
+; GFX1132GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1132GISEL-NEXT: s_cbranch_scc1 .LBB10_1
; GFX1132GISEL-NEXT: ; %bb.2:
; GFX1132GISEL-NEXT: v_dual_mov_b32 v3, s1 :: v_dual_mov_b32 v2, s0
@@ -4465,7 +4491,7 @@ define amdgpu_kernel void @divergent_cfg_i64(ptr addrspace(1) %out, i64 %in, i64
; GFX1164DAGISEL-NEXT: v_and_b32_e32 v0, 0x3ff, v0
; GFX1164DAGISEL-NEXT: s_mov_b64 s[8:9], exec
; GFX1164DAGISEL-NEXT: ; implicit-def: $sgpr6_sgpr7
-; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164DAGISEL-NEXT: v_cmpx_lt_u32_e32 15, v0
; GFX1164DAGISEL-NEXT: s_xor_b64 s[8:9], exec, s[8:9]
; GFX1164DAGISEL-NEXT: s_cbranch_execz .LBB11_2
@@ -4495,6 +4521,7 @@ define amdgpu_kernel void @divergent_cfg_i64(ptr addrspace(1) %out, i64 %in, i64
; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v1, s5
; GFX1164DAGISEL-NEXT: ; %bb.4: ; %endif
; GFX1164DAGISEL-NEXT: s_or_b64 exec, exec, s[2:3]
+; GFX1164DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164DAGISEL-NEXT: v_mov_b32_e32 v2, 0
; GFX1164DAGISEL-NEXT: global_store_b64 v2, v[0:1], s[0:1]
; GFX1164DAGISEL-NEXT: s_endpgm
@@ -4505,7 +4532,7 @@ define amdgpu_kernel void @divergent_cfg_i64(ptr addrspace(1) %out, i64 %in, i64
; GFX1164GISEL-NEXT: v_and_b32_e32 v0, 0x3ff, v0
; GFX1164GISEL-NEXT: s_mov_b64 s[8:9], exec
; GFX1164GISEL-NEXT: ; implicit-def: $sgpr6_sgpr7
-; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1164GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164GISEL-NEXT: v_cmpx_le_u32_e32 16, v0
; GFX1164GISEL-NEXT: s_xor_b64 s[8:9], exec, s[8:9]
; GFX1164GISEL-NEXT: s_cbranch_execz .LBB11_2
@@ -4523,6 +4550,7 @@ define amdgpu_kernel void @divergent_cfg_i64(ptr addrspace(1) %out, i64 %in, i64
; GFX1164GISEL-NEXT: v_mov_b32_e32 v0, s6
; GFX1164GISEL-NEXT: v_mov_b32_e32 v1, s7
; GFX1164GISEL-NEXT: s_xor_b64 exec, exec, s[2:3]
+; GFX1164GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164GISEL-NEXT: s_cbranch_execz .LBB11_4
; GFX1164GISEL-NEXT: ; %bb.3: ; %if
; GFX1164GISEL-NEXT: s_load_b64 s[4:5], s[4:5], 0x34
@@ -4537,6 +4565,7 @@ define amdgpu_kernel void @divergent_cfg_i64(ptr addrspace(1) %out, i64 %in, i64
; GFX1164GISEL-NEXT: v_mov_b32_e32 v1, s5
; GFX1164GISEL-NEXT: .LBB11_4: ; %endif
; GFX1164GISEL-NEXT: s_or_b64 exec, exec, s[2:3]
+; GFX1164GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164GISEL-NEXT: v_mov_b32_e32 v2, 0
; GFX1164GISEL-NEXT: global_store_b64 v2, v[0:1], s[0:1]
; GFX1164GISEL-NEXT: s_endpgm
@@ -4549,9 +4578,10 @@ define amdgpu_kernel void @divergent_cfg_i64(ptr addrspace(1) %out, i64 %in, i64
; GFX1132DAGISEL-NEXT: v_and_b32_e32 v0, 0x3ff, v0
; GFX1132DAGISEL-NEXT: s_mov_b32 s8, exec_lo
; GFX1132DAGISEL-NEXT: ; implicit-def: $sgpr6_sgpr7
-; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132DAGISEL-NEXT: v_cmpx_lt_u32_e32 15, v0
; GFX1132DAGISEL-NEXT: s_xor_b32 s8, exec_lo, s8
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1132DAGISEL-NEXT: s_cbranch_execz .LBB11_2
; GFX1132DAGISEL-NEXT: ; %bb.1: ; %else
; GFX1132DAGISEL-NEXT: s_mov_b32 s6, exec_lo
@@ -4578,6 +4608,7 @@ define amdgpu_kernel void @divergent_cfg_i64(ptr addrspace(1) %out, i64 %in, i64
; GFX1132DAGISEL-NEXT: v_dual_mov_b32 v0, s4 :: v_dual_mov_b32 v1, s5
; GFX1132DAGISEL-NEXT: ; %bb.4: ; %endif
; GFX1132DAGISEL-NEXT: s_or_b32 exec_lo, exec_lo, s2
+; GFX1132DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1132DAGISEL-NEXT: v_mov_b32_e32 v2, 0
; GFX1132DAGISEL-NEXT: global_store_b64 v2, v[0:1], s[0:1]
; GFX1132DAGISEL-NEXT: s_endpgm
@@ -4588,9 +4619,10 @@ define amdgpu_kernel void @divergent_cfg_i64(ptr addrspace(1) %out, i64 %in, i64
; GFX1132GISEL-NEXT: v_and_b32_e32 v0, 0x3ff, v0
; GFX1132GISEL-NEXT: s_mov_b32 s8, exec_lo
; GFX1132GISEL-NEXT: ; implicit-def: $sgpr6_sgpr7
-; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1132GISEL-NEXT: v_cmpx_le_u32_e32 16, v0
; GFX1132GISEL-NEXT: s_xor_b32 s8, exec_lo, s8
+; GFX1132GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1132GISEL-NEXT: s_cbranch_execz .LBB11_2
; GFX1132GISEL-NEXT: ; %bb.1: ; %else
; GFX1132GISEL-NEXT: s_mov_b32 s6, exec_lo
@@ -4605,6 +4637,7 @@ define amdgpu_kernel void @divergent_cfg_i64(ptr addrspace(1) %out, i64 %in, i64
; GFX1132GISEL-NEXT: s_or_saveexec_b32 s2, s8
; GFX1132GISEL-NEXT: v_dual_mov_b32 v0, s6 :: v_dual_mov_b32 v1, s7
; GFX1132GISEL-NEXT: s_xor_b32 exec_lo, exec_lo, s2
+; GFX1132GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1132GISEL-NEXT: s_cbranch_execz .LBB11_4
; GFX1132GISEL-NEXT: ; %bb.3: ; %if
; GFX1132GISEL-NEXT: s_load_b64 s[4:5], s[4:5], 0x34
@@ -4619,6 +4652,7 @@ define amdgpu_kernel void @divergent_cfg_i64(ptr addrspace(1) %out, i64 %in, i64
; GFX1132GISEL-NEXT: v_dual_mov_b32 v0, s4 :: v_dual_mov_b32 v1, s5
; GFX1132GISEL-NEXT: .LBB11_4: ; %endif
; GFX1132GISEL-NEXT: s_or_b32 exec_lo, exec_lo, s2
+; GFX1132GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1132GISEL-NEXT: v_mov_b32_e32 v2, 0
; GFX1132GISEL-NEXT: global_store_b64 v2, v[0:1], s[0:1]
; GFX1132GISEL-NEXT: s_endpgm
diff --git a/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.s.barrier.ll b/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.s.barrier.ll
index 916c0a2e87e9f7..a635f283093f81 100644
--- a/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.s.barrier.ll
+++ b/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.s.barrier.ll
@@ -127,7 +127,7 @@ define amdgpu_kernel void @test_barrier(ptr addrspace(1) %out, i32 %size) #0 {
; VARIANT4-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; VARIANT4-NEXT: v_lshlrev_b64_e32 v[0:1], 2, v[0:1]
; VARIANT4-NEXT: v_add_co_u32 v0, vcc_lo, s0, v0
-; VARIANT4-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; VARIANT4-NEXT: s_delay_alu instid0(VALU_DEP_2)
; VARIANT4-NEXT: v_add_co_ci_u32_e64 v1, null, s1, v1, vcc_lo
; VARIANT4-NEXT: s_barrier_wait -1
; VARIANT4-NEXT: global_load_b32 v0, v[0:1], off
@@ -149,7 +149,7 @@ define amdgpu_kernel void @test_barrier(ptr addrspace(1) %out, i32 %size) #0 {
; VARIANT5-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; VARIANT5-NEXT: v_lshlrev_b64_e32 v[0:1], 2, v[0:1]
; VARIANT5-NEXT: v_add_co_u32 v0, vcc_lo, s0, v0
-; VARIANT5-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; VARIANT5-NEXT: s_delay_alu instid0(VALU_DEP_2)
; VARIANT5-NEXT: v_add_co_ci_u32_e64 v1, null, s1, v1, vcc_lo
; VARIANT5-NEXT: s_barrier_wait -1
; VARIANT5-NEXT: global_load_b32 v0, v[0:1], off
@@ -170,7 +170,7 @@ define amdgpu_kernel void @test_barrier(ptr addrspace(1) %out, i32 %size) #0 {
; VARIANT6-NEXT: s_barrier_signal -1
; VARIANT6-NEXT: v_ashrrev_i32_e32 v1, 31, v0
; VARIANT6-NEXT: v_lshlrev_b64_e32 v[0:1], 2, v[0:1]
-; VARIANT6-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; VARIANT6-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; VARIANT6-NEXT: v_add_co_u32 v0, vcc_lo, v2, v0
; VARIANT6-NEXT: v_add_co_ci_u32_e64 v1, null, v3, v1, vcc_lo
; VARIANT6-NEXT: s_barrier_wait -1
diff --git a/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.s.barrier.signal.isfirst.ll b/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.s.barrier.signal.isfirst.ll
index 0347d1035800a9..94a8f94c80ecf2 100644
--- a/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.s.barrier.signal.isfirst.ll
+++ b/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.s.barrier.signal.isfirst.ll
@@ -11,6 +11,7 @@ define i1 @func1() {
; GFX12-SDAG-NEXT: s_wait_bvhcnt 0x0
; GFX12-SDAG-NEXT: s_wait_kmcnt 0x0
; GFX12-SDAG-NEXT: s_cmp_eq_u32 0, 0
+; GFX12-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_2)
; GFX12-SDAG-NEXT: s_barrier_signal_isfirst -1
; GFX12-SDAG-NEXT: s_wait_kmcnt 0x0
; GFX12-SDAG-NEXT: s_cselect_b32 s0, 1, 0
@@ -26,6 +27,7 @@ define i1 @func1() {
; GFX12-GISEL-NEXT: s_wait_bvhcnt 0x0
; GFX12-GISEL-NEXT: s_wait_kmcnt 0x0
; GFX12-GISEL-NEXT: s_cmp_eq_u32 0, 0
+; GFX12-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_2)
; GFX12-GISEL-NEXT: s_barrier_signal_isfirst -1
; GFX12-GISEL-NEXT: s_wait_kmcnt 0x0
; GFX12-GISEL-NEXT: s_cselect_b32 s0, 1, 0
@@ -45,6 +47,7 @@ define i1 @signal_isfirst_same_barrier_wait() {
; GFX12-SDAG-NEXT: s_wait_bvhcnt 0x0
; GFX12-SDAG-NEXT: s_wait_kmcnt 0x0
; GFX12-SDAG-NEXT: s_cmp_eq_u32 0, 0
+; GFX12-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_2)
; GFX12-SDAG-NEXT: s_barrier_signal_isfirst -1
; GFX12-SDAG-NEXT: s_barrier_wait -1
; GFX12-SDAG-NEXT: s_cselect_b32 s0, 1, 0
@@ -60,6 +63,7 @@ define i1 @signal_isfirst_same_barrier_wait() {
; GFX12-GISEL-NEXT: s_wait_bvhcnt 0x0
; GFX12-GISEL-NEXT: s_wait_kmcnt 0x0
; GFX12-GISEL-NEXT: s_cmp_eq_u32 0, 0
+; GFX12-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_2)
; GFX12-GISEL-NEXT: s_barrier_signal_isfirst -1
; GFX12-GISEL-NEXT: s_barrier_wait -1
; GFX12-GISEL-NEXT: s_cselect_b32 s0, 1, 0
@@ -80,6 +84,7 @@ define i1 @signal_isfirst_different_barrier_wait() {
; GFX12-SDAG-NEXT: s_wait_bvhcnt 0x0
; GFX12-SDAG-NEXT: s_wait_kmcnt 0x0
; GFX12-SDAG-NEXT: s_cmp_eq_u32 0, 0
+; GFX12-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
; GFX12-SDAG-NEXT: s_barrier_signal_isfirst -1
; GFX12-SDAG-NEXT: s_barrier_wait 0
; GFX12-SDAG-NEXT: s_wait_kmcnt 0x0
@@ -96,6 +101,7 @@ define i1 @signal_isfirst_different_barrier_wait() {
; GFX12-GISEL-NEXT: s_wait_bvhcnt 0x0
; GFX12-GISEL-NEXT: s_wait_kmcnt 0x0
; GFX12-GISEL-NEXT: s_cmp_eq_u32 0, 0
+; GFX12-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
; GFX12-GISEL-NEXT: s_barrier_signal_isfirst -1
; GFX12-GISEL-NEXT: s_barrier_wait 0
; GFX12-GISEL-NEXT: s_wait_kmcnt 0x0
diff --git a/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.s.prefetch.data.ll b/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.s.prefetch.data.ll
index 7eca56c22aa5d3..ab1f20c4718192 100644
--- a/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.s.prefetch.data.ll
+++ b/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.s.prefetch.data.ll
@@ -219,10 +219,9 @@ define amdgpu_ps void @prefetch_data_vgpr_imm_base_sgpr_len(ptr addrspace(4) %pt
; GISEL-LABEL: prefetch_data_vgpr_imm_base_sgpr_len:
; GISEL: ; %bb.0: ; %entry
; GISEL-NEXT: v_add_co_u32 v0, vcc_lo, 0x200, v0
-; GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
+; GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GISEL-NEXT: v_readfirstlane_b32 s2, v0
-; GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GISEL-NEXT: v_readfirstlane_b32 s3, v1
; GISEL-NEXT: s_prefetch_data s[2:3], 0x0, s0, 0
; GISEL-NEXT: s_endpgm
@@ -245,10 +244,9 @@ define amdgpu_ps void @prefetch_data_vgpr_imm_base_sgpr_len(ptr addrspace(4) %pt
; GFX1250-GISEL-NEXT: v_nop
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, 0x200, v0
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_readfirstlane_b32 s2, v0
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_readfirstlane_b32 s3, v1
; GFX1250-GISEL-NEXT: s_prefetch_data s[2:3], 0x0, s0, 0
; GFX1250-GISEL-NEXT: s_endpgm
@@ -263,10 +261,9 @@ define amdgpu_ps void @prefetch_data_vgpr_imm_base_sgpr_len(ptr addrspace(4) %pt
; GFX1310-GISEL-LABEL: prefetch_data_vgpr_imm_base_sgpr_len:
; GFX1310-GISEL: ; %bb.0: ; %entry
; GFX1310-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, 0x200, v0
-; GFX1310-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1310-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
+; GFX1310-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1310-GISEL-NEXT: v_readfirstlane_b32 s2, v0
-; GFX1310-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX1310-GISEL-NEXT: v_readfirstlane_b32 s3, v1
; GFX1310-GISEL-NEXT: s_prefetch_data s[2:3], 0x0, s0, 0
; GFX1310-GISEL-NEXT: s_endpgm
diff --git a/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.s.prefetch.inst.ll b/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.s.prefetch.inst.ll
index ffc21667cae075..28b30cb4c1ef7e 100644
--- a/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.s.prefetch.inst.ll
+++ b/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.s.prefetch.inst.ll
@@ -111,10 +111,9 @@ define amdgpu_ps void @prefetch_inst_vgpr_imm_base_sgpr_len(ptr addrspace(4) %pt
; GFX12-GISEL-LABEL: prefetch_inst_vgpr_imm_base_sgpr_len:
; GFX12-GISEL: ; %bb.0:
; GFX12-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, 0x200, v0
-; GFX12-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX12-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
+; GFX12-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX12-GISEL-NEXT: v_readfirstlane_b32 s2, v0
-; GFX12-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX12-GISEL-NEXT: v_readfirstlane_b32 s3, v1
; GFX12-GISEL-NEXT: s_prefetch_inst s[2:3], 0x0, s0, 0
; GFX12-GISEL-NEXT: s_endpgm
@@ -137,10 +136,9 @@ define amdgpu_ps void @prefetch_inst_vgpr_imm_base_sgpr_len(ptr addrspace(4) %pt
; GFX1250-GISEL-NEXT: v_nop
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, 0x200, v0
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_readfirstlane_b32 s2, v0
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX1250-GISEL-NEXT: v_readfirstlane_b32 s3, v1
; GFX1250-GISEL-NEXT: s_prefetch_inst s[2:3], 0x0, s0, 0
; GFX1250-GISEL-NEXT: s_endpgm
diff --git a/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.s.ttracedata.ll b/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.s.ttracedata.ll
index 0529cf20f80b5a..45ed691a00c15a 100644
--- a/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.s.ttracedata.ll
+++ b/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.s.ttracedata.ll
@@ -9,6 +9,7 @@ define amdgpu_cs void @ttracedata_c() {
; GFX11-LABEL: ttracedata_c:
; GFX11: ; %bb.0:
; GFX11-NEXT: s_mov_b32 m0, 0xf4240
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_ttracedata
; GFX11-NEXT: s_endpgm
call void @llvm.amdgcn.s.ttracedata(i32 1000000)
@@ -19,6 +20,7 @@ define amdgpu_cs void @ttracedata_s(i32 inreg %val) {
; GFX11-LABEL: ttracedata_s:
; GFX11: ; %bb.0:
; GFX11-NEXT: s_mov_b32 m0, s0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_ttracedata
; GFX11-NEXT: s_endpgm
call void @llvm.amdgcn.s.ttracedata(i32 %val)
@@ -30,6 +32,7 @@ define amdgpu_cs void @ttracedata_v(i32 %val) {
; GFX11: ; %bb.0:
; GFX11-NEXT: v_readfirstlane_b32 s0, v0
; GFX11-NEXT: s_mov_b32 m0, s0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_ttracedata
; GFX11-NEXT: s_endpgm
call void @llvm.amdgcn.s.ttracedata(i32 %val)
diff --git a/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.set.inactive.chain.arg.ll b/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.set.inactive.chain.arg.ll
index 6743fd14747789..b4079fe00bd9ae 100644
--- a/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.set.inactive.chain.arg.ll
+++ b/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.set.inactive.chain.arg.ll
@@ -15,11 +15,12 @@ define amdgpu_cs_chain void @set_inactive_chain_arg(ptr addrspace(1) %out, i32 %
; GFX11-NEXT: s_or_saveexec_b32 s0, -1
; GFX11-NEXT: v_mov_b32_e32 v0, v10
; GFX11-NEXT: s_mov_b32 exec_lo, s0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_or_saveexec_b32 s0, -1
; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: v_cndmask_b32_e64 v0, v0, v11, s0
; GFX11-NEXT: s_mov_b32 exec_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: v_mov_b32_e32 v1, v0
; GFX11-NEXT: global_store_b32 v[8:9], v1, off
; GFX11-NEXT: s_endpgm
@@ -43,11 +44,12 @@ define amdgpu_cs_chain void @set_inactive_chain_arg(ptr addrspace(1) %out, i32 %
; GFX11_W64-NEXT: s_or_saveexec_b64 s[0:1], -1
; GFX11_W64-NEXT: v_mov_b32_e32 v0, v10
; GFX11_W64-NEXT: s_mov_b64 exec, s[0:1]
+; GFX11_W64-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11_W64-NEXT: s_or_saveexec_b64 s[0:1], -1
; GFX11_W64-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11_W64-NEXT: v_cndmask_b32_e64 v0, v0, v11, s[0:1]
; GFX11_W64-NEXT: s_mov_b64 exec, s[0:1]
-; GFX11_W64-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX11_W64-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11_W64-NEXT: v_mov_b32_e32 v1, v0
; GFX11_W64-NEXT: global_store_b32 v[8:9], v1, off
; GFX11_W64-NEXT: s_endpgm
@@ -77,12 +79,14 @@ define amdgpu_cs_chain void @set_inactive_chain_arg_64(ptr addrspace(1) %out, i6
; GISEL11-NEXT: s_or_saveexec_b32 s0, -1
; GISEL11-NEXT: v_dual_mov_b32 v0, v10 :: v_dual_mov_b32 v1, v11
; GISEL11-NEXT: s_mov_b32 exec_lo, s0
+; GISEL11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GISEL11-NEXT: s_or_saveexec_b32 s0, -1
; GISEL11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GISEL11-NEXT: v_cndmask_b32_e64 v0, v0, v12, s0
-; GISEL11-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_1)
+; GISEL11-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GISEL11-NEXT: v_cndmask_b32_e64 v1, v1, v13, s0
; GISEL11-NEXT: s_mov_b32 exec_lo, s0
+; GISEL11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GISEL11-NEXT: v_dual_mov_b32 v2, v0 :: v_dual_mov_b32 v3, v1
; GISEL11-NEXT: global_store_b64 v[8:9], v[2:3], off
; GISEL11-NEXT: s_endpgm
@@ -93,12 +97,14 @@ define amdgpu_cs_chain void @set_inactive_chain_arg_64(ptr addrspace(1) %out, i6
; DAGISEL11-NEXT: s_or_saveexec_b32 s0, -1
; DAGISEL11-NEXT: v_dual_mov_b32 v0, v11 :: v_dual_mov_b32 v1, v10
; DAGISEL11-NEXT: s_mov_b32 exec_lo, s0
+; DAGISEL11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; DAGISEL11-NEXT: s_or_saveexec_b32 s0, -1
; DAGISEL11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; DAGISEL11-NEXT: v_cndmask_b32_e64 v2, v0, v13, s0
-; DAGISEL11-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_1)
+; DAGISEL11-NEXT: s_delay_alu instid0(VALU_DEP_2)
; DAGISEL11-NEXT: v_cndmask_b32_e64 v1, v1, v12, s0
; DAGISEL11-NEXT: s_mov_b32 exec_lo, s0
+; DAGISEL11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; DAGISEL11-NEXT: v_dual_mov_b32 v3, v1 :: v_dual_mov_b32 v4, v2
; DAGISEL11-NEXT: global_store_b64 v[8:9], v[3:4], off
; DAGISEL11-NEXT: s_endpgm
@@ -142,12 +148,14 @@ define amdgpu_cs_chain void @set_inactive_chain_arg_64(ptr addrspace(1) %out, i6
; GISEL11_W64-NEXT: v_mov_b32_e32 v0, v10
; GISEL11_W64-NEXT: v_mov_b32_e32 v1, v11
; GISEL11_W64-NEXT: s_mov_b64 exec, s[0:1]
+; GISEL11_W64-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GISEL11_W64-NEXT: s_or_saveexec_b64 s[0:1], -1
; GISEL11_W64-NEXT: s_delay_alu instid0(VALU_DEP_2) | instid1(SALU_CYCLE_1)
; GISEL11_W64-NEXT: v_cndmask_b32_e64 v0, v0, v12, s[0:1]
-; GISEL11_W64-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_2)
+; GISEL11_W64-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GISEL11_W64-NEXT: v_cndmask_b32_e64 v1, v1, v13, s[0:1]
; GISEL11_W64-NEXT: s_mov_b64 exec, s[0:1]
+; GISEL11_W64-NEXT: s_delay_alu instid0(VALU_DEP_2) | instid1(SALU_CYCLE_1)
; GISEL11_W64-NEXT: v_mov_b32_e32 v2, v0
; GISEL11_W64-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GISEL11_W64-NEXT: v_mov_b32_e32 v3, v1
@@ -161,12 +169,14 @@ define amdgpu_cs_chain void @set_inactive_chain_arg_64(ptr addrspace(1) %out, i6
; DAGISEL11_W64-NEXT: v_mov_b32_e32 v0, v11
; DAGISEL11_W64-NEXT: v_mov_b32_e32 v1, v10
; DAGISEL11_W64-NEXT: s_mov_b64 exec, s[0:1]
+; DAGISEL11_W64-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; DAGISEL11_W64-NEXT: s_or_saveexec_b64 s[0:1], -1
; DAGISEL11_W64-NEXT: s_delay_alu instid0(VALU_DEP_2) | instid1(SALU_CYCLE_1)
; DAGISEL11_W64-NEXT: v_cndmask_b32_e64 v2, v0, v13, s[0:1]
-; DAGISEL11_W64-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_1)
+; DAGISEL11_W64-NEXT: s_delay_alu instid0(VALU_DEP_2)
; DAGISEL11_W64-NEXT: v_cndmask_b32_e64 v1, v1, v12, s[0:1]
; DAGISEL11_W64-NEXT: s_mov_b64 exec, s[0:1]
+; DAGISEL11_W64-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; DAGISEL11_W64-NEXT: v_mov_b32_e32 v3, v1
; DAGISEL11_W64-NEXT: s_delay_alu instid0(VALU_DEP_3)
; DAGISEL11_W64-NEXT: v_mov_b32_e32 v4, v2
@@ -217,13 +227,15 @@ define amdgpu_cs_chain void @set_inactive_chain_arg_dpp(ptr addrspace(1) %out, i
; GFX11-NEXT: s_or_saveexec_b32 s0, -1
; GFX11-NEXT: v_mov_b32_e32 v0, v10
; GFX11-NEXT: s_mov_b32 exec_lo, s0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_or_saveexec_b32 s0, -1
; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: v_cndmask_b32_e64 v0, v0, v11, s0
; GFX11-NEXT: v_mov_b32_e32 v1, 0
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_mov_b32_dpp v1, v0 row_xmask:1 row_mask:0xf bank_mask:0xf
; GFX11-NEXT: s_mov_b32 exec_lo, s0
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: v_mov_b32_e32 v2, v1
; GFX11-NEXT: global_store_b32 v[8:9], v2, off
; GFX11-NEXT: s_endpgm
@@ -249,13 +261,15 @@ define amdgpu_cs_chain void @set_inactive_chain_arg_dpp(ptr addrspace(1) %out, i
; GFX11_W64-NEXT: s_or_saveexec_b64 s[0:1], -1
; GFX11_W64-NEXT: v_mov_b32_e32 v0, v10
; GFX11_W64-NEXT: s_mov_b64 exec, s[0:1]
+; GFX11_W64-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11_W64-NEXT: s_or_saveexec_b64 s[0:1], -1
; GFX11_W64-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11_W64-NEXT: v_cndmask_b32_e64 v0, v0, v11, s[0:1]
; GFX11_W64-NEXT: v_mov_b32_e32 v1, 0
-; GFX11_W64-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
+; GFX11_W64-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11_W64-NEXT: v_mov_b32_dpp v1, v0 row_xmask:1 row_mask:0xf bank_mask:0xf
; GFX11_W64-NEXT: s_mov_b64 exec, s[0:1]
+; GFX11_W64-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11_W64-NEXT: v_mov_b32_e32 v2, v1
; GFX11_W64-NEXT: global_store_b32 v[8:9], v2, off
; GFX11_W64-NEXT: s_endpgm
@@ -305,9 +319,10 @@ define amdgpu_cs_chain void @set_inactive_chain_arg_call(ptr addrspace(1) %out,
; GISEL11-NEXT: s_waitcnt lgkmcnt(0)
; GISEL11-NEXT: s_swappc_b64 s[30:31], s[0:1]
; GISEL11-NEXT: s_or_saveexec_b32 s0, -1
-; GISEL11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
+; GISEL11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GISEL11-NEXT: v_cndmask_b32_e64 v12, v40, v43, s0
; GISEL11-NEXT: s_mov_b32 exec_lo, s0
+; GISEL11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GISEL11-NEXT: v_mov_b32_e32 v0, v12
; GISEL11-NEXT: global_store_b32 v[41:42], v0, off
; GISEL11-NEXT: s_endpgm
@@ -333,9 +348,10 @@ define amdgpu_cs_chain void @set_inactive_chain_arg_call(ptr addrspace(1) %out,
; DAGISEL11-NEXT: s_waitcnt lgkmcnt(0)
; DAGISEL11-NEXT: s_swappc_b64 s[30:31], s[0:1]
; DAGISEL11-NEXT: s_or_saveexec_b32 s0, -1
-; DAGISEL11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
+; DAGISEL11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; DAGISEL11-NEXT: v_cndmask_b32_e64 v12, v40, v43, s0
; DAGISEL11-NEXT: s_mov_b32 exec_lo, s0
+; DAGISEL11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; DAGISEL11-NEXT: v_mov_b32_e32 v0, v12
; DAGISEL11-NEXT: global_store_b32 v[41:42], v0, off
; DAGISEL11-NEXT: s_endpgm
@@ -440,9 +456,10 @@ define amdgpu_cs_chain void @set_inactive_chain_arg_call(ptr addrspace(1) %out,
; GISEL11_W64-NEXT: s_waitcnt lgkmcnt(0)
; GISEL11_W64-NEXT: s_swappc_b64 s[30:31], s[0:1]
; GISEL11_W64-NEXT: s_or_saveexec_b64 s[0:1], -1
-; GISEL11_W64-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
+; GISEL11_W64-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GISEL11_W64-NEXT: v_cndmask_b32_e64 v12, v40, v43, s[0:1]
; GISEL11_W64-NEXT: s_mov_b64 exec, s[0:1]
+; GISEL11_W64-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GISEL11_W64-NEXT: v_mov_b32_e32 v0, v12
; GISEL11_W64-NEXT: global_store_b32 v[41:42], v0, off
; GISEL11_W64-NEXT: s_endpgm
@@ -475,9 +492,10 @@ define amdgpu_cs_chain void @set_inactive_chain_arg_call(ptr addrspace(1) %out,
; DAGISEL11_W64-NEXT: s_waitcnt lgkmcnt(0)
; DAGISEL11_W64-NEXT: s_swappc_b64 s[30:31], s[0:1]
; DAGISEL11_W64-NEXT: s_or_saveexec_b64 s[0:1], -1
-; DAGISEL11_W64-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
+; DAGISEL11_W64-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; DAGISEL11_W64-NEXT: v_cndmask_b32_e64 v12, v40, v43, s[0:1]
; DAGISEL11_W64-NEXT: s_mov_b64 exec, s[0:1]
+; DAGISEL11_W64-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; DAGISEL11_W64-NEXT: v_mov_b32_e32 v0, v12
; DAGISEL11_W64-NEXT: global_store_b32 v[41:42], v0, off
; DAGISEL11_W64-NEXT: s_endpgm
@@ -586,9 +604,10 @@ define amdgpu_cs_chain void @set_inactive_chain_arg_last_vgpr(ptr addrspace(1) %
; GISEL11-NEXT: s_waitcnt lgkmcnt(0)
; GISEL11-NEXT: s_swappc_b64 s[30:31], s[0:1]
; GISEL11-NEXT: s_or_saveexec_b32 s0, -1
-; GISEL11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
+; GISEL11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GISEL11-NEXT: v_cndmask_b32_e64 v12, v40, v43, s0
; GISEL11-NEXT: s_mov_b32 exec_lo, s0
+; GISEL11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GISEL11-NEXT: v_mov_b32_e32 v0, v12
; GISEL11-NEXT: global_store_b32 v[41:42], v0, off
; GISEL11-NEXT: s_endpgm
@@ -614,9 +633,10 @@ define amdgpu_cs_chain void @set_inactive_chain_arg_last_vgpr(ptr addrspace(1) %
; DAGISEL11-NEXT: s_waitcnt lgkmcnt(0)
; DAGISEL11-NEXT: s_swappc_b64 s[30:31], s[0:1]
; DAGISEL11-NEXT: s_or_saveexec_b32 s0, -1
-; DAGISEL11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
+; DAGISEL11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; DAGISEL11-NEXT: v_cndmask_b32_e64 v12, v40, v43, s0
; DAGISEL11-NEXT: s_mov_b32 exec_lo, s0
+; DAGISEL11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; DAGISEL11-NEXT: v_mov_b32_e32 v0, v12
; DAGISEL11-NEXT: global_store_b32 v[41:42], v0, off
; DAGISEL11-NEXT: s_endpgm
@@ -721,9 +741,10 @@ define amdgpu_cs_chain void @set_inactive_chain_arg_last_vgpr(ptr addrspace(1) %
; GISEL11_W64-NEXT: s_waitcnt lgkmcnt(0)
; GISEL11_W64-NEXT: s_swappc_b64 s[30:31], s[0:1]
; GISEL11_W64-NEXT: s_or_saveexec_b64 s[0:1], -1
-; GISEL11_W64-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
+; GISEL11_W64-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GISEL11_W64-NEXT: v_cndmask_b32_e64 v12, v40, v43, s[0:1]
; GISEL11_W64-NEXT: s_mov_b64 exec, s[0:1]
+; GISEL11_W64-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GISEL11_W64-NEXT: v_mov_b32_e32 v0, v12
; GISEL11_W64-NEXT: global_store_b32 v[41:42], v0, off
; GISEL11_W64-NEXT: s_endpgm
@@ -756,9 +777,10 @@ define amdgpu_cs_chain void @set_inactive_chain_arg_last_vgpr(ptr addrspace(1) %
; DAGISEL11_W64-NEXT: s_waitcnt lgkmcnt(0)
; DAGISEL11_W64-NEXT: s_swappc_b64 s[30:31], s[0:1]
; DAGISEL11_W64-NEXT: s_or_saveexec_b64 s[0:1], -1
-; DAGISEL11_W64-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
+; DAGISEL11_W64-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; DAGISEL11_W64-NEXT: v_cndmask_b32_e64 v12, v40, v43, s[0:1]
; DAGISEL11_W64-NEXT: s_mov_b64 exec, s[0:1]
+; DAGISEL11_W64-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; DAGISEL11_W64-NEXT: v_mov_b32_e32 v0, v12
; DAGISEL11_W64-NEXT: global_store_b32 v[41:42], v0, off
; DAGISEL11_W64-NEXT: s_endpgm
@@ -851,13 +873,15 @@ define amdgpu_cs_chain void @set_inactive_chain_arg_active_demanded(ptr addrspac
; GISEL11-NEXT: s_or_saveexec_b32 s0, -1
; GISEL11-NEXT: v_mov_b32_e32 v0, v10
; GISEL11-NEXT: s_mov_b32 exec_lo, s0
+; GISEL11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GISEL11-NEXT: v_or_b32_e32 v1, 0xffff0000, v11
; GISEL11-NEXT: s_or_saveexec_b32 s0, -1
; GISEL11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GISEL11-NEXT: v_cndmask_b32_e64 v0, v0, v1, s0
-; GISEL11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
+; GISEL11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GISEL11-NEXT: v_and_b32_e32 v0, 0xffff, v0
; GISEL11-NEXT: s_mov_b32 exec_lo, s0
+; GISEL11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GISEL11-NEXT: v_mov_b32_e32 v1, v0
; GISEL11-NEXT: global_store_b32 v[8:9], v1, off
; GISEL11-NEXT: s_endpgm
@@ -868,12 +892,14 @@ define amdgpu_cs_chain void @set_inactive_chain_arg_active_demanded(ptr addrspac
; DAGISEL11-NEXT: s_or_saveexec_b32 s0, -1
; DAGISEL11-NEXT: v_mov_b32_e32 v0, v10
; DAGISEL11-NEXT: s_mov_b32 exec_lo, s0
+; DAGISEL11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; DAGISEL11-NEXT: s_or_saveexec_b32 s0, -1
; DAGISEL11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; DAGISEL11-NEXT: v_cndmask_b32_e64 v0, v0, v11, s0
-; DAGISEL11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
+; DAGISEL11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; DAGISEL11-NEXT: v_and_b32_e32 v0, 0xffff, v0
; DAGISEL11-NEXT: s_mov_b32 exec_lo, s0
+; DAGISEL11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; DAGISEL11-NEXT: v_mov_b32_e32 v1, v0
; DAGISEL11-NEXT: global_store_b32 v[8:9], v1, off
; DAGISEL11-NEXT: s_endpgm
@@ -913,13 +939,14 @@ define amdgpu_cs_chain void @set_inactive_chain_arg_active_demanded(ptr addrspac
; GISEL11_W64-NEXT: s_or_saveexec_b64 s[0:1], -1
; GISEL11_W64-NEXT: v_mov_b32_e32 v0, v10
; GISEL11_W64-NEXT: s_mov_b64 exec, s[0:1]
+; GISEL11_W64-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_3) | instid1(VALU_DEP_1)
; GISEL11_W64-NEXT: v_or_b32_e32 v1, 0xffff0000, v11
; GISEL11_W64-NEXT: s_or_saveexec_b64 s[0:1], -1
; GISEL11_W64-NEXT: s_waitcnt_depctr depctr_va_vdst(0)
; GISEL11_W64-NEXT: v_cndmask_b32_e64 v0, v0, v1, s[0:1]
-; GISEL11_W64-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GISEL11_W64-NEXT: v_and_b32_e32 v0, 0xffff, v0
; GISEL11_W64-NEXT: s_mov_b64 exec, s[0:1]
+; GISEL11_W64-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GISEL11_W64-NEXT: v_mov_b32_e32 v1, v0
; GISEL11_W64-NEXT: global_store_b32 v[8:9], v1, off
; GISEL11_W64-NEXT: s_endpgm
@@ -930,12 +957,14 @@ define amdgpu_cs_chain void @set_inactive_chain_arg_active_demanded(ptr addrspac
; DAGISEL11_W64-NEXT: s_or_saveexec_b64 s[0:1], -1
; DAGISEL11_W64-NEXT: v_mov_b32_e32 v0, v10
; DAGISEL11_W64-NEXT: s_mov_b64 exec, s[0:1]
+; DAGISEL11_W64-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; DAGISEL11_W64-NEXT: s_or_saveexec_b64 s[0:1], -1
; DAGISEL11_W64-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; DAGISEL11_W64-NEXT: v_cndmask_b32_e64 v0, v0, v11, s[0:1]
-; DAGISEL11_W64-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
+; DAGISEL11_W64-NEXT: s_delay_alu instid0(VALU_DEP_1)
; DAGISEL11_W64-NEXT: v_and_b32_e32 v0, 0xffff, v0
; DAGISEL11_W64-NEXT: s_mov_b64 exec, s[0:1]
+; DAGISEL11_W64-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; DAGISEL11_W64-NEXT: v_mov_b32_e32 v1, v0
; DAGISEL11_W64-NEXT: global_store_b32 v[8:9], v1, off
; DAGISEL11_W64-NEXT: s_endpgm
diff --git a/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.struct.atomic.buffer.load.ll b/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.struct.atomic.buffer.load.ll
index 7f795e83411650..955c1643c0f58c 100644
--- a/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.struct.atomic.buffer.load.ll
+++ b/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.struct.atomic.buffer.load.ll
@@ -26,7 +26,7 @@ define amdgpu_kernel void @struct_atomic_buffer_load_i32(<4 x i32> %addr, i32 %i
; GFX11-NEXT: s_waitcnt vmcnt(0)
; GFX11-NEXT: v_cmp_ne_u32_e32 vcc_lo, v2, v0
; GFX11-NEXT: s_or_b32 s4, vcc_lo, s4
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
; GFX11-NEXT: s_cbranch_execnz .LBB0_1
; GFX11-NEXT: ; %bb.2: ; %bb2
@@ -52,7 +52,7 @@ define amdgpu_kernel void @struct_atomic_buffer_load_i32(<4 x i32> %addr, i32 %i
; GFX12-NEXT: s_wait_loadcnt 0x0
; GFX12-NEXT: v_cmp_ne_u32_e32 vcc_lo, v2, v0
; GFX12-NEXT: s_or_b32 s4, vcc_lo, s4
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
; GFX12-NEXT: s_cbranch_execnz .LBB0_1
; GFX12-NEXT: ; %bb.2: ; %bb2
@@ -81,7 +81,7 @@ define amdgpu_kernel void @struct_atomic_buffer_load_i32_const_idx(<4 x i32> %ad
; GFX11-NEXT: s_waitcnt vmcnt(0)
; GFX11-NEXT: v_cmp_ne_u32_e32 vcc_lo, v2, v0
; GFX11-NEXT: s_or_b32 s4, vcc_lo, s4
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
; GFX11-NEXT: s_cbranch_execnz .LBB1_1
; GFX11-NEXT: ; %bb.2: ; %bb2
@@ -105,7 +105,7 @@ define amdgpu_kernel void @struct_atomic_buffer_load_i32_const_idx(<4 x i32> %ad
; GFX12-NEXT: s_wait_loadcnt 0x0
; GFX12-NEXT: v_cmp_ne_u32_e32 vcc_lo, v2, v0
; GFX12-NEXT: s_or_b32 s4, vcc_lo, s4
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
; GFX12-NEXT: s_cbranch_execnz .LBB1_1
; GFX12-NEXT: ; %bb.2: ; %bb2
@@ -137,7 +137,7 @@ define amdgpu_kernel void @struct_atomic_buffer_load_i32_off(<4 x i32> %addr, i3
; GFX11-NEXT: s_waitcnt vmcnt(0)
; GFX11-NEXT: v_cmp_ne_u32_e32 vcc_lo, v2, v0
; GFX11-NEXT: s_or_b32 s4, vcc_lo, s4
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
; GFX11-NEXT: s_cbranch_execnz .LBB2_1
; GFX11-NEXT: ; %bb.2: ; %bb2
@@ -163,7 +163,7 @@ define amdgpu_kernel void @struct_atomic_buffer_load_i32_off(<4 x i32> %addr, i3
; GFX12-NEXT: s_wait_loadcnt 0x0
; GFX12-NEXT: v_cmp_ne_u32_e32 vcc_lo, v2, v0
; GFX12-NEXT: s_or_b32 s4, vcc_lo, s4
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
; GFX12-NEXT: s_cbranch_execnz .LBB2_1
; GFX12-NEXT: ; %bb.2: ; %bb2
@@ -195,7 +195,7 @@ define amdgpu_kernel void @struct_atomic_buffer_load_i32_soff(<4 x i32> %addr, i
; GFX11-NEXT: s_waitcnt vmcnt(0)
; GFX11-NEXT: v_cmp_ne_u32_e32 vcc_lo, v2, v0
; GFX11-NEXT: s_or_b32 s4, vcc_lo, s4
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
; GFX11-NEXT: s_cbranch_execnz .LBB3_1
; GFX11-NEXT: ; %bb.2: ; %bb2
@@ -222,7 +222,7 @@ define amdgpu_kernel void @struct_atomic_buffer_load_i32_soff(<4 x i32> %addr, i
; GFX12-NEXT: s_wait_loadcnt 0x0
; GFX12-NEXT: v_cmp_ne_u32_e32 vcc_lo, v2, v0
; GFX12-NEXT: s_or_b32 s4, vcc_lo, s4
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
; GFX12-NEXT: s_cbranch_execnz .LBB3_1
; GFX12-NEXT: ; %bb.2: ; %bb2
@@ -253,7 +253,7 @@ define amdgpu_kernel void @struct_atomic_buffer_load_i32_dlc(<4 x i32> %addr, i3
; GFX11-NEXT: s_waitcnt vmcnt(0)
; GFX11-NEXT: v_cmp_ne_u32_e32 vcc_lo, v2, v0
; GFX11-NEXT: s_or_b32 s4, vcc_lo, s4
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
; GFX11-NEXT: s_cbranch_execnz .LBB4_1
; GFX11-NEXT: ; %bb.2: ; %bb2
@@ -279,7 +279,7 @@ define amdgpu_kernel void @struct_atomic_buffer_load_i32_dlc(<4 x i32> %addr, i3
; GFX12-NEXT: s_wait_loadcnt 0x0
; GFX12-NEXT: v_cmp_ne_u32_e32 vcc_lo, v2, v0
; GFX12-NEXT: s_or_b32 s4, vcc_lo, s4
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
; GFX12-NEXT: s_cbranch_execnz .LBB4_1
; GFX12-NEXT: ; %bb.2: ; %bb2
@@ -313,6 +313,7 @@ define amdgpu_kernel void @struct_nonatomic_buffer_load_i32(<4 x i32> %addr, i32
; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_or_b32 s0, s1, s0
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_execnz .LBB5_1
; GFX11-NEXT: ; %bb.2: ; %bb2
; GFX11-NEXT: s_endpgm
@@ -340,6 +341,7 @@ define amdgpu_kernel void @struct_nonatomic_buffer_load_i32(<4 x i32> %addr, i32
; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_or_b32 s0, s1, s0
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB5_1
; GFX12-NEXT: ; %bb.2: ; %bb2
; GFX12-NEXT: s_endpgm
@@ -370,7 +372,7 @@ define amdgpu_kernel void @struct_atomic_buffer_load_i64(<4 x i32> %addr, i32 %i
; GFX11-NEXT: s_waitcnt vmcnt(0)
; GFX11-NEXT: v_cmp_ne_u64_e32 vcc_lo, v[3:4], v[0:1]
; GFX11-NEXT: s_or_b32 s4, vcc_lo, s4
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
; GFX11-NEXT: s_cbranch_execnz .LBB6_1
; GFX11-NEXT: ; %bb.2: ; %bb2
@@ -397,7 +399,7 @@ define amdgpu_kernel void @struct_atomic_buffer_load_i64(<4 x i32> %addr, i32 %i
; GFX12-SDAG-TRUE16-NEXT: s_wait_loadcnt 0x0
; GFX12-SDAG-TRUE16-NEXT: v_cmp_ne_u64_e32 vcc_lo, v[4:5], v[0:1]
; GFX12-SDAG-TRUE16-NEXT: s_or_b32 s4, vcc_lo, s4
-; GFX12-SDAG-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX12-SDAG-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-SDAG-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
; GFX12-SDAG-TRUE16-NEXT: s_cbranch_execnz .LBB6_1
; GFX12-SDAG-TRUE16-NEXT: ; %bb.2: ; %bb2
@@ -424,7 +426,7 @@ define amdgpu_kernel void @struct_atomic_buffer_load_i64(<4 x i32> %addr, i32 %i
; GFX12-SDAG-FAKE16-NEXT: s_wait_loadcnt 0x0
; GFX12-SDAG-FAKE16-NEXT: v_cmp_ne_u64_e32 vcc_lo, v[4:5], v[0:1]
; GFX12-SDAG-FAKE16-NEXT: s_or_b32 s4, vcc_lo, s4
-; GFX12-SDAG-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX12-SDAG-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-SDAG-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
; GFX12-SDAG-FAKE16-NEXT: s_cbranch_execnz .LBB6_1
; GFX12-SDAG-FAKE16-NEXT: ; %bb.2: ; %bb2
@@ -451,7 +453,7 @@ define amdgpu_kernel void @struct_atomic_buffer_load_i64(<4 x i32> %addr, i32 %i
; GFX12-GISEL-TRUE16-NEXT: s_wait_loadcnt 0x0
; GFX12-GISEL-TRUE16-NEXT: v_cmp_ne_u64_e32 vcc_lo, v[4:5], v[0:1]
; GFX12-GISEL-TRUE16-NEXT: s_or_b32 s4, vcc_lo, s4
-; GFX12-GISEL-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX12-GISEL-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-GISEL-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
; GFX12-GISEL-TRUE16-NEXT: s_cbranch_execnz .LBB6_1
; GFX12-GISEL-TRUE16-NEXT: ; %bb.2: ; %bb2
@@ -478,7 +480,7 @@ define amdgpu_kernel void @struct_atomic_buffer_load_i64(<4 x i32> %addr, i32 %i
; GFX12-GISEL-FAKE16-NEXT: s_wait_loadcnt 0x0
; GFX12-GISEL-FAKE16-NEXT: v_cmp_ne_u64_e32 vcc_lo, v[4:5], v[0:1]
; GFX12-GISEL-FAKE16-NEXT: s_or_b32 s4, vcc_lo, s4
-; GFX12-GISEL-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX12-GISEL-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-GISEL-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
; GFX12-GISEL-FAKE16-NEXT: s_cbranch_execnz .LBB6_1
; GFX12-GISEL-FAKE16-NEXT: ; %bb.2: ; %bb2
@@ -511,7 +513,7 @@ define amdgpu_kernel void @struct_atomic_buffer_load_v2i16(<4 x i32> %addr, i32
; GFX11-NEXT: s_waitcnt vmcnt(0)
; GFX11-NEXT: v_cmp_ne_u32_e32 vcc_lo, v2, v0
; GFX11-NEXT: s_or_b32 s4, vcc_lo, s4
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
; GFX11-NEXT: s_cbranch_execnz .LBB7_1
; GFX11-NEXT: ; %bb.2: ; %bb2
@@ -537,7 +539,7 @@ define amdgpu_kernel void @struct_atomic_buffer_load_v2i16(<4 x i32> %addr, i32
; GFX12-NEXT: s_wait_loadcnt 0x0
; GFX12-NEXT: v_cmp_ne_u32_e32 vcc_lo, v2, v0
; GFX12-NEXT: s_or_b32 s4, vcc_lo, s4
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
; GFX12-NEXT: s_cbranch_execnz .LBB7_1
; GFX12-NEXT: ; %bb.2: ; %bb2
@@ -573,6 +575,7 @@ define amdgpu_kernel void @struct_atomic_buffer_load_v4i16(<4 x i32> %addr, i32
; GFX11-SDAG-TRUE16-NEXT: v_cmp_ne_u32_e32 vcc_lo, v2, v0
; GFX11-SDAG-TRUE16-NEXT: s_or_b32 s4, vcc_lo, s4
; GFX11-SDAG-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
+; GFX11-SDAG-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-SDAG-TRUE16-NEXT: s_cbranch_execnz .LBB8_1
; GFX11-SDAG-TRUE16-NEXT: ; %bb.2: ; %bb2
; GFX11-SDAG-TRUE16-NEXT: s_endpgm
@@ -595,7 +598,7 @@ define amdgpu_kernel void @struct_atomic_buffer_load_v4i16(<4 x i32> %addr, i32
; GFX11-SDAG-FAKE16-NEXT: v_lshl_or_b32 v2, v3, 16, v2
; GFX11-SDAG-FAKE16-NEXT: v_cmp_ne_u32_e32 vcc_lo, v2, v0
; GFX11-SDAG-FAKE16-NEXT: s_or_b32 s4, vcc_lo, s4
-; GFX11-SDAG-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-SDAG-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-SDAG-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
; GFX11-SDAG-FAKE16-NEXT: s_cbranch_execnz .LBB8_1
; GFX11-SDAG-FAKE16-NEXT: ; %bb.2: ; %bb2
@@ -621,6 +624,7 @@ define amdgpu_kernel void @struct_atomic_buffer_load_v4i16(<4 x i32> %addr, i32
; GFX11-GISEL-TRUE16-NEXT: v_cmp_ne_u32_e32 vcc_lo, s5, v0
; GFX11-GISEL-TRUE16-NEXT: s_or_b32 s4, vcc_lo, s4
; GFX11-GISEL-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
+; GFX11-GISEL-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-GISEL-TRUE16-NEXT: s_cbranch_execnz .LBB8_1
; GFX11-GISEL-TRUE16-NEXT: ; %bb.2: ; %bb2
; GFX11-GISEL-TRUE16-NEXT: s_endpgm
@@ -645,6 +649,7 @@ define amdgpu_kernel void @struct_atomic_buffer_load_v4i16(<4 x i32> %addr, i32
; GFX11-GISEL-FAKE16-NEXT: v_cmp_ne_u32_e32 vcc_lo, s5, v0
; GFX11-GISEL-FAKE16-NEXT: s_or_b32 s4, vcc_lo, s4
; GFX11-GISEL-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
+; GFX11-GISEL-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-GISEL-FAKE16-NEXT: s_cbranch_execnz .LBB8_1
; GFX11-GISEL-FAKE16-NEXT: ; %bb.2: ; %bb2
; GFX11-GISEL-FAKE16-NEXT: s_endpgm
@@ -669,6 +674,7 @@ define amdgpu_kernel void @struct_atomic_buffer_load_v4i16(<4 x i32> %addr, i32
; GFX11-GISEL-NEXT: v_cmp_ne_u32_e32 vcc_lo, s5, v0
; GFX11-GISEL-NEXT: s_or_b32 s4, vcc_lo, s4
; GFX11-GISEL-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
+; GFX11-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-GISEL-NEXT: s_cbranch_execnz .LBB8_1
; GFX11-GISEL-NEXT: ; %bb.2: ; %bb2
; GFX11-GISEL-NEXT: s_endpgm
@@ -696,6 +702,7 @@ define amdgpu_kernel void @struct_atomic_buffer_load_v4i16(<4 x i32> %addr, i32
; GFX12-SDAG-TRUE16-NEXT: v_cmp_ne_u32_e32 vcc_lo, v2, v0
; GFX12-SDAG-TRUE16-NEXT: s_or_b32 s4, vcc_lo, s4
; GFX12-SDAG-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
+; GFX12-SDAG-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-SDAG-TRUE16-NEXT: s_cbranch_execnz .LBB8_1
; GFX12-SDAG-TRUE16-NEXT: ; %bb.2: ; %bb2
; GFX12-SDAG-TRUE16-NEXT: s_endpgm
@@ -723,7 +730,7 @@ define amdgpu_kernel void @struct_atomic_buffer_load_v4i16(<4 x i32> %addr, i32
; GFX12-SDAG-FAKE16-NEXT: v_lshl_or_b32 v2, v3, 16, v2
; GFX12-SDAG-FAKE16-NEXT: v_cmp_ne_u32_e32 vcc_lo, v2, v0
; GFX12-SDAG-FAKE16-NEXT: s_or_b32 s4, vcc_lo, s4
-; GFX12-SDAG-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX12-SDAG-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-SDAG-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
; GFX12-SDAG-FAKE16-NEXT: s_cbranch_execnz .LBB8_1
; GFX12-SDAG-FAKE16-NEXT: ; %bb.2: ; %bb2
@@ -754,6 +761,7 @@ define amdgpu_kernel void @struct_atomic_buffer_load_v4i16(<4 x i32> %addr, i32
; GFX12-GISEL-TRUE16-NEXT: v_cmp_ne_u32_e32 vcc_lo, s5, v0
; GFX12-GISEL-TRUE16-NEXT: s_or_b32 s4, vcc_lo, s4
; GFX12-GISEL-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
+; GFX12-GISEL-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-GISEL-TRUE16-NEXT: s_cbranch_execnz .LBB8_1
; GFX12-GISEL-TRUE16-NEXT: ; %bb.2: ; %bb2
; GFX12-GISEL-TRUE16-NEXT: s_endpgm
@@ -783,6 +791,7 @@ define amdgpu_kernel void @struct_atomic_buffer_load_v4i16(<4 x i32> %addr, i32
; GFX12-GISEL-FAKE16-NEXT: v_cmp_ne_u32_e32 vcc_lo, s5, v0
; GFX12-GISEL-FAKE16-NEXT: s_or_b32 s4, vcc_lo, s4
; GFX12-GISEL-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
+; GFX12-GISEL-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-GISEL-FAKE16-NEXT: s_cbranch_execnz .LBB8_1
; GFX12-GISEL-FAKE16-NEXT: ; %bb.2: ; %bb2
; GFX12-GISEL-FAKE16-NEXT: s_endpgm
@@ -815,7 +824,7 @@ define amdgpu_kernel void @struct_atomic_buffer_load_v4i32(<4 x i32> %addr, i32
; GFX11-NEXT: s_waitcnt vmcnt(0)
; GFX11-NEXT: v_cmp_ne_u32_e32 vcc_lo, v5, v0
; GFX11-NEXT: s_or_b32 s4, vcc_lo, s4
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
; GFX11-NEXT: s_cbranch_execnz .LBB9_1
; GFX11-NEXT: ; %bb.2: ; %bb2
@@ -841,7 +850,7 @@ define amdgpu_kernel void @struct_atomic_buffer_load_v4i32(<4 x i32> %addr, i32
; GFX12-NEXT: s_wait_loadcnt 0x0
; GFX12-NEXT: v_cmp_ne_u32_e32 vcc_lo, v5, v0
; GFX12-NEXT: s_or_b32 s4, vcc_lo, s4
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
; GFX12-NEXT: s_cbranch_execnz .LBB9_1
; GFX12-NEXT: ; %bb.2: ; %bb2
@@ -876,7 +885,7 @@ define amdgpu_kernel void @struct_atomic_buffer_load_ptr(<4 x i32> %addr, i32 %i
; GFX11-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
; GFX11-NEXT: v_cmp_ne_u32_e32 vcc_lo, v2, v0
; GFX11-NEXT: s_or_b32 s4, vcc_lo, s4
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
; GFX11-NEXT: s_cbranch_execnz .LBB10_1
; GFX11-NEXT: ; %bb.2: ; %bb2
@@ -904,7 +913,7 @@ define amdgpu_kernel void @struct_atomic_buffer_load_ptr(<4 x i32> %addr, i32 %i
; GFX12-NEXT: s_wait_loadcnt_dscnt 0x0
; GFX12-NEXT: v_cmp_ne_u32_e32 vcc_lo, v2, v0
; GFX12-NEXT: s_or_b32 s4, vcc_lo, s4
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
; GFX12-NEXT: s_cbranch_execnz .LBB10_1
; GFX12-NEXT: ; %bb.2: ; %bb2
diff --git a/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.struct.buffer.store.ll b/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.struct.buffer.store.ll
index b5212ff1b7d8ce..8e0974e7f67420 100644
--- a/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.struct.buffer.store.ll
+++ b/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.struct.buffer.store.ll
@@ -362,6 +362,7 @@ define amdgpu_ps void @struct_buffer_store_f16(<4 x i32> inreg %rsrc, float %v1,
; GFX12-TRUE16-NEXT: v_nop
; GFX12-TRUE16-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX12-TRUE16-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 2, 2), 0 ; msbs: dst=0 src0=0 src1=0 src2=0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: v_cvt_f16_f32_e32 v0.l, v0
; GFX12-TRUE16-NEXT: buffer_store_b16 v0, v1, s[0:3], null idxen
; GFX12-TRUE16-NEXT: s_endpgm
@@ -373,6 +374,7 @@ define amdgpu_ps void @struct_buffer_store_f16(<4 x i32> inreg %rsrc, float %v1,
; GFX12-FAKE16-NEXT: v_nop
; GFX12-FAKE16-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX12-FAKE16-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 2, 2), 0 ; msbs: dst=0 src0=0 src1=0 src2=0
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: v_cvt_f16_f32_e32 v0, v0
; GFX12-FAKE16-NEXT: buffer_store_b16 v0, v1, s[0:3], null idxen
; GFX12-FAKE16-NEXT: s_endpgm
diff --git a/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.struct.ptr.atomic.buffer.load.ll b/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.struct.ptr.atomic.buffer.load.ll
index f29842a69ea118..4a8c256930bb27 100644
--- a/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.struct.ptr.atomic.buffer.load.ll
+++ b/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.struct.ptr.atomic.buffer.load.ll
@@ -26,7 +26,7 @@ define amdgpu_kernel void @struct_ptr_atomic_buffer_load_i32(ptr addrspace(8) %p
; GFX11-NEXT: s_waitcnt vmcnt(0)
; GFX11-NEXT: v_cmp_ne_u32_e32 vcc_lo, v2, v0
; GFX11-NEXT: s_or_b32 s4, vcc_lo, s4
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
; GFX11-NEXT: s_cbranch_execnz .LBB0_1
; GFX11-NEXT: ; %bb.2: ; %bb2
@@ -52,7 +52,7 @@ define amdgpu_kernel void @struct_ptr_atomic_buffer_load_i32(ptr addrspace(8) %p
; GFX12-NEXT: s_wait_loadcnt 0x0
; GFX12-NEXT: v_cmp_ne_u32_e32 vcc_lo, v2, v0
; GFX12-NEXT: s_or_b32 s4, vcc_lo, s4
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
; GFX12-NEXT: s_cbranch_execnz .LBB0_1
; GFX12-NEXT: ; %bb.2: ; %bb2
@@ -81,7 +81,7 @@ define amdgpu_kernel void @struct_ptr_atomic_buffer_load_i32_const_idx(ptr addrs
; GFX11-NEXT: s_waitcnt vmcnt(0)
; GFX11-NEXT: v_cmp_ne_u32_e32 vcc_lo, v2, v0
; GFX11-NEXT: s_or_b32 s4, vcc_lo, s4
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
; GFX11-NEXT: s_cbranch_execnz .LBB1_1
; GFX11-NEXT: ; %bb.2: ; %bb2
@@ -105,7 +105,7 @@ define amdgpu_kernel void @struct_ptr_atomic_buffer_load_i32_const_idx(ptr addrs
; GFX12-NEXT: s_wait_loadcnt 0x0
; GFX12-NEXT: v_cmp_ne_u32_e32 vcc_lo, v2, v0
; GFX12-NEXT: s_or_b32 s4, vcc_lo, s4
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
; GFX12-NEXT: s_cbranch_execnz .LBB1_1
; GFX12-NEXT: ; %bb.2: ; %bb2
@@ -137,7 +137,7 @@ define amdgpu_kernel void @struct_ptr_atomic_buffer_load_i32_off(ptr addrspace(8
; GFX11-NEXT: s_waitcnt vmcnt(0)
; GFX11-NEXT: v_cmp_ne_u32_e32 vcc_lo, v2, v0
; GFX11-NEXT: s_or_b32 s4, vcc_lo, s4
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
; GFX11-NEXT: s_cbranch_execnz .LBB2_1
; GFX11-NEXT: ; %bb.2: ; %bb2
@@ -163,7 +163,7 @@ define amdgpu_kernel void @struct_ptr_atomic_buffer_load_i32_off(ptr addrspace(8
; GFX12-NEXT: s_wait_loadcnt 0x0
; GFX12-NEXT: v_cmp_ne_u32_e32 vcc_lo, v2, v0
; GFX12-NEXT: s_or_b32 s4, vcc_lo, s4
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
; GFX12-NEXT: s_cbranch_execnz .LBB2_1
; GFX12-NEXT: ; %bb.2: ; %bb2
@@ -195,7 +195,7 @@ define amdgpu_kernel void @struct_ptr_atomic_buffer_load_i32_soff(ptr addrspace(
; GFX11-NEXT: s_waitcnt vmcnt(0)
; GFX11-NEXT: v_cmp_ne_u32_e32 vcc_lo, v2, v0
; GFX11-NEXT: s_or_b32 s4, vcc_lo, s4
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
; GFX11-NEXT: s_cbranch_execnz .LBB3_1
; GFX11-NEXT: ; %bb.2: ; %bb2
@@ -222,7 +222,7 @@ define amdgpu_kernel void @struct_ptr_atomic_buffer_load_i32_soff(ptr addrspace(
; GFX12-NEXT: s_wait_loadcnt 0x0
; GFX12-NEXT: v_cmp_ne_u32_e32 vcc_lo, v2, v0
; GFX12-NEXT: s_or_b32 s4, vcc_lo, s4
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
; GFX12-NEXT: s_cbranch_execnz .LBB3_1
; GFX12-NEXT: ; %bb.2: ; %bb2
@@ -253,7 +253,7 @@ define amdgpu_kernel void @struct_ptr_atomic_buffer_load_i32_dlc(ptr addrspace(8
; GFX11-NEXT: s_waitcnt vmcnt(0)
; GFX11-NEXT: v_cmp_ne_u32_e32 vcc_lo, v2, v0
; GFX11-NEXT: s_or_b32 s4, vcc_lo, s4
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
; GFX11-NEXT: s_cbranch_execnz .LBB4_1
; GFX11-NEXT: ; %bb.2: ; %bb2
@@ -279,7 +279,7 @@ define amdgpu_kernel void @struct_ptr_atomic_buffer_load_i32_dlc(ptr addrspace(8
; GFX12-NEXT: s_wait_loadcnt 0x0
; GFX12-NEXT: v_cmp_ne_u32_e32 vcc_lo, v2, v0
; GFX12-NEXT: s_or_b32 s4, vcc_lo, s4
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
; GFX12-NEXT: s_cbranch_execnz .LBB4_1
; GFX12-NEXT: ; %bb.2: ; %bb2
@@ -313,6 +313,7 @@ define amdgpu_kernel void @struct_ptr_nonatomic_buffer_load_i32(ptr addrspace(8)
; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_or_b32 s0, s1, s0
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_execnz .LBB5_1
; GFX11-NEXT: ; %bb.2: ; %bb2
; GFX11-NEXT: s_endpgm
@@ -340,6 +341,7 @@ define amdgpu_kernel void @struct_ptr_nonatomic_buffer_load_i32(ptr addrspace(8)
; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_or_b32 s0, s1, s0
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB5_1
; GFX12-NEXT: ; %bb.2: ; %bb2
; GFX12-NEXT: s_endpgm
@@ -370,7 +372,7 @@ define amdgpu_kernel void @struct_ptr_atomic_buffer_load_i64(ptr addrspace(8) %p
; GFX11-NEXT: s_waitcnt vmcnt(0)
; GFX11-NEXT: v_cmp_ne_u64_e32 vcc_lo, v[3:4], v[0:1]
; GFX11-NEXT: s_or_b32 s4, vcc_lo, s4
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
; GFX11-NEXT: s_cbranch_execnz .LBB6_1
; GFX11-NEXT: ; %bb.2: ; %bb2
@@ -397,7 +399,7 @@ define amdgpu_kernel void @struct_ptr_atomic_buffer_load_i64(ptr addrspace(8) %p
; GFX12-SDAG-TRUE16-NEXT: s_wait_loadcnt 0x0
; GFX12-SDAG-TRUE16-NEXT: v_cmp_ne_u64_e32 vcc_lo, v[4:5], v[0:1]
; GFX12-SDAG-TRUE16-NEXT: s_or_b32 s4, vcc_lo, s4
-; GFX12-SDAG-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX12-SDAG-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-SDAG-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
; GFX12-SDAG-TRUE16-NEXT: s_cbranch_execnz .LBB6_1
; GFX12-SDAG-TRUE16-NEXT: ; %bb.2: ; %bb2
@@ -424,7 +426,7 @@ define amdgpu_kernel void @struct_ptr_atomic_buffer_load_i64(ptr addrspace(8) %p
; GFX12-SDAG-FAKE16-NEXT: s_wait_loadcnt 0x0
; GFX12-SDAG-FAKE16-NEXT: v_cmp_ne_u64_e32 vcc_lo, v[4:5], v[0:1]
; GFX12-SDAG-FAKE16-NEXT: s_or_b32 s4, vcc_lo, s4
-; GFX12-SDAG-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX12-SDAG-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-SDAG-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
; GFX12-SDAG-FAKE16-NEXT: s_cbranch_execnz .LBB6_1
; GFX12-SDAG-FAKE16-NEXT: ; %bb.2: ; %bb2
@@ -451,7 +453,7 @@ define amdgpu_kernel void @struct_ptr_atomic_buffer_load_i64(ptr addrspace(8) %p
; GFX12-GISEL-TRUE16-NEXT: s_wait_loadcnt 0x0
; GFX12-GISEL-TRUE16-NEXT: v_cmp_ne_u64_e32 vcc_lo, v[4:5], v[0:1]
; GFX12-GISEL-TRUE16-NEXT: s_or_b32 s4, vcc_lo, s4
-; GFX12-GISEL-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX12-GISEL-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-GISEL-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
; GFX12-GISEL-TRUE16-NEXT: s_cbranch_execnz .LBB6_1
; GFX12-GISEL-TRUE16-NEXT: ; %bb.2: ; %bb2
@@ -478,7 +480,7 @@ define amdgpu_kernel void @struct_ptr_atomic_buffer_load_i64(ptr addrspace(8) %p
; GFX12-GISEL-FAKE16-NEXT: s_wait_loadcnt 0x0
; GFX12-GISEL-FAKE16-NEXT: v_cmp_ne_u64_e32 vcc_lo, v[4:5], v[0:1]
; GFX12-GISEL-FAKE16-NEXT: s_or_b32 s4, vcc_lo, s4
-; GFX12-GISEL-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX12-GISEL-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-GISEL-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
; GFX12-GISEL-FAKE16-NEXT: s_cbranch_execnz .LBB6_1
; GFX12-GISEL-FAKE16-NEXT: ; %bb.2: ; %bb2
@@ -511,7 +513,7 @@ define amdgpu_kernel void @struct_ptr_atomic_buffer_load_v2i16(ptr addrspace(8)
; GFX11-NEXT: s_waitcnt vmcnt(0)
; GFX11-NEXT: v_cmp_ne_u32_e32 vcc_lo, v2, v0
; GFX11-NEXT: s_or_b32 s4, vcc_lo, s4
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
; GFX11-NEXT: s_cbranch_execnz .LBB7_1
; GFX11-NEXT: ; %bb.2: ; %bb2
@@ -537,7 +539,7 @@ define amdgpu_kernel void @struct_ptr_atomic_buffer_load_v2i16(ptr addrspace(8)
; GFX12-NEXT: s_wait_loadcnt 0x0
; GFX12-NEXT: v_cmp_ne_u32_e32 vcc_lo, v2, v0
; GFX12-NEXT: s_or_b32 s4, vcc_lo, s4
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
; GFX12-NEXT: s_cbranch_execnz .LBB7_1
; GFX12-NEXT: ; %bb.2: ; %bb2
@@ -573,6 +575,7 @@ define amdgpu_kernel void @struct_ptr_atomic_buffer_load_v4i16(ptr addrspace(8)
; GFX11-SDAG-TRUE16-NEXT: v_cmp_ne_u32_e32 vcc_lo, v2, v0
; GFX11-SDAG-TRUE16-NEXT: s_or_b32 s4, vcc_lo, s4
; GFX11-SDAG-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
+; GFX11-SDAG-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-SDAG-TRUE16-NEXT: s_cbranch_execnz .LBB8_1
; GFX11-SDAG-TRUE16-NEXT: ; %bb.2: ; %bb2
; GFX11-SDAG-TRUE16-NEXT: s_endpgm
@@ -595,7 +598,7 @@ define amdgpu_kernel void @struct_ptr_atomic_buffer_load_v4i16(ptr addrspace(8)
; GFX11-SDAG-FAKE16-NEXT: v_lshl_or_b32 v2, v3, 16, v2
; GFX11-SDAG-FAKE16-NEXT: v_cmp_ne_u32_e32 vcc_lo, v2, v0
; GFX11-SDAG-FAKE16-NEXT: s_or_b32 s4, vcc_lo, s4
-; GFX11-SDAG-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-SDAG-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-SDAG-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
; GFX11-SDAG-FAKE16-NEXT: s_cbranch_execnz .LBB8_1
; GFX11-SDAG-FAKE16-NEXT: ; %bb.2: ; %bb2
@@ -621,6 +624,7 @@ define amdgpu_kernel void @struct_ptr_atomic_buffer_load_v4i16(ptr addrspace(8)
; GFX11-GISEL-TRUE16-NEXT: v_cmp_ne_u32_e32 vcc_lo, s5, v0
; GFX11-GISEL-TRUE16-NEXT: s_or_b32 s4, vcc_lo, s4
; GFX11-GISEL-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
+; GFX11-GISEL-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-GISEL-TRUE16-NEXT: s_cbranch_execnz .LBB8_1
; GFX11-GISEL-TRUE16-NEXT: ; %bb.2: ; %bb2
; GFX11-GISEL-TRUE16-NEXT: s_endpgm
@@ -645,6 +649,7 @@ define amdgpu_kernel void @struct_ptr_atomic_buffer_load_v4i16(ptr addrspace(8)
; GFX11-GISEL-FAKE16-NEXT: v_cmp_ne_u32_e32 vcc_lo, s5, v0
; GFX11-GISEL-FAKE16-NEXT: s_or_b32 s4, vcc_lo, s4
; GFX11-GISEL-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
+; GFX11-GISEL-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-GISEL-FAKE16-NEXT: s_cbranch_execnz .LBB8_1
; GFX11-GISEL-FAKE16-NEXT: ; %bb.2: ; %bb2
; GFX11-GISEL-FAKE16-NEXT: s_endpgm
@@ -669,6 +674,7 @@ define amdgpu_kernel void @struct_ptr_atomic_buffer_load_v4i16(ptr addrspace(8)
; GFX11-GISEL-NEXT: v_cmp_ne_u32_e32 vcc_lo, s5, v0
; GFX11-GISEL-NEXT: s_or_b32 s4, vcc_lo, s4
; GFX11-GISEL-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
+; GFX11-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-GISEL-NEXT: s_cbranch_execnz .LBB8_1
; GFX11-GISEL-NEXT: ; %bb.2: ; %bb2
; GFX11-GISEL-NEXT: s_endpgm
@@ -696,6 +702,7 @@ define amdgpu_kernel void @struct_ptr_atomic_buffer_load_v4i16(ptr addrspace(8)
; GFX12-SDAG-TRUE16-NEXT: v_cmp_ne_u32_e32 vcc_lo, v2, v0
; GFX12-SDAG-TRUE16-NEXT: s_or_b32 s4, vcc_lo, s4
; GFX12-SDAG-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
+; GFX12-SDAG-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-SDAG-TRUE16-NEXT: s_cbranch_execnz .LBB8_1
; GFX12-SDAG-TRUE16-NEXT: ; %bb.2: ; %bb2
; GFX12-SDAG-TRUE16-NEXT: s_endpgm
@@ -723,7 +730,7 @@ define amdgpu_kernel void @struct_ptr_atomic_buffer_load_v4i16(ptr addrspace(8)
; GFX12-SDAG-FAKE16-NEXT: v_lshl_or_b32 v2, v3, 16, v2
; GFX12-SDAG-FAKE16-NEXT: v_cmp_ne_u32_e32 vcc_lo, v2, v0
; GFX12-SDAG-FAKE16-NEXT: s_or_b32 s4, vcc_lo, s4
-; GFX12-SDAG-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX12-SDAG-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-SDAG-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
; GFX12-SDAG-FAKE16-NEXT: s_cbranch_execnz .LBB8_1
; GFX12-SDAG-FAKE16-NEXT: ; %bb.2: ; %bb2
@@ -754,6 +761,7 @@ define amdgpu_kernel void @struct_ptr_atomic_buffer_load_v4i16(ptr addrspace(8)
; GFX12-GISEL-TRUE16-NEXT: v_cmp_ne_u32_e32 vcc_lo, s5, v0
; GFX12-GISEL-TRUE16-NEXT: s_or_b32 s4, vcc_lo, s4
; GFX12-GISEL-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
+; GFX12-GISEL-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-GISEL-TRUE16-NEXT: s_cbranch_execnz .LBB8_1
; GFX12-GISEL-TRUE16-NEXT: ; %bb.2: ; %bb2
; GFX12-GISEL-TRUE16-NEXT: s_endpgm
@@ -783,6 +791,7 @@ define amdgpu_kernel void @struct_ptr_atomic_buffer_load_v4i16(ptr addrspace(8)
; GFX12-GISEL-FAKE16-NEXT: v_cmp_ne_u32_e32 vcc_lo, s5, v0
; GFX12-GISEL-FAKE16-NEXT: s_or_b32 s4, vcc_lo, s4
; GFX12-GISEL-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
+; GFX12-GISEL-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-GISEL-FAKE16-NEXT: s_cbranch_execnz .LBB8_1
; GFX12-GISEL-FAKE16-NEXT: ; %bb.2: ; %bb2
; GFX12-GISEL-FAKE16-NEXT: s_endpgm
@@ -815,7 +824,7 @@ define amdgpu_kernel void @struct_ptr_atomic_buffer_load_v4i32(ptr addrspace(8)
; GFX11-NEXT: s_waitcnt vmcnt(0)
; GFX11-NEXT: v_cmp_ne_u32_e32 vcc_lo, v5, v0
; GFX11-NEXT: s_or_b32 s4, vcc_lo, s4
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
; GFX11-NEXT: s_cbranch_execnz .LBB9_1
; GFX11-NEXT: ; %bb.2: ; %bb2
@@ -841,7 +850,7 @@ define amdgpu_kernel void @struct_ptr_atomic_buffer_load_v4i32(ptr addrspace(8)
; GFX12-NEXT: s_wait_loadcnt 0x0
; GFX12-NEXT: v_cmp_ne_u32_e32 vcc_lo, v5, v0
; GFX12-NEXT: s_or_b32 s4, vcc_lo, s4
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
; GFX12-NEXT: s_cbranch_execnz .LBB9_1
; GFX12-NEXT: ; %bb.2: ; %bb2
@@ -876,7 +885,7 @@ define amdgpu_kernel void @struct_ptr_atomic_buffer_load_ptr(ptr addrspace(8) %p
; GFX11-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
; GFX11-NEXT: v_cmp_ne_u32_e32 vcc_lo, v2, v0
; GFX11-NEXT: s_or_b32 s4, vcc_lo, s4
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
; GFX11-NEXT: s_cbranch_execnz .LBB10_1
; GFX11-NEXT: ; %bb.2: ; %bb2
@@ -904,7 +913,7 @@ define amdgpu_kernel void @struct_ptr_atomic_buffer_load_ptr(ptr addrspace(8) %p
; GFX12-NEXT: s_wait_loadcnt_dscnt 0x0
; GFX12-NEXT: v_cmp_ne_u32_e32 vcc_lo, v2, v0
; GFX12-NEXT: s_or_b32 s4, vcc_lo, s4
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s4
; GFX12-NEXT: s_cbranch_execnz .LBB10_1
; GFX12-NEXT: ; %bb.2: ; %bb2
diff --git a/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.tensor.load.store.ll b/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.tensor.load.store.ll
index 5d85ef0c8239c9..6e462dafd7dd94 100644
--- a/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.tensor.load.store.ll
+++ b/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.tensor.load.store.ll
@@ -333,6 +333,7 @@ define amdgpu_ps void @tensor_load_to_lds_with_asyncmark(<4 x i32> inreg %D0, <8
; GFX1250-NEXT: tensor_load_to_lds s[0:3], s[4:11]
; GFX1250-NEXT: ; asyncmark
; GFX1250-NEXT: ; wait_asyncmark(0)
+; GFX1250-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1250-NEXT: s_wait_tensorcnt 0x0
; GFX1250-NEXT: s_endpgm
call void @llvm.amdgcn.tensor.load.to.lds(<4 x i32> %D0, <8 x i32> %D1, <4 x i32> zeroinitializer, <4 x i32> zeroinitializer, <8 x i32> zeroinitializer, i32 0)
@@ -351,6 +352,7 @@ define amdgpu_ps void @tensor_store_from_lds_with_asyncmark(<4 x i32> inreg %D0,
; GFX1250-NEXT: tensor_store_from_lds s[0:3], s[4:11]
; GFX1250-NEXT: ; asyncmark
; GFX1250-NEXT: ; wait_asyncmark(0)
+; GFX1250-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1250-NEXT: s_wait_tensorcnt 0x0
; GFX1250-NEXT: s_endpgm
call void @llvm.amdgcn.tensor.store.from.lds(<4 x i32> %D0, <8 x i32> %D1, <4 x i32> zeroinitializer, <4 x i32> zeroinitializer, <8 x i32> zeroinitializer, i32 0)
@@ -371,10 +373,12 @@ define amdgpu_ps void @tensor_load_to_lds_two_asyncmarks(<4 x i32> inreg %D0a, <
; GFX1250-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-NEXT: tensor_load_to_lds s[0:3], s[4:11]
; GFX1250-NEXT: ; asyncmark
+; GFX1250-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1250-NEXT: s_wait_tensorcnt 0xa
; GFX1250-NEXT: tensor_load_to_lds s[12:15], s[16:23]
; GFX1250-NEXT: ; asyncmark
; GFX1250-NEXT: ; wait_asyncmark(1)
+; GFX1250-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1250-NEXT: s_wait_tensorcnt 0x1
; GFX1250-NEXT: ds_load_b32 v1, v0
; GFX1250-NEXT: ; wait_asyncmark(0)
@@ -418,6 +422,7 @@ define void @tensor_and_async_lds_with_asyncmark(<4 x i32> inreg %D0, <8 x i32>
; GFX1250-NEXT: ; asyncmark
; GFX1250-NEXT: ; wait_asyncmark(0)
; GFX1250-NEXT: s_wait_asynccnt 0x0
+; GFX1250-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1250-NEXT: s_wait_tensorcnt 0x0
; GFX1250-NEXT: s_set_pc_i64 s[30:31]
call void @llvm.amdgcn.global.load.async.to.lds.b32(ptr addrspace(1) %src, ptr addrspace(3) %dst, i32 0, i32 0)
@@ -461,6 +466,7 @@ define void @tensor_or_async_lds_diamonds(i32 inreg %cond1, i32 inreg %cond2, <4
; GFX1250-SDAG-NEXT: s_and_b32 s0, s0, exec_lo
; GFX1250-SDAG-NEXT: s_cselect_b32 s0, 1, 0
; GFX1250-SDAG-NEXT: s_cmp_lg_u32 s0, 1
+; GFX1250-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1250-SDAG-NEXT: s_cbranch_scc1 .LBB14_4
; GFX1250-SDAG-NEXT: ; %bb.3: ; %t1
; GFX1250-SDAG-NEXT: tensor_load_to_lds s[12:15], s[4:11]
@@ -478,14 +484,17 @@ define void @tensor_or_async_lds_diamonds(i32 inreg %cond1, i32 inreg %cond2, <4
; GFX1250-SDAG-NEXT: s_and_b32 s0, s0, exec_lo
; GFX1250-SDAG-NEXT: s_cselect_b32 s0, 1, 0
; GFX1250-SDAG-NEXT: s_cmp_lg_u32 s0, 1
+; GFX1250-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1250-SDAG-NEXT: s_cbranch_scc1 .LBB14_8
; GFX1250-SDAG-NEXT: ; %bb.7: ; %t2
+; GFX1250-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1250-SDAG-NEXT: s_wait_tensorcnt 0xa
; GFX1250-SDAG-NEXT: tensor_load_to_lds s[12:15], s[4:11]
; GFX1250-SDAG-NEXT: ; asyncmark
; GFX1250-SDAG-NEXT: .LBB14_8: ; %merge2
; GFX1250-SDAG-NEXT: ; wait_asyncmark(1)
; GFX1250-SDAG-NEXT: s_wait_asynccnt 0x0
+; GFX1250-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1250-SDAG-NEXT: s_wait_tensorcnt 0x0
; GFX1250-SDAG-NEXT: s_set_pc_i64 s[30:31]
;
@@ -516,6 +525,7 @@ define void @tensor_or_async_lds_diamonds(i32 inreg %cond1, i32 inreg %cond2, <4
; GFX1250-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1250-GISEL-NEXT: s_xor_b32 s0, s0, 1
; GFX1250-GISEL-NEXT: s_cmp_lg_u32 s0, 0
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1250-GISEL-NEXT: s_cbranch_scc1 .LBB14_4
; GFX1250-GISEL-NEXT: ; %bb.3: ; %t1
; GFX1250-GISEL-NEXT: tensor_load_to_lds s[12:15], s[4:11]
@@ -532,14 +542,17 @@ define void @tensor_or_async_lds_diamonds(i32 inreg %cond1, i32 inreg %cond2, <4
; GFX1250-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1250-GISEL-NEXT: s_xor_b32 s0, s0, 1
; GFX1250-GISEL-NEXT: s_cmp_lg_u32 s0, 0
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1250-GISEL-NEXT: s_cbranch_scc1 .LBB14_8
; GFX1250-GISEL-NEXT: ; %bb.7: ; %t2
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1250-GISEL-NEXT: s_wait_tensorcnt 0xa
; GFX1250-GISEL-NEXT: tensor_load_to_lds s[12:15], s[4:11]
; GFX1250-GISEL-NEXT: ; asyncmark
; GFX1250-GISEL-NEXT: .LBB14_8: ; %merge2
; GFX1250-GISEL-NEXT: ; wait_asyncmark(1)
; GFX1250-GISEL-NEXT: s_wait_asynccnt 0x0
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1250-GISEL-NEXT: s_wait_tensorcnt 0x0
; GFX1250-GISEL-NEXT: s_set_pc_i64 s[30:31]
entry:
diff --git a/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.wave.shuffle.ll b/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.wave.shuffle.ll
index 7116f937b6d8a1..6061c5c96bdad6 100644
--- a/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.wave.shuffle.ll
+++ b/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.wave.shuffle.ll
@@ -93,6 +93,7 @@ define float @test_wave_shuffle_float(float %val, i32 %idx) {
; GFX11-W64-NEXT: s_xor_saveexec_b64 s[0:1], -1
; GFX11-W64-NEXT: scratch_store_b32 off, v2, s32 ; 4-byte Folded Spill
; GFX11-W64-NEXT: s_mov_b64 exec, s[0:1]
+; GFX11-W64-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-W64-NEXT: v_lshlrev_b32_e32 v3, 2, v1
; GFX11-W64-NEXT: ; kill: def $vgpr0 killed $vgpr0 killed $exec
; GFX11-W64-NEXT: ; kill: def $vgpr3 killed $vgpr3 killed $exec
@@ -187,6 +188,7 @@ define float @test_wave_shuffle_float(float %val, i32 %idx) {
; GFX11-W64-GISEL-NEXT: s_xor_saveexec_b64 s[0:1], -1
; GFX11-W64-GISEL-NEXT: scratch_store_b32 off, v2, s32 ; 4-byte Folded Spill
; GFX11-W64-GISEL-NEXT: s_mov_b64 exec, s[0:1]
+; GFX11-W64-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-W64-GISEL-NEXT: v_lshlrev_b32_e32 v3, 2, v1
; GFX11-W64-GISEL-NEXT: ; kill: def $vgpr0 killed $vgpr0 killed $exec
; GFX11-W64-GISEL-NEXT: ; kill: def $vgpr3 killed $vgpr3 killed $exec
diff --git a/llvm/test/CodeGen/AMDGPU/llvm.exp2.bf16.ll b/llvm/test/CodeGen/AMDGPU/llvm.exp2.bf16.ll
index 448f56fbbd5892..2e4a00896a8ce9 100644
--- a/llvm/test/CodeGen/AMDGPU/llvm.exp2.bf16.ll
+++ b/llvm/test/CodeGen/AMDGPU/llvm.exp2.bf16.ll
@@ -688,42 +688,42 @@ define <2 x bfloat> @v_exp2_v2bf16(<2 x bfloat> %in) {
; GFX1200-SDAG-TRUE16-NEXT: s_wait_kmcnt 0x0
; GFX1200-SDAG-TRUE16-NEXT: v_and_b32_e32 v1, 0xffff0000, v0
; GFX1200-SDAG-TRUE16-NEXT: v_lshlrev_b32_e32 v0, 16, v0
-; GFX1200-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
+; GFX1200-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_2) | instid1(VALU_DEP_1)
; GFX1200-SDAG-TRUE16-NEXT: v_cmp_gt_f32_e64 s0, 0xc2fc0000, v0
; GFX1200-SDAG-TRUE16-NEXT: s_wait_alu depctr_va_sdst(0)
; GFX1200-SDAG-TRUE16-NEXT: v_cndmask_b32_e64 v3, 0, 0x42800000, s0
-; GFX1200-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_2) | instid1(VALU_DEP_3)
; GFX1200-SDAG-TRUE16-NEXT: v_add_f32_e32 v0, v0, v3
; GFX1200-SDAG-TRUE16-NEXT: v_cmp_gt_f32_e32 vcc_lo, 0xc2fc0000, v1
; GFX1200-SDAG-TRUE16-NEXT: v_cndmask_b32_e64 v3, 0, 0xffffffc0, s0
+; GFX1200-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_2) | instid1(VALU_DEP_1)
; GFX1200-SDAG-TRUE16-NEXT: v_exp_f32_e32 v0, v0
; GFX1200-SDAG-TRUE16-NEXT: s_wait_alu depctr_va_vcc(0)
; GFX1200-SDAG-TRUE16-NEXT: v_cndmask_b32_e64 v2, 0, 0x42800000, vcc_lo
-; GFX1200-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(TRANS32_DEP_1)
; GFX1200-SDAG-TRUE16-NEXT: v_add_f32_e32 v1, v1, v2
; GFX1200-SDAG-TRUE16-NEXT: v_cndmask_b32_e64 v2, 0, 0xffffffc0, vcc_lo
+; GFX1200-SDAG-TRUE16-NEXT: s_delay_alu instid0(TRANS32_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_3)
; GFX1200-SDAG-TRUE16-NEXT: v_ldexp_f32 v0, v0, v3
-; GFX1200-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1200-SDAG-TRUE16-NEXT: v_exp_f32_e32 v1, v1
+; GFX1200-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX1200-SDAG-TRUE16-NEXT: v_bfe_u32 v3, v0, 16, 1
; GFX1200-SDAG-TRUE16-NEXT: v_or_b32_e32 v5, 0x400000, v0
-; GFX1200-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(TRANS32_DEP_1)
; GFX1200-SDAG-TRUE16-NEXT: v_add3_u32 v3, v3, v0, 0x7fff
+; GFX1200-SDAG-TRUE16-NEXT: s_delay_alu instid0(TRANS32_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1200-SDAG-TRUE16-NEXT: v_ldexp_f32 v1, v1, v2
-; GFX1200-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_2) | instid1(VALU_DEP_3)
; GFX1200-SDAG-TRUE16-NEXT: v_bfe_u32 v2, v1, 16, 1
; GFX1200-SDAG-TRUE16-NEXT: v_or_b32_e32 v4, 0x400000, v1
; GFX1200-SDAG-TRUE16-NEXT: v_cmp_u_f32_e32 vcc_lo, v1, v1
+; GFX1200-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX1200-SDAG-TRUE16-NEXT: v_add3_u32 v2, v2, v1, 0x7fff
; GFX1200-SDAG-TRUE16-NEXT: s_wait_alu depctr_va_vcc(0)
-; GFX1200-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_3) | instid1(VALU_DEP_3)
; GFX1200-SDAG-TRUE16-NEXT: v_cndmask_b32_e32 v1, v2, v4, vcc_lo
; GFX1200-SDAG-TRUE16-NEXT: v_cmp_u_f32_e32 vcc_lo, v0, v0
; GFX1200-SDAG-TRUE16-NEXT: s_wait_alu depctr_va_vcc(0)
; GFX1200-SDAG-TRUE16-NEXT: v_cndmask_b32_e32 v0, v3, v5, vcc_lo
+; GFX1200-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1200-SDAG-TRUE16-NEXT: v_lshrrev_b32_e32 v1, 16, v1
-; GFX1200-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1200-SDAG-TRUE16-NEXT: v_lshrrev_b32_e32 v0, 16, v0
+; GFX1200-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX1200-SDAG-TRUE16-NEXT: v_mov_b16_e32 v0.h, v1.l
; GFX1200-SDAG-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -736,39 +736,39 @@ define <2 x bfloat> @v_exp2_v2bf16(<2 x bfloat> %in) {
; GFX1200-SDAG-FAKE16-NEXT: s_wait_kmcnt 0x0
; GFX1200-SDAG-FAKE16-NEXT: v_lshlrev_b32_e32 v1, 16, v0
; GFX1200-SDAG-FAKE16-NEXT: v_and_b32_e32 v0, 0xffff0000, v0
-; GFX1200-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
+; GFX1200-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_2) | instid1(VALU_DEP_1)
; GFX1200-SDAG-FAKE16-NEXT: v_cmp_gt_f32_e64 s0, 0xc2fc0000, v0
; GFX1200-SDAG-FAKE16-NEXT: s_wait_alu depctr_va_sdst(0)
; GFX1200-SDAG-FAKE16-NEXT: v_cndmask_b32_e64 v3, 0, 0x42800000, s0
-; GFX1200-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_2) | instid1(VALU_DEP_3)
; GFX1200-SDAG-FAKE16-NEXT: v_add_f32_e32 v0, v0, v3
; GFX1200-SDAG-FAKE16-NEXT: v_cmp_gt_f32_e32 vcc_lo, 0xc2fc0000, v1
; GFX1200-SDAG-FAKE16-NEXT: v_cndmask_b32_e64 v3, 0, 0xffffffc0, s0
+; GFX1200-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_2) | instid1(VALU_DEP_1)
; GFX1200-SDAG-FAKE16-NEXT: v_exp_f32_e32 v0, v0
; GFX1200-SDAG-FAKE16-NEXT: s_wait_alu depctr_va_vcc(0)
; GFX1200-SDAG-FAKE16-NEXT: v_cndmask_b32_e64 v2, 0, 0x42800000, vcc_lo
-; GFX1200-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(TRANS32_DEP_1)
; GFX1200-SDAG-FAKE16-NEXT: v_add_f32_e32 v1, v1, v2
; GFX1200-SDAG-FAKE16-NEXT: v_cndmask_b32_e64 v2, 0, 0xffffffc0, vcc_lo
+; GFX1200-SDAG-FAKE16-NEXT: s_delay_alu instid0(TRANS32_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_3)
; GFX1200-SDAG-FAKE16-NEXT: v_ldexp_f32 v0, v0, v3
-; GFX1200-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1200-SDAG-FAKE16-NEXT: v_exp_f32_e32 v1, v1
+; GFX1200-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX1200-SDAG-FAKE16-NEXT: v_bfe_u32 v3, v0, 16, 1
; GFX1200-SDAG-FAKE16-NEXT: v_or_b32_e32 v5, 0x400000, v0
-; GFX1200-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(TRANS32_DEP_1)
; GFX1200-SDAG-FAKE16-NEXT: v_add3_u32 v3, v3, v0, 0x7fff
+; GFX1200-SDAG-FAKE16-NEXT: s_delay_alu instid0(TRANS32_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1200-SDAG-FAKE16-NEXT: v_ldexp_f32 v1, v1, v2
-; GFX1200-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_2) | instid1(VALU_DEP_3)
; GFX1200-SDAG-FAKE16-NEXT: v_bfe_u32 v2, v1, 16, 1
; GFX1200-SDAG-FAKE16-NEXT: v_or_b32_e32 v4, 0x400000, v1
; GFX1200-SDAG-FAKE16-NEXT: v_cmp_u_f32_e32 vcc_lo, v1, v1
+; GFX1200-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX1200-SDAG-FAKE16-NEXT: v_add3_u32 v2, v2, v1, 0x7fff
; GFX1200-SDAG-FAKE16-NEXT: s_wait_alu depctr_va_vcc(0)
-; GFX1200-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_3) | instid1(VALU_DEP_1)
; GFX1200-SDAG-FAKE16-NEXT: v_cndmask_b32_e32 v1, v2, v4, vcc_lo
; GFX1200-SDAG-FAKE16-NEXT: v_cmp_u_f32_e32 vcc_lo, v0, v0
; GFX1200-SDAG-FAKE16-NEXT: s_wait_alu depctr_va_vcc(0)
; GFX1200-SDAG-FAKE16-NEXT: v_cndmask_b32_e32 v0, v3, v5, vcc_lo
+; GFX1200-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1200-SDAG-FAKE16-NEXT: v_perm_b32 v0, v0, v1, 0x7060302
; GFX1200-SDAG-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -784,37 +784,37 @@ define <2 x bfloat> @v_exp2_v2bf16(<2 x bfloat> %in) {
; GFX1200-GI-TRUE16-NEXT: v_lshlrev_b32_e32 v1, 16, v1
; GFX1200-GI-TRUE16-NEXT: v_cmp_gt_f32_e64 s0, 0xc2fc0000, v1
; GFX1200-GI-TRUE16-NEXT: s_wait_alu depctr_va_sdst(0)
-; GFX1200-GI-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1200-GI-TRUE16-NEXT: v_cndmask_b32_e64 v3, 0, 0x42800000, s0
+; GFX1200-GI-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1200-GI-TRUE16-NEXT: v_dual_add_f32 v1, v1, v3 :: v_dual_lshlrev_b32 v0, 16, v0
-; GFX1200-GI-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_3)
; GFX1200-GI-TRUE16-NEXT: v_cmp_gt_f32_e32 vcc_lo, 0xc2fc0000, v0
; GFX1200-GI-TRUE16-NEXT: v_cndmask_b32_e64 v3, 0, 0xffffffc0, s0
+; GFX1200-GI-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_2) | instid1(VALU_DEP_1)
; GFX1200-GI-TRUE16-NEXT: v_exp_f32_e32 v1, v1
; GFX1200-GI-TRUE16-NEXT: s_wait_alu depctr_va_vcc(0)
; GFX1200-GI-TRUE16-NEXT: v_cndmask_b32_e64 v2, 0, 0x42800000, vcc_lo
-; GFX1200-GI-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(TRANS32_DEP_1)
; GFX1200-GI-TRUE16-NEXT: v_add_f32_e32 v0, v0, v2
; GFX1200-GI-TRUE16-NEXT: v_cndmask_b32_e64 v2, 0, 0xffffffc0, vcc_lo
+; GFX1200-GI-TRUE16-NEXT: s_delay_alu instid0(TRANS32_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_3)
; GFX1200-GI-TRUE16-NEXT: v_ldexp_f32 v1, v1, v3
-; GFX1200-GI-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1200-GI-TRUE16-NEXT: v_exp_f32_e32 v0, v0
+; GFX1200-GI-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX1200-GI-TRUE16-NEXT: v_bfe_u32 v3, v1, 16, 1
; GFX1200-GI-TRUE16-NEXT: v_or_b32_e32 v5, 0x400000, v1
-; GFX1200-GI-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(TRANS32_DEP_1)
; GFX1200-GI-TRUE16-NEXT: v_add3_u32 v3, v3, v1, 0x7fff
+; GFX1200-GI-TRUE16-NEXT: s_delay_alu instid0(TRANS32_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1200-GI-TRUE16-NEXT: v_ldexp_f32 v0, v0, v2
-; GFX1200-GI-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_2) | instid1(VALU_DEP_3)
; GFX1200-GI-TRUE16-NEXT: v_bfe_u32 v2, v0, 16, 1
; GFX1200-GI-TRUE16-NEXT: v_or_b32_e32 v4, 0x400000, v0
; GFX1200-GI-TRUE16-NEXT: v_cmp_u_f32_e32 vcc_lo, 0, v0
+; GFX1200-GI-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX1200-GI-TRUE16-NEXT: v_add3_u32 v2, v2, v0, 0x7fff
; GFX1200-GI-TRUE16-NEXT: s_wait_alu depctr_va_vcc(0)
-; GFX1200-GI-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_3) | instid1(VALU_DEP_3)
; GFX1200-GI-TRUE16-NEXT: v_cndmask_b32_e32 v2, v2, v4, vcc_lo
; GFX1200-GI-TRUE16-NEXT: v_cmp_u_f32_e32 vcc_lo, 0, v1
; GFX1200-GI-TRUE16-NEXT: s_wait_alu depctr_va_vcc(0)
; GFX1200-GI-TRUE16-NEXT: v_cndmask_b32_e32 v0, v3, v5, vcc_lo
+; GFX1200-GI-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX1200-GI-TRUE16-NEXT: v_mov_b16_e32 v0.l, v2.h
; GFX1200-GI-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -830,37 +830,37 @@ define <2 x bfloat> @v_exp2_v2bf16(<2 x bfloat> %in) {
; GFX1200-GI-FAKE16-NEXT: v_lshlrev_b32_e32 v1, 16, v1
; GFX1200-GI-FAKE16-NEXT: v_cmp_gt_f32_e64 s0, 0xc2fc0000, v1
; GFX1200-GI-FAKE16-NEXT: s_wait_alu depctr_va_sdst(0)
-; GFX1200-GI-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1200-GI-FAKE16-NEXT: v_cndmask_b32_e64 v3, 0, 0x42800000, s0
+; GFX1200-GI-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1200-GI-FAKE16-NEXT: v_dual_add_f32 v1, v1, v3 :: v_dual_lshlrev_b32 v0, 16, v0
-; GFX1200-GI-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_3)
; GFX1200-GI-FAKE16-NEXT: v_cmp_gt_f32_e32 vcc_lo, 0xc2fc0000, v0
; GFX1200-GI-FAKE16-NEXT: v_cndmask_b32_e64 v3, 0, 0xffffffc0, s0
+; GFX1200-GI-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_2) | instid1(VALU_DEP_1)
; GFX1200-GI-FAKE16-NEXT: v_exp_f32_e32 v1, v1
; GFX1200-GI-FAKE16-NEXT: s_wait_alu depctr_va_vcc(0)
; GFX1200-GI-FAKE16-NEXT: v_cndmask_b32_e64 v2, 0, 0x42800000, vcc_lo
-; GFX1200-GI-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(TRANS32_DEP_1)
; GFX1200-GI-FAKE16-NEXT: v_add_f32_e32 v0, v0, v2
; GFX1200-GI-FAKE16-NEXT: v_cndmask_b32_e64 v2, 0, 0xffffffc0, vcc_lo
+; GFX1200-GI-FAKE16-NEXT: s_delay_alu instid0(TRANS32_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_3)
; GFX1200-GI-FAKE16-NEXT: v_ldexp_f32 v1, v1, v3
-; GFX1200-GI-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1200-GI-FAKE16-NEXT: v_exp_f32_e32 v0, v0
+; GFX1200-GI-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX1200-GI-FAKE16-NEXT: v_bfe_u32 v3, v1, 16, 1
; GFX1200-GI-FAKE16-NEXT: v_or_b32_e32 v5, 0x400000, v1
-; GFX1200-GI-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(TRANS32_DEP_1)
; GFX1200-GI-FAKE16-NEXT: v_add3_u32 v3, v3, v1, 0x7fff
+; GFX1200-GI-FAKE16-NEXT: s_delay_alu instid0(TRANS32_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1200-GI-FAKE16-NEXT: v_ldexp_f32 v0, v0, v2
-; GFX1200-GI-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_2) | instid1(VALU_DEP_3)
; GFX1200-GI-FAKE16-NEXT: v_bfe_u32 v2, v0, 16, 1
; GFX1200-GI-FAKE16-NEXT: v_or_b32_e32 v4, 0x400000, v0
; GFX1200-GI-FAKE16-NEXT: v_cmp_u_f32_e32 vcc_lo, 0, v0
+; GFX1200-GI-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX1200-GI-FAKE16-NEXT: v_add3_u32 v2, v2, v0, 0x7fff
; GFX1200-GI-FAKE16-NEXT: s_wait_alu depctr_va_vcc(0)
-; GFX1200-GI-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_3) | instid1(VALU_DEP_1)
; GFX1200-GI-FAKE16-NEXT: v_cndmask_b32_e32 v0, v2, v4, vcc_lo
; GFX1200-GI-FAKE16-NEXT: v_cmp_u_f32_e32 vcc_lo, 0, v1
; GFX1200-GI-FAKE16-NEXT: s_wait_alu depctr_va_vcc(0)
; GFX1200-GI-FAKE16-NEXT: v_cndmask_b32_e32 v1, v3, v5, vcc_lo
+; GFX1200-GI-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1200-GI-FAKE16-NEXT: v_perm_b32 v0, v1, v0, 0x7060302
; GFX1200-GI-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -925,33 +925,32 @@ define <2 x bfloat> @v_exp2_fabs_v2bf16(<2 x bfloat> %in) {
; GFX1200-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1200-SDAG-TRUE16-NEXT: v_and_b16 v1.l, 0x7fff, v1.l
; GFX1200-SDAG-TRUE16-NEXT: v_lshlrev_b32_e32 v1, 16, v1
-; GFX1200-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
+; GFX1200-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_2) | instid1(VALU_DEP_1)
; GFX1200-SDAG-TRUE16-NEXT: v_cmp_gt_f32_e64 s0, 0xc2fc0000, v1
; GFX1200-SDAG-TRUE16-NEXT: s_wait_alu depctr_va_sdst(0)
; GFX1200-SDAG-TRUE16-NEXT: v_cndmask_b32_e64 v2, 0, 0x42800000, s0
-; GFX1200-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1200-SDAG-TRUE16-NEXT: v_dual_add_f32 v1, v1, v2 :: v_dual_lshlrev_b32 v0, 16, v0
+; GFX1200-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_3)
; GFX1200-SDAG-TRUE16-NEXT: v_cmp_gt_f32_e32 vcc_lo, 0xc2fc0000, v0
; GFX1200-SDAG-TRUE16-NEXT: v_cndmask_b32_e64 v2, 0, 0xffffffc0, s0
-; GFX1200-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_3) | instid1(VALU_DEP_2)
; GFX1200-SDAG-TRUE16-NEXT: v_exp_f32_e32 v1, v1
; GFX1200-SDAG-TRUE16-NEXT: s_wait_alu depctr_va_vcc(0)
; GFX1200-SDAG-TRUE16-NEXT: v_cndmask_b32_e64 v4, 0, 0x42800000, vcc_lo
; GFX1200-SDAG-TRUE16-NEXT: v_cndmask_b32_e64 v3, 0, 0xffffffc0, vcc_lo
+; GFX1200-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(TRANS32_DEP_1)
; GFX1200-SDAG-TRUE16-NEXT: v_add_f32_e32 v0, v0, v4
-; GFX1200-SDAG-TRUE16-NEXT: s_delay_alu instid0(TRANS32_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1200-SDAG-TRUE16-NEXT: v_ldexp_f32 v1, v1, v2
+; GFX1200-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1200-SDAG-TRUE16-NEXT: v_exp_f32_e32 v0, v0
-; GFX1200-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_2) | instid1(VALU_DEP_3)
; GFX1200-SDAG-TRUE16-NEXT: v_bfe_u32 v2, v1, 16, 1
; GFX1200-SDAG-TRUE16-NEXT: v_or_b32_e32 v4, 0x400000, v1
; GFX1200-SDAG-TRUE16-NEXT: v_cmp_u_f32_e32 vcc_lo, v1, v1
+; GFX1200-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(TRANS32_DEP_1)
; GFX1200-SDAG-TRUE16-NEXT: v_add3_u32 v2, v2, v1, 0x7fff
-; GFX1200-SDAG-TRUE16-NEXT: s_delay_alu instid0(TRANS32_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1200-SDAG-TRUE16-NEXT: v_ldexp_f32 v0, v0, v3
+; GFX1200-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX1200-SDAG-TRUE16-NEXT: v_bfe_u32 v3, v0, 16, 1
; GFX1200-SDAG-TRUE16-NEXT: v_or_b32_e32 v5, 0x400000, v0
-; GFX1200-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX1200-SDAG-TRUE16-NEXT: v_add3_u32 v3, v3, v0, 0x7fff
; GFX1200-SDAG-TRUE16-NEXT: s_wait_alu depctr_va_vcc(0)
; GFX1200-SDAG-TRUE16-NEXT: v_cndmask_b32_e32 v1, v2, v4, vcc_lo
@@ -976,41 +975,40 @@ define <2 x bfloat> @v_exp2_fabs_v2bf16(<2 x bfloat> %in) {
; GFX1200-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1200-SDAG-FAKE16-NEXT: v_and_b32_e32 v1, 0x7fff, v1
; GFX1200-SDAG-FAKE16-NEXT: v_lshlrev_b32_e32 v1, 16, v1
-; GFX1200-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
+; GFX1200-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_2) | instid1(VALU_DEP_1)
; GFX1200-SDAG-FAKE16-NEXT: v_cmp_gt_f32_e64 s0, 0xc2fc0000, v1
; GFX1200-SDAG-FAKE16-NEXT: s_wait_alu depctr_va_sdst(0)
; GFX1200-SDAG-FAKE16-NEXT: v_cndmask_b32_e64 v3, 0, 0x42800000, s0
-; GFX1200-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX1200-SDAG-FAKE16-NEXT: v_dual_add_f32 v1, v1, v3 :: v_dual_and_b32 v0, 0x7fff, v0
; GFX1200-SDAG-FAKE16-NEXT: v_cndmask_b32_e64 v3, 0, 0xffffffc0, s0
+; GFX1200-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(TRANS32_DEP_1)
; GFX1200-SDAG-FAKE16-NEXT: v_exp_f32_e32 v1, v1
-; GFX1200-SDAG-FAKE16-NEXT: s_delay_alu instid0(TRANS32_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1200-SDAG-FAKE16-NEXT: v_ldexp_f32 v1, v1, v3
+; GFX1200-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX1200-SDAG-FAKE16-NEXT: v_bfe_u32 v3, v1, 16, 1
; GFX1200-SDAG-FAKE16-NEXT: v_or_b32_e32 v5, 0x400000, v1
-; GFX1200-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX1200-SDAG-FAKE16-NEXT: v_add3_u32 v3, v3, v1, 0x7fff
; GFX1200-SDAG-FAKE16-NEXT: v_lshlrev_b32_e32 v0, 16, v0
+; GFX1200-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_2) | instid1(VALU_DEP_1)
; GFX1200-SDAG-FAKE16-NEXT: v_cmp_gt_f32_e32 vcc_lo, 0xc2fc0000, v0
; GFX1200-SDAG-FAKE16-NEXT: s_wait_alu depctr_va_vcc(0)
; GFX1200-SDAG-FAKE16-NEXT: v_cndmask_b32_e64 v2, 0, 0x42800000, vcc_lo
-; GFX1200-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX1200-SDAG-FAKE16-NEXT: v_add_f32_e32 v0, v0, v2
; GFX1200-SDAG-FAKE16-NEXT: v_cndmask_b32_e64 v2, 0, 0xffffffc0, vcc_lo
+; GFX1200-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(TRANS32_DEP_1)
; GFX1200-SDAG-FAKE16-NEXT: v_exp_f32_e32 v0, v0
-; GFX1200-SDAG-FAKE16-NEXT: s_delay_alu instid0(TRANS32_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1200-SDAG-FAKE16-NEXT: v_ldexp_f32 v0, v0, v2
+; GFX1200-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_2) | instid1(VALU_DEP_3)
; GFX1200-SDAG-FAKE16-NEXT: v_bfe_u32 v2, v0, 16, 1
; GFX1200-SDAG-FAKE16-NEXT: v_or_b32_e32 v4, 0x400000, v0
; GFX1200-SDAG-FAKE16-NEXT: v_cmp_u_f32_e32 vcc_lo, v0, v0
-; GFX1200-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX1200-SDAG-FAKE16-NEXT: v_add3_u32 v2, v2, v0, 0x7fff
; GFX1200-SDAG-FAKE16-NEXT: s_wait_alu depctr_va_vcc(0)
+; GFX1200-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_3) | instid1(VALU_DEP_1)
; GFX1200-SDAG-FAKE16-NEXT: v_cndmask_b32_e32 v0, v2, v4, vcc_lo
; GFX1200-SDAG-FAKE16-NEXT: v_cmp_u_f32_e32 vcc_lo, v1, v1
; GFX1200-SDAG-FAKE16-NEXT: s_wait_alu depctr_va_vcc(0)
; GFX1200-SDAG-FAKE16-NEXT: v_cndmask_b32_e32 v1, v3, v5, vcc_lo
-; GFX1200-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1200-SDAG-FAKE16-NEXT: v_perm_b32 v0, v1, v0, 0x7060302
; GFX1200-SDAG-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -1025,40 +1023,39 @@ define <2 x bfloat> @v_exp2_fabs_v2bf16(<2 x bfloat> %in) {
; GFX1200-GI-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1200-GI-TRUE16-NEXT: v_mov_b16_e32 v1.l, v0.h
; GFX1200-GI-TRUE16-NEXT: v_lshlrev_b32_e32 v1, 16, v1
-; GFX1200-GI-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
+; GFX1200-GI-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_2) | instid1(VALU_DEP_1)
; GFX1200-GI-TRUE16-NEXT: v_cmp_gt_f32_e64 s0, 0xc2fc0000, v1
; GFX1200-GI-TRUE16-NEXT: s_wait_alu depctr_va_sdst(0)
; GFX1200-GI-TRUE16-NEXT: v_cndmask_b32_e64 v3, 0, 0x42800000, s0
-; GFX1200-GI-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX1200-GI-TRUE16-NEXT: v_dual_add_f32 v1, v1, v3 :: v_dual_lshlrev_b32 v0, 16, v0
; GFX1200-GI-TRUE16-NEXT: v_cndmask_b32_e64 v3, 0, 0xffffffc0, s0
+; GFX1200-GI-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1200-GI-TRUE16-NEXT: v_exp_f32_e32 v1, v1
-; GFX1200-GI-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_2) | instid1(TRANS32_DEP_1)
; GFX1200-GI-TRUE16-NEXT: v_cmp_gt_f32_e32 vcc_lo, 0xc2fc0000, v0
; GFX1200-GI-TRUE16-NEXT: s_wait_alu depctr_va_vcc(0)
; GFX1200-GI-TRUE16-NEXT: v_cndmask_b32_e64 v2, 0, 0x42800000, vcc_lo
+; GFX1200-GI-TRUE16-NEXT: s_delay_alu instid0(TRANS32_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1200-GI-TRUE16-NEXT: v_ldexp_f32 v1, v1, v3
-; GFX1200-GI-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_3)
; GFX1200-GI-TRUE16-NEXT: v_add_f32_e32 v0, v0, v2
; GFX1200-GI-TRUE16-NEXT: v_cndmask_b32_e64 v2, 0, 0xffffffc0, vcc_lo
+; GFX1200-GI-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_1) | instid1(VALU_DEP_4)
; GFX1200-GI-TRUE16-NEXT: v_bfe_u32 v3, v1, 16, 1
; GFX1200-GI-TRUE16-NEXT: v_or_b32_e32 v5, 0x400000, v1
-; GFX1200-GI-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1200-GI-TRUE16-NEXT: v_exp_f32_e32 v0, v0
+; GFX1200-GI-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(TRANS32_DEP_1)
; GFX1200-GI-TRUE16-NEXT: v_add3_u32 v3, v3, v1, 0x7fff
-; GFX1200-GI-TRUE16-NEXT: s_delay_alu instid0(TRANS32_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1200-GI-TRUE16-NEXT: v_ldexp_f32 v0, v0, v2
+; GFX1200-GI-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_2) | instid1(VALU_DEP_3)
; GFX1200-GI-TRUE16-NEXT: v_bfe_u32 v2, v0, 16, 1
; GFX1200-GI-TRUE16-NEXT: v_or_b32_e32 v4, 0x400000, v0
; GFX1200-GI-TRUE16-NEXT: v_cmp_u_f32_e32 vcc_lo, 0, v0
-; GFX1200-GI-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX1200-GI-TRUE16-NEXT: v_add3_u32 v2, v2, v0, 0x7fff
; GFX1200-GI-TRUE16-NEXT: s_wait_alu depctr_va_vcc(0)
+; GFX1200-GI-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_3) | instid1(VALU_DEP_3)
; GFX1200-GI-TRUE16-NEXT: v_cndmask_b32_e32 v2, v2, v4, vcc_lo
; GFX1200-GI-TRUE16-NEXT: v_cmp_u_f32_e32 vcc_lo, 0, v1
; GFX1200-GI-TRUE16-NEXT: s_wait_alu depctr_va_vcc(0)
; GFX1200-GI-TRUE16-NEXT: v_cndmask_b32_e32 v0, v3, v5, vcc_lo
-; GFX1200-GI-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX1200-GI-TRUE16-NEXT: v_mov_b16_e32 v0.l, v2.h
; GFX1200-GI-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -1073,41 +1070,40 @@ define <2 x bfloat> @v_exp2_fabs_v2bf16(<2 x bfloat> %in) {
; GFX1200-GI-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1200-GI-FAKE16-NEXT: v_lshrrev_b32_e32 v1, 16, v0
; GFX1200-GI-FAKE16-NEXT: v_lshlrev_b32_e32 v1, 16, v1
-; GFX1200-GI-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
+; GFX1200-GI-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_2) | instid1(VALU_DEP_1)
; GFX1200-GI-FAKE16-NEXT: v_cmp_gt_f32_e64 s0, 0xc2fc0000, v1
; GFX1200-GI-FAKE16-NEXT: s_wait_alu depctr_va_sdst(0)
; GFX1200-GI-FAKE16-NEXT: v_cndmask_b32_e64 v3, 0, 0x42800000, s0
-; GFX1200-GI-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX1200-GI-FAKE16-NEXT: v_add_f32_e32 v1, v1, v3
; GFX1200-GI-FAKE16-NEXT: v_cndmask_b32_e64 v3, 0, 0xffffffc0, s0
+; GFX1200-GI-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(TRANS32_DEP_1)
; GFX1200-GI-FAKE16-NEXT: v_exp_f32_e32 v1, v1
-; GFX1200-GI-FAKE16-NEXT: s_delay_alu instid0(TRANS32_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1200-GI-FAKE16-NEXT: v_ldexp_f32 v1, v1, v3
+; GFX1200-GI-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX1200-GI-FAKE16-NEXT: v_bfe_u32 v3, v1, 16, 1
; GFX1200-GI-FAKE16-NEXT: v_or_b32_e32 v5, 0x400000, v1
-; GFX1200-GI-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX1200-GI-FAKE16-NEXT: v_add3_u32 v3, v3, v1, 0x7fff
; GFX1200-GI-FAKE16-NEXT: v_lshlrev_b32_e32 v0, 16, v0
+; GFX1200-GI-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_2) | instid1(VALU_DEP_1)
; GFX1200-GI-FAKE16-NEXT: v_cmp_gt_f32_e32 vcc_lo, 0xc2fc0000, v0
; GFX1200-GI-FAKE16-NEXT: s_wait_alu depctr_va_vcc(0)
; GFX1200-GI-FAKE16-NEXT: v_cndmask_b32_e64 v2, 0, 0x42800000, vcc_lo
-; GFX1200-GI-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX1200-GI-FAKE16-NEXT: v_add_f32_e32 v0, v0, v2
; GFX1200-GI-FAKE16-NEXT: v_cndmask_b32_e64 v2, 0, 0xffffffc0, vcc_lo
+; GFX1200-GI-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(TRANS32_DEP_1)
; GFX1200-GI-FAKE16-NEXT: v_exp_f32_e32 v0, v0
-; GFX1200-GI-FAKE16-NEXT: s_delay_alu instid0(TRANS32_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1200-GI-FAKE16-NEXT: v_ldexp_f32 v0, v0, v2
+; GFX1200-GI-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_2) | instid1(VALU_DEP_3)
; GFX1200-GI-FAKE16-NEXT: v_bfe_u32 v2, v0, 16, 1
; GFX1200-GI-FAKE16-NEXT: v_or_b32_e32 v4, 0x400000, v0
; GFX1200-GI-FAKE16-NEXT: v_cmp_u_f32_e32 vcc_lo, 0, v0
-; GFX1200-GI-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX1200-GI-FAKE16-NEXT: v_add3_u32 v2, v2, v0, 0x7fff
; GFX1200-GI-FAKE16-NEXT: s_wait_alu depctr_va_vcc(0)
+; GFX1200-GI-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_3) | instid1(VALU_DEP_1)
; GFX1200-GI-FAKE16-NEXT: v_cndmask_b32_e32 v0, v2, v4, vcc_lo
; GFX1200-GI-FAKE16-NEXT: v_cmp_u_f32_e32 vcc_lo, 0, v1
; GFX1200-GI-FAKE16-NEXT: s_wait_alu depctr_va_vcc(0)
; GFX1200-GI-FAKE16-NEXT: v_cndmask_b32_e32 v1, v3, v5, vcc_lo
-; GFX1200-GI-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1200-GI-FAKE16-NEXT: v_perm_b32 v0, v1, v0, 0x7060302
; GFX1200-GI-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -1176,33 +1172,32 @@ define <2 x bfloat> @v_exp2_fneg_fabs_v2bf16(<2 x bfloat> %in) {
; GFX1200-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1200-SDAG-TRUE16-NEXT: v_or_b16 v1.l, 0x8000, v1.l
; GFX1200-SDAG-TRUE16-NEXT: v_lshlrev_b32_e32 v1, 16, v1
-; GFX1200-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
+; GFX1200-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_2) | instid1(VALU_DEP_1)
; GFX1200-SDAG-TRUE16-NEXT: v_cmp_gt_f32_e64 s0, 0xc2fc0000, v1
; GFX1200-SDAG-TRUE16-NEXT: s_wait_alu depctr_va_sdst(0)
; GFX1200-SDAG-TRUE16-NEXT: v_cndmask_b32_e64 v2, 0, 0x42800000, s0
-; GFX1200-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1200-SDAG-TRUE16-NEXT: v_dual_add_f32 v1, v1, v2 :: v_dual_lshlrev_b32 v0, 16, v0
+; GFX1200-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_3)
; GFX1200-SDAG-TRUE16-NEXT: v_cmp_gt_f32_e32 vcc_lo, 0xc2fc0000, v0
; GFX1200-SDAG-TRUE16-NEXT: v_cndmask_b32_e64 v2, 0, 0xffffffc0, s0
-; GFX1200-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_3) | instid1(VALU_DEP_2)
; GFX1200-SDAG-TRUE16-NEXT: v_exp_f32_e32 v1, v1
; GFX1200-SDAG-TRUE16-NEXT: s_wait_alu depctr_va_vcc(0)
; GFX1200-SDAG-TRUE16-NEXT: v_cndmask_b32_e64 v4, 0, 0x42800000, vcc_lo
; GFX1200-SDAG-TRUE16-NEXT: v_cndmask_b32_e64 v3, 0, 0xffffffc0, vcc_lo
+; GFX1200-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(TRANS32_DEP_1)
; GFX1200-SDAG-TRUE16-NEXT: v_add_f32_e32 v0, v0, v4
-; GFX1200-SDAG-TRUE16-NEXT: s_delay_alu instid0(TRANS32_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1200-SDAG-TRUE16-NEXT: v_ldexp_f32 v1, v1, v2
+; GFX1200-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1200-SDAG-TRUE16-NEXT: v_exp_f32_e32 v0, v0
-; GFX1200-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_2) | instid1(VALU_DEP_3)
; GFX1200-SDAG-TRUE16-NEXT: v_bfe_u32 v2, v1, 16, 1
; GFX1200-SDAG-TRUE16-NEXT: v_or_b32_e32 v4, 0x400000, v1
; GFX1200-SDAG-TRUE16-NEXT: v_cmp_u_f32_e32 vcc_lo, v1, v1
+; GFX1200-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(TRANS32_DEP_1)
; GFX1200-SDAG-TRUE16-NEXT: v_add3_u32 v2, v2, v1, 0x7fff
-; GFX1200-SDAG-TRUE16-NEXT: s_delay_alu instid0(TRANS32_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1200-SDAG-TRUE16-NEXT: v_ldexp_f32 v0, v0, v3
+; GFX1200-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX1200-SDAG-TRUE16-NEXT: v_bfe_u32 v3, v0, 16, 1
; GFX1200-SDAG-TRUE16-NEXT: v_or_b32_e32 v5, 0x400000, v0
-; GFX1200-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX1200-SDAG-TRUE16-NEXT: v_add3_u32 v3, v3, v0, 0x7fff
; GFX1200-SDAG-TRUE16-NEXT: s_wait_alu depctr_va_vcc(0)
; GFX1200-SDAG-TRUE16-NEXT: v_cndmask_b32_e32 v1, v2, v4, vcc_lo
@@ -1228,40 +1223,39 @@ define <2 x bfloat> @v_exp2_fneg_fabs_v2bf16(<2 x bfloat> %in) {
; GFX1200-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1200-SDAG-FAKE16-NEXT: v_or_b32_e32 v1, 0x8000, v1
; GFX1200-SDAG-FAKE16-NEXT: v_lshlrev_b32_e32 v1, 16, v1
-; GFX1200-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
+; GFX1200-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_2) | instid1(VALU_DEP_1)
; GFX1200-SDAG-FAKE16-NEXT: v_cmp_gt_f32_e64 s0, 0xc2fc0000, v1
; GFX1200-SDAG-FAKE16-NEXT: s_wait_alu depctr_va_sdst(0)
; GFX1200-SDAG-FAKE16-NEXT: v_cndmask_b32_e64 v3, 0, 0x42800000, s0
-; GFX1200-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1200-SDAG-FAKE16-NEXT: v_dual_add_f32 v1, v1, v3 :: v_dual_lshlrev_b32 v0, 16, v0
+; GFX1200-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_3)
; GFX1200-SDAG-FAKE16-NEXT: v_cmp_gt_f32_e32 vcc_lo, 0xc2fc0000, v0
; GFX1200-SDAG-FAKE16-NEXT: v_cndmask_b32_e64 v3, 0, 0xffffffc0, s0
-; GFX1200-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_2) | instid1(VALU_DEP_1)
; GFX1200-SDAG-FAKE16-NEXT: v_exp_f32_e32 v1, v1
; GFX1200-SDAG-FAKE16-NEXT: s_wait_alu depctr_va_vcc(0)
; GFX1200-SDAG-FAKE16-NEXT: v_cndmask_b32_e64 v2, 0, 0x42800000, vcc_lo
+; GFX1200-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(TRANS32_DEP_1)
; GFX1200-SDAG-FAKE16-NEXT: v_add_f32_e32 v0, v0, v2
; GFX1200-SDAG-FAKE16-NEXT: v_cndmask_b32_e64 v2, 0, 0xffffffc0, vcc_lo
-; GFX1200-SDAG-FAKE16-NEXT: s_delay_alu instid0(TRANS32_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_3)
; GFX1200-SDAG-FAKE16-NEXT: v_ldexp_f32 v1, v1, v3
+; GFX1200-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1200-SDAG-FAKE16-NEXT: v_exp_f32_e32 v0, v0
-; GFX1200-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX1200-SDAG-FAKE16-NEXT: v_bfe_u32 v3, v1, 16, 1
; GFX1200-SDAG-FAKE16-NEXT: v_or_b32_e32 v5, 0x400000, v1
+; GFX1200-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(TRANS32_DEP_1)
; GFX1200-SDAG-FAKE16-NEXT: v_add3_u32 v3, v3, v1, 0x7fff
-; GFX1200-SDAG-FAKE16-NEXT: s_delay_alu instid0(TRANS32_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1200-SDAG-FAKE16-NEXT: v_ldexp_f32 v0, v0, v2
+; GFX1200-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_2) | instid1(VALU_DEP_3)
; GFX1200-SDAG-FAKE16-NEXT: v_bfe_u32 v2, v0, 16, 1
; GFX1200-SDAG-FAKE16-NEXT: v_or_b32_e32 v4, 0x400000, v0
; GFX1200-SDAG-FAKE16-NEXT: v_cmp_u_f32_e32 vcc_lo, v0, v0
-; GFX1200-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX1200-SDAG-FAKE16-NEXT: v_add3_u32 v2, v2, v0, 0x7fff
; GFX1200-SDAG-FAKE16-NEXT: s_wait_alu depctr_va_vcc(0)
+; GFX1200-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_3) | instid1(VALU_DEP_1)
; GFX1200-SDAG-FAKE16-NEXT: v_cndmask_b32_e32 v0, v2, v4, vcc_lo
; GFX1200-SDAG-FAKE16-NEXT: v_cmp_u_f32_e32 vcc_lo, v1, v1
; GFX1200-SDAG-FAKE16-NEXT: s_wait_alu depctr_va_vcc(0)
; GFX1200-SDAG-FAKE16-NEXT: v_cndmask_b32_e32 v1, v3, v5, vcc_lo
-; GFX1200-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1200-SDAG-FAKE16-NEXT: v_perm_b32 v0, v1, v0, 0x7060302
; GFX1200-SDAG-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -1276,40 +1270,39 @@ define <2 x bfloat> @v_exp2_fneg_fabs_v2bf16(<2 x bfloat> %in) {
; GFX1200-GI-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1200-GI-TRUE16-NEXT: v_mov_b16_e32 v1.l, v0.h
; GFX1200-GI-TRUE16-NEXT: v_lshlrev_b32_e32 v1, 16, v1
-; GFX1200-GI-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
+; GFX1200-GI-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_2) | instid1(VALU_DEP_1)
; GFX1200-GI-TRUE16-NEXT: v_cmp_gt_f32_e64 s0, 0xc2fc0000, v1
; GFX1200-GI-TRUE16-NEXT: s_wait_alu depctr_va_sdst(0)
; GFX1200-GI-TRUE16-NEXT: v_cndmask_b32_e64 v3, 0, 0x42800000, s0
-; GFX1200-GI-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1200-GI-TRUE16-NEXT: v_dual_add_f32 v1, v1, v3 :: v_dual_lshlrev_b32 v0, 16, v0
+; GFX1200-GI-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_3)
; GFX1200-GI-TRUE16-NEXT: v_cmp_gt_f32_e32 vcc_lo, 0xc2fc0000, v0
; GFX1200-GI-TRUE16-NEXT: v_cndmask_b32_e64 v3, 0, 0xffffffc0, s0
-; GFX1200-GI-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_2) | instid1(VALU_DEP_1)
; GFX1200-GI-TRUE16-NEXT: v_exp_f32_e32 v1, v1
; GFX1200-GI-TRUE16-NEXT: s_wait_alu depctr_va_vcc(0)
; GFX1200-GI-TRUE16-NEXT: v_cndmask_b32_e64 v2, 0, 0x42800000, vcc_lo
+; GFX1200-GI-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(TRANS32_DEP_1)
; GFX1200-GI-TRUE16-NEXT: v_add_f32_e32 v0, v0, v2
; GFX1200-GI-TRUE16-NEXT: v_cndmask_b32_e64 v2, 0, 0xffffffc0, vcc_lo
-; GFX1200-GI-TRUE16-NEXT: s_delay_alu instid0(TRANS32_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_3)
; GFX1200-GI-TRUE16-NEXT: v_ldexp_f32 v1, v1, v3
+; GFX1200-GI-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1200-GI-TRUE16-NEXT: v_exp_f32_e32 v0, v0
-; GFX1200-GI-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX1200-GI-TRUE16-NEXT: v_bfe_u32 v3, v1, 16, 1
; GFX1200-GI-TRUE16-NEXT: v_or_b32_e32 v5, 0x400000, v1
+; GFX1200-GI-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(TRANS32_DEP_1)
; GFX1200-GI-TRUE16-NEXT: v_add3_u32 v3, v3, v1, 0x7fff
-; GFX1200-GI-TRUE16-NEXT: s_delay_alu instid0(TRANS32_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1200-GI-TRUE16-NEXT: v_ldexp_f32 v0, v0, v2
+; GFX1200-GI-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_2) | instid1(VALU_DEP_3)
; GFX1200-GI-TRUE16-NEXT: v_bfe_u32 v2, v0, 16, 1
; GFX1200-GI-TRUE16-NEXT: v_or_b32_e32 v4, 0x400000, v0
; GFX1200-GI-TRUE16-NEXT: v_cmp_u_f32_e32 vcc_lo, 0, v0
-; GFX1200-GI-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX1200-GI-TRUE16-NEXT: v_add3_u32 v2, v2, v0, 0x7fff
; GFX1200-GI-TRUE16-NEXT: s_wait_alu depctr_va_vcc(0)
+; GFX1200-GI-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_3) | instid1(VALU_DEP_3)
; GFX1200-GI-TRUE16-NEXT: v_cndmask_b32_e32 v2, v2, v4, vcc_lo
; GFX1200-GI-TRUE16-NEXT: v_cmp_u_f32_e32 vcc_lo, 0, v1
; GFX1200-GI-TRUE16-NEXT: s_wait_alu depctr_va_vcc(0)
; GFX1200-GI-TRUE16-NEXT: v_cndmask_b32_e32 v0, v3, v5, vcc_lo
-; GFX1200-GI-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX1200-GI-TRUE16-NEXT: v_mov_b16_e32 v0.l, v2.h
; GFX1200-GI-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -1324,40 +1317,39 @@ define <2 x bfloat> @v_exp2_fneg_fabs_v2bf16(<2 x bfloat> %in) {
; GFX1200-GI-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1200-GI-FAKE16-NEXT: v_lshrrev_b32_e32 v1, 16, v0
; GFX1200-GI-FAKE16-NEXT: v_lshlrev_b32_e32 v1, 16, v1
-; GFX1200-GI-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
+; GFX1200-GI-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_2) | instid1(VALU_DEP_1)
; GFX1200-GI-FAKE16-NEXT: v_cmp_gt_f32_e64 s0, 0xc2fc0000, v1
; GFX1200-GI-FAKE16-NEXT: s_wait_alu depctr_va_sdst(0)
; GFX1200-GI-FAKE16-NEXT: v_cndmask_b32_e64 v3, 0, 0x42800000, s0
-; GFX1200-GI-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1200-GI-FAKE16-NEXT: v_dual_add_f32 v1, v1, v3 :: v_dual_lshlrev_b32 v0, 16, v0
+; GFX1200-GI-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_3)
; GFX1200-GI-FAKE16-NEXT: v_cmp_gt_f32_e32 vcc_lo, 0xc2fc0000, v0
; GFX1200-GI-FAKE16-NEXT: v_cndmask_b32_e64 v3, 0, 0xffffffc0, s0
-; GFX1200-GI-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_2) | instid1(VALU_DEP_1)
; GFX1200-GI-FAKE16-NEXT: v_exp_f32_e32 v1, v1
; GFX1200-GI-FAKE16-NEXT: s_wait_alu depctr_va_vcc(0)
; GFX1200-GI-FAKE16-NEXT: v_cndmask_b32_e64 v2, 0, 0x42800000, vcc_lo
+; GFX1200-GI-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(TRANS32_DEP_1)
; GFX1200-GI-FAKE16-NEXT: v_add_f32_e32 v0, v0, v2
; GFX1200-GI-FAKE16-NEXT: v_cndmask_b32_e64 v2, 0, 0xffffffc0, vcc_lo
-; GFX1200-GI-FAKE16-NEXT: s_delay_alu instid0(TRANS32_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_3)
; GFX1200-GI-FAKE16-NEXT: v_ldexp_f32 v1, v1, v3
+; GFX1200-GI-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1200-GI-FAKE16-NEXT: v_exp_f32_e32 v0, v0
-; GFX1200-GI-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX1200-GI-FAKE16-NEXT: v_bfe_u32 v3, v1, 16, 1
; GFX1200-GI-FAKE16-NEXT: v_or_b32_e32 v5, 0x400000, v1
+; GFX1200-GI-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(TRANS32_DEP_1)
; GFX1200-GI-FAKE16-NEXT: v_add3_u32 v3, v3, v1, 0x7fff
-; GFX1200-GI-FAKE16-NEXT: s_delay_alu instid0(TRANS32_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1200-GI-FAKE16-NEXT: v_ldexp_f32 v0, v0, v2
+; GFX1200-GI-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_2) | instid1(VALU_DEP_3)
; GFX1200-GI-FAKE16-NEXT: v_bfe_u32 v2, v0, 16, 1
; GFX1200-GI-FAKE16-NEXT: v_or_b32_e32 v4, 0x400000, v0
; GFX1200-GI-FAKE16-NEXT: v_cmp_u_f32_e32 vcc_lo, 0, v0
-; GFX1200-GI-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX1200-GI-FAKE16-NEXT: v_add3_u32 v2, v2, v0, 0x7fff
; GFX1200-GI-FAKE16-NEXT: s_wait_alu depctr_va_vcc(0)
+; GFX1200-GI-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_3) | instid1(VALU_DEP_1)
; GFX1200-GI-FAKE16-NEXT: v_cndmask_b32_e32 v0, v2, v4, vcc_lo
; GFX1200-GI-FAKE16-NEXT: v_cmp_u_f32_e32 vcc_lo, 0, v1
; GFX1200-GI-FAKE16-NEXT: s_wait_alu depctr_va_vcc(0)
; GFX1200-GI-FAKE16-NEXT: v_cndmask_b32_e32 v1, v3, v5, vcc_lo
-; GFX1200-GI-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1200-GI-FAKE16-NEXT: v_perm_b32 v0, v1, v0, 0x7060302
; GFX1200-GI-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -1427,33 +1419,32 @@ define <2 x bfloat> @v_exp2_fneg_v2bf16(<2 x bfloat> %in) {
; GFX1200-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1200-SDAG-TRUE16-NEXT: v_xor_b16 v1.l, 0x8000, v1.l
; GFX1200-SDAG-TRUE16-NEXT: v_lshlrev_b32_e32 v1, 16, v1
-; GFX1200-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
+; GFX1200-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_2) | instid1(VALU_DEP_1)
; GFX1200-SDAG-TRUE16-NEXT: v_cmp_gt_f32_e64 s0, 0xc2fc0000, v1
; GFX1200-SDAG-TRUE16-NEXT: s_wait_alu depctr_va_sdst(0)
; GFX1200-SDAG-TRUE16-NEXT: v_cndmask_b32_e64 v2, 0, 0x42800000, s0
-; GFX1200-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1200-SDAG-TRUE16-NEXT: v_dual_add_f32 v1, v1, v2 :: v_dual_lshlrev_b32 v0, 16, v0
+; GFX1200-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_3)
; GFX1200-SDAG-TRUE16-NEXT: v_cmp_gt_f32_e32 vcc_lo, 0xc2fc0000, v0
; GFX1200-SDAG-TRUE16-NEXT: v_cndmask_b32_e64 v2, 0, 0xffffffc0, s0
-; GFX1200-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_3) | instid1(VALU_DEP_2)
; GFX1200-SDAG-TRUE16-NEXT: v_exp_f32_e32 v1, v1
; GFX1200-SDAG-TRUE16-NEXT: s_wait_alu depctr_va_vcc(0)
; GFX1200-SDAG-TRUE16-NEXT: v_cndmask_b32_e64 v4, 0, 0x42800000, vcc_lo
; GFX1200-SDAG-TRUE16-NEXT: v_cndmask_b32_e64 v3, 0, 0xffffffc0, vcc_lo
+; GFX1200-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(TRANS32_DEP_1)
; GFX1200-SDAG-TRUE16-NEXT: v_add_f32_e32 v0, v0, v4
-; GFX1200-SDAG-TRUE16-NEXT: s_delay_alu instid0(TRANS32_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1200-SDAG-TRUE16-NEXT: v_ldexp_f32 v1, v1, v2
+; GFX1200-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1200-SDAG-TRUE16-NEXT: v_exp_f32_e32 v0, v0
-; GFX1200-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_2) | instid1(VALU_DEP_3)
; GFX1200-SDAG-TRUE16-NEXT: v_bfe_u32 v2, v1, 16, 1
; GFX1200-SDAG-TRUE16-NEXT: v_or_b32_e32 v4, 0x400000, v1
; GFX1200-SDAG-TRUE16-NEXT: v_cmp_u_f32_e32 vcc_lo, v1, v1
+; GFX1200-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(TRANS32_DEP_1)
; GFX1200-SDAG-TRUE16-NEXT: v_add3_u32 v2, v2, v1, 0x7fff
-; GFX1200-SDAG-TRUE16-NEXT: s_delay_alu instid0(TRANS32_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1200-SDAG-TRUE16-NEXT: v_ldexp_f32 v0, v0, v3
+; GFX1200-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX1200-SDAG-TRUE16-NEXT: v_bfe_u32 v3, v0, 16, 1
; GFX1200-SDAG-TRUE16-NEXT: v_or_b32_e32 v5, 0x400000, v0
-; GFX1200-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX1200-SDAG-TRUE16-NEXT: v_add3_u32 v3, v3, v0, 0x7fff
; GFX1200-SDAG-TRUE16-NEXT: s_wait_alu depctr_va_vcc(0)
; GFX1200-SDAG-TRUE16-NEXT: v_cndmask_b32_e32 v1, v2, v4, vcc_lo
@@ -1479,40 +1470,39 @@ define <2 x bfloat> @v_exp2_fneg_v2bf16(<2 x bfloat> %in) {
; GFX1200-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1200-SDAG-FAKE16-NEXT: v_xor_b32_e32 v1, 0x8000, v1
; GFX1200-SDAG-FAKE16-NEXT: v_lshlrev_b32_e32 v1, 16, v1
-; GFX1200-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
+; GFX1200-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_2) | instid1(VALU_DEP_1)
; GFX1200-SDAG-FAKE16-NEXT: v_cmp_gt_f32_e64 s0, 0xc2fc0000, v1
; GFX1200-SDAG-FAKE16-NEXT: s_wait_alu depctr_va_sdst(0)
; GFX1200-SDAG-FAKE16-NEXT: v_cndmask_b32_e64 v3, 0, 0x42800000, s0
-; GFX1200-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1200-SDAG-FAKE16-NEXT: v_dual_add_f32 v1, v1, v3 :: v_dual_lshlrev_b32 v0, 16, v0
+; GFX1200-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_3)
; GFX1200-SDAG-FAKE16-NEXT: v_cmp_gt_f32_e32 vcc_lo, 0xc2fc0000, v0
; GFX1200-SDAG-FAKE16-NEXT: v_cndmask_b32_e64 v3, 0, 0xffffffc0, s0
-; GFX1200-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_2) | instid1(VALU_DEP_1)
; GFX1200-SDAG-FAKE16-NEXT: v_exp_f32_e32 v1, v1
; GFX1200-SDAG-FAKE16-NEXT: s_wait_alu depctr_va_vcc(0)
; GFX1200-SDAG-FAKE16-NEXT: v_cndmask_b32_e64 v2, 0, 0x42800000, vcc_lo
+; GFX1200-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(TRANS32_DEP_1)
; GFX1200-SDAG-FAKE16-NEXT: v_add_f32_e32 v0, v0, v2
; GFX1200-SDAG-FAKE16-NEXT: v_cndmask_b32_e64 v2, 0, 0xffffffc0, vcc_lo
-; GFX1200-SDAG-FAKE16-NEXT: s_delay_alu instid0(TRANS32_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_3)
; GFX1200-SDAG-FAKE16-NEXT: v_ldexp_f32 v1, v1, v3
+; GFX1200-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1200-SDAG-FAKE16-NEXT: v_exp_f32_e32 v0, v0
-; GFX1200-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX1200-SDAG-FAKE16-NEXT: v_bfe_u32 v3, v1, 16, 1
; GFX1200-SDAG-FAKE16-NEXT: v_or_b32_e32 v5, 0x400000, v1
+; GFX1200-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(TRANS32_DEP_1)
; GFX1200-SDAG-FAKE16-NEXT: v_add3_u32 v3, v3, v1, 0x7fff
-; GFX1200-SDAG-FAKE16-NEXT: s_delay_alu instid0(TRANS32_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1200-SDAG-FAKE16-NEXT: v_ldexp_f32 v0, v0, v2
+; GFX1200-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_2) | instid1(VALU_DEP_3)
; GFX1200-SDAG-FAKE16-NEXT: v_bfe_u32 v2, v0, 16, 1
; GFX1200-SDAG-FAKE16-NEXT: v_or_b32_e32 v4, 0x400000, v0
; GFX1200-SDAG-FAKE16-NEXT: v_cmp_u_f32_e32 vcc_lo, v0, v0
-; GFX1200-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX1200-SDAG-FAKE16-NEXT: v_add3_u32 v2, v2, v0, 0x7fff
; GFX1200-SDAG-FAKE16-NEXT: s_wait_alu depctr_va_vcc(0)
+; GFX1200-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_3) | instid1(VALU_DEP_1)
; GFX1200-SDAG-FAKE16-NEXT: v_cndmask_b32_e32 v0, v2, v4, vcc_lo
; GFX1200-SDAG-FAKE16-NEXT: v_cmp_u_f32_e32 vcc_lo, v1, v1
; GFX1200-SDAG-FAKE16-NEXT: s_wait_alu depctr_va_vcc(0)
; GFX1200-SDAG-FAKE16-NEXT: v_cndmask_b32_e32 v1, v3, v5, vcc_lo
-; GFX1200-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1200-SDAG-FAKE16-NEXT: v_perm_b32 v0, v1, v0, 0x7060302
; GFX1200-SDAG-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -1527,40 +1517,39 @@ define <2 x bfloat> @v_exp2_fneg_v2bf16(<2 x bfloat> %in) {
; GFX1200-GI-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1200-GI-TRUE16-NEXT: v_mov_b16_e32 v1.l, v0.h
; GFX1200-GI-TRUE16-NEXT: v_lshlrev_b32_e32 v1, 16, v1
-; GFX1200-GI-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
+; GFX1200-GI-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_2) | instid1(VALU_DEP_1)
; GFX1200-GI-TRUE16-NEXT: v_cmp_gt_f32_e64 s0, 0xc2fc0000, v1
; GFX1200-GI-TRUE16-NEXT: s_wait_alu depctr_va_sdst(0)
; GFX1200-GI-TRUE16-NEXT: v_cndmask_b32_e64 v3, 0, 0x42800000, s0
-; GFX1200-GI-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1200-GI-TRUE16-NEXT: v_dual_add_f32 v1, v1, v3 :: v_dual_lshlrev_b32 v0, 16, v0
+; GFX1200-GI-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_3)
; GFX1200-GI-TRUE16-NEXT: v_cmp_gt_f32_e32 vcc_lo, 0xc2fc0000, v0
; GFX1200-GI-TRUE16-NEXT: v_cndmask_b32_e64 v3, 0, 0xffffffc0, s0
-; GFX1200-GI-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_2) | instid1(VALU_DEP_1)
; GFX1200-GI-TRUE16-NEXT: v_exp_f32_e32 v1, v1
; GFX1200-GI-TRUE16-NEXT: s_wait_alu depctr_va_vcc(0)
; GFX1200-GI-TRUE16-NEXT: v_cndmask_b32_e64 v2, 0, 0x42800000, vcc_lo
+; GFX1200-GI-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(TRANS32_DEP_1)
; GFX1200-GI-TRUE16-NEXT: v_add_f32_e32 v0, v0, v2
; GFX1200-GI-TRUE16-NEXT: v_cndmask_b32_e64 v2, 0, 0xffffffc0, vcc_lo
-; GFX1200-GI-TRUE16-NEXT: s_delay_alu instid0(TRANS32_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_3)
; GFX1200-GI-TRUE16-NEXT: v_ldexp_f32 v1, v1, v3
+; GFX1200-GI-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1200-GI-TRUE16-NEXT: v_exp_f32_e32 v0, v0
-; GFX1200-GI-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX1200-GI-TRUE16-NEXT: v_bfe_u32 v3, v1, 16, 1
; GFX1200-GI-TRUE16-NEXT: v_or_b32_e32 v5, 0x400000, v1
+; GFX1200-GI-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(TRANS32_DEP_1)
; GFX1200-GI-TRUE16-NEXT: v_add3_u32 v3, v3, v1, 0x7fff
-; GFX1200-GI-TRUE16-NEXT: s_delay_alu instid0(TRANS32_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1200-GI-TRUE16-NEXT: v_ldexp_f32 v0, v0, v2
+; GFX1200-GI-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_2) | instid1(VALU_DEP_3)
; GFX1200-GI-TRUE16-NEXT: v_bfe_u32 v2, v0, 16, 1
; GFX1200-GI-TRUE16-NEXT: v_or_b32_e32 v4, 0x400000, v0
; GFX1200-GI-TRUE16-NEXT: v_cmp_u_f32_e32 vcc_lo, 0, v0
-; GFX1200-GI-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX1200-GI-TRUE16-NEXT: v_add3_u32 v2, v2, v0, 0x7fff
; GFX1200-GI-TRUE16-NEXT: s_wait_alu depctr_va_vcc(0)
+; GFX1200-GI-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_3) | instid1(VALU_DEP_3)
; GFX1200-GI-TRUE16-NEXT: v_cndmask_b32_e32 v2, v2, v4, vcc_lo
; GFX1200-GI-TRUE16-NEXT: v_cmp_u_f32_e32 vcc_lo, 0, v1
; GFX1200-GI-TRUE16-NEXT: s_wait_alu depctr_va_vcc(0)
; GFX1200-GI-TRUE16-NEXT: v_cndmask_b32_e32 v0, v3, v5, vcc_lo
-; GFX1200-GI-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX1200-GI-TRUE16-NEXT: v_mov_b16_e32 v0.l, v2.h
; GFX1200-GI-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -1575,40 +1564,39 @@ define <2 x bfloat> @v_exp2_fneg_v2bf16(<2 x bfloat> %in) {
; GFX1200-GI-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1200-GI-FAKE16-NEXT: v_lshrrev_b32_e32 v1, 16, v0
; GFX1200-GI-FAKE16-NEXT: v_lshlrev_b32_e32 v1, 16, v1
-; GFX1200-GI-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
+; GFX1200-GI-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_2) | instid1(VALU_DEP_1)
; GFX1200-GI-FAKE16-NEXT: v_cmp_gt_f32_e64 s0, 0xc2fc0000, v1
; GFX1200-GI-FAKE16-NEXT: s_wait_alu depctr_va_sdst(0)
; GFX1200-GI-FAKE16-NEXT: v_cndmask_b32_e64 v3, 0, 0x42800000, s0
-; GFX1200-GI-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1200-GI-FAKE16-NEXT: v_dual_add_f32 v1, v1, v3 :: v_dual_lshlrev_b32 v0, 16, v0
+; GFX1200-GI-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_3)
; GFX1200-GI-FAKE16-NEXT: v_cmp_gt_f32_e32 vcc_lo, 0xc2fc0000, v0
; GFX1200-GI-FAKE16-NEXT: v_cndmask_b32_e64 v3, 0, 0xffffffc0, s0
-; GFX1200-GI-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_2) | instid1(VALU_DEP_1)
; GFX1200-GI-FAKE16-NEXT: v_exp_f32_e32 v1, v1
; GFX1200-GI-FAKE16-NEXT: s_wait_alu depctr_va_vcc(0)
; GFX1200-GI-FAKE16-NEXT: v_cndmask_b32_e64 v2, 0, 0x42800000, vcc_lo
+; GFX1200-GI-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(TRANS32_DEP_1)
; GFX1200-GI-FAKE16-NEXT: v_add_f32_e32 v0, v0, v2
; GFX1200-GI-FAKE16-NEXT: v_cndmask_b32_e64 v2, 0, 0xffffffc0, vcc_lo
-; GFX1200-GI-FAKE16-NEXT: s_delay_alu instid0(TRANS32_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_3)
; GFX1200-GI-FAKE16-NEXT: v_ldexp_f32 v1, v1, v3
+; GFX1200-GI-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1200-GI-FAKE16-NEXT: v_exp_f32_e32 v0, v0
-; GFX1200-GI-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX1200-GI-FAKE16-NEXT: v_bfe_u32 v3, v1, 16, 1
; GFX1200-GI-FAKE16-NEXT: v_or_b32_e32 v5, 0x400000, v1
+; GFX1200-GI-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(TRANS32_DEP_1)
; GFX1200-GI-FAKE16-NEXT: v_add3_u32 v3, v3, v1, 0x7fff
-; GFX1200-GI-FAKE16-NEXT: s_delay_alu instid0(TRANS32_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1200-GI-FAKE16-NEXT: v_ldexp_f32 v0, v0, v2
+; GFX1200-GI-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_2) | instid1(VALU_DEP_3)
; GFX1200-GI-FAKE16-NEXT: v_bfe_u32 v2, v0, 16, 1
; GFX1200-GI-FAKE16-NEXT: v_or_b32_e32 v4, 0x400000, v0
; GFX1200-GI-FAKE16-NEXT: v_cmp_u_f32_e32 vcc_lo, 0, v0
-; GFX1200-GI-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX1200-GI-FAKE16-NEXT: v_add3_u32 v2, v2, v0, 0x7fff
; GFX1200-GI-FAKE16-NEXT: s_wait_alu depctr_va_vcc(0)
+; GFX1200-GI-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_3) | instid1(VALU_DEP_1)
; GFX1200-GI-FAKE16-NEXT: v_cndmask_b32_e32 v0, v2, v4, vcc_lo
; GFX1200-GI-FAKE16-NEXT: v_cmp_u_f32_e32 vcc_lo, 0, v1
; GFX1200-GI-FAKE16-NEXT: s_wait_alu depctr_va_vcc(0)
; GFX1200-GI-FAKE16-NEXT: v_cndmask_b32_e32 v1, v3, v5, vcc_lo
-; GFX1200-GI-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1200-GI-FAKE16-NEXT: v_perm_b32 v0, v1, v0, 0x7060302
; GFX1200-GI-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -1674,42 +1662,42 @@ define <2 x bfloat> @v_exp2_v2bf16_fast(<2 x bfloat> %in) {
; GFX1200-SDAG-TRUE16-NEXT: s_wait_kmcnt 0x0
; GFX1200-SDAG-TRUE16-NEXT: v_and_b32_e32 v1, 0xffff0000, v0
; GFX1200-SDAG-TRUE16-NEXT: v_lshlrev_b32_e32 v0, 16, v0
-; GFX1200-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
+; GFX1200-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_2) | instid1(VALU_DEP_1)
; GFX1200-SDAG-TRUE16-NEXT: v_cmp_gt_f32_e64 s0, 0xc2fc0000, v0
; GFX1200-SDAG-TRUE16-NEXT: s_wait_alu depctr_va_sdst(0)
; GFX1200-SDAG-TRUE16-NEXT: v_cndmask_b32_e64 v3, 0, 0x42800000, s0
-; GFX1200-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_2) | instid1(VALU_DEP_3)
; GFX1200-SDAG-TRUE16-NEXT: v_add_f32_e32 v0, v0, v3
; GFX1200-SDAG-TRUE16-NEXT: v_cmp_gt_f32_e32 vcc_lo, 0xc2fc0000, v1
; GFX1200-SDAG-TRUE16-NEXT: v_cndmask_b32_e64 v3, 0, 0xffffffc0, s0
+; GFX1200-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_2) | instid1(VALU_DEP_1)
; GFX1200-SDAG-TRUE16-NEXT: v_exp_f32_e32 v0, v0
; GFX1200-SDAG-TRUE16-NEXT: s_wait_alu depctr_va_vcc(0)
; GFX1200-SDAG-TRUE16-NEXT: v_cndmask_b32_e64 v2, 0, 0x42800000, vcc_lo
-; GFX1200-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(TRANS32_DEP_1)
; GFX1200-SDAG-TRUE16-NEXT: v_add_f32_e32 v1, v1, v2
; GFX1200-SDAG-TRUE16-NEXT: v_cndmask_b32_e64 v2, 0, 0xffffffc0, vcc_lo
+; GFX1200-SDAG-TRUE16-NEXT: s_delay_alu instid0(TRANS32_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_3)
; GFX1200-SDAG-TRUE16-NEXT: v_ldexp_f32 v0, v0, v3
-; GFX1200-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1200-SDAG-TRUE16-NEXT: v_exp_f32_e32 v1, v1
+; GFX1200-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX1200-SDAG-TRUE16-NEXT: v_bfe_u32 v3, v0, 16, 1
; GFX1200-SDAG-TRUE16-NEXT: v_or_b32_e32 v5, 0x400000, v0
-; GFX1200-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(TRANS32_DEP_1)
; GFX1200-SDAG-TRUE16-NEXT: v_add3_u32 v3, v3, v0, 0x7fff
+; GFX1200-SDAG-TRUE16-NEXT: s_delay_alu instid0(TRANS32_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1200-SDAG-TRUE16-NEXT: v_ldexp_f32 v1, v1, v2
-; GFX1200-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_2) | instid1(VALU_DEP_3)
; GFX1200-SDAG-TRUE16-NEXT: v_bfe_u32 v2, v1, 16, 1
; GFX1200-SDAG-TRUE16-NEXT: v_or_b32_e32 v4, 0x400000, v1
; GFX1200-SDAG-TRUE16-NEXT: v_cmp_u_f32_e32 vcc_lo, v1, v1
+; GFX1200-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX1200-SDAG-TRUE16-NEXT: v_add3_u32 v2, v2, v1, 0x7fff
; GFX1200-SDAG-TRUE16-NEXT: s_wait_alu depctr_va_vcc(0)
-; GFX1200-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_3) | instid1(VALU_DEP_3)
; GFX1200-SDAG-TRUE16-NEXT: v_cndmask_b32_e32 v1, v2, v4, vcc_lo
; GFX1200-SDAG-TRUE16-NEXT: v_cmp_u_f32_e32 vcc_lo, v0, v0
; GFX1200-SDAG-TRUE16-NEXT: s_wait_alu depctr_va_vcc(0)
; GFX1200-SDAG-TRUE16-NEXT: v_cndmask_b32_e32 v0, v3, v5, vcc_lo
+; GFX1200-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1200-SDAG-TRUE16-NEXT: v_lshrrev_b32_e32 v1, 16, v1
-; GFX1200-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1200-SDAG-TRUE16-NEXT: v_lshrrev_b32_e32 v0, 16, v0
+; GFX1200-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX1200-SDAG-TRUE16-NEXT: v_mov_b16_e32 v0.h, v1.l
; GFX1200-SDAG-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -1722,39 +1710,39 @@ define <2 x bfloat> @v_exp2_v2bf16_fast(<2 x bfloat> %in) {
; GFX1200-SDAG-FAKE16-NEXT: s_wait_kmcnt 0x0
; GFX1200-SDAG-FAKE16-NEXT: v_lshlrev_b32_e32 v1, 16, v0
; GFX1200-SDAG-FAKE16-NEXT: v_and_b32_e32 v0, 0xffff0000, v0
-; GFX1200-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
+; GFX1200-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_2) | instid1(VALU_DEP_1)
; GFX1200-SDAG-FAKE16-NEXT: v_cmp_gt_f32_e64 s0, 0xc2fc0000, v0
; GFX1200-SDAG-FAKE16-NEXT: s_wait_alu depctr_va_sdst(0)
; GFX1200-SDAG-FAKE16-NEXT: v_cndmask_b32_e64 v3, 0, 0x42800000, s0
-; GFX1200-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_2) | instid1(VALU_DEP_3)
; GFX1200-SDAG-FAKE16-NEXT: v_add_f32_e32 v0, v0, v3
; GFX1200-SDAG-FAKE16-NEXT: v_cmp_gt_f32_e32 vcc_lo, 0xc2fc0000, v1
; GFX1200-SDAG-FAKE16-NEXT: v_cndmask_b32_e64 v3, 0, 0xffffffc0, s0
+; GFX1200-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_2) | instid1(VALU_DEP_1)
; GFX1200-SDAG-FAKE16-NEXT: v_exp_f32_e32 v0, v0
; GFX1200-SDAG-FAKE16-NEXT: s_wait_alu depctr_va_vcc(0)
; GFX1200-SDAG-FAKE16-NEXT: v_cndmask_b32_e64 v2, 0, 0x42800000, vcc_lo
-; GFX1200-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(TRANS32_DEP_1)
; GFX1200-SDAG-FAKE16-NEXT: v_add_f32_e32 v1, v1, v2
; GFX1200-SDAG-FAKE16-NEXT: v_cndmask_b32_e64 v2, 0, 0xffffffc0, vcc_lo
+; GFX1200-SDAG-FAKE16-NEXT: s_delay_alu instid0(TRANS32_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_3)
; GFX1200-SDAG-FAKE16-NEXT: v_ldexp_f32 v0, v0, v3
-; GFX1200-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1200-SDAG-FAKE16-NEXT: v_exp_f32_e32 v1, v1
+; GFX1200-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX1200-SDAG-FAKE16-NEXT: v_bfe_u32 v3, v0, 16, 1
; GFX1200-SDAG-FAKE16-NEXT: v_or_b32_e32 v5, 0x400000, v0
-; GFX1200-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(TRANS32_DEP_1)
; GFX1200-SDAG-FAKE16-NEXT: v_add3_u32 v3, v3, v0, 0x7fff
+; GFX1200-SDAG-FAKE16-NEXT: s_delay_alu instid0(TRANS32_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1200-SDAG-FAKE16-NEXT: v_ldexp_f32 v1, v1, v2
-; GFX1200-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_2) | instid1(VALU_DEP_3)
; GFX1200-SDAG-FAKE16-NEXT: v_bfe_u32 v2, v1, 16, 1
; GFX1200-SDAG-FAKE16-NEXT: v_or_b32_e32 v4, 0x400000, v1
; GFX1200-SDAG-FAKE16-NEXT: v_cmp_u_f32_e32 vcc_lo, v1, v1
+; GFX1200-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX1200-SDAG-FAKE16-NEXT: v_add3_u32 v2, v2, v1, 0x7fff
; GFX1200-SDAG-FAKE16-NEXT: s_wait_alu depctr_va_vcc(0)
-; GFX1200-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_3) | instid1(VALU_DEP_1)
; GFX1200-SDAG-FAKE16-NEXT: v_cndmask_b32_e32 v1, v2, v4, vcc_lo
; GFX1200-SDAG-FAKE16-NEXT: v_cmp_u_f32_e32 vcc_lo, v0, v0
; GFX1200-SDAG-FAKE16-NEXT: s_wait_alu depctr_va_vcc(0)
; GFX1200-SDAG-FAKE16-NEXT: v_cndmask_b32_e32 v0, v3, v5, vcc_lo
+; GFX1200-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1200-SDAG-FAKE16-NEXT: v_perm_b32 v0, v0, v1, 0x7060302
; GFX1200-SDAG-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -1770,29 +1758,28 @@ define <2 x bfloat> @v_exp2_v2bf16_fast(<2 x bfloat> %in) {
; GFX1200-GI-TRUE16-NEXT: v_lshlrev_b32_e32 v1, 16, v1
; GFX1200-GI-TRUE16-NEXT: v_cmp_gt_f32_e64 s0, 0xc2fc0000, v1
; GFX1200-GI-TRUE16-NEXT: s_wait_alu depctr_va_sdst(0)
-; GFX1200-GI-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1200-GI-TRUE16-NEXT: v_cndmask_b32_e64 v3, 0, 0x42800000, s0
+; GFX1200-GI-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1200-GI-TRUE16-NEXT: v_dual_add_f32 v1, v1, v3 :: v_dual_lshlrev_b32 v0, 16, v0
-; GFX1200-GI-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_3)
; GFX1200-GI-TRUE16-NEXT: v_cmp_gt_f32_e32 vcc_lo, 0xc2fc0000, v0
; GFX1200-GI-TRUE16-NEXT: v_cndmask_b32_e64 v3, 0, 0xffffffc0, s0
+; GFX1200-GI-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_2) | instid1(VALU_DEP_1)
; GFX1200-GI-TRUE16-NEXT: v_exp_f32_e32 v1, v1
; GFX1200-GI-TRUE16-NEXT: s_wait_alu depctr_va_vcc(0)
; GFX1200-GI-TRUE16-NEXT: v_cndmask_b32_e64 v2, 0, 0x42800000, vcc_lo
-; GFX1200-GI-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(TRANS32_DEP_1)
; GFX1200-GI-TRUE16-NEXT: v_add_f32_e32 v0, v0, v2
; GFX1200-GI-TRUE16-NEXT: v_cndmask_b32_e64 v2, 0, 0xffffffc0, vcc_lo
+; GFX1200-GI-TRUE16-NEXT: s_delay_alu instid0(TRANS32_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_3)
; GFX1200-GI-TRUE16-NEXT: v_ldexp_f32 v1, v1, v3
-; GFX1200-GI-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1200-GI-TRUE16-NEXT: v_exp_f32_e32 v0, v0
+; GFX1200-GI-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(TRANS32_DEP_1)
; GFX1200-GI-TRUE16-NEXT: v_bfe_u32 v3, v1, 16, 1
-; GFX1200-GI-TRUE16-NEXT: s_delay_alu instid0(TRANS32_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1200-GI-TRUE16-NEXT: v_ldexp_f32 v0, v0, v2
+; GFX1200-GI-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1200-GI-TRUE16-NEXT: v_bfe_u32 v2, v0, 16, 1
-; GFX1200-GI-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_4)
; GFX1200-GI-TRUE16-NEXT: v_add3_u32 v2, v2, v0, 0x7fff
+; GFX1200-GI-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1200-GI-TRUE16-NEXT: v_add3_u32 v0, v3, v1, 0x7fff
-; GFX1200-GI-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX1200-GI-TRUE16-NEXT: v_mov_b16_e32 v0.l, v2.h
; GFX1200-GI-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -1808,29 +1795,28 @@ define <2 x bfloat> @v_exp2_v2bf16_fast(<2 x bfloat> %in) {
; GFX1200-GI-FAKE16-NEXT: v_lshlrev_b32_e32 v1, 16, v1
; GFX1200-GI-FAKE16-NEXT: v_cmp_gt_f32_e64 s0, 0xc2fc0000, v1
; GFX1200-GI-FAKE16-NEXT: s_wait_alu depctr_va_sdst(0)
-; GFX1200-GI-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1200-GI-FAKE16-NEXT: v_cndmask_b32_e64 v3, 0, 0x42800000, s0
+; GFX1200-GI-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1200-GI-FAKE16-NEXT: v_dual_add_f32 v1, v1, v3 :: v_dual_lshlrev_b32 v0, 16, v0
-; GFX1200-GI-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_3)
; GFX1200-GI-FAKE16-NEXT: v_cmp_gt_f32_e32 vcc_lo, 0xc2fc0000, v0
; GFX1200-GI-FAKE16-NEXT: v_cndmask_b32_e64 v3, 0, 0xffffffc0, s0
+; GFX1200-GI-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_2) | instid1(VALU_DEP_1)
; GFX1200-GI-FAKE16-NEXT: v_exp_f32_e32 v1, v1
; GFX1200-GI-FAKE16-NEXT: s_wait_alu depctr_va_vcc(0)
; GFX1200-GI-FAKE16-NEXT: v_cndmask_b32_e64 v2, 0, 0x42800000, vcc_lo
-; GFX1200-GI-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(TRANS32_DEP_1)
; GFX1200-GI-FAKE16-NEXT: v_add_f32_e32 v0, v0, v2
; GFX1200-GI-FAKE16-NEXT: v_cndmask_b32_e64 v2, 0, 0xffffffc0, vcc_lo
+; GFX1200-GI-FAKE16-NEXT: s_delay_alu instid0(TRANS32_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_3)
; GFX1200-GI-FAKE16-NEXT: v_ldexp_f32 v1, v1, v3
-; GFX1200-GI-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1200-GI-FAKE16-NEXT: v_exp_f32_e32 v0, v0
+; GFX1200-GI-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1200-GI-FAKE16-NEXT: v_bfe_u32 v3, v1, 16, 1
-; GFX1200-GI-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(TRANS32_DEP_1)
; GFX1200-GI-FAKE16-NEXT: v_add3_u32 v1, v3, v1, 0x7fff
+; GFX1200-GI-FAKE16-NEXT: s_delay_alu instid0(TRANS32_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1200-GI-FAKE16-NEXT: v_ldexp_f32 v0, v0, v2
-; GFX1200-GI-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1200-GI-FAKE16-NEXT: v_bfe_u32 v2, v0, 16, 1
+; GFX1200-GI-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1200-GI-FAKE16-NEXT: v_add3_u32 v0, v2, v0, 0x7fff
-; GFX1200-GI-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1200-GI-FAKE16-NEXT: v_perm_b32 v0, v1, v0, 0x7060302
; GFX1200-GI-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
diff --git a/llvm/test/CodeGen/AMDGPU/llvm.fptrunc.round.ll b/llvm/test/CodeGen/AMDGPU/llvm.fptrunc.round.ll
index 2ab76c99095f82..d52c2bb6c6f263 100644
--- a/llvm/test/CodeGen/AMDGPU/llvm.fptrunc.round.ll
+++ b/llvm/test/CodeGen/AMDGPU/llvm.fptrunc.round.ll
@@ -46,12 +46,14 @@ define amdgpu_gs half @v_fptrunc_round_f32_to_f16_upward(float %a) {
; GFX11-SDAG-LABEL: v_fptrunc_round_f32_to_f16_upward:
; GFX11-SDAG: ; %bb.0:
; GFX11-SDAG-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_MODE, 2, 1), 1
+; GFX11-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-SDAG-NEXT: v_cvt_f16_f32_e64 v0.l, v0
; GFX11-SDAG-NEXT: ; return to shader part epilog
;
; GFX11-GISEL-LABEL: v_fptrunc_round_f32_to_f16_upward:
; GFX11-GISEL: ; %bb.0:
; GFX11-GISEL-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_MODE, 2, 1), 1
+; GFX11-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-GISEL-NEXT: v_cvt_f16_f32_e64 v0.l, v0
; GFX11-GISEL-NEXT: ; return to shader part epilog
;
@@ -64,6 +66,7 @@ define amdgpu_gs half @v_fptrunc_round_f32_to_f16_upward(float %a) {
; GFX12-LABEL: v_fptrunc_round_f32_to_f16_upward:
; GFX12: ; %bb.0:
; GFX12-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 2, 1), 1
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: v_cvt_f16_f32_e64 v0.l, v0
; GFX12-NEXT: ; return to shader part epilog
%res = call half @llvm.fptrunc.round.f16.f32(float %a, metadata !"round.upward")
@@ -80,12 +83,14 @@ define amdgpu_gs half @v_fptrunc_round_f32_to_f16_downward(float %a) {
; GFX11-SDAG-LABEL: v_fptrunc_round_f32_to_f16_downward:
; GFX11-SDAG: ; %bb.0:
; GFX11-SDAG-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_MODE, 3, 1), 1
+; GFX11-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-SDAG-NEXT: v_cvt_f16_f32_e64 v0.l, v0
; GFX11-SDAG-NEXT: ; return to shader part epilog
;
; GFX11-GISEL-LABEL: v_fptrunc_round_f32_to_f16_downward:
; GFX11-GISEL: ; %bb.0:
; GFX11-GISEL-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_MODE, 3, 1), 1
+; GFX11-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-GISEL-NEXT: v_cvt_f16_f32_e64 v0.l, v0
; GFX11-GISEL-NEXT: ; return to shader part epilog
;
@@ -98,6 +103,7 @@ define amdgpu_gs half @v_fptrunc_round_f32_to_f16_downward(float %a) {
; GFX12-LABEL: v_fptrunc_round_f32_to_f16_downward:
; GFX12: ; %bb.0:
; GFX12-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 3, 1), 1
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: v_cvt_f16_f32_e64 v0.l, v0
; GFX12-NEXT: ; return to shader part epilog
%res = call half @llvm.fptrunc.round.f16.f32(float %a, metadata !"round.downward")
@@ -269,13 +275,15 @@ define amdgpu_gs void @v_fptrunc_round_f32_to_f16_upward_multiple_calls(float %a
; GFX11-SDAG-LABEL: v_fptrunc_round_f32_to_f16_upward_multiple_calls:
; GFX11-SDAG: ; %bb.0:
; GFX11-SDAG-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_MODE, 2, 1), 1
+; GFX11-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
; GFX11-SDAG-NEXT: v_cvt_f16_f32_e64 v0.l, v0
; GFX11-SDAG-NEXT: v_cvt_f16_f32_e64 v0.h, v1
; GFX11-SDAG-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_MODE, 2, 2), 2
; GFX11-SDAG-NEXT: v_cvt_f16_f32_e64 v1.l, v1
; GFX11-SDAG-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_MODE, 3, 1), 0
-; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_2) | instid1(SALU_CYCLE_1)
; GFX11-SDAG-NEXT: v_add_f16_e32 v0.l, v0.l, v0.h
+; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-SDAG-NEXT: v_add_f16_e32 v0.l, v1.l, v0.l
; GFX11-SDAG-NEXT: global_store_b16 v[2:3], v0, off
; GFX11-SDAG-NEXT: s_endpgm
@@ -283,13 +291,15 @@ define amdgpu_gs void @v_fptrunc_round_f32_to_f16_upward_multiple_calls(float %a
; GFX11-GISEL-LABEL: v_fptrunc_round_f32_to_f16_upward_multiple_calls:
; GFX11-GISEL: ; %bb.0:
; GFX11-GISEL-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_MODE, 2, 1), 1
+; GFX11-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
; GFX11-GISEL-NEXT: v_cvt_f16_f32_e64 v0.l, v0
; GFX11-GISEL-NEXT: v_cvt_f16_f32_e64 v0.h, v1
; GFX11-GISEL-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_MODE, 2, 2), 2
; GFX11-GISEL-NEXT: v_cvt_f16_f32_e64 v1.l, v1
; GFX11-GISEL-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_MODE, 3, 1), 0
-; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instid1(SALU_CYCLE_1)
; GFX11-GISEL-NEXT: v_add_f16_e32 v0.l, v0.l, v0.h
+; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-GISEL-NEXT: v_add_f16_e32 v0.l, v1.l, v0.l
; GFX11-GISEL-NEXT: global_store_b16 v[2:3], v0, off
; GFX11-GISEL-NEXT: s_endpgm
@@ -310,13 +320,15 @@ define amdgpu_gs void @v_fptrunc_round_f32_to_f16_upward_multiple_calls(float %a
; GFX12-LABEL: v_fptrunc_round_f32_to_f16_upward_multiple_calls:
; GFX12: ; %bb.0:
; GFX12-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 2, 1), 1
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: v_cvt_f16_f32_e64 v0.l, v0
; GFX12-NEXT: v_cvt_f16_f32_e64 v0.h, v1
; GFX12-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 2, 2), 2
; GFX12-NEXT: v_cvt_f16_f32_e64 v1.l, v1
; GFX12-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 3, 1), 0
-; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_2) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: v_add_f16_e32 v0.l, v0.l, v0.h
+; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-NEXT: v_add_f16_e32 v0.l, v1.l, v0.l
; GFX12-NEXT: global_store_b16 v[2:3], v0, off
; GFX12-NEXT: s_endpgm
@@ -346,13 +358,15 @@ define amdgpu_gs void @v_fptrunc_round_f32_to_f16_downward_multiple_calls(float
; GFX11-SDAG-LABEL: v_fptrunc_round_f32_to_f16_downward_multiple_calls:
; GFX11-SDAG: ; %bb.0:
; GFX11-SDAG-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_MODE, 2, 1), 1
+; GFX11-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX11-SDAG-NEXT: v_cvt_f16_f32_e64 v4.l, v0
; GFX11-SDAG-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_MODE, 2, 2), 2
; GFX11-SDAG-NEXT: v_cvt_f16_f32_e64 v0.l, v0
; GFX11-SDAG-NEXT: v_cvt_f16_f32_e64 v0.h, v1
; GFX11-SDAG-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_MODE, 3, 1), 0
-; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_2) | instid1(SALU_CYCLE_1)
; GFX11-SDAG-NEXT: v_add_f16_e32 v0.l, v4.l, v0.l
+; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-SDAG-NEXT: v_add_f16_e32 v0.l, v0.h, v0.l
; GFX11-SDAG-NEXT: global_store_b16 v[2:3], v0, off
; GFX11-SDAG-NEXT: s_endpgm
@@ -360,13 +374,15 @@ define amdgpu_gs void @v_fptrunc_round_f32_to_f16_downward_multiple_calls(float
; GFX11-GISEL-LABEL: v_fptrunc_round_f32_to_f16_downward_multiple_calls:
; GFX11-GISEL: ; %bb.0:
; GFX11-GISEL-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_MODE, 2, 1), 1
+; GFX11-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX11-GISEL-NEXT: v_cvt_f16_f32_e64 v4.l, v0
; GFX11-GISEL-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_MODE, 2, 2), 2
; GFX11-GISEL-NEXT: v_cvt_f16_f32_e64 v0.l, v0
; GFX11-GISEL-NEXT: v_cvt_f16_f32_e64 v0.h, v1
; GFX11-GISEL-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_MODE, 3, 1), 0
-; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instid1(SALU_CYCLE_1)
; GFX11-GISEL-NEXT: v_add_f16_e32 v0.l, v4.l, v0.l
+; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-GISEL-NEXT: v_add_f16_e32 v0.l, v0.h, v0.l
; GFX11-GISEL-NEXT: global_store_b16 v[2:3], v0, off
; GFX11-GISEL-NEXT: s_endpgm
@@ -387,13 +403,15 @@ define amdgpu_gs void @v_fptrunc_round_f32_to_f16_downward_multiple_calls(float
; GFX12-LABEL: v_fptrunc_round_f32_to_f16_downward_multiple_calls:
; GFX12: ; %bb.0:
; GFX12-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 2, 1), 1
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: v_cvt_f16_f32_e64 v4.l, v0
; GFX12-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 2, 2), 2
; GFX12-NEXT: v_cvt_f16_f32_e64 v0.l, v0
; GFX12-NEXT: v_cvt_f16_f32_e64 v0.h, v1
; GFX12-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 3, 1), 0
-; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_2) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: v_add_f16_e32 v0.l, v4.l, v0.l
+; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-NEXT: v_add_f16_e32 v0.l, v0.h, v0.l
; GFX12-NEXT: global_store_b16 v[2:3], v0, off
; GFX12-NEXT: s_endpgm
@@ -424,10 +442,12 @@ define amdgpu_gs void @v_fptrunc_round_f32_to_f16_towardzero_multiple_calls(floa
; GFX11-SDAG-NEXT: v_cvt_pk_rtz_f16_f32_e64 v4, v0, s0
; GFX11-SDAG-NEXT: v_cvt_pk_rtz_f16_f32_e64 v5, v1, s0
; GFX11-SDAG-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_MODE, 2, 1), 1
+; GFX11-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-SDAG-NEXT: v_cvt_f16_f32_e64 v0.l, v1
; GFX11-SDAG-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_MODE, 2, 2), 0
-; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_2) | instid1(SALU_CYCLE_1)
; GFX11-SDAG-NEXT: v_add_f16_e32 v0.h, v4.l, v5.l
+; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-SDAG-NEXT: v_add_f16_e32 v0.l, v0.l, v0.h
; GFX11-SDAG-NEXT: global_store_b16 v[2:3], v0, off
; GFX11-SDAG-NEXT: s_endpgm
@@ -437,10 +457,12 @@ define amdgpu_gs void @v_fptrunc_round_f32_to_f16_towardzero_multiple_calls(floa
; GFX11-GISEL-NEXT: v_cvt_pk_rtz_f16_f32_e32 v4, v0, v0
; GFX11-GISEL-NEXT: v_cvt_pk_rtz_f16_f32_e32 v5, v1, v0
; GFX11-GISEL-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_MODE, 2, 1), 1
+; GFX11-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-GISEL-NEXT: v_cvt_f16_f32_e64 v0.l, v1
; GFX11-GISEL-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_MODE, 2, 2), 0
-; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instid1(SALU_CYCLE_1)
; GFX11-GISEL-NEXT: v_add_f16_e32 v0.h, v4.l, v5.l
+; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-GISEL-NEXT: v_add_f16_e32 v0.l, v0.l, v0.h
; GFX11-GISEL-NEXT: global_store_b16 v[2:3], v0, off
; GFX11-GISEL-NEXT: s_endpgm
@@ -462,10 +484,12 @@ define amdgpu_gs void @v_fptrunc_round_f32_to_f16_towardzero_multiple_calls(floa
; GFX12-SDAG-NEXT: v_cvt_pk_rtz_f16_f32_e64 v4, v0, s0
; GFX12-SDAG-NEXT: v_cvt_pk_rtz_f16_f32_e64 v5, v1, s0
; GFX12-SDAG-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 2, 1), 1
+; GFX12-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-SDAG-NEXT: v_cvt_f16_f32_e64 v0.l, v1
; GFX12-SDAG-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 2, 2), 0
-; GFX12-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX12-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_2) | instid1(SALU_CYCLE_1)
; GFX12-SDAG-NEXT: v_add_f16_e32 v0.h, v4.l, v5.l
+; GFX12-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-SDAG-NEXT: v_add_f16_e32 v0.l, v0.l, v0.h
; GFX12-SDAG-NEXT: global_store_b16 v[2:3], v0, off
; GFX12-SDAG-NEXT: s_endpgm
@@ -475,10 +499,12 @@ define amdgpu_gs void @v_fptrunc_round_f32_to_f16_towardzero_multiple_calls(floa
; GFX12-GISEL-NEXT: v_cvt_pk_rtz_f16_f32_e32 v4, v0, v0
; GFX12-GISEL-NEXT: v_cvt_pk_rtz_f16_f32_e32 v5, v1, v0
; GFX12-GISEL-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 2, 1), 1
+; GFX12-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-GISEL-NEXT: v_cvt_f16_f32_e64 v0.l, v1
; GFX12-GISEL-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 2, 2), 0
-; GFX12-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX12-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instid1(SALU_CYCLE_1)
; GFX12-GISEL-NEXT: v_add_f16_e32 v0.h, v4.l, v5.l
+; GFX12-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-GISEL-NEXT: v_add_f16_e32 v0.l, v0.l, v0.h
; GFX12-GISEL-NEXT: global_store_b16 v[2:3], v0, off
; GFX12-GISEL-NEXT: s_endpgm
@@ -504,17 +530,18 @@ define amdgpu_gs i32 @s_fptrunc_round_f32_to_f16_upward(float inreg %a, ptr addr
; GFX11-SDAG-LABEL: s_fptrunc_round_f32_to_f16_upward:
; GFX11-SDAG: ; %bb.0:
; GFX11-SDAG-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_MODE, 2, 1), 1
+; GFX11-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-SDAG-NEXT: v_cvt_f16_f32_e64 v0.l, s0
-; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-SDAG-NEXT: v_and_b32_e32 v0, 0xffff, v0
+; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-SDAG-NEXT: v_readfirstlane_b32 s0, v0
; GFX11-SDAG-NEXT: ; return to shader part epilog
;
; GFX11-GISEL-LABEL: s_fptrunc_round_f32_to_f16_upward:
; GFX11-GISEL: ; %bb.0:
; GFX11-GISEL-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_MODE, 2, 1), 1
+; GFX11-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-GISEL-NEXT: v_cvt_f16_f32_e64 v0.l, s0
-; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-GISEL-NEXT: v_readfirstlane_b32 s0, v0
; GFX11-GISEL-NEXT: s_and_b32 s0, 0xffff, s0
; GFX11-GISEL-NEXT: ; return to shader part epilog
@@ -531,8 +558,8 @@ define amdgpu_gs i32 @s_fptrunc_round_f32_to_f16_upward(float inreg %a, ptr addr
; GFX12-LABEL: s_fptrunc_round_f32_to_f16_upward:
; GFX12: ; %bb.0:
; GFX12-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 2, 1), 1
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_3)
; GFX12-NEXT: s_cvt_f16_f32 s0, s0
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_3)
; GFX12-NEXT: s_and_b32 s0, 0xffff, s0
; GFX12-NEXT: ; return to shader part epilog
%res = call half @llvm.fptrunc.round.f16.f32(float %a, metadata !"round.upward")
@@ -554,17 +581,18 @@ define amdgpu_gs i32 @s_fptrunc_round_f32_to_f16_downward(float inreg %a, ptr ad
; GFX11-SDAG-LABEL: s_fptrunc_round_f32_to_f16_downward:
; GFX11-SDAG: ; %bb.0:
; GFX11-SDAG-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_MODE, 3, 1), 1
+; GFX11-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-SDAG-NEXT: v_cvt_f16_f32_e64 v0.l, s0
-; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-SDAG-NEXT: v_and_b32_e32 v0, 0xffff, v0
+; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-SDAG-NEXT: v_readfirstlane_b32 s0, v0
; GFX11-SDAG-NEXT: ; return to shader part epilog
;
; GFX11-GISEL-LABEL: s_fptrunc_round_f32_to_f16_downward:
; GFX11-GISEL: ; %bb.0:
; GFX11-GISEL-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_MODE, 3, 1), 1
+; GFX11-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-GISEL-NEXT: v_cvt_f16_f32_e64 v0.l, s0
-; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-GISEL-NEXT: v_readfirstlane_b32 s0, v0
; GFX11-GISEL-NEXT: s_and_b32 s0, 0xffff, s0
; GFX11-GISEL-NEXT: ; return to shader part epilog
@@ -581,8 +609,8 @@ define amdgpu_gs i32 @s_fptrunc_round_f32_to_f16_downward(float inreg %a, ptr ad
; GFX12-LABEL: s_fptrunc_round_f32_to_f16_downward:
; GFX12: ; %bb.0:
; GFX12-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 3, 1), 1
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_3)
; GFX12-NEXT: s_cvt_f16_f32 s0, s0
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_3)
; GFX12-NEXT: s_and_b32 s0, 0xffff, s0
; GFX12-NEXT: ; return to shader part epilog
%res = call half @llvm.fptrunc.round.f16.f32(float %a, metadata !"round.downward")
@@ -653,13 +681,15 @@ define amdgpu_gs void @s_fptrunc_round_f32_to_f16_upward_multiple_calls(float in
; GFX11-SDAG-LABEL: s_fptrunc_round_f32_to_f16_upward_multiple_calls:
; GFX11-SDAG: ; %bb.0:
; GFX11-SDAG-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_MODE, 2, 1), 1
+; GFX11-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
; GFX11-SDAG-NEXT: v_cvt_f16_f32_e64 v2.l, s0
; GFX11-SDAG-NEXT: v_cvt_f16_f32_e64 v2.h, s1
; GFX11-SDAG-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_MODE, 2, 2), 2
; GFX11-SDAG-NEXT: v_cvt_f16_f32_e64 v3.l, s1
; GFX11-SDAG-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_MODE, 3, 1), 0
-; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_2) | instid1(SALU_CYCLE_1)
; GFX11-SDAG-NEXT: v_add_f16_e32 v2.l, v2.l, v2.h
+; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-SDAG-NEXT: v_add_f16_e32 v2.l, v3.l, v2.l
; GFX11-SDAG-NEXT: global_store_b16 v[0:1], v2, off
; GFX11-SDAG-NEXT: s_endpgm
@@ -667,13 +697,15 @@ define amdgpu_gs void @s_fptrunc_round_f32_to_f16_upward_multiple_calls(float in
; GFX11-GISEL-LABEL: s_fptrunc_round_f32_to_f16_upward_multiple_calls:
; GFX11-GISEL: ; %bb.0:
; GFX11-GISEL-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_MODE, 2, 1), 1
+; GFX11-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
; GFX11-GISEL-NEXT: v_cvt_f16_f32_e64 v2.l, s0
; GFX11-GISEL-NEXT: v_cvt_f16_f32_e64 v2.h, s1
; GFX11-GISEL-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_MODE, 2, 2), 2
; GFX11-GISEL-NEXT: v_cvt_f16_f32_e64 v3.l, s1
; GFX11-GISEL-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_MODE, 3, 1), 0
-; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instid1(SALU_CYCLE_1)
; GFX11-GISEL-NEXT: v_add_f16_e32 v2.l, v2.l, v2.h
+; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-GISEL-NEXT: v_add_f16_e32 v2.l, v3.l, v2.l
; GFX11-GISEL-NEXT: global_store_b16 v[0:1], v2, off
; GFX11-GISEL-NEXT: s_endpgm
@@ -696,14 +728,16 @@ define amdgpu_gs void @s_fptrunc_round_f32_to_f16_upward_multiple_calls(float in
; GFX12-LABEL: s_fptrunc_round_f32_to_f16_upward_multiple_calls:
; GFX12: ; %bb.0:
; GFX12-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 2, 1), 1
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cvt_f16_f32 s0, s0
; GFX12-NEXT: s_cvt_f16_f32 s2, s1
; GFX12-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 2, 2), 2
; GFX12-NEXT: s_cvt_f16_f32 s1, s1
; GFX12-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 3, 1), 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_3)
; GFX12-NEXT: s_add_f16 s0, s0, s2
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_3) | instskip(NEXT) | instid1(SALU_CYCLE_3)
; GFX12-NEXT: s_add_f16 s0, s1, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_3)
; GFX12-NEXT: v_mov_b16_e32 v2.l, s0
; GFX12-NEXT: global_store_b16 v[0:1], v2, off
; GFX12-NEXT: s_endpgm
@@ -728,15 +762,16 @@ define amdgpu_gs <2 x half> @v_fptrunc_round_v2f32_to_v2f16_upward(<2 x float> %
; GFX11-SDAG-LABEL: v_fptrunc_round_v2f32_to_v2f16_upward:
; GFX11-SDAG: ; %bb.0:
; GFX11-SDAG-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_MODE, 2, 1), 1
+; GFX11-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-SDAG-NEXT: v_cvt_f16_f32_e64 v1.h, v1
; GFX11-SDAG-NEXT: v_cvt_f16_f32_e64 v1.l, v0
-; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-SDAG-NEXT: v_mov_b32_e32 v0, v1
; GFX11-SDAG-NEXT: ; return to shader part epilog
;
; GFX11-GISEL-LABEL: v_fptrunc_round_v2f32_to_v2f16_upward:
; GFX11-GISEL: ; %bb.0:
; GFX11-GISEL-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_MODE, 2, 1), 1
+; GFX11-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-GISEL-NEXT: v_cvt_f16_f32_e64 v0.l, v0
; GFX11-GISEL-NEXT: v_cvt_f16_f32_e64 v0.h, v1
; GFX11-GISEL-NEXT: ; return to shader part epilog
@@ -752,15 +787,16 @@ define amdgpu_gs <2 x half> @v_fptrunc_round_v2f32_to_v2f16_upward(<2 x float> %
; GFX12-SDAG-LABEL: v_fptrunc_round_v2f32_to_v2f16_upward:
; GFX12-SDAG: ; %bb.0:
; GFX12-SDAG-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 2, 1), 1
+; GFX12-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX12-SDAG-NEXT: v_cvt_f16_f32_e64 v1.h, v1
; GFX12-SDAG-NEXT: v_cvt_f16_f32_e64 v1.l, v0
-; GFX12-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-SDAG-NEXT: v_mov_b32_e32 v0, v1
; GFX12-SDAG-NEXT: ; return to shader part epilog
;
; GFX12-GISEL-LABEL: v_fptrunc_round_v2f32_to_v2f16_upward:
; GFX12-GISEL: ; %bb.0:
; GFX12-GISEL-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 2, 1), 1
+; GFX12-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-GISEL-NEXT: v_cvt_f16_f32_e64 v0.l, v0
; GFX12-GISEL-NEXT: v_cvt_f16_f32_e64 v0.h, v1
; GFX12-GISEL-NEXT: ; return to shader part epilog
@@ -780,15 +816,16 @@ define amdgpu_gs <2 x half> @v_fptrunc_round_v2f32_to_v2f16_downward(<2 x float>
; GFX11-SDAG-LABEL: v_fptrunc_round_v2f32_to_v2f16_downward:
; GFX11-SDAG: ; %bb.0:
; GFX11-SDAG-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_MODE, 3, 1), 1
+; GFX11-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-SDAG-NEXT: v_cvt_f16_f32_e64 v1.h, v1
; GFX11-SDAG-NEXT: v_cvt_f16_f32_e64 v1.l, v0
-; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-SDAG-NEXT: v_mov_b32_e32 v0, v1
; GFX11-SDAG-NEXT: ; return to shader part epilog
;
; GFX11-GISEL-LABEL: v_fptrunc_round_v2f32_to_v2f16_downward:
; GFX11-GISEL: ; %bb.0:
; GFX11-GISEL-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_MODE, 3, 1), 1
+; GFX11-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-GISEL-NEXT: v_cvt_f16_f32_e64 v0.l, v0
; GFX11-GISEL-NEXT: v_cvt_f16_f32_e64 v0.h, v1
; GFX11-GISEL-NEXT: ; return to shader part epilog
@@ -804,15 +841,16 @@ define amdgpu_gs <2 x half> @v_fptrunc_round_v2f32_to_v2f16_downward(<2 x float>
; GFX12-SDAG-LABEL: v_fptrunc_round_v2f32_to_v2f16_downward:
; GFX12-SDAG: ; %bb.0:
; GFX12-SDAG-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 3, 1), 1
+; GFX12-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX12-SDAG-NEXT: v_cvt_f16_f32_e64 v1.h, v1
; GFX12-SDAG-NEXT: v_cvt_f16_f32_e64 v1.l, v0
-; GFX12-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-SDAG-NEXT: v_mov_b32_e32 v0, v1
; GFX12-SDAG-NEXT: ; return to shader part epilog
;
; GFX12-GISEL-LABEL: v_fptrunc_round_v2f32_to_v2f16_downward:
; GFX12-GISEL: ; %bb.0:
; GFX12-GISEL-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 3, 1), 1
+; GFX12-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-GISEL-NEXT: v_cvt_f16_f32_e64 v0.l, v0
; GFX12-GISEL-NEXT: v_cvt_f16_f32_e64 v0.h, v1
; GFX12-GISEL-NEXT: ; return to shader part epilog
@@ -843,6 +881,7 @@ define amdgpu_gs void @v_fptrunc_round_v2f32_to_v2f16_upward_multiple_calls(<2 x
; GFX11-SDAG-LABEL: v_fptrunc_round_v2f32_to_v2f16_upward_multiple_calls:
; GFX11-SDAG: ; %bb.0:
; GFX11-SDAG-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_MODE, 2, 1), 1
+; GFX11-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_4) | instid1(SALU_CYCLE_1)
; GFX11-SDAG-NEXT: v_cvt_f16_f32_e64 v1.h, v1
; GFX11-SDAG-NEXT: v_cvt_f16_f32_e64 v1.l, v0
; GFX11-SDAG-NEXT: v_cvt_f16_f32_e64 v0.h, v3
@@ -851,8 +890,8 @@ define amdgpu_gs void @v_fptrunc_round_v2f32_to_v2f16_upward_multiple_calls(<2 x
; GFX11-SDAG-NEXT: v_cvt_f16_f32_e64 v3.h, v3
; GFX11-SDAG-NEXT: v_cvt_f16_f32_e64 v3.l, v2
; GFX11-SDAG-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_MODE, 3, 1), 0
+; GFX11-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-SDAG-NEXT: v_pk_add_f16 v0, v1, v0
-; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-SDAG-NEXT: v_pk_add_f16 v0, v3, v0
; GFX11-SDAG-NEXT: global_store_b32 v[4:5], v0, off
; GFX11-SDAG-NEXT: s_endpgm
@@ -860,6 +899,7 @@ define amdgpu_gs void @v_fptrunc_round_v2f32_to_v2f16_upward_multiple_calls(<2 x
; GFX11-GISEL-LABEL: v_fptrunc_round_v2f32_to_v2f16_upward_multiple_calls:
; GFX11-GISEL: ; %bb.0:
; GFX11-GISEL-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_MODE, 2, 1), 1
+; GFX11-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_4) | instid1(SALU_CYCLE_1)
; GFX11-GISEL-NEXT: v_cvt_f16_f32_e64 v0.l, v0
; GFX11-GISEL-NEXT: v_cvt_f16_f32_e64 v0.h, v1
; GFX11-GISEL-NEXT: v_cvt_f16_f32_e64 v1.l, v2
@@ -868,8 +908,8 @@ define amdgpu_gs void @v_fptrunc_round_v2f32_to_v2f16_upward_multiple_calls(<2 x
; GFX11-GISEL-NEXT: v_cvt_f16_f32_e64 v2.l, v2
; GFX11-GISEL-NEXT: v_cvt_f16_f32_e64 v2.h, v3
; GFX11-GISEL-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_MODE, 3, 1), 0
+; GFX11-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-GISEL-NEXT: v_pk_add_f16 v0, v0, v1
-; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-GISEL-NEXT: v_pk_add_f16 v0, v2, v0
; GFX11-GISEL-NEXT: global_store_b32 v[4:5], v0, off
; GFX11-GISEL-NEXT: s_endpgm
@@ -896,6 +936,7 @@ define amdgpu_gs void @v_fptrunc_round_v2f32_to_v2f16_upward_multiple_calls(<2 x
; GFX12-SDAG-LABEL: v_fptrunc_round_v2f32_to_v2f16_upward_multiple_calls:
; GFX12-SDAG: ; %bb.0:
; GFX12-SDAG-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 2, 1), 1
+; GFX12-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_4) | instid1(SALU_CYCLE_1)
; GFX12-SDAG-NEXT: v_cvt_f16_f32_e64 v1.h, v1
; GFX12-SDAG-NEXT: v_cvt_f16_f32_e64 v1.l, v0
; GFX12-SDAG-NEXT: v_cvt_f16_f32_e64 v0.h, v3
@@ -904,8 +945,8 @@ define amdgpu_gs void @v_fptrunc_round_v2f32_to_v2f16_upward_multiple_calls(<2 x
; GFX12-SDAG-NEXT: v_cvt_f16_f32_e64 v3.h, v3
; GFX12-SDAG-NEXT: v_cvt_f16_f32_e64 v3.l, v2
; GFX12-SDAG-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 3, 1), 0
+; GFX12-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX12-SDAG-NEXT: v_pk_add_f16 v0, v1, v0
-; GFX12-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-SDAG-NEXT: v_pk_add_f16 v0, v3, v0
; GFX12-SDAG-NEXT: global_store_b32 v[4:5], v0, off
; GFX12-SDAG-NEXT: s_endpgm
@@ -913,6 +954,7 @@ define amdgpu_gs void @v_fptrunc_round_v2f32_to_v2f16_upward_multiple_calls(<2 x
; GFX12-GISEL-LABEL: v_fptrunc_round_v2f32_to_v2f16_upward_multiple_calls:
; GFX12-GISEL: ; %bb.0:
; GFX12-GISEL-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 2, 1), 1
+; GFX12-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_4) | instid1(SALU_CYCLE_1)
; GFX12-GISEL-NEXT: v_cvt_f16_f32_e64 v0.l, v0
; GFX12-GISEL-NEXT: v_cvt_f16_f32_e64 v0.h, v1
; GFX12-GISEL-NEXT: v_cvt_f16_f32_e64 v1.l, v2
@@ -921,8 +963,8 @@ define amdgpu_gs void @v_fptrunc_round_v2f32_to_v2f16_upward_multiple_calls(<2 x
; GFX12-GISEL-NEXT: v_cvt_f16_f32_e64 v2.l, v2
; GFX12-GISEL-NEXT: v_cvt_f16_f32_e64 v2.h, v3
; GFX12-GISEL-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 3, 1), 0
+; GFX12-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX12-GISEL-NEXT: v_pk_add_f16 v0, v0, v1
-; GFX12-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-GISEL-NEXT: v_pk_add_f16 v0, v2, v0
; GFX12-GISEL-NEXT: global_store_b32 v[4:5], v0, off
; GFX12-GISEL-NEXT: s_endpgm
@@ -952,23 +994,25 @@ define amdgpu_gs <2 x i32> @s_fptrunc_round_v2f32_to_v2f16_upward(<2 x float> in
; GFX11-SDAG-LABEL: s_fptrunc_round_v2f32_to_v2f16_upward:
; GFX11-SDAG: ; %bb.0:
; GFX11-SDAG-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_MODE, 2, 1), 1
+; GFX11-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX11-SDAG-NEXT: v_cvt_f16_f32_e64 v0.l, s0
; GFX11-SDAG-NEXT: v_cvt_f16_f32_e64 v1.l, s1
-; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-SDAG-NEXT: v_and_b32_e32 v0, 0xffff, v0
-; GFX11-SDAG-NEXT: v_and_b32_e32 v1, 0xffff, v1
; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX11-SDAG-NEXT: v_and_b32_e32 v1, 0xffff, v1
; GFX11-SDAG-NEXT: v_readfirstlane_b32 s0, v0
+; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX11-SDAG-NEXT: v_readfirstlane_b32 s1, v1
; GFX11-SDAG-NEXT: ; return to shader part epilog
;
; GFX11-GISEL-LABEL: s_fptrunc_round_v2f32_to_v2f16_upward:
; GFX11-GISEL: ; %bb.0:
; GFX11-GISEL-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_MODE, 2, 1), 1
+; GFX11-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX11-GISEL-NEXT: v_cvt_f16_f32_e64 v0.l, s0
; GFX11-GISEL-NEXT: v_cvt_f16_f32_e64 v1.l, s1
-; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-GISEL-NEXT: v_readfirstlane_b32 s0, v0
+; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX11-GISEL-NEXT: v_readfirstlane_b32 s1, v1
; GFX11-GISEL-NEXT: s_and_b32 s0, 0xffff, s0
; GFX11-GISEL-NEXT: s_and_b32 s1, 0xffff, s1
@@ -991,10 +1035,11 @@ define amdgpu_gs <2 x i32> @s_fptrunc_round_v2f32_to_v2f16_upward(<2 x float> in
; GFX12-LABEL: s_fptrunc_round_v2f32_to_v2f16_upward:
; GFX12: ; %bb.0:
; GFX12-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 2, 1), 1
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_2)
; GFX12-NEXT: s_cvt_f16_f32 s0, s0
; GFX12-NEXT: s_cvt_f16_f32 s1, s1
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_2) | instskip(NEXT) | instid1(SALU_CYCLE_2)
; GFX12-NEXT: s_and_b32 s0, 0xffff, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_2)
; GFX12-NEXT: s_and_b32 s1, 0xffff, s1
; GFX12-NEXT: ; return to shader part epilog
%res = call <2 x half> @llvm.fptrunc.round.v2f16.v2f32(<2 x float> %a, metadata !"round.upward")
@@ -1020,23 +1065,25 @@ define amdgpu_gs <2 x i32> @s_fptrunc_round_v2f32_to_v2f16_downward(<2 x float>
; GFX11-SDAG-LABEL: s_fptrunc_round_v2f32_to_v2f16_downward:
; GFX11-SDAG: ; %bb.0:
; GFX11-SDAG-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_MODE, 3, 1), 1
+; GFX11-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX11-SDAG-NEXT: v_cvt_f16_f32_e64 v0.l, s0
; GFX11-SDAG-NEXT: v_cvt_f16_f32_e64 v1.l, s1
-; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-SDAG-NEXT: v_and_b32_e32 v0, 0xffff, v0
-; GFX11-SDAG-NEXT: v_and_b32_e32 v1, 0xffff, v1
; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX11-SDAG-NEXT: v_and_b32_e32 v1, 0xffff, v1
; GFX11-SDAG-NEXT: v_readfirstlane_b32 s0, v0
+; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX11-SDAG-NEXT: v_readfirstlane_b32 s1, v1
; GFX11-SDAG-NEXT: ; return to shader part epilog
;
; GFX11-GISEL-LABEL: s_fptrunc_round_v2f32_to_v2f16_downward:
; GFX11-GISEL: ; %bb.0:
; GFX11-GISEL-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_MODE, 3, 1), 1
+; GFX11-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX11-GISEL-NEXT: v_cvt_f16_f32_e64 v0.l, s0
; GFX11-GISEL-NEXT: v_cvt_f16_f32_e64 v1.l, s1
-; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-GISEL-NEXT: v_readfirstlane_b32 s0, v0
+; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX11-GISEL-NEXT: v_readfirstlane_b32 s1, v1
; GFX11-GISEL-NEXT: s_and_b32 s0, 0xffff, s0
; GFX11-GISEL-NEXT: s_and_b32 s1, 0xffff, s1
@@ -1059,10 +1106,11 @@ define amdgpu_gs <2 x i32> @s_fptrunc_round_v2f32_to_v2f16_downward(<2 x float>
; GFX12-LABEL: s_fptrunc_round_v2f32_to_v2f16_downward:
; GFX12: ; %bb.0:
; GFX12-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 3, 1), 1
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_2)
; GFX12-NEXT: s_cvt_f16_f32 s0, s0
; GFX12-NEXT: s_cvt_f16_f32 s1, s1
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_2) | instskip(NEXT) | instid1(SALU_CYCLE_2)
; GFX12-NEXT: s_and_b32 s0, 0xffff, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_2)
; GFX12-NEXT: s_and_b32 s1, 0xffff, s1
; GFX12-NEXT: ; return to shader part epilog
%res = call <2 x half> @llvm.fptrunc.round.v2f16.v2f32(<2 x float> %a, metadata !"round.downward")
@@ -1160,6 +1208,7 @@ define amdgpu_gs void @s_fptrunc_round_v2f32_to_v2f16_upward_multiple_calls(<2 x
; GFX11-SDAG-LABEL: s_fptrunc_round_v2f32_to_v2f16_upward_multiple_calls:
; GFX11-SDAG: ; %bb.0:
; GFX11-SDAG-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_MODE, 2, 1), 1
+; GFX11-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_4) | instid1(SALU_CYCLE_1)
; GFX11-SDAG-NEXT: v_cvt_f16_f32_e64 v2.h, s1
; GFX11-SDAG-NEXT: v_cvt_f16_f32_e64 v2.l, s0
; GFX11-SDAG-NEXT: v_cvt_f16_f32_e64 v3.h, s3
@@ -1168,8 +1217,8 @@ define amdgpu_gs void @s_fptrunc_round_v2f32_to_v2f16_upward_multiple_calls(<2 x
; GFX11-SDAG-NEXT: v_cvt_f16_f32_e64 v4.h, s3
; GFX11-SDAG-NEXT: v_cvt_f16_f32_e64 v4.l, s2
; GFX11-SDAG-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_MODE, 3, 1), 0
+; GFX11-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-SDAG-NEXT: v_pk_add_f16 v2, v2, v3
-; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-SDAG-NEXT: v_pk_add_f16 v2, v4, v2
; GFX11-SDAG-NEXT: global_store_b32 v[0:1], v2, off
; GFX11-SDAG-NEXT: s_endpgm
@@ -1177,12 +1226,13 @@ define amdgpu_gs void @s_fptrunc_round_v2f32_to_v2f16_upward_multiple_calls(<2 x
; GFX11-GISEL-LABEL: s_fptrunc_round_v2f32_to_v2f16_upward_multiple_calls:
; GFX11-GISEL: ; %bb.0:
; GFX11-GISEL-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_MODE, 2, 1), 1
+; GFX11-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(VALU_DEP_3)
; GFX11-GISEL-NEXT: v_cvt_f16_f32_e64 v2.l, s0
; GFX11-GISEL-NEXT: v_cvt_f16_f32_e64 v3.l, s1
; GFX11-GISEL-NEXT: v_cvt_f16_f32_e64 v4.l, s3
-; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_1) | instid1(VALU_DEP_4)
; GFX11-GISEL-NEXT: v_mov_b16_e32 v5.l, v2.l
; GFX11-GISEL-NEXT: v_cvt_f16_f32_e64 v2.l, s2
+; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX11-GISEL-NEXT: v_mov_b16_e32 v6.l, v3.l
; GFX11-GISEL-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_MODE, 2, 2), 2
; GFX11-GISEL-NEXT: v_cvt_f16_f32_e64 v3.l, s3
@@ -1198,6 +1248,7 @@ define amdgpu_gs void @s_fptrunc_round_v2f32_to_v2f16_upward_multiple_calls(<2 x
; GFX11-GISEL-NEXT: v_readfirstlane_b32 s2, v2
; GFX11-GISEL-NEXT: v_readfirstlane_b32 s3, v3
; GFX11-GISEL-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_MODE, 3, 1), 0
+; GFX11-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-GISEL-NEXT: v_pk_add_f16 v2, s0, s1
; GFX11-GISEL-NEXT: s_pack_ll_b32_b16 s0, s2, s3
; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
@@ -1237,6 +1288,7 @@ define amdgpu_gs void @s_fptrunc_round_v2f32_to_v2f16_upward_multiple_calls(<2 x
; GFX12-SDAG-LABEL: s_fptrunc_round_v2f32_to_v2f16_upward_multiple_calls:
; GFX12-SDAG: ; %bb.0:
; GFX12-SDAG-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 2, 1), 1
+; GFX12-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_4) | instid1(SALU_CYCLE_1)
; GFX12-SDAG-NEXT: s_cvt_f16_f32 s1, s1
; GFX12-SDAG-NEXT: s_cvt_f16_f32 s0, s0
; GFX12-SDAG-NEXT: s_cvt_f16_f32 s4, s3
@@ -1247,10 +1299,10 @@ define amdgpu_gs void @s_fptrunc_round_v2f32_to_v2f16_upward_multiple_calls(<2 x
; GFX12-SDAG-NEXT: s_pack_ll_b32_b16 s0, s0, s1
; GFX12-SDAG-NEXT: s_pack_ll_b32_b16 s1, s5, s4
; GFX12-SDAG-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 3, 1), 0
+; GFX12-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(VALU_DEP_1)
; GFX12-SDAG-NEXT: v_pk_add_f16 v2, s0, s1
; GFX12-SDAG-NEXT: s_pack_ll_b32_b16 s0, s2, s3
; GFX12-SDAG-NEXT: s_wait_alu depctr_sa_sdst(0)
-; GFX12-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-SDAG-NEXT: v_pk_add_f16 v2, s0, v2
; GFX12-SDAG-NEXT: global_store_b32 v[0:1], v2, off
; GFX12-SDAG-NEXT: s_endpgm
@@ -1258,6 +1310,7 @@ define amdgpu_gs void @s_fptrunc_round_v2f32_to_v2f16_upward_multiple_calls(<2 x
; GFX12-GISEL-LABEL: s_fptrunc_round_v2f32_to_v2f16_upward_multiple_calls:
; GFX12-GISEL: ; %bb.0:
; GFX12-GISEL-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 2, 1), 1
+; GFX12-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_4) | instid1(SALU_CYCLE_1)
; GFX12-GISEL-NEXT: s_cvt_f16_f32 s0, s0
; GFX12-GISEL-NEXT: s_cvt_f16_f32 s1, s1
; GFX12-GISEL-NEXT: s_cvt_f16_f32 s4, s2
@@ -1266,13 +1319,14 @@ define amdgpu_gs void @s_fptrunc_round_v2f32_to_v2f16_upward_multiple_calls(<2 x
; GFX12-GISEL-NEXT: s_cvt_f16_f32 s2, s2
; GFX12-GISEL-NEXT: s_cvt_f16_f32 s3, s3
; GFX12-GISEL-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 3, 1), 0
+; GFX12-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_2)
; GFX12-GISEL-NEXT: s_add_f16 s0, s0, s4
; GFX12-GISEL-NEXT: s_add_f16 s1, s1, s5
-; GFX12-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_2) | instskip(NEXT) | instid1(SALU_CYCLE_2)
; GFX12-GISEL-NEXT: s_add_f16 s0, s2, s0
+; GFX12-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_2) | instskip(NEXT) | instid1(SALU_CYCLE_3)
; GFX12-GISEL-NEXT: s_add_f16 s1, s3, s1
-; GFX12-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_3) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-GISEL-NEXT: s_pack_ll_b32_b16 s0, s0, s1
+; GFX12-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-GISEL-NEXT: v_mov_b32_e32 v2, s0
; GFX12-GISEL-NEXT: global_store_b32 v[0:1], v2, off
; GFX12-GISEL-NEXT: s_endpgm
@@ -1298,16 +1352,17 @@ define amdgpu_gs <3 x half> @v_fptrunc_round_v3f32_to_v3f16_upward(<3 x float> %
; GFX11-SDAG-LABEL: v_fptrunc_round_v3f32_to_v3f16_upward:
; GFX11-SDAG: ; %bb.0:
; GFX11-SDAG-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_MODE, 2, 1), 1
+; GFX11-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(VALU_DEP_2)
; GFX11-SDAG-NEXT: v_cvt_f16_f32_e64 v3.h, v1
; GFX11-SDAG-NEXT: v_cvt_f16_f32_e64 v3.l, v0
; GFX11-SDAG-NEXT: v_cvt_f16_f32_e64 v1.l, v2
-; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX11-SDAG-NEXT: v_mov_b32_e32 v0, v3
; GFX11-SDAG-NEXT: ; return to shader part epilog
;
; GFX11-GISEL-LABEL: v_fptrunc_round_v3f32_to_v3f16_upward:
; GFX11-GISEL: ; %bb.0:
; GFX11-GISEL-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_MODE, 2, 1), 1
+; GFX11-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-GISEL-NEXT: v_cvt_f16_f32_e64 v0.l, v0
; GFX11-GISEL-NEXT: v_cvt_f16_f32_e64 v0.h, v1
; GFX11-GISEL-NEXT: v_cvt_f16_f32_e64 v1.l, v2
@@ -1327,16 +1382,17 @@ define amdgpu_gs <3 x half> @v_fptrunc_round_v3f32_to_v3f16_upward(<3 x float> %
; GFX12-SDAG-LABEL: v_fptrunc_round_v3f32_to_v3f16_upward:
; GFX12-SDAG: ; %bb.0:
; GFX12-SDAG-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 2, 1), 1
+; GFX12-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(VALU_DEP_2)
; GFX12-SDAG-NEXT: v_cvt_f16_f32_e64 v3.h, v1
; GFX12-SDAG-NEXT: v_cvt_f16_f32_e64 v3.l, v0
; GFX12-SDAG-NEXT: v_cvt_f16_f32_e64 v1.l, v2
-; GFX12-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX12-SDAG-NEXT: v_mov_b32_e32 v0, v3
; GFX12-SDAG-NEXT: ; return to shader part epilog
;
; GFX12-GISEL-LABEL: v_fptrunc_round_v3f32_to_v3f16_upward:
; GFX12-GISEL: ; %bb.0:
; GFX12-GISEL-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 2, 1), 1
+; GFX12-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-GISEL-NEXT: v_cvt_f16_f32_e64 v0.l, v0
; GFX12-GISEL-NEXT: v_cvt_f16_f32_e64 v0.h, v1
; GFX12-GISEL-NEXT: v_cvt_f16_f32_e64 v1.l, v2
@@ -1358,16 +1414,17 @@ define amdgpu_gs <3 x half> @v_fptrunc_round_v3f32_to_v3f16_downward(<3 x float>
; GFX11-SDAG-LABEL: v_fptrunc_round_v3f32_to_v3f16_downward:
; GFX11-SDAG: ; %bb.0:
; GFX11-SDAG-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_MODE, 3, 1), 1
+; GFX11-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(VALU_DEP_2)
; GFX11-SDAG-NEXT: v_cvt_f16_f32_e64 v3.h, v1
; GFX11-SDAG-NEXT: v_cvt_f16_f32_e64 v3.l, v0
; GFX11-SDAG-NEXT: v_cvt_f16_f32_e64 v1.l, v2
-; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX11-SDAG-NEXT: v_mov_b32_e32 v0, v3
; GFX11-SDAG-NEXT: ; return to shader part epilog
;
; GFX11-GISEL-LABEL: v_fptrunc_round_v3f32_to_v3f16_downward:
; GFX11-GISEL: ; %bb.0:
; GFX11-GISEL-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_MODE, 3, 1), 1
+; GFX11-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-GISEL-NEXT: v_cvt_f16_f32_e64 v0.l, v0
; GFX11-GISEL-NEXT: v_cvt_f16_f32_e64 v0.h, v1
; GFX11-GISEL-NEXT: v_cvt_f16_f32_e64 v1.l, v2
@@ -1387,16 +1444,17 @@ define amdgpu_gs <3 x half> @v_fptrunc_round_v3f32_to_v3f16_downward(<3 x float>
; GFX12-SDAG-LABEL: v_fptrunc_round_v3f32_to_v3f16_downward:
; GFX12-SDAG: ; %bb.0:
; GFX12-SDAG-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 3, 1), 1
+; GFX12-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(VALU_DEP_2)
; GFX12-SDAG-NEXT: v_cvt_f16_f32_e64 v3.h, v1
; GFX12-SDAG-NEXT: v_cvt_f16_f32_e64 v3.l, v0
; GFX12-SDAG-NEXT: v_cvt_f16_f32_e64 v1.l, v2
-; GFX12-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX12-SDAG-NEXT: v_mov_b32_e32 v0, v3
; GFX12-SDAG-NEXT: ; return to shader part epilog
;
; GFX12-GISEL-LABEL: v_fptrunc_round_v3f32_to_v3f16_downward:
; GFX12-GISEL: ; %bb.0:
; GFX12-GISEL-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 3, 1), 1
+; GFX12-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-GISEL-NEXT: v_cvt_f16_f32_e64 v0.l, v0
; GFX12-GISEL-NEXT: v_cvt_f16_f32_e64 v0.h, v1
; GFX12-GISEL-NEXT: v_cvt_f16_f32_e64 v1.l, v2
@@ -1420,17 +1478,18 @@ define amdgpu_gs <4 x half> @v_fptrunc_round_v4f32_to_v4f16_upward(<4 x float> %
; GFX11-SDAG-LABEL: v_fptrunc_round_v4f32_to_v4f16_upward:
; GFX11-SDAG: ; %bb.0:
; GFX11-SDAG-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_MODE, 2, 1), 1
+; GFX11-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_3) | instid1(VALU_DEP_1)
; GFX11-SDAG-NEXT: v_cvt_f16_f32_e64 v3.h, v3
; GFX11-SDAG-NEXT: v_cvt_f16_f32_e64 v1.h, v1
; GFX11-SDAG-NEXT: v_cvt_f16_f32_e64 v1.l, v0
; GFX11-SDAG-NEXT: v_cvt_f16_f32_e64 v3.l, v2
-; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-SDAG-NEXT: v_dual_mov_b32 v0, v1 :: v_dual_mov_b32 v1, v3
; GFX11-SDAG-NEXT: ; return to shader part epilog
;
; GFX11-GISEL-LABEL: v_fptrunc_round_v4f32_to_v4f16_upward:
; GFX11-GISEL: ; %bb.0:
; GFX11-GISEL-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_MODE, 2, 1), 1
+; GFX11-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-GISEL-NEXT: v_cvt_f16_f32_e64 v0.l, v0
; GFX11-GISEL-NEXT: v_cvt_f16_f32_e64 v0.h, v1
; GFX11-GISEL-NEXT: v_cvt_f16_f32_e64 v1.l, v2
@@ -1451,17 +1510,18 @@ define amdgpu_gs <4 x half> @v_fptrunc_round_v4f32_to_v4f16_upward(<4 x float> %
; GFX12-SDAG-LABEL: v_fptrunc_round_v4f32_to_v4f16_upward:
; GFX12-SDAG: ; %bb.0:
; GFX12-SDAG-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 2, 1), 1
+; GFX12-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_3) | instid1(VALU_DEP_1)
; GFX12-SDAG-NEXT: v_cvt_f16_f32_e64 v3.h, v3
; GFX12-SDAG-NEXT: v_cvt_f16_f32_e64 v1.h, v1
; GFX12-SDAG-NEXT: v_cvt_f16_f32_e64 v1.l, v0
; GFX12-SDAG-NEXT: v_cvt_f16_f32_e64 v3.l, v2
-; GFX12-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-SDAG-NEXT: v_dual_mov_b32 v0, v1 :: v_dual_mov_b32 v1, v3
; GFX12-SDAG-NEXT: ; return to shader part epilog
;
; GFX12-GISEL-LABEL: v_fptrunc_round_v4f32_to_v4f16_upward:
; GFX12-GISEL: ; %bb.0:
; GFX12-GISEL-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 2, 1), 1
+; GFX12-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-GISEL-NEXT: v_cvt_f16_f32_e64 v0.l, v0
; GFX12-GISEL-NEXT: v_cvt_f16_f32_e64 v0.h, v1
; GFX12-GISEL-NEXT: v_cvt_f16_f32_e64 v1.l, v2
@@ -1486,17 +1546,18 @@ define amdgpu_gs <4 x half> @v_fptrunc_round_v4f32_to_v4f16_downward(<4 x float>
; GFX11-SDAG-LABEL: v_fptrunc_round_v4f32_to_v4f16_downward:
; GFX11-SDAG: ; %bb.0:
; GFX11-SDAG-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_MODE, 3, 1), 1
+; GFX11-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_3) | instid1(VALU_DEP_1)
; GFX11-SDAG-NEXT: v_cvt_f16_f32_e64 v3.h, v3
; GFX11-SDAG-NEXT: v_cvt_f16_f32_e64 v1.h, v1
; GFX11-SDAG-NEXT: v_cvt_f16_f32_e64 v1.l, v0
; GFX11-SDAG-NEXT: v_cvt_f16_f32_e64 v3.l, v2
-; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-SDAG-NEXT: v_dual_mov_b32 v0, v1 :: v_dual_mov_b32 v1, v3
; GFX11-SDAG-NEXT: ; return to shader part epilog
;
; GFX11-GISEL-LABEL: v_fptrunc_round_v4f32_to_v4f16_downward:
; GFX11-GISEL: ; %bb.0:
; GFX11-GISEL-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_MODE, 3, 1), 1
+; GFX11-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-GISEL-NEXT: v_cvt_f16_f32_e64 v0.l, v0
; GFX11-GISEL-NEXT: v_cvt_f16_f32_e64 v0.h, v1
; GFX11-GISEL-NEXT: v_cvt_f16_f32_e64 v1.l, v2
@@ -1517,17 +1578,18 @@ define amdgpu_gs <4 x half> @v_fptrunc_round_v4f32_to_v4f16_downward(<4 x float>
; GFX12-SDAG-LABEL: v_fptrunc_round_v4f32_to_v4f16_downward:
; GFX12-SDAG: ; %bb.0:
; GFX12-SDAG-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 3, 1), 1
+; GFX12-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_3) | instid1(VALU_DEP_1)
; GFX12-SDAG-NEXT: v_cvt_f16_f32_e64 v3.h, v3
; GFX12-SDAG-NEXT: v_cvt_f16_f32_e64 v1.h, v1
; GFX12-SDAG-NEXT: v_cvt_f16_f32_e64 v1.l, v0
; GFX12-SDAG-NEXT: v_cvt_f16_f32_e64 v3.l, v2
-; GFX12-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-SDAG-NEXT: v_dual_mov_b32 v0, v1 :: v_dual_mov_b32 v1, v3
; GFX12-SDAG-NEXT: ; return to shader part epilog
;
; GFX12-GISEL-LABEL: v_fptrunc_round_v4f32_to_v4f16_downward:
; GFX12-GISEL: ; %bb.0:
; GFX12-GISEL-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 3, 1), 1
+; GFX12-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-GISEL-NEXT: v_cvt_f16_f32_e64 v0.l, v0
; GFX12-GISEL-NEXT: v_cvt_f16_f32_e64 v0.h, v1
; GFX12-GISEL-NEXT: v_cvt_f16_f32_e64 v1.l, v2
@@ -1558,6 +1620,7 @@ define amdgpu_gs <8 x half> @v_fptrunc_round_v8f32_to_v8f16_upward(<8 x float> %
; GFX11-SDAG-LABEL: v_fptrunc_round_v8f32_to_v8f16_upward:
; GFX11-SDAG: ; %bb.0:
; GFX11-SDAG-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_MODE, 2, 1), 1
+; GFX11-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-SDAG-NEXT: v_cvt_f16_f32_e64 v7.h, v7
; GFX11-SDAG-NEXT: v_cvt_f16_f32_e64 v5.h, v5
; GFX11-SDAG-NEXT: v_cvt_f16_f32_e64 v3.h, v3
@@ -1574,6 +1637,7 @@ define amdgpu_gs <8 x half> @v_fptrunc_round_v8f32_to_v8f16_upward(<8 x float> %
; GFX11-GISEL-LABEL: v_fptrunc_round_v8f32_to_v8f16_upward:
; GFX11-GISEL: ; %bb.0:
; GFX11-GISEL-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_MODE, 2, 1), 1
+; GFX11-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-GISEL-NEXT: v_cvt_f16_f32_e64 v0.l, v0
; GFX11-GISEL-NEXT: v_cvt_f16_f32_e64 v0.h, v1
; GFX11-GISEL-NEXT: v_cvt_f16_f32_e64 v1.l, v2
@@ -1604,6 +1668,7 @@ define amdgpu_gs <8 x half> @v_fptrunc_round_v8f32_to_v8f16_upward(<8 x float> %
; GFX12-SDAG-LABEL: v_fptrunc_round_v8f32_to_v8f16_upward:
; GFX12-SDAG: ; %bb.0:
; GFX12-SDAG-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 2, 1), 1
+; GFX12-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-SDAG-NEXT: v_cvt_f16_f32_e64 v7.h, v7
; GFX12-SDAG-NEXT: v_cvt_f16_f32_e64 v5.h, v5
; GFX12-SDAG-NEXT: v_cvt_f16_f32_e64 v3.h, v3
@@ -1620,6 +1685,7 @@ define amdgpu_gs <8 x half> @v_fptrunc_round_v8f32_to_v8f16_upward(<8 x float> %
; GFX12-GISEL-LABEL: v_fptrunc_round_v8f32_to_v8f16_upward:
; GFX12-GISEL: ; %bb.0:
; GFX12-GISEL-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 2, 1), 1
+; GFX12-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-GISEL-NEXT: v_cvt_f16_f32_e64 v0.l, v0
; GFX12-GISEL-NEXT: v_cvt_f16_f32_e64 v0.h, v1
; GFX12-GISEL-NEXT: v_cvt_f16_f32_e64 v1.l, v2
@@ -1654,6 +1720,7 @@ define amdgpu_gs <8 x half> @v_fptrunc_round_v8f32_to_v8f16_downward(<8 x float>
; GFX11-SDAG-LABEL: v_fptrunc_round_v8f32_to_v8f16_downward:
; GFX11-SDAG: ; %bb.0:
; GFX11-SDAG-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_MODE, 3, 1), 1
+; GFX11-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-SDAG-NEXT: v_cvt_f16_f32_e64 v7.h, v7
; GFX11-SDAG-NEXT: v_cvt_f16_f32_e64 v5.h, v5
; GFX11-SDAG-NEXT: v_cvt_f16_f32_e64 v3.h, v3
@@ -1670,6 +1737,7 @@ define amdgpu_gs <8 x half> @v_fptrunc_round_v8f32_to_v8f16_downward(<8 x float>
; GFX11-GISEL-LABEL: v_fptrunc_round_v8f32_to_v8f16_downward:
; GFX11-GISEL: ; %bb.0:
; GFX11-GISEL-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_MODE, 3, 1), 1
+; GFX11-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-GISEL-NEXT: v_cvt_f16_f32_e64 v0.l, v0
; GFX11-GISEL-NEXT: v_cvt_f16_f32_e64 v0.h, v1
; GFX11-GISEL-NEXT: v_cvt_f16_f32_e64 v1.l, v2
@@ -1700,6 +1768,7 @@ define amdgpu_gs <8 x half> @v_fptrunc_round_v8f32_to_v8f16_downward(<8 x float>
; GFX12-SDAG-LABEL: v_fptrunc_round_v8f32_to_v8f16_downward:
; GFX12-SDAG: ; %bb.0:
; GFX12-SDAG-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 3, 1), 1
+; GFX12-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-SDAG-NEXT: v_cvt_f16_f32_e64 v7.h, v7
; GFX12-SDAG-NEXT: v_cvt_f16_f32_e64 v5.h, v5
; GFX12-SDAG-NEXT: v_cvt_f16_f32_e64 v3.h, v3
@@ -1716,6 +1785,7 @@ define amdgpu_gs <8 x half> @v_fptrunc_round_v8f32_to_v8f16_downward(<8 x float>
; GFX12-GISEL-LABEL: v_fptrunc_round_v8f32_to_v8f16_downward:
; GFX12-GISEL: ; %bb.0:
; GFX12-GISEL-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 3, 1), 1
+; GFX12-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-GISEL-NEXT: v_cvt_f16_f32_e64 v0.l, v0
; GFX12-GISEL-NEXT: v_cvt_f16_f32_e64 v0.h, v1
; GFX12-GISEL-NEXT: v_cvt_f16_f32_e64 v1.l, v2
@@ -1744,15 +1814,36 @@ define amdgpu_gs float @v_fptrunc_round_f64_to_f32_tonearest(double %a) {
}
define amdgpu_gs float @v_fptrunc_round_f64_to_f32_upward(double %a) {
-; CHECK-LABEL: v_fptrunc_round_f64_to_f32_upward:
-; CHECK: ; %bb.0:
-; CHECK-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_MODE, 2, 1), 1
-; CHECK-NEXT: v_cvt_f32_f64_e32 v0, v[0:1]
-; CHECK-NEXT: ; return to shader part epilog
+; SDAG-LABEL: v_fptrunc_round_f64_to_f32_upward:
+; SDAG: ; %bb.0:
+; SDAG-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_MODE, 2, 1), 1
+; SDAG-NEXT: v_cvt_f32_f64_e32 v0, v[0:1]
+; SDAG-NEXT: ; return to shader part epilog
+;
+; GFX11-SDAG-LABEL: v_fptrunc_round_f64_to_f32_upward:
+; GFX11-SDAG: ; %bb.0:
+; GFX11-SDAG-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_MODE, 2, 1), 1
+; GFX11-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-SDAG-NEXT: v_cvt_f32_f64_e32 v0, v[0:1]
+; GFX11-SDAG-NEXT: ; return to shader part epilog
+;
+; GFX11-GISEL-LABEL: v_fptrunc_round_f64_to_f32_upward:
+; GFX11-GISEL: ; %bb.0:
+; GFX11-GISEL-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_MODE, 2, 1), 1
+; GFX11-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-GISEL-NEXT: v_cvt_f32_f64_e32 v0, v[0:1]
+; GFX11-GISEL-NEXT: ; return to shader part epilog
+;
+; GISEL-LABEL: v_fptrunc_round_f64_to_f32_upward:
+; GISEL: ; %bb.0:
+; GISEL-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_MODE, 2, 1), 1
+; GISEL-NEXT: v_cvt_f32_f64_e32 v0, v[0:1]
+; GISEL-NEXT: ; return to shader part epilog
;
; GFX12-LABEL: v_fptrunc_round_f64_to_f32_upward:
; GFX12: ; %bb.0:
; GFX12-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 2, 1), 1
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: v_cvt_f32_f64_e32 v0, v[0:1]
; GFX12-NEXT: ; return to shader part epilog
%res = call float @llvm.fptrunc.round.f32.f64(double %a, metadata !"round.upward")
@@ -1760,15 +1851,36 @@ define amdgpu_gs float @v_fptrunc_round_f64_to_f32_upward(double %a) {
}
define amdgpu_gs float @v_fptrunc_round_f64_to_f32_downward(double %a) {
-; CHECK-LABEL: v_fptrunc_round_f64_to_f32_downward:
-; CHECK: ; %bb.0:
-; CHECK-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_MODE, 3, 1), 1
-; CHECK-NEXT: v_cvt_f32_f64_e32 v0, v[0:1]
-; CHECK-NEXT: ; return to shader part epilog
+; SDAG-LABEL: v_fptrunc_round_f64_to_f32_downward:
+; SDAG: ; %bb.0:
+; SDAG-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_MODE, 3, 1), 1
+; SDAG-NEXT: v_cvt_f32_f64_e32 v0, v[0:1]
+; SDAG-NEXT: ; return to shader part epilog
+;
+; GFX11-SDAG-LABEL: v_fptrunc_round_f64_to_f32_downward:
+; GFX11-SDAG: ; %bb.0:
+; GFX11-SDAG-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_MODE, 3, 1), 1
+; GFX11-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-SDAG-NEXT: v_cvt_f32_f64_e32 v0, v[0:1]
+; GFX11-SDAG-NEXT: ; return to shader part epilog
+;
+; GFX11-GISEL-LABEL: v_fptrunc_round_f64_to_f32_downward:
+; GFX11-GISEL: ; %bb.0:
+; GFX11-GISEL-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_MODE, 3, 1), 1
+; GFX11-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-GISEL-NEXT: v_cvt_f32_f64_e32 v0, v[0:1]
+; GFX11-GISEL-NEXT: ; return to shader part epilog
+;
+; GISEL-LABEL: v_fptrunc_round_f64_to_f32_downward:
+; GISEL: ; %bb.0:
+; GISEL-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_MODE, 3, 1), 1
+; GISEL-NEXT: v_cvt_f32_f64_e32 v0, v[0:1]
+; GISEL-NEXT: ; return to shader part epilog
;
; GFX12-LABEL: v_fptrunc_round_f64_to_f32_downward:
; GFX12: ; %bb.0:
; GFX12-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 3, 1), 1
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: v_cvt_f32_f64_e32 v0, v[0:1]
; GFX12-NEXT: ; return to shader part epilog
%res = call float @llvm.fptrunc.round.f32.f64(double %a, metadata !"round.downward")
@@ -1776,15 +1888,36 @@ define amdgpu_gs float @v_fptrunc_round_f64_to_f32_downward(double %a) {
}
define amdgpu_gs float @v_fptrunc_round_f64_to_f32_towardzero(double %a) {
-; CHECK-LABEL: v_fptrunc_round_f64_to_f32_towardzero:
-; CHECK: ; %bb.0:
-; CHECK-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_MODE, 2, 2), 3
-; CHECK-NEXT: v_cvt_f32_f64_e32 v0, v[0:1]
-; CHECK-NEXT: ; return to shader part epilog
+; SDAG-LABEL: v_fptrunc_round_f64_to_f32_towardzero:
+; SDAG: ; %bb.0:
+; SDAG-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_MODE, 2, 2), 3
+; SDAG-NEXT: v_cvt_f32_f64_e32 v0, v[0:1]
+; SDAG-NEXT: ; return to shader part epilog
+;
+; GFX11-SDAG-LABEL: v_fptrunc_round_f64_to_f32_towardzero:
+; GFX11-SDAG: ; %bb.0:
+; GFX11-SDAG-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_MODE, 2, 2), 3
+; GFX11-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-SDAG-NEXT: v_cvt_f32_f64_e32 v0, v[0:1]
+; GFX11-SDAG-NEXT: ; return to shader part epilog
+;
+; GFX11-GISEL-LABEL: v_fptrunc_round_f64_to_f32_towardzero:
+; GFX11-GISEL: ; %bb.0:
+; GFX11-GISEL-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_MODE, 2, 2), 3
+; GFX11-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-GISEL-NEXT: v_cvt_f32_f64_e32 v0, v[0:1]
+; GFX11-GISEL-NEXT: ; return to shader part epilog
+;
+; GISEL-LABEL: v_fptrunc_round_f64_to_f32_towardzero:
+; GISEL: ; %bb.0:
+; GISEL-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_MODE, 2, 2), 3
+; GISEL-NEXT: v_cvt_f32_f64_e32 v0, v[0:1]
+; GISEL-NEXT: ; return to shader part epilog
;
; GFX12-LABEL: v_fptrunc_round_f64_to_f32_towardzero:
; GFX12: ; %bb.0:
; GFX12-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 2, 2), 3
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: v_cvt_f32_f64_e32 v0, v[0:1]
; GFX12-NEXT: ; return to shader part epilog
%res = call float @llvm.fptrunc.round.f32.f64(double %a, metadata !"round.towardzero")
@@ -1804,7 +1937,7 @@ define amdgpu_gs float @s_fptrunc_round_f64_to_f32_upward(double inreg %a) {
; GFX11-SDAG: ; %bb.0:
; GFX11-SDAG-NEXT: v_dual_mov_b32 v0, s0 :: v_dual_mov_b32 v1, s1
; GFX11-SDAG-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_MODE, 2, 1), 1
-; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-SDAG-NEXT: v_cvt_f32_f64_e32 v0, v[0:1]
; GFX11-SDAG-NEXT: ; return to shader part epilog
;
@@ -1812,7 +1945,7 @@ define amdgpu_gs float @s_fptrunc_round_f64_to_f32_upward(double inreg %a) {
; GFX11-GISEL: ; %bb.0:
; GFX11-GISEL-NEXT: v_dual_mov_b32 v0, s0 :: v_dual_mov_b32 v1, s1
; GFX11-GISEL-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_MODE, 2, 1), 1
-; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-GISEL-NEXT: v_cvt_f32_f64_e32 v0, v[0:1]
; GFX11-GISEL-NEXT: ; return to shader part epilog
;
@@ -1828,7 +1961,7 @@ define amdgpu_gs float @s_fptrunc_round_f64_to_f32_upward(double inreg %a) {
; GFX12: ; %bb.0:
; GFX12-NEXT: v_dual_mov_b32 v0, s0 :: v_dual_mov_b32 v1, s1
; GFX12-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 2, 1), 1
-; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: v_cvt_f32_f64_e32 v0, v[0:1]
; GFX12-NEXT: ; return to shader part epilog
%res = call float @llvm.fptrunc.round.f32.f64(double %a, metadata !"round.upward")
@@ -1848,7 +1981,7 @@ define amdgpu_gs float @s_fptrunc_round_f64_to_f32_downward(double inreg %a) {
; GFX11-SDAG: ; %bb.0:
; GFX11-SDAG-NEXT: v_dual_mov_b32 v0, s0 :: v_dual_mov_b32 v1, s1
; GFX11-SDAG-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_MODE, 3, 1), 1
-; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-SDAG-NEXT: v_cvt_f32_f64_e32 v0, v[0:1]
; GFX11-SDAG-NEXT: ; return to shader part epilog
;
@@ -1856,7 +1989,7 @@ define amdgpu_gs float @s_fptrunc_round_f64_to_f32_downward(double inreg %a) {
; GFX11-GISEL: ; %bb.0:
; GFX11-GISEL-NEXT: v_dual_mov_b32 v0, s0 :: v_dual_mov_b32 v1, s1
; GFX11-GISEL-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_MODE, 3, 1), 1
-; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-GISEL-NEXT: v_cvt_f32_f64_e32 v0, v[0:1]
; GFX11-GISEL-NEXT: ; return to shader part epilog
;
@@ -1872,7 +2005,7 @@ define amdgpu_gs float @s_fptrunc_round_f64_to_f32_downward(double inreg %a) {
; GFX12: ; %bb.0:
; GFX12-NEXT: v_dual_mov_b32 v0, s0 :: v_dual_mov_b32 v1, s1
; GFX12-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 3, 1), 1
-; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: v_cvt_f32_f64_e32 v0, v[0:1]
; GFX12-NEXT: ; return to shader part epilog
%res = call float @llvm.fptrunc.round.f32.f64(double %a, metadata !"round.downward")
diff --git a/llvm/test/CodeGen/AMDGPU/llvm.is.fpclass.ll b/llvm/test/CodeGen/AMDGPU/llvm.is.fpclass.ll
index 17f7a401e96c84..6892f144a7e10d 100644
--- a/llvm/test/CodeGen/AMDGPU/llvm.is.fpclass.ll
+++ b/llvm/test/CodeGen/AMDGPU/llvm.is.fpclass.ll
@@ -147,8 +147,8 @@ define amdgpu_kernel void @sgpr_isnan_f32(ptr addrspace(1) %out, float %x) {
; GFX11GLISEL-NEXT: s_waitcnt lgkmcnt(0)
; GFX11GLISEL-NEXT: v_cmp_class_f32_e64 s2, s2, 3
; GFX11GLISEL-NEXT: s_cmp_lg_u32 s2, 0
+; GFX11GLISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11GLISEL-NEXT: s_cselect_b32 s2, -1, 0
-; GFX11GLISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11GLISEL-NEXT: v_mov_b32_e32 v0, s2
; GFX11GLISEL-NEXT: global_store_b32 v1, v0, s[0:1]
; GFX11GLISEL-NEXT: s_endpgm
@@ -278,8 +278,8 @@ define amdgpu_kernel void @sgpr_isnan_f64(ptr addrspace(1) %out, double %x) {
; GFX11GLISEL-NEXT: s_waitcnt lgkmcnt(0)
; GFX11GLISEL-NEXT: v_cmp_class_f64_e64 s2, s[2:3], 3
; GFX11GLISEL-NEXT: s_cmp_lg_u32 s2, 0
+; GFX11GLISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11GLISEL-NEXT: s_cselect_b32 s2, -1, 0
-; GFX11GLISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11GLISEL-NEXT: v_mov_b32_e32 v0, s2
; GFX11GLISEL-NEXT: global_store_b32 v1, v0, s[0:1]
; GFX11GLISEL-NEXT: s_endpgm
diff --git a/llvm/test/CodeGen/AMDGPU/llvm.log.ll b/llvm/test/CodeGen/AMDGPU/llvm.log.ll
index 0b8c2a75d68069..323f8f57db8323 100644
--- a/llvm/test/CodeGen/AMDGPU/llvm.log.ll
+++ b/llvm/test/CodeGen/AMDGPU/llvm.log.ll
@@ -219,23 +219,23 @@ define amdgpu_kernel void @s_log_f32(ptr addrspace(1) %out, float %in) {
; GFX1100-SDAG-NEXT: s_load_b32 s0, s[4:5], 0x2c
; GFX1100-SDAG-NEXT: s_waitcnt lgkmcnt(0)
; GFX1100-SDAG-NEXT: v_cmp_gt_f32_e64 s1, 0x800000, s0
-; GFX1100-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
; GFX1100-SDAG-NEXT: v_cndmask_b32_e64 v0, 0, 0x41b17218, s1
; GFX1100-SDAG-NEXT: s_and_b32 s1, s1, exec_lo
; GFX1100-SDAG-NEXT: s_cselect_b32 s1, 32, 0
+; GFX1100-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX1100-SDAG-NEXT: v_ldexp_f32 v1, s0, s1
; GFX1100-SDAG-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
-; GFX1100-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_3) | instid1(VALU_DEP_2)
; GFX1100-SDAG-NEXT: v_log_f32_e32 v1, v1
; GFX1100-SDAG-NEXT: s_waitcnt_depctr depctr_va_vdst(0)
; GFX1100-SDAG-NEXT: v_mul_f32_e32 v2, 0x3f317217, v1
; GFX1100-SDAG-NEXT: v_cmp_gt_f32_e64 vcc_lo, 0x7f800000, |v1|
+; GFX1100-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1100-SDAG-NEXT: v_fma_f32 v3, 0x3f317217, v1, -v2
-; GFX1100-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1100-SDAG-NEXT: v_fmamk_f32 v3, v1, 0x3377d1cf, v3
-; GFX1100-SDAG-NEXT: v_add_f32_e32 v2, v2, v3
; GFX1100-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1100-SDAG-NEXT: v_add_f32_e32 v2, v2, v3
; GFX1100-SDAG-NEXT: v_dual_cndmask_b32 v1, v1, v2 :: v_dual_mov_b32 v2, 0
+; GFX1100-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1100-SDAG-NEXT: v_sub_f32_e32 v0, v1, v0
; GFX1100-SDAG-NEXT: s_waitcnt lgkmcnt(0)
; GFX1100-SDAG-NEXT: global_store_b32 v2, v0, s[0:1]
@@ -247,23 +247,24 @@ define amdgpu_kernel void @s_log_f32(ptr addrspace(1) %out, float %in) {
; GFX1100-GISEL-NEXT: s_waitcnt lgkmcnt(0)
; GFX1100-GISEL-NEXT: v_cmp_gt_f32_e64 s1, 0x800000, s0
; GFX1100-GISEL-NEXT: s_cmp_lg_u32 s1, 0
+; GFX1100-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
; GFX1100-GISEL-NEXT: s_cselect_b32 s1, 1, 0
; GFX1100-GISEL-NEXT: s_cselect_b32 s2, 1, 0
; GFX1100-GISEL-NEXT: s_lshl_b32 s1, s1, 5
-; GFX1100-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX1100-GISEL-NEXT: v_ldexp_f32 v0, s0, s1
; GFX1100-GISEL-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
+; GFX1100-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_4) | instid1(VALU_DEP_3)
; GFX1100-GISEL-NEXT: v_log_f32_e32 v0, v0
; GFX1100-GISEL-NEXT: s_waitcnt_depctr depctr_va_vdst(0)
; GFX1100-GISEL-NEXT: v_mul_f32_e32 v1, 0x3f317217, v0
; GFX1100-GISEL-NEXT: v_cmp_gt_f32_e64 s3, 0x7f800000, |v0|
; GFX1100-GISEL-NEXT: v_readfirstlane_b32 s4, v0
-; GFX1100-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX1100-GISEL-NEXT: v_fma_f32 v2, 0x3f317217, v0, -v1
; GFX1100-GISEL-NEXT: s_cmp_lg_u32 s3, 0
-; GFX1100-GISEL-NEXT: v_fmac_f32_e32 v2, 0x3377d1cf, v0
; GFX1100-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1100-GISEL-NEXT: v_fmac_f32_e32 v2, 0x3377d1cf, v0
; GFX1100-GISEL-NEXT: v_add_f32_e32 v1, v1, v2
+; GFX1100-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_3) | instid1(SALU_CYCLE_1)
; GFX1100-GISEL-NEXT: v_readfirstlane_b32 s5, v1
; GFX1100-GISEL-NEXT: v_mov_b32_e32 v1, 0
; GFX1100-GISEL-NEXT: s_cselect_b32 s3, s5, s4
@@ -552,23 +553,23 @@ define amdgpu_kernel void @s_log_contract_f32(ptr addrspace(1) %out, float %in)
; GFX1100-SDAG-NEXT: s_load_b32 s0, s[4:5], 0x2c
; GFX1100-SDAG-NEXT: s_waitcnt lgkmcnt(0)
; GFX1100-SDAG-NEXT: v_cmp_gt_f32_e64 s1, 0x800000, s0
-; GFX1100-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
; GFX1100-SDAG-NEXT: v_cndmask_b32_e64 v0, 0, 0x41b17218, s1
; GFX1100-SDAG-NEXT: s_and_b32 s1, s1, exec_lo
; GFX1100-SDAG-NEXT: s_cselect_b32 s1, 32, 0
+; GFX1100-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX1100-SDAG-NEXT: v_ldexp_f32 v1, s0, s1
; GFX1100-SDAG-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
-; GFX1100-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_3) | instid1(VALU_DEP_2)
; GFX1100-SDAG-NEXT: v_log_f32_e32 v1, v1
; GFX1100-SDAG-NEXT: s_waitcnt_depctr depctr_va_vdst(0)
; GFX1100-SDAG-NEXT: v_mul_f32_e32 v2, 0x3f317217, v1
; GFX1100-SDAG-NEXT: v_cmp_gt_f32_e64 vcc_lo, 0x7f800000, |v1|
+; GFX1100-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1100-SDAG-NEXT: v_fma_f32 v3, 0x3f317217, v1, -v2
-; GFX1100-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1100-SDAG-NEXT: v_fmamk_f32 v3, v1, 0x3377d1cf, v3
-; GFX1100-SDAG-NEXT: v_add_f32_e32 v2, v2, v3
; GFX1100-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1100-SDAG-NEXT: v_add_f32_e32 v2, v2, v3
; GFX1100-SDAG-NEXT: v_dual_cndmask_b32 v1, v1, v2 :: v_dual_mov_b32 v2, 0
+; GFX1100-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1100-SDAG-NEXT: v_sub_f32_e32 v0, v1, v0
; GFX1100-SDAG-NEXT: s_waitcnt lgkmcnt(0)
; GFX1100-SDAG-NEXT: global_store_b32 v2, v0, s[0:1]
@@ -580,23 +581,24 @@ define amdgpu_kernel void @s_log_contract_f32(ptr addrspace(1) %out, float %in)
; GFX1100-GISEL-NEXT: s_waitcnt lgkmcnt(0)
; GFX1100-GISEL-NEXT: v_cmp_gt_f32_e64 s1, 0x800000, s0
; GFX1100-GISEL-NEXT: s_cmp_lg_u32 s1, 0
+; GFX1100-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
; GFX1100-GISEL-NEXT: s_cselect_b32 s1, 1, 0
; GFX1100-GISEL-NEXT: s_cselect_b32 s2, 1, 0
; GFX1100-GISEL-NEXT: s_lshl_b32 s1, s1, 5
-; GFX1100-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX1100-GISEL-NEXT: v_ldexp_f32 v0, s0, s1
; GFX1100-GISEL-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
+; GFX1100-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_4) | instid1(VALU_DEP_3)
; GFX1100-GISEL-NEXT: v_log_f32_e32 v0, v0
; GFX1100-GISEL-NEXT: s_waitcnt_depctr depctr_va_vdst(0)
; GFX1100-GISEL-NEXT: v_mul_f32_e32 v1, 0x3f317217, v0
; GFX1100-GISEL-NEXT: v_cmp_gt_f32_e64 s3, 0x7f800000, |v0|
; GFX1100-GISEL-NEXT: v_readfirstlane_b32 s4, v0
-; GFX1100-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX1100-GISEL-NEXT: v_fma_f32 v2, 0x3f317217, v0, -v1
; GFX1100-GISEL-NEXT: s_cmp_lg_u32 s3, 0
-; GFX1100-GISEL-NEXT: v_fmac_f32_e32 v2, 0x3377d1cf, v0
; GFX1100-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1100-GISEL-NEXT: v_fmac_f32_e32 v2, 0x3377d1cf, v0
; GFX1100-GISEL-NEXT: v_add_f32_e32 v1, v1, v2
+; GFX1100-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_3) | instid1(SALU_CYCLE_1)
; GFX1100-GISEL-NEXT: v_readfirstlane_b32 s5, v1
; GFX1100-GISEL-NEXT: v_mov_b32_e32 v1, 0
; GFX1100-GISEL-NEXT: s_cselect_b32 s3, s5, s4
@@ -996,7 +998,6 @@ define amdgpu_kernel void @s_log_v2f32(ptr addrspace(1) %out, <2 x float> %in) {
; GFX1100-SDAG-NEXT: s_waitcnt lgkmcnt(0)
; GFX1100-SDAG-NEXT: v_cmp_gt_f32_e64 s4, 0x800000, s3
; GFX1100-SDAG-NEXT: v_cmp_gt_f32_e64 s5, 0x800000, s2
-; GFX1100-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX1100-SDAG-NEXT: v_cndmask_b32_e64 v0, 0, 0x41b17218, s4
; GFX1100-SDAG-NEXT: s_and_b32 s4, s4, exec_lo
; GFX1100-SDAG-NEXT: s_cselect_b32 s4, 32, 0
@@ -1033,45 +1034,47 @@ define amdgpu_kernel void @s_log_v2f32(ptr addrspace(1) %out, <2 x float> %in) {
; GFX1100-GISEL-NEXT: s_waitcnt lgkmcnt(0)
; GFX1100-GISEL-NEXT: v_cmp_gt_f32_e64 s4, 0x800000, s2
; GFX1100-GISEL-NEXT: s_cmp_lg_u32 s4, 0
+; GFX1100-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
; GFX1100-GISEL-NEXT: s_cselect_b32 s5, 1, 0
; GFX1100-GISEL-NEXT: s_cselect_b32 s4, 1, 0
; GFX1100-GISEL-NEXT: s_lshl_b32 s5, s5, 5
-; GFX1100-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1100-GISEL-NEXT: v_ldexp_f32 v0, s2, s5
+; GFX1100-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_4) | instid1(VALU_DEP_3)
; GFX1100-GISEL-NEXT: v_log_f32_e32 v0, v0
; GFX1100-GISEL-NEXT: s_waitcnt_depctr depctr_va_vdst(0)
; GFX1100-GISEL-NEXT: v_mul_f32_e32 v1, 0x3f317217, v0
; GFX1100-GISEL-NEXT: v_cmp_gt_f32_e64 s2, 0x7f800000, |v0|
; GFX1100-GISEL-NEXT: v_readfirstlane_b32 s5, v0
-; GFX1100-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_2) | instid1(VALU_DEP_2)
; GFX1100-GISEL-NEXT: v_fma_f32 v2, 0x3f317217, v0, -v1
; GFX1100-GISEL-NEXT: s_cmp_lg_u32 s2, 0
; GFX1100-GISEL-NEXT: v_cmp_gt_f32_e64 s2, 0x800000, s3
+; GFX1100-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1100-GISEL-NEXT: v_fmac_f32_e32 v2, 0x3377d1cf, v0
-; GFX1100-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1100-GISEL-NEXT: v_add_f32_e32 v1, v1, v2
+; GFX1100-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
; GFX1100-GISEL-NEXT: v_readfirstlane_b32 s6, v1
; GFX1100-GISEL-NEXT: s_cselect_b32 s5, s6, s5
; GFX1100-GISEL-NEXT: s_cmp_lg_u32 s4, 0
; GFX1100-GISEL-NEXT: s_cselect_b32 s4, 0x41b17218, 0
; GFX1100-GISEL-NEXT: s_cmp_lg_u32 s2, 0
+; GFX1100-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
; GFX1100-GISEL-NEXT: s_cselect_b32 s6, 1, 0
; GFX1100-GISEL-NEXT: s_cselect_b32 s2, 1, 0
; GFX1100-GISEL-NEXT: s_lshl_b32 s6, s6, 5
-; GFX1100-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1100-GISEL-NEXT: v_ldexp_f32 v0, s3, s6
+; GFX1100-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_4) | instid1(VALU_DEP_3)
; GFX1100-GISEL-NEXT: v_log_f32_e32 v0, v0
; GFX1100-GISEL-NEXT: s_waitcnt_depctr depctr_va_vdst(0)
; GFX1100-GISEL-NEXT: v_mul_f32_e32 v1, 0x3f317217, v0
; GFX1100-GISEL-NEXT: v_cmp_gt_f32_e64 s3, 0x7f800000, |v0|
; GFX1100-GISEL-NEXT: v_readfirstlane_b32 s6, v0
-; GFX1100-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX1100-GISEL-NEXT: v_fma_f32 v2, 0x3f317217, v0, -v1
; GFX1100-GISEL-NEXT: s_cmp_lg_u32 s3, 0
+; GFX1100-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX1100-GISEL-NEXT: v_fmac_f32_e32 v2, 0x3377d1cf, v0
; GFX1100-GISEL-NEXT: v_sub_f32_e64 v0, s5, s4
-; GFX1100-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1100-GISEL-NEXT: v_dual_add_f32 v1, v1, v2 :: v_dual_mov_b32 v2, 0
+; GFX1100-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
; GFX1100-GISEL-NEXT: v_readfirstlane_b32 s7, v1
; GFX1100-GISEL-NEXT: s_cselect_b32 s3, s7, s6
; GFX1100-GISEL-NEXT: s_cmp_lg_u32 s2, 0
@@ -1633,7 +1636,6 @@ define amdgpu_kernel void @s_log_v3f32(ptr addrspace(1) %out, <3 x float> %in) {
; GFX1100-SDAG-NEXT: v_cmp_gt_f32_e64 s3, 0x800000, s2
; GFX1100-SDAG-NEXT: v_cmp_gt_f32_e64 s6, 0x800000, s1
; GFX1100-SDAG-NEXT: v_cmp_gt_f32_e64 s7, 0x800000, s0
-; GFX1100-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX1100-SDAG-NEXT: v_cndmask_b32_e64 v0, 0, 0x41b17218, s3
; GFX1100-SDAG-NEXT: s_and_b32 s3, s3, exec_lo
; GFX1100-SDAG-NEXT: v_cndmask_b32_e64 v1, 0, 0x41b17218, s6
@@ -1685,68 +1687,71 @@ define amdgpu_kernel void @s_log_v3f32(ptr addrspace(1) %out, <3 x float> %in) {
; GFX1100-GISEL-NEXT: s_waitcnt lgkmcnt(0)
; GFX1100-GISEL-NEXT: v_cmp_gt_f32_e64 s3, 0x800000, s0
; GFX1100-GISEL-NEXT: s_cmp_lg_u32 s3, 0
+; GFX1100-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
; GFX1100-GISEL-NEXT: s_cselect_b32 s6, 1, 0
; GFX1100-GISEL-NEXT: s_cselect_b32 s3, 1, 0
; GFX1100-GISEL-NEXT: s_lshl_b32 s6, s6, 5
-; GFX1100-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1100-GISEL-NEXT: v_ldexp_f32 v0, s0, s6
+; GFX1100-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_4) | instid1(VALU_DEP_3)
; GFX1100-GISEL-NEXT: v_log_f32_e32 v0, v0
; GFX1100-GISEL-NEXT: s_waitcnt_depctr depctr_va_vdst(0)
; GFX1100-GISEL-NEXT: v_mul_f32_e32 v1, 0x3f317217, v0
; GFX1100-GISEL-NEXT: v_cmp_gt_f32_e64 s0, 0x7f800000, |v0|
; GFX1100-GISEL-NEXT: v_readfirstlane_b32 s6, v0
-; GFX1100-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_2) | instid1(VALU_DEP_2)
; GFX1100-GISEL-NEXT: v_fma_f32 v2, 0x3f317217, v0, -v1
; GFX1100-GISEL-NEXT: s_cmp_lg_u32 s0, 0
; GFX1100-GISEL-NEXT: v_cmp_gt_f32_e64 s0, 0x800000, s1
+; GFX1100-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1100-GISEL-NEXT: v_fmac_f32_e32 v2, 0x3377d1cf, v0
-; GFX1100-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1100-GISEL-NEXT: v_add_f32_e32 v1, v1, v2
+; GFX1100-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
; GFX1100-GISEL-NEXT: v_readfirstlane_b32 s7, v1
; GFX1100-GISEL-NEXT: s_cselect_b32 s6, s7, s6
; GFX1100-GISEL-NEXT: s_cmp_lg_u32 s3, 0
; GFX1100-GISEL-NEXT: s_cselect_b32 s3, 0x41b17218, 0
; GFX1100-GISEL-NEXT: s_cmp_lg_u32 s0, 0
+; GFX1100-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
; GFX1100-GISEL-NEXT: s_cselect_b32 s7, 1, 0
; GFX1100-GISEL-NEXT: s_cselect_b32 s0, 1, 0
; GFX1100-GISEL-NEXT: s_lshl_b32 s7, s7, 5
-; GFX1100-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1100-GISEL-NEXT: v_ldexp_f32 v0, s1, s7
+; GFX1100-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_4) | instid1(VALU_DEP_3)
; GFX1100-GISEL-NEXT: v_log_f32_e32 v0, v0
; GFX1100-GISEL-NEXT: s_waitcnt_depctr depctr_va_vdst(0)
; GFX1100-GISEL-NEXT: v_mul_f32_e32 v1, 0x3f317217, v0
; GFX1100-GISEL-NEXT: v_cmp_gt_f32_e64 s1, 0x7f800000, |v0|
; GFX1100-GISEL-NEXT: v_readfirstlane_b32 s7, v0
-; GFX1100-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_2) | instid1(VALU_DEP_2)
; GFX1100-GISEL-NEXT: v_fma_f32 v2, 0x3f317217, v0, -v1
; GFX1100-GISEL-NEXT: s_cmp_lg_u32 s1, 0
; GFX1100-GISEL-NEXT: v_cmp_gt_f32_e64 s1, 0x800000, s2
+; GFX1100-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1100-GISEL-NEXT: v_fmac_f32_e32 v2, 0x3377d1cf, v0
-; GFX1100-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1100-GISEL-NEXT: v_add_f32_e32 v1, v1, v2
+; GFX1100-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
; GFX1100-GISEL-NEXT: v_readfirstlane_b32 s8, v1
; GFX1100-GISEL-NEXT: s_cselect_b32 s7, s8, s7
; GFX1100-GISEL-NEXT: s_cmp_lg_u32 s0, 0
; GFX1100-GISEL-NEXT: s_cselect_b32 s8, 0x41b17218, 0
; GFX1100-GISEL-NEXT: s_cmp_lg_u32 s1, 0
+; GFX1100-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
; GFX1100-GISEL-NEXT: s_cselect_b32 s0, 1, 0
; GFX1100-GISEL-NEXT: s_cselect_b32 s9, 1, 0
; GFX1100-GISEL-NEXT: s_lshl_b32 s0, s0, 5
-; GFX1100-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX1100-GISEL-NEXT: v_ldexp_f32 v0, s2, s0
; GFX1100-GISEL-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
+; GFX1100-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_4) | instid1(VALU_DEP_3)
; GFX1100-GISEL-NEXT: v_log_f32_e32 v0, v0
; GFX1100-GISEL-NEXT: s_waitcnt_depctr depctr_va_vdst(0)
; GFX1100-GISEL-NEXT: v_mul_f32_e32 v1, 0x3f317217, v0
; GFX1100-GISEL-NEXT: v_cmp_gt_f32_e64 s2, 0x7f800000, |v0|
; GFX1100-GISEL-NEXT: v_readfirstlane_b32 s4, v0
-; GFX1100-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX1100-GISEL-NEXT: v_fma_f32 v2, 0x3f317217, v0, -v1
; GFX1100-GISEL-NEXT: s_cmp_lg_u32 s2, 0
+; GFX1100-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX1100-GISEL-NEXT: v_fmac_f32_e32 v2, 0x3377d1cf, v0
; GFX1100-GISEL-NEXT: v_sub_f32_e64 v0, s6, s3
-; GFX1100-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1100-GISEL-NEXT: v_add_f32_e32 v1, v1, v2
+; GFX1100-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_3) | instid1(SALU_CYCLE_1)
; GFX1100-GISEL-NEXT: v_readfirstlane_b32 s5, v1
; GFX1100-GISEL-NEXT: v_sub_f32_e64 v1, s7, s8
; GFX1100-GISEL-NEXT: s_cselect_b32 s2, s5, s4
@@ -2464,7 +2469,6 @@ define amdgpu_kernel void @s_log_v4f32(ptr addrspace(1) %out, <4 x float> %in) {
; GFX1100-SDAG-NEXT: v_cmp_gt_f32_e64 s7, 0x800000, s2
; GFX1100-SDAG-NEXT: v_cmp_gt_f32_e64 s8, 0x800000, s1
; GFX1100-SDAG-NEXT: v_cmp_gt_f32_e64 s9, 0x800000, s0
-; GFX1100-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_4)
; GFX1100-SDAG-NEXT: v_cndmask_b32_e64 v0, 0, 0x41b17218, s6
; GFX1100-SDAG-NEXT: s_and_b32 s6, s6, exec_lo
; GFX1100-SDAG-NEXT: s_cselect_b32 s6, 32, 0
@@ -2524,91 +2528,95 @@ define amdgpu_kernel void @s_log_v4f32(ptr addrspace(1) %out, <4 x float> %in) {
; GFX1100-GISEL-NEXT: s_waitcnt lgkmcnt(0)
; GFX1100-GISEL-NEXT: v_cmp_gt_f32_e64 s6, 0x800000, s0
; GFX1100-GISEL-NEXT: s_cmp_lg_u32 s6, 0
+; GFX1100-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
; GFX1100-GISEL-NEXT: s_cselect_b32 s7, 1, 0
; GFX1100-GISEL-NEXT: s_cselect_b32 s6, 1, 0
; GFX1100-GISEL-NEXT: s_lshl_b32 s7, s7, 5
-; GFX1100-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1100-GISEL-NEXT: v_ldexp_f32 v0, s0, s7
+; GFX1100-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_4) | instid1(VALU_DEP_3)
; GFX1100-GISEL-NEXT: v_log_f32_e32 v0, v0
; GFX1100-GISEL-NEXT: s_waitcnt_depctr depctr_va_vdst(0)
; GFX1100-GISEL-NEXT: v_mul_f32_e32 v1, 0x3f317217, v0
; GFX1100-GISEL-NEXT: v_cmp_gt_f32_e64 s0, 0x7f800000, |v0|
; GFX1100-GISEL-NEXT: v_readfirstlane_b32 s7, v0
-; GFX1100-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_2) | instid1(VALU_DEP_2)
; GFX1100-GISEL-NEXT: v_fma_f32 v2, 0x3f317217, v0, -v1
; GFX1100-GISEL-NEXT: s_cmp_lg_u32 s0, 0
; GFX1100-GISEL-NEXT: v_cmp_gt_f32_e64 s0, 0x800000, s1
+; GFX1100-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1100-GISEL-NEXT: v_fmac_f32_e32 v2, 0x3377d1cf, v0
-; GFX1100-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1100-GISEL-NEXT: v_add_f32_e32 v1, v1, v2
+; GFX1100-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
; GFX1100-GISEL-NEXT: v_readfirstlane_b32 s8, v1
; GFX1100-GISEL-NEXT: s_cselect_b32 s7, s8, s7
; GFX1100-GISEL-NEXT: s_cmp_lg_u32 s6, 0
; GFX1100-GISEL-NEXT: s_cselect_b32 s6, 0x41b17218, 0
; GFX1100-GISEL-NEXT: s_cmp_lg_u32 s0, 0
+; GFX1100-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
; GFX1100-GISEL-NEXT: s_cselect_b32 s8, 1, 0
; GFX1100-GISEL-NEXT: s_cselect_b32 s0, 1, 0
; GFX1100-GISEL-NEXT: s_lshl_b32 s8, s8, 5
-; GFX1100-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1100-GISEL-NEXT: v_ldexp_f32 v0, s1, s8
+; GFX1100-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_4) | instid1(VALU_DEP_3)
; GFX1100-GISEL-NEXT: v_log_f32_e32 v0, v0
; GFX1100-GISEL-NEXT: s_waitcnt_depctr depctr_va_vdst(0)
; GFX1100-GISEL-NEXT: v_mul_f32_e32 v1, 0x3f317217, v0
; GFX1100-GISEL-NEXT: v_cmp_gt_f32_e64 s1, 0x7f800000, |v0|
; GFX1100-GISEL-NEXT: v_readfirstlane_b32 s8, v0
-; GFX1100-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_2) | instid1(VALU_DEP_2)
; GFX1100-GISEL-NEXT: v_fma_f32 v2, 0x3f317217, v0, -v1
; GFX1100-GISEL-NEXT: s_cmp_lg_u32 s1, 0
; GFX1100-GISEL-NEXT: v_cmp_gt_f32_e64 s1, 0x800000, s2
+; GFX1100-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1100-GISEL-NEXT: v_fmac_f32_e32 v2, 0x3377d1cf, v0
-; GFX1100-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1100-GISEL-NEXT: v_add_f32_e32 v1, v1, v2
+; GFX1100-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
; GFX1100-GISEL-NEXT: v_readfirstlane_b32 s9, v1
; GFX1100-GISEL-NEXT: s_cselect_b32 s8, s9, s8
; GFX1100-GISEL-NEXT: s_cmp_lg_u32 s0, 0
; GFX1100-GISEL-NEXT: s_cselect_b32 s9, 0x41b17218, 0
; GFX1100-GISEL-NEXT: s_cmp_lg_u32 s1, 0
+; GFX1100-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
; GFX1100-GISEL-NEXT: s_cselect_b32 s1, 1, 0
; GFX1100-GISEL-NEXT: s_cselect_b32 s0, 1, 0
; GFX1100-GISEL-NEXT: s_lshl_b32 s1, s1, 5
-; GFX1100-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1100-GISEL-NEXT: v_ldexp_f32 v0, s2, s1
+; GFX1100-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_4) | instid1(VALU_DEP_3)
; GFX1100-GISEL-NEXT: v_log_f32_e32 v0, v0
; GFX1100-GISEL-NEXT: s_waitcnt_depctr depctr_va_vdst(0)
; GFX1100-GISEL-NEXT: v_mul_f32_e32 v1, 0x3f317217, v0
; GFX1100-GISEL-NEXT: v_cmp_gt_f32_e64 s1, 0x7f800000, |v0|
; GFX1100-GISEL-NEXT: v_readfirstlane_b32 s2, v0
-; GFX1100-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_2) | instid1(VALU_DEP_2)
; GFX1100-GISEL-NEXT: v_fma_f32 v2, 0x3f317217, v0, -v1
; GFX1100-GISEL-NEXT: s_cmp_lg_u32 s1, 0
; GFX1100-GISEL-NEXT: v_cmp_gt_f32_e64 s1, 0x800000, s3
+; GFX1100-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1100-GISEL-NEXT: v_fmac_f32_e32 v2, 0x3377d1cf, v0
-; GFX1100-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1100-GISEL-NEXT: v_add_f32_e32 v1, v1, v2
+; GFX1100-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
; GFX1100-GISEL-NEXT: v_readfirstlane_b32 s10, v1
; GFX1100-GISEL-NEXT: s_cselect_b32 s2, s10, s2
; GFX1100-GISEL-NEXT: s_cmp_lg_u32 s0, 0
; GFX1100-GISEL-NEXT: s_cselect_b32 s10, 0x41b17218, 0
; GFX1100-GISEL-NEXT: s_cmp_lg_u32 s1, 0
+; GFX1100-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
; GFX1100-GISEL-NEXT: s_cselect_b32 s0, 1, 0
; GFX1100-GISEL-NEXT: s_cselect_b32 s11, 1, 0
; GFX1100-GISEL-NEXT: s_lshl_b32 s0, s0, 5
-; GFX1100-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX1100-GISEL-NEXT: v_ldexp_f32 v0, s3, s0
; GFX1100-GISEL-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
+; GFX1100-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_4) | instid1(VALU_DEP_3)
; GFX1100-GISEL-NEXT: v_log_f32_e32 v0, v0
; GFX1100-GISEL-NEXT: s_waitcnt_depctr depctr_va_vdst(0)
; GFX1100-GISEL-NEXT: v_mul_f32_e32 v1, 0x3f317217, v0
; GFX1100-GISEL-NEXT: v_cmp_gt_f32_e64 s3, 0x7f800000, |v0|
; GFX1100-GISEL-NEXT: v_readfirstlane_b32 s4, v0
-; GFX1100-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX1100-GISEL-NEXT: v_fma_f32 v2, 0x3f317217, v0, -v1
; GFX1100-GISEL-NEXT: s_cmp_lg_u32 s3, 0
+; GFX1100-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX1100-GISEL-NEXT: v_fmac_f32_e32 v2, 0x3377d1cf, v0
; GFX1100-GISEL-NEXT: v_sub_f32_e64 v0, s7, s6
-; GFX1100-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX1100-GISEL-NEXT: v_add_f32_e32 v1, v1, v2
; GFX1100-GISEL-NEXT: v_sub_f32_e64 v2, s2, s10
+; GFX1100-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_3) | instid1(SALU_CYCLE_1)
; GFX1100-GISEL-NEXT: v_readfirstlane_b32 s5, v1
; GFX1100-GISEL-NEXT: v_sub_f32_e64 v1, s8, s9
; GFX1100-GISEL-NEXT: s_cselect_b32 s3, s5, s4
@@ -3148,21 +3156,21 @@ define float @v_log_fabs_f32(float %in) {
; GFX1100-SDAG: ; %bb.0:
; GFX1100-SDAG-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX1100-SDAG-NEXT: v_cmp_gt_f32_e64 s0, 0x800000, |v0|
-; GFX1100-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1100-SDAG-NEXT: v_cndmask_b32_e64 v1, 0, 32, s0
+; GFX1100-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1100-SDAG-NEXT: v_ldexp_f32 v0, |v0|, v1
-; GFX1100-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_3) | instid1(VALU_DEP_2)
; GFX1100-SDAG-NEXT: v_log_f32_e32 v0, v0
; GFX1100-SDAG-NEXT: s_waitcnt_depctr depctr_va_vdst(0)
; GFX1100-SDAG-NEXT: v_mul_f32_e32 v1, 0x3f317217, v0
; GFX1100-SDAG-NEXT: v_cmp_gt_f32_e64 vcc_lo, 0x7f800000, |v0|
+; GFX1100-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1100-SDAG-NEXT: v_fma_f32 v2, 0x3f317217, v0, -v1
-; GFX1100-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1100-SDAG-NEXT: v_fmamk_f32 v2, v0, 0x3377d1cf, v2
+; GFX1100-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1100-SDAG-NEXT: v_add_f32_e32 v1, v1, v2
-; GFX1100-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX1100-SDAG-NEXT: v_cndmask_b32_e32 v0, v0, v1, vcc_lo
; GFX1100-SDAG-NEXT: v_cndmask_b32_e64 v1, 0, 0x41b17218, s0
+; GFX1100-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1100-SDAG-NEXT: v_sub_f32_e32 v0, v0, v1
; GFX1100-SDAG-NEXT: s_setpc_b64 s[30:31]
;
@@ -3170,23 +3178,22 @@ define float @v_log_fabs_f32(float %in) {
; GFX1100-GISEL: ; %bb.0:
; GFX1100-GISEL-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX1100-GISEL-NEXT: v_cmp_gt_f32_e64 s0, 0x800000, |v0|
-; GFX1100-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1100-GISEL-NEXT: v_cndmask_b32_e64 v1, 0, 1, s0
-; GFX1100-GISEL-NEXT: v_lshlrev_b32_e32 v1, 5, v1
; GFX1100-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1100-GISEL-NEXT: v_lshlrev_b32_e32 v1, 5, v1
; GFX1100-GISEL-NEXT: v_ldexp_f32 v0, |v0|, v1
+; GFX1100-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_3) | instid1(VALU_DEP_2)
; GFX1100-GISEL-NEXT: v_log_f32_e32 v0, v0
; GFX1100-GISEL-NEXT: s_waitcnt_depctr depctr_va_vdst(0)
; GFX1100-GISEL-NEXT: v_mul_f32_e32 v1, 0x3f317217, v0
; GFX1100-GISEL-NEXT: v_cmp_gt_f32_e64 vcc_lo, 0x7f800000, |v0|
-; GFX1100-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1100-GISEL-NEXT: v_fma_f32 v2, 0x3f317217, v0, -v1
-; GFX1100-GISEL-NEXT: v_fmac_f32_e32 v2, 0x3377d1cf, v0
; GFX1100-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1100-GISEL-NEXT: v_fmac_f32_e32 v2, 0x3377d1cf, v0
; GFX1100-GISEL-NEXT: v_add_f32_e32 v1, v1, v2
+; GFX1100-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX1100-GISEL-NEXT: v_cndmask_b32_e32 v0, v0, v1, vcc_lo
; GFX1100-GISEL-NEXT: v_cndmask_b32_e64 v1, 0, 0x41b17218, s0
-; GFX1100-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1100-GISEL-NEXT: v_sub_f32_e32 v0, v0, v1
; GFX1100-GISEL-NEXT: s_setpc_b64 s[30:31]
;
@@ -3350,21 +3357,21 @@ define float @v_log_fneg_fabs_f32(float %in) {
; GFX1100-SDAG: ; %bb.0:
; GFX1100-SDAG-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX1100-SDAG-NEXT: v_cmp_lt_f32_e64 s0, 0x80800000, |v0|
-; GFX1100-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1100-SDAG-NEXT: v_cndmask_b32_e64 v1, 0, 32, s0
+; GFX1100-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1100-SDAG-NEXT: v_ldexp_f32 v0, -|v0|, v1
-; GFX1100-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_3) | instid1(VALU_DEP_2)
; GFX1100-SDAG-NEXT: v_log_f32_e32 v0, v0
; GFX1100-SDAG-NEXT: s_waitcnt_depctr depctr_va_vdst(0)
; GFX1100-SDAG-NEXT: v_mul_f32_e32 v1, 0x3f317217, v0
; GFX1100-SDAG-NEXT: v_cmp_gt_f32_e64 vcc_lo, 0x7f800000, |v0|
+; GFX1100-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1100-SDAG-NEXT: v_fma_f32 v2, 0x3f317217, v0, -v1
-; GFX1100-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1100-SDAG-NEXT: v_fmamk_f32 v2, v0, 0x3377d1cf, v2
+; GFX1100-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1100-SDAG-NEXT: v_add_f32_e32 v1, v1, v2
-; GFX1100-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX1100-SDAG-NEXT: v_cndmask_b32_e32 v0, v0, v1, vcc_lo
; GFX1100-SDAG-NEXT: v_cndmask_b32_e64 v1, 0, 0x41b17218, s0
+; GFX1100-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1100-SDAG-NEXT: v_sub_f32_e32 v0, v0, v1
; GFX1100-SDAG-NEXT: s_setpc_b64 s[30:31]
;
@@ -3372,23 +3379,22 @@ define float @v_log_fneg_fabs_f32(float %in) {
; GFX1100-GISEL: ; %bb.0:
; GFX1100-GISEL-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX1100-GISEL-NEXT: v_cmp_gt_f32_e64 s0, 0x800000, -|v0|
-; GFX1100-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1100-GISEL-NEXT: v_cndmask_b32_e64 v1, 0, 1, s0
-; GFX1100-GISEL-NEXT: v_lshlrev_b32_e32 v1, 5, v1
; GFX1100-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1100-GISEL-NEXT: v_lshlrev_b32_e32 v1, 5, v1
; GFX1100-GISEL-NEXT: v_ldexp_f32 v0, -|v0|, v1
+; GFX1100-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_3) | instid1(VALU_DEP_2)
; GFX1100-GISEL-NEXT: v_log_f32_e32 v0, v0
; GFX1100-GISEL-NEXT: s_waitcnt_depctr depctr_va_vdst(0)
; GFX1100-GISEL-NEXT: v_mul_f32_e32 v1, 0x3f317217, v0
; GFX1100-GISEL-NEXT: v_cmp_gt_f32_e64 vcc_lo, 0x7f800000, |v0|
-; GFX1100-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1100-GISEL-NEXT: v_fma_f32 v2, 0x3f317217, v0, -v1
-; GFX1100-GISEL-NEXT: v_fmac_f32_e32 v2, 0x3377d1cf, v0
; GFX1100-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1100-GISEL-NEXT: v_fmac_f32_e32 v2, 0x3377d1cf, v0
; GFX1100-GISEL-NEXT: v_add_f32_e32 v1, v1, v2
+; GFX1100-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX1100-GISEL-NEXT: v_cndmask_b32_e32 v0, v0, v1, vcc_lo
; GFX1100-GISEL-NEXT: v_cndmask_b32_e64 v1, 0, 0x41b17218, s0
-; GFX1100-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1100-GISEL-NEXT: v_sub_f32_e32 v0, v0, v1
; GFX1100-GISEL-NEXT: s_setpc_b64 s[30:31]
;
@@ -3575,23 +3581,22 @@ define float @v_log_fneg_f32(float %in) {
; GFX1100-GISEL: ; %bb.0:
; GFX1100-GISEL-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX1100-GISEL-NEXT: v_cmp_gt_f32_e64 s0, 0x800000, -v0
-; GFX1100-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1100-GISEL-NEXT: v_cndmask_b32_e64 v1, 0, 1, s0
-; GFX1100-GISEL-NEXT: v_lshlrev_b32_e32 v1, 5, v1
; GFX1100-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1100-GISEL-NEXT: v_lshlrev_b32_e32 v1, 5, v1
; GFX1100-GISEL-NEXT: v_ldexp_f32 v0, -v0, v1
+; GFX1100-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_3) | instid1(VALU_DEP_2)
; GFX1100-GISEL-NEXT: v_log_f32_e32 v0, v0
; GFX1100-GISEL-NEXT: s_waitcnt_depctr depctr_va_vdst(0)
; GFX1100-GISEL-NEXT: v_mul_f32_e32 v1, 0x3f317217, v0
; GFX1100-GISEL-NEXT: v_cmp_gt_f32_e64 vcc_lo, 0x7f800000, |v0|
-; GFX1100-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1100-GISEL-NEXT: v_fma_f32 v2, 0x3f317217, v0, -v1
-; GFX1100-GISEL-NEXT: v_fmac_f32_e32 v2, 0x3377d1cf, v0
; GFX1100-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1100-GISEL-NEXT: v_fmac_f32_e32 v2, 0x3377d1cf, v0
; GFX1100-GISEL-NEXT: v_add_f32_e32 v1, v1, v2
+; GFX1100-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX1100-GISEL-NEXT: v_cndmask_b32_e32 v0, v0, v1, vcc_lo
; GFX1100-GISEL-NEXT: v_cndmask_b32_e64 v1, 0, 0x41b17218, s0
-; GFX1100-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1100-GISEL-NEXT: v_sub_f32_e32 v0, v0, v1
; GFX1100-GISEL-NEXT: s_setpc_b64 s[30:31]
;
@@ -4317,11 +4322,10 @@ define float @v_fabs_log_f32_afn(float %in) {
; GFX1100-SDAG: ; %bb.0:
; GFX1100-SDAG-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX1100-SDAG-NEXT: v_cmp_gt_f32_e64 s0, 0x800000, |v0|
-; GFX1100-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX1100-SDAG-NEXT: v_cndmask_b32_e64 v2, 0, 32, s0
; GFX1100-SDAG-NEXT: v_cndmask_b32_e64 v1, 0, 0xc1b17218, s0
+; GFX1100-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1100-SDAG-NEXT: v_ldexp_f32 v0, |v0|, v2
-; GFX1100-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1100-SDAG-NEXT: v_log_f32_e32 v0, v0
; GFX1100-SDAG-NEXT: s_waitcnt_depctr depctr_va_vdst(0)
; GFX1100-SDAG-NEXT: v_fmamk_f32 v0, v0, 0x3f317218, v1
@@ -4331,11 +4335,11 @@ define float @v_fabs_log_f32_afn(float %in) {
; GFX1100-GISEL: ; %bb.0:
; GFX1100-GISEL-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX1100-GISEL-NEXT: v_cmp_gt_f32_e64 s0, 0x800000, |v0|
-; GFX1100-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1100-GISEL-NEXT: v_cndmask_b32_e64 v1, 0, 1, s0
-; GFX1100-GISEL-NEXT: v_lshlrev_b32_e32 v1, 5, v1
; GFX1100-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1100-GISEL-NEXT: v_lshlrev_b32_e32 v1, 5, v1
; GFX1100-GISEL-NEXT: v_ldexp_f32 v0, |v0|, v1
+; GFX1100-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1100-GISEL-NEXT: v_log_f32_e32 v1, v0
; GFX1100-GISEL-NEXT: v_cndmask_b32_e64 v0, 0, 0xc1b17218, s0
; GFX1100-GISEL-NEXT: s_waitcnt_depctr depctr_va_vdst(0)
diff --git a/llvm/test/CodeGen/AMDGPU/llvm.log10.ll b/llvm/test/CodeGen/AMDGPU/llvm.log10.ll
index 8fcc16c2e6f56f..2255abc6e34e60 100644
--- a/llvm/test/CodeGen/AMDGPU/llvm.log10.ll
+++ b/llvm/test/CodeGen/AMDGPU/llvm.log10.ll
@@ -219,23 +219,23 @@ define amdgpu_kernel void @s_log10_f32(ptr addrspace(1) %out, float %in) {
; GFX1100-SDAG-NEXT: s_load_b32 s0, s[4:5], 0x2c
; GFX1100-SDAG-NEXT: s_waitcnt lgkmcnt(0)
; GFX1100-SDAG-NEXT: v_cmp_gt_f32_e64 s1, 0x800000, s0
-; GFX1100-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
; GFX1100-SDAG-NEXT: v_cndmask_b32_e64 v0, 0, 0x411a209b, s1
; GFX1100-SDAG-NEXT: s_and_b32 s1, s1, exec_lo
; GFX1100-SDAG-NEXT: s_cselect_b32 s1, 32, 0
+; GFX1100-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX1100-SDAG-NEXT: v_ldexp_f32 v1, s0, s1
; GFX1100-SDAG-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
-; GFX1100-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_3) | instid1(VALU_DEP_2)
; GFX1100-SDAG-NEXT: v_log_f32_e32 v1, v1
; GFX1100-SDAG-NEXT: s_waitcnt_depctr depctr_va_vdst(0)
; GFX1100-SDAG-NEXT: v_mul_f32_e32 v2, 0x3e9a209a, v1
; GFX1100-SDAG-NEXT: v_cmp_gt_f32_e64 vcc_lo, 0x7f800000, |v1|
+; GFX1100-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1100-SDAG-NEXT: v_fma_f32 v3, 0x3e9a209a, v1, -v2
-; GFX1100-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1100-SDAG-NEXT: v_fmamk_f32 v3, v1, 0x3284fbcf, v3
-; GFX1100-SDAG-NEXT: v_add_f32_e32 v2, v2, v3
; GFX1100-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1100-SDAG-NEXT: v_add_f32_e32 v2, v2, v3
; GFX1100-SDAG-NEXT: v_dual_cndmask_b32 v1, v1, v2 :: v_dual_mov_b32 v2, 0
+; GFX1100-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1100-SDAG-NEXT: v_sub_f32_e32 v0, v1, v0
; GFX1100-SDAG-NEXT: s_waitcnt lgkmcnt(0)
; GFX1100-SDAG-NEXT: global_store_b32 v2, v0, s[0:1]
@@ -247,23 +247,24 @@ define amdgpu_kernel void @s_log10_f32(ptr addrspace(1) %out, float %in) {
; GFX1100-GISEL-NEXT: s_waitcnt lgkmcnt(0)
; GFX1100-GISEL-NEXT: v_cmp_gt_f32_e64 s1, 0x800000, s0
; GFX1100-GISEL-NEXT: s_cmp_lg_u32 s1, 0
+; GFX1100-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
; GFX1100-GISEL-NEXT: s_cselect_b32 s1, 1, 0
; GFX1100-GISEL-NEXT: s_cselect_b32 s2, 1, 0
; GFX1100-GISEL-NEXT: s_lshl_b32 s1, s1, 5
-; GFX1100-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX1100-GISEL-NEXT: v_ldexp_f32 v0, s0, s1
; GFX1100-GISEL-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
+; GFX1100-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_4) | instid1(VALU_DEP_3)
; GFX1100-GISEL-NEXT: v_log_f32_e32 v0, v0
; GFX1100-GISEL-NEXT: s_waitcnt_depctr depctr_va_vdst(0)
; GFX1100-GISEL-NEXT: v_mul_f32_e32 v1, 0x3e9a209a, v0
; GFX1100-GISEL-NEXT: v_cmp_gt_f32_e64 s3, 0x7f800000, |v0|
; GFX1100-GISEL-NEXT: v_readfirstlane_b32 s4, v0
-; GFX1100-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX1100-GISEL-NEXT: v_fma_f32 v2, 0x3e9a209a, v0, -v1
; GFX1100-GISEL-NEXT: s_cmp_lg_u32 s3, 0
-; GFX1100-GISEL-NEXT: v_fmac_f32_e32 v2, 0x3284fbcf, v0
; GFX1100-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1100-GISEL-NEXT: v_fmac_f32_e32 v2, 0x3284fbcf, v0
; GFX1100-GISEL-NEXT: v_add_f32_e32 v1, v1, v2
+; GFX1100-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_3) | instid1(SALU_CYCLE_1)
; GFX1100-GISEL-NEXT: v_readfirstlane_b32 s5, v1
; GFX1100-GISEL-NEXT: v_mov_b32_e32 v1, 0
; GFX1100-GISEL-NEXT: s_cselect_b32 s3, s5, s4
@@ -552,23 +553,23 @@ define amdgpu_kernel void @s_log10_contract_f32(ptr addrspace(1) %out, float %in
; GFX1100-SDAG-NEXT: s_load_b32 s0, s[4:5], 0x2c
; GFX1100-SDAG-NEXT: s_waitcnt lgkmcnt(0)
; GFX1100-SDAG-NEXT: v_cmp_gt_f32_e64 s1, 0x800000, s0
-; GFX1100-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
; GFX1100-SDAG-NEXT: v_cndmask_b32_e64 v0, 0, 0x411a209b, s1
; GFX1100-SDAG-NEXT: s_and_b32 s1, s1, exec_lo
; GFX1100-SDAG-NEXT: s_cselect_b32 s1, 32, 0
+; GFX1100-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX1100-SDAG-NEXT: v_ldexp_f32 v1, s0, s1
; GFX1100-SDAG-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
-; GFX1100-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_3) | instid1(VALU_DEP_2)
; GFX1100-SDAG-NEXT: v_log_f32_e32 v1, v1
; GFX1100-SDAG-NEXT: s_waitcnt_depctr depctr_va_vdst(0)
; GFX1100-SDAG-NEXT: v_mul_f32_e32 v2, 0x3e9a209a, v1
; GFX1100-SDAG-NEXT: v_cmp_gt_f32_e64 vcc_lo, 0x7f800000, |v1|
+; GFX1100-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1100-SDAG-NEXT: v_fma_f32 v3, 0x3e9a209a, v1, -v2
-; GFX1100-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1100-SDAG-NEXT: v_fmamk_f32 v3, v1, 0x3284fbcf, v3
-; GFX1100-SDAG-NEXT: v_add_f32_e32 v2, v2, v3
; GFX1100-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1100-SDAG-NEXT: v_add_f32_e32 v2, v2, v3
; GFX1100-SDAG-NEXT: v_dual_cndmask_b32 v1, v1, v2 :: v_dual_mov_b32 v2, 0
+; GFX1100-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1100-SDAG-NEXT: v_sub_f32_e32 v0, v1, v0
; GFX1100-SDAG-NEXT: s_waitcnt lgkmcnt(0)
; GFX1100-SDAG-NEXT: global_store_b32 v2, v0, s[0:1]
@@ -580,23 +581,24 @@ define amdgpu_kernel void @s_log10_contract_f32(ptr addrspace(1) %out, float %in
; GFX1100-GISEL-NEXT: s_waitcnt lgkmcnt(0)
; GFX1100-GISEL-NEXT: v_cmp_gt_f32_e64 s1, 0x800000, s0
; GFX1100-GISEL-NEXT: s_cmp_lg_u32 s1, 0
+; GFX1100-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
; GFX1100-GISEL-NEXT: s_cselect_b32 s1, 1, 0
; GFX1100-GISEL-NEXT: s_cselect_b32 s2, 1, 0
; GFX1100-GISEL-NEXT: s_lshl_b32 s1, s1, 5
-; GFX1100-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX1100-GISEL-NEXT: v_ldexp_f32 v0, s0, s1
; GFX1100-GISEL-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
+; GFX1100-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_4) | instid1(VALU_DEP_3)
; GFX1100-GISEL-NEXT: v_log_f32_e32 v0, v0
; GFX1100-GISEL-NEXT: s_waitcnt_depctr depctr_va_vdst(0)
; GFX1100-GISEL-NEXT: v_mul_f32_e32 v1, 0x3e9a209a, v0
; GFX1100-GISEL-NEXT: v_cmp_gt_f32_e64 s3, 0x7f800000, |v0|
; GFX1100-GISEL-NEXT: v_readfirstlane_b32 s4, v0
-; GFX1100-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX1100-GISEL-NEXT: v_fma_f32 v2, 0x3e9a209a, v0, -v1
; GFX1100-GISEL-NEXT: s_cmp_lg_u32 s3, 0
-; GFX1100-GISEL-NEXT: v_fmac_f32_e32 v2, 0x3284fbcf, v0
; GFX1100-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1100-GISEL-NEXT: v_fmac_f32_e32 v2, 0x3284fbcf, v0
; GFX1100-GISEL-NEXT: v_add_f32_e32 v1, v1, v2
+; GFX1100-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_3) | instid1(SALU_CYCLE_1)
; GFX1100-GISEL-NEXT: v_readfirstlane_b32 s5, v1
; GFX1100-GISEL-NEXT: v_mov_b32_e32 v1, 0
; GFX1100-GISEL-NEXT: s_cselect_b32 s3, s5, s4
@@ -996,7 +998,6 @@ define amdgpu_kernel void @s_log10_v2f32(ptr addrspace(1) %out, <2 x float> %in)
; GFX1100-SDAG-NEXT: s_waitcnt lgkmcnt(0)
; GFX1100-SDAG-NEXT: v_cmp_gt_f32_e64 s4, 0x800000, s3
; GFX1100-SDAG-NEXT: v_cmp_gt_f32_e64 s5, 0x800000, s2
-; GFX1100-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX1100-SDAG-NEXT: v_cndmask_b32_e64 v0, 0, 0x411a209b, s4
; GFX1100-SDAG-NEXT: s_and_b32 s4, s4, exec_lo
; GFX1100-SDAG-NEXT: s_cselect_b32 s4, 32, 0
@@ -1033,45 +1034,47 @@ define amdgpu_kernel void @s_log10_v2f32(ptr addrspace(1) %out, <2 x float> %in)
; GFX1100-GISEL-NEXT: s_waitcnt lgkmcnt(0)
; GFX1100-GISEL-NEXT: v_cmp_gt_f32_e64 s4, 0x800000, s2
; GFX1100-GISEL-NEXT: s_cmp_lg_u32 s4, 0
+; GFX1100-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
; GFX1100-GISEL-NEXT: s_cselect_b32 s5, 1, 0
; GFX1100-GISEL-NEXT: s_cselect_b32 s4, 1, 0
; GFX1100-GISEL-NEXT: s_lshl_b32 s5, s5, 5
-; GFX1100-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1100-GISEL-NEXT: v_ldexp_f32 v0, s2, s5
+; GFX1100-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_4) | instid1(VALU_DEP_3)
; GFX1100-GISEL-NEXT: v_log_f32_e32 v0, v0
; GFX1100-GISEL-NEXT: s_waitcnt_depctr depctr_va_vdst(0)
; GFX1100-GISEL-NEXT: v_mul_f32_e32 v1, 0x3e9a209a, v0
; GFX1100-GISEL-NEXT: v_cmp_gt_f32_e64 s2, 0x7f800000, |v0|
; GFX1100-GISEL-NEXT: v_readfirstlane_b32 s5, v0
-; GFX1100-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_2) | instid1(VALU_DEP_2)
; GFX1100-GISEL-NEXT: v_fma_f32 v2, 0x3e9a209a, v0, -v1
; GFX1100-GISEL-NEXT: s_cmp_lg_u32 s2, 0
; GFX1100-GISEL-NEXT: v_cmp_gt_f32_e64 s2, 0x800000, s3
+; GFX1100-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1100-GISEL-NEXT: v_fmac_f32_e32 v2, 0x3284fbcf, v0
-; GFX1100-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1100-GISEL-NEXT: v_add_f32_e32 v1, v1, v2
+; GFX1100-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
; GFX1100-GISEL-NEXT: v_readfirstlane_b32 s6, v1
; GFX1100-GISEL-NEXT: s_cselect_b32 s5, s6, s5
; GFX1100-GISEL-NEXT: s_cmp_lg_u32 s4, 0
; GFX1100-GISEL-NEXT: s_cselect_b32 s4, 0x411a209b, 0
; GFX1100-GISEL-NEXT: s_cmp_lg_u32 s2, 0
+; GFX1100-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
; GFX1100-GISEL-NEXT: s_cselect_b32 s6, 1, 0
; GFX1100-GISEL-NEXT: s_cselect_b32 s2, 1, 0
; GFX1100-GISEL-NEXT: s_lshl_b32 s6, s6, 5
-; GFX1100-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1100-GISEL-NEXT: v_ldexp_f32 v0, s3, s6
+; GFX1100-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_4) | instid1(VALU_DEP_3)
; GFX1100-GISEL-NEXT: v_log_f32_e32 v0, v0
; GFX1100-GISEL-NEXT: s_waitcnt_depctr depctr_va_vdst(0)
; GFX1100-GISEL-NEXT: v_mul_f32_e32 v1, 0x3e9a209a, v0
; GFX1100-GISEL-NEXT: v_cmp_gt_f32_e64 s3, 0x7f800000, |v0|
; GFX1100-GISEL-NEXT: v_readfirstlane_b32 s6, v0
-; GFX1100-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX1100-GISEL-NEXT: v_fma_f32 v2, 0x3e9a209a, v0, -v1
; GFX1100-GISEL-NEXT: s_cmp_lg_u32 s3, 0
+; GFX1100-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX1100-GISEL-NEXT: v_fmac_f32_e32 v2, 0x3284fbcf, v0
; GFX1100-GISEL-NEXT: v_sub_f32_e64 v0, s5, s4
-; GFX1100-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1100-GISEL-NEXT: v_dual_add_f32 v1, v1, v2 :: v_dual_mov_b32 v2, 0
+; GFX1100-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
; GFX1100-GISEL-NEXT: v_readfirstlane_b32 s7, v1
; GFX1100-GISEL-NEXT: s_cselect_b32 s3, s7, s6
; GFX1100-GISEL-NEXT: s_cmp_lg_u32 s2, 0
@@ -1633,7 +1636,6 @@ define amdgpu_kernel void @s_log10_v3f32(ptr addrspace(1) %out, <3 x float> %in)
; GFX1100-SDAG-NEXT: v_cmp_gt_f32_e64 s3, 0x800000, s2
; GFX1100-SDAG-NEXT: v_cmp_gt_f32_e64 s6, 0x800000, s1
; GFX1100-SDAG-NEXT: v_cmp_gt_f32_e64 s7, 0x800000, s0
-; GFX1100-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX1100-SDAG-NEXT: v_cndmask_b32_e64 v0, 0, 0x411a209b, s3
; GFX1100-SDAG-NEXT: s_and_b32 s3, s3, exec_lo
; GFX1100-SDAG-NEXT: v_cndmask_b32_e64 v1, 0, 0x411a209b, s6
@@ -1685,68 +1687,71 @@ define amdgpu_kernel void @s_log10_v3f32(ptr addrspace(1) %out, <3 x float> %in)
; GFX1100-GISEL-NEXT: s_waitcnt lgkmcnt(0)
; GFX1100-GISEL-NEXT: v_cmp_gt_f32_e64 s3, 0x800000, s0
; GFX1100-GISEL-NEXT: s_cmp_lg_u32 s3, 0
+; GFX1100-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
; GFX1100-GISEL-NEXT: s_cselect_b32 s6, 1, 0
; GFX1100-GISEL-NEXT: s_cselect_b32 s3, 1, 0
; GFX1100-GISEL-NEXT: s_lshl_b32 s6, s6, 5
-; GFX1100-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1100-GISEL-NEXT: v_ldexp_f32 v0, s0, s6
+; GFX1100-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_4) | instid1(VALU_DEP_3)
; GFX1100-GISEL-NEXT: v_log_f32_e32 v0, v0
; GFX1100-GISEL-NEXT: s_waitcnt_depctr depctr_va_vdst(0)
; GFX1100-GISEL-NEXT: v_mul_f32_e32 v1, 0x3e9a209a, v0
; GFX1100-GISEL-NEXT: v_cmp_gt_f32_e64 s0, 0x7f800000, |v0|
; GFX1100-GISEL-NEXT: v_readfirstlane_b32 s6, v0
-; GFX1100-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_2) | instid1(VALU_DEP_2)
; GFX1100-GISEL-NEXT: v_fma_f32 v2, 0x3e9a209a, v0, -v1
; GFX1100-GISEL-NEXT: s_cmp_lg_u32 s0, 0
; GFX1100-GISEL-NEXT: v_cmp_gt_f32_e64 s0, 0x800000, s1
+; GFX1100-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1100-GISEL-NEXT: v_fmac_f32_e32 v2, 0x3284fbcf, v0
-; GFX1100-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1100-GISEL-NEXT: v_add_f32_e32 v1, v1, v2
+; GFX1100-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
; GFX1100-GISEL-NEXT: v_readfirstlane_b32 s7, v1
; GFX1100-GISEL-NEXT: s_cselect_b32 s6, s7, s6
; GFX1100-GISEL-NEXT: s_cmp_lg_u32 s3, 0
; GFX1100-GISEL-NEXT: s_cselect_b32 s3, 0x411a209b, 0
; GFX1100-GISEL-NEXT: s_cmp_lg_u32 s0, 0
+; GFX1100-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
; GFX1100-GISEL-NEXT: s_cselect_b32 s7, 1, 0
; GFX1100-GISEL-NEXT: s_cselect_b32 s0, 1, 0
; GFX1100-GISEL-NEXT: s_lshl_b32 s7, s7, 5
-; GFX1100-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1100-GISEL-NEXT: v_ldexp_f32 v0, s1, s7
+; GFX1100-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_4) | instid1(VALU_DEP_3)
; GFX1100-GISEL-NEXT: v_log_f32_e32 v0, v0
; GFX1100-GISEL-NEXT: s_waitcnt_depctr depctr_va_vdst(0)
; GFX1100-GISEL-NEXT: v_mul_f32_e32 v1, 0x3e9a209a, v0
; GFX1100-GISEL-NEXT: v_cmp_gt_f32_e64 s1, 0x7f800000, |v0|
; GFX1100-GISEL-NEXT: v_readfirstlane_b32 s7, v0
-; GFX1100-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_2) | instid1(VALU_DEP_2)
; GFX1100-GISEL-NEXT: v_fma_f32 v2, 0x3e9a209a, v0, -v1
; GFX1100-GISEL-NEXT: s_cmp_lg_u32 s1, 0
; GFX1100-GISEL-NEXT: v_cmp_gt_f32_e64 s1, 0x800000, s2
+; GFX1100-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1100-GISEL-NEXT: v_fmac_f32_e32 v2, 0x3284fbcf, v0
-; GFX1100-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1100-GISEL-NEXT: v_add_f32_e32 v1, v1, v2
+; GFX1100-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
; GFX1100-GISEL-NEXT: v_readfirstlane_b32 s8, v1
; GFX1100-GISEL-NEXT: s_cselect_b32 s7, s8, s7
; GFX1100-GISEL-NEXT: s_cmp_lg_u32 s0, 0
; GFX1100-GISEL-NEXT: s_cselect_b32 s8, 0x411a209b, 0
; GFX1100-GISEL-NEXT: s_cmp_lg_u32 s1, 0
+; GFX1100-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
; GFX1100-GISEL-NEXT: s_cselect_b32 s0, 1, 0
; GFX1100-GISEL-NEXT: s_cselect_b32 s9, 1, 0
; GFX1100-GISEL-NEXT: s_lshl_b32 s0, s0, 5
-; GFX1100-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX1100-GISEL-NEXT: v_ldexp_f32 v0, s2, s0
; GFX1100-GISEL-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
+; GFX1100-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_4) | instid1(VALU_DEP_3)
; GFX1100-GISEL-NEXT: v_log_f32_e32 v0, v0
; GFX1100-GISEL-NEXT: s_waitcnt_depctr depctr_va_vdst(0)
; GFX1100-GISEL-NEXT: v_mul_f32_e32 v1, 0x3e9a209a, v0
; GFX1100-GISEL-NEXT: v_cmp_gt_f32_e64 s2, 0x7f800000, |v0|
; GFX1100-GISEL-NEXT: v_readfirstlane_b32 s4, v0
-; GFX1100-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX1100-GISEL-NEXT: v_fma_f32 v2, 0x3e9a209a, v0, -v1
; GFX1100-GISEL-NEXT: s_cmp_lg_u32 s2, 0
+; GFX1100-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX1100-GISEL-NEXT: v_fmac_f32_e32 v2, 0x3284fbcf, v0
; GFX1100-GISEL-NEXT: v_sub_f32_e64 v0, s6, s3
-; GFX1100-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1100-GISEL-NEXT: v_add_f32_e32 v1, v1, v2
+; GFX1100-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_3) | instid1(SALU_CYCLE_1)
; GFX1100-GISEL-NEXT: v_readfirstlane_b32 s5, v1
; GFX1100-GISEL-NEXT: v_sub_f32_e64 v1, s7, s8
; GFX1100-GISEL-NEXT: s_cselect_b32 s2, s5, s4
@@ -2464,7 +2469,6 @@ define amdgpu_kernel void @s_log10_v4f32(ptr addrspace(1) %out, <4 x float> %in)
; GFX1100-SDAG-NEXT: v_cmp_gt_f32_e64 s7, 0x800000, s2
; GFX1100-SDAG-NEXT: v_cmp_gt_f32_e64 s8, 0x800000, s1
; GFX1100-SDAG-NEXT: v_cmp_gt_f32_e64 s9, 0x800000, s0
-; GFX1100-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_4)
; GFX1100-SDAG-NEXT: v_cndmask_b32_e64 v0, 0, 0x411a209b, s6
; GFX1100-SDAG-NEXT: s_and_b32 s6, s6, exec_lo
; GFX1100-SDAG-NEXT: s_cselect_b32 s6, 32, 0
@@ -2524,91 +2528,95 @@ define amdgpu_kernel void @s_log10_v4f32(ptr addrspace(1) %out, <4 x float> %in)
; GFX1100-GISEL-NEXT: s_waitcnt lgkmcnt(0)
; GFX1100-GISEL-NEXT: v_cmp_gt_f32_e64 s6, 0x800000, s0
; GFX1100-GISEL-NEXT: s_cmp_lg_u32 s6, 0
+; GFX1100-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
; GFX1100-GISEL-NEXT: s_cselect_b32 s7, 1, 0
; GFX1100-GISEL-NEXT: s_cselect_b32 s6, 1, 0
; GFX1100-GISEL-NEXT: s_lshl_b32 s7, s7, 5
-; GFX1100-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1100-GISEL-NEXT: v_ldexp_f32 v0, s0, s7
+; GFX1100-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_4) | instid1(VALU_DEP_3)
; GFX1100-GISEL-NEXT: v_log_f32_e32 v0, v0
; GFX1100-GISEL-NEXT: s_waitcnt_depctr depctr_va_vdst(0)
; GFX1100-GISEL-NEXT: v_mul_f32_e32 v1, 0x3e9a209a, v0
; GFX1100-GISEL-NEXT: v_cmp_gt_f32_e64 s0, 0x7f800000, |v0|
; GFX1100-GISEL-NEXT: v_readfirstlane_b32 s7, v0
-; GFX1100-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_2) | instid1(VALU_DEP_2)
; GFX1100-GISEL-NEXT: v_fma_f32 v2, 0x3e9a209a, v0, -v1
; GFX1100-GISEL-NEXT: s_cmp_lg_u32 s0, 0
; GFX1100-GISEL-NEXT: v_cmp_gt_f32_e64 s0, 0x800000, s1
+; GFX1100-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1100-GISEL-NEXT: v_fmac_f32_e32 v2, 0x3284fbcf, v0
-; GFX1100-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1100-GISEL-NEXT: v_add_f32_e32 v1, v1, v2
+; GFX1100-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
; GFX1100-GISEL-NEXT: v_readfirstlane_b32 s8, v1
; GFX1100-GISEL-NEXT: s_cselect_b32 s7, s8, s7
; GFX1100-GISEL-NEXT: s_cmp_lg_u32 s6, 0
; GFX1100-GISEL-NEXT: s_cselect_b32 s6, 0x411a209b, 0
; GFX1100-GISEL-NEXT: s_cmp_lg_u32 s0, 0
+; GFX1100-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
; GFX1100-GISEL-NEXT: s_cselect_b32 s8, 1, 0
; GFX1100-GISEL-NEXT: s_cselect_b32 s0, 1, 0
; GFX1100-GISEL-NEXT: s_lshl_b32 s8, s8, 5
-; GFX1100-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1100-GISEL-NEXT: v_ldexp_f32 v0, s1, s8
+; GFX1100-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_4) | instid1(VALU_DEP_3)
; GFX1100-GISEL-NEXT: v_log_f32_e32 v0, v0
; GFX1100-GISEL-NEXT: s_waitcnt_depctr depctr_va_vdst(0)
; GFX1100-GISEL-NEXT: v_mul_f32_e32 v1, 0x3e9a209a, v0
; GFX1100-GISEL-NEXT: v_cmp_gt_f32_e64 s1, 0x7f800000, |v0|
; GFX1100-GISEL-NEXT: v_readfirstlane_b32 s8, v0
-; GFX1100-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_2) | instid1(VALU_DEP_2)
; GFX1100-GISEL-NEXT: v_fma_f32 v2, 0x3e9a209a, v0, -v1
; GFX1100-GISEL-NEXT: s_cmp_lg_u32 s1, 0
; GFX1100-GISEL-NEXT: v_cmp_gt_f32_e64 s1, 0x800000, s2
+; GFX1100-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1100-GISEL-NEXT: v_fmac_f32_e32 v2, 0x3284fbcf, v0
-; GFX1100-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1100-GISEL-NEXT: v_add_f32_e32 v1, v1, v2
+; GFX1100-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
; GFX1100-GISEL-NEXT: v_readfirstlane_b32 s9, v1
; GFX1100-GISEL-NEXT: s_cselect_b32 s8, s9, s8
; GFX1100-GISEL-NEXT: s_cmp_lg_u32 s0, 0
; GFX1100-GISEL-NEXT: s_cselect_b32 s9, 0x411a209b, 0
; GFX1100-GISEL-NEXT: s_cmp_lg_u32 s1, 0
+; GFX1100-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
; GFX1100-GISEL-NEXT: s_cselect_b32 s1, 1, 0
; GFX1100-GISEL-NEXT: s_cselect_b32 s0, 1, 0
; GFX1100-GISEL-NEXT: s_lshl_b32 s1, s1, 5
-; GFX1100-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1100-GISEL-NEXT: v_ldexp_f32 v0, s2, s1
+; GFX1100-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_4) | instid1(VALU_DEP_3)
; GFX1100-GISEL-NEXT: v_log_f32_e32 v0, v0
; GFX1100-GISEL-NEXT: s_waitcnt_depctr depctr_va_vdst(0)
; GFX1100-GISEL-NEXT: v_mul_f32_e32 v1, 0x3e9a209a, v0
; GFX1100-GISEL-NEXT: v_cmp_gt_f32_e64 s1, 0x7f800000, |v0|
; GFX1100-GISEL-NEXT: v_readfirstlane_b32 s2, v0
-; GFX1100-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_2) | instid1(VALU_DEP_2)
; GFX1100-GISEL-NEXT: v_fma_f32 v2, 0x3e9a209a, v0, -v1
; GFX1100-GISEL-NEXT: s_cmp_lg_u32 s1, 0
; GFX1100-GISEL-NEXT: v_cmp_gt_f32_e64 s1, 0x800000, s3
+; GFX1100-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1100-GISEL-NEXT: v_fmac_f32_e32 v2, 0x3284fbcf, v0
-; GFX1100-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1100-GISEL-NEXT: v_add_f32_e32 v1, v1, v2
+; GFX1100-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
; GFX1100-GISEL-NEXT: v_readfirstlane_b32 s10, v1
; GFX1100-GISEL-NEXT: s_cselect_b32 s2, s10, s2
; GFX1100-GISEL-NEXT: s_cmp_lg_u32 s0, 0
; GFX1100-GISEL-NEXT: s_cselect_b32 s10, 0x411a209b, 0
; GFX1100-GISEL-NEXT: s_cmp_lg_u32 s1, 0
+; GFX1100-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
; GFX1100-GISEL-NEXT: s_cselect_b32 s0, 1, 0
; GFX1100-GISEL-NEXT: s_cselect_b32 s11, 1, 0
; GFX1100-GISEL-NEXT: s_lshl_b32 s0, s0, 5
-; GFX1100-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX1100-GISEL-NEXT: v_ldexp_f32 v0, s3, s0
; GFX1100-GISEL-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
+; GFX1100-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_4) | instid1(VALU_DEP_3)
; GFX1100-GISEL-NEXT: v_log_f32_e32 v0, v0
; GFX1100-GISEL-NEXT: s_waitcnt_depctr depctr_va_vdst(0)
; GFX1100-GISEL-NEXT: v_mul_f32_e32 v1, 0x3e9a209a, v0
; GFX1100-GISEL-NEXT: v_cmp_gt_f32_e64 s3, 0x7f800000, |v0|
; GFX1100-GISEL-NEXT: v_readfirstlane_b32 s4, v0
-; GFX1100-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX1100-GISEL-NEXT: v_fma_f32 v2, 0x3e9a209a, v0, -v1
; GFX1100-GISEL-NEXT: s_cmp_lg_u32 s3, 0
+; GFX1100-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX1100-GISEL-NEXT: v_fmac_f32_e32 v2, 0x3284fbcf, v0
; GFX1100-GISEL-NEXT: v_sub_f32_e64 v0, s7, s6
-; GFX1100-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX1100-GISEL-NEXT: v_add_f32_e32 v1, v1, v2
; GFX1100-GISEL-NEXT: v_sub_f32_e64 v2, s2, s10
+; GFX1100-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_3) | instid1(SALU_CYCLE_1)
; GFX1100-GISEL-NEXT: v_readfirstlane_b32 s5, v1
; GFX1100-GISEL-NEXT: v_sub_f32_e64 v1, s8, s9
; GFX1100-GISEL-NEXT: s_cselect_b32 s3, s5, s4
@@ -3148,21 +3156,21 @@ define float @v_log10_fabs_f32(float %in) {
; GFX1100-SDAG: ; %bb.0:
; GFX1100-SDAG-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX1100-SDAG-NEXT: v_cmp_gt_f32_e64 s0, 0x800000, |v0|
-; GFX1100-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1100-SDAG-NEXT: v_cndmask_b32_e64 v1, 0, 32, s0
+; GFX1100-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1100-SDAG-NEXT: v_ldexp_f32 v0, |v0|, v1
-; GFX1100-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_3) | instid1(VALU_DEP_2)
; GFX1100-SDAG-NEXT: v_log_f32_e32 v0, v0
; GFX1100-SDAG-NEXT: s_waitcnt_depctr depctr_va_vdst(0)
; GFX1100-SDAG-NEXT: v_mul_f32_e32 v1, 0x3e9a209a, v0
; GFX1100-SDAG-NEXT: v_cmp_gt_f32_e64 vcc_lo, 0x7f800000, |v0|
+; GFX1100-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1100-SDAG-NEXT: v_fma_f32 v2, 0x3e9a209a, v0, -v1
-; GFX1100-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1100-SDAG-NEXT: v_fmamk_f32 v2, v0, 0x3284fbcf, v2
+; GFX1100-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1100-SDAG-NEXT: v_add_f32_e32 v1, v1, v2
-; GFX1100-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX1100-SDAG-NEXT: v_cndmask_b32_e32 v0, v0, v1, vcc_lo
; GFX1100-SDAG-NEXT: v_cndmask_b32_e64 v1, 0, 0x411a209b, s0
+; GFX1100-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1100-SDAG-NEXT: v_sub_f32_e32 v0, v0, v1
; GFX1100-SDAG-NEXT: s_setpc_b64 s[30:31]
;
@@ -3170,23 +3178,22 @@ define float @v_log10_fabs_f32(float %in) {
; GFX1100-GISEL: ; %bb.0:
; GFX1100-GISEL-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX1100-GISEL-NEXT: v_cmp_gt_f32_e64 s0, 0x800000, |v0|
-; GFX1100-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1100-GISEL-NEXT: v_cndmask_b32_e64 v1, 0, 1, s0
-; GFX1100-GISEL-NEXT: v_lshlrev_b32_e32 v1, 5, v1
; GFX1100-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1100-GISEL-NEXT: v_lshlrev_b32_e32 v1, 5, v1
; GFX1100-GISEL-NEXT: v_ldexp_f32 v0, |v0|, v1
+; GFX1100-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_3) | instid1(VALU_DEP_2)
; GFX1100-GISEL-NEXT: v_log_f32_e32 v0, v0
; GFX1100-GISEL-NEXT: s_waitcnt_depctr depctr_va_vdst(0)
; GFX1100-GISEL-NEXT: v_mul_f32_e32 v1, 0x3e9a209a, v0
; GFX1100-GISEL-NEXT: v_cmp_gt_f32_e64 vcc_lo, 0x7f800000, |v0|
-; GFX1100-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1100-GISEL-NEXT: v_fma_f32 v2, 0x3e9a209a, v0, -v1
-; GFX1100-GISEL-NEXT: v_fmac_f32_e32 v2, 0x3284fbcf, v0
; GFX1100-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1100-GISEL-NEXT: v_fmac_f32_e32 v2, 0x3284fbcf, v0
; GFX1100-GISEL-NEXT: v_add_f32_e32 v1, v1, v2
+; GFX1100-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX1100-GISEL-NEXT: v_cndmask_b32_e32 v0, v0, v1, vcc_lo
; GFX1100-GISEL-NEXT: v_cndmask_b32_e64 v1, 0, 0x411a209b, s0
-; GFX1100-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1100-GISEL-NEXT: v_sub_f32_e32 v0, v0, v1
; GFX1100-GISEL-NEXT: s_setpc_b64 s[30:31]
;
@@ -3350,21 +3357,21 @@ define float @v_log10_fneg_fabs_f32(float %in) {
; GFX1100-SDAG: ; %bb.0:
; GFX1100-SDAG-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX1100-SDAG-NEXT: v_cmp_lt_f32_e64 s0, 0x80800000, |v0|
-; GFX1100-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1100-SDAG-NEXT: v_cndmask_b32_e64 v1, 0, 32, s0
+; GFX1100-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1100-SDAG-NEXT: v_ldexp_f32 v0, -|v0|, v1
-; GFX1100-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_3) | instid1(VALU_DEP_2)
; GFX1100-SDAG-NEXT: v_log_f32_e32 v0, v0
; GFX1100-SDAG-NEXT: s_waitcnt_depctr depctr_va_vdst(0)
; GFX1100-SDAG-NEXT: v_mul_f32_e32 v1, 0x3e9a209a, v0
; GFX1100-SDAG-NEXT: v_cmp_gt_f32_e64 vcc_lo, 0x7f800000, |v0|
+; GFX1100-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1100-SDAG-NEXT: v_fma_f32 v2, 0x3e9a209a, v0, -v1
-; GFX1100-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1100-SDAG-NEXT: v_fmamk_f32 v2, v0, 0x3284fbcf, v2
+; GFX1100-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1100-SDAG-NEXT: v_add_f32_e32 v1, v1, v2
-; GFX1100-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX1100-SDAG-NEXT: v_cndmask_b32_e32 v0, v0, v1, vcc_lo
; GFX1100-SDAG-NEXT: v_cndmask_b32_e64 v1, 0, 0x411a209b, s0
+; GFX1100-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1100-SDAG-NEXT: v_sub_f32_e32 v0, v0, v1
; GFX1100-SDAG-NEXT: s_setpc_b64 s[30:31]
;
@@ -3372,23 +3379,22 @@ define float @v_log10_fneg_fabs_f32(float %in) {
; GFX1100-GISEL: ; %bb.0:
; GFX1100-GISEL-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX1100-GISEL-NEXT: v_cmp_gt_f32_e64 s0, 0x800000, -|v0|
-; GFX1100-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1100-GISEL-NEXT: v_cndmask_b32_e64 v1, 0, 1, s0
-; GFX1100-GISEL-NEXT: v_lshlrev_b32_e32 v1, 5, v1
; GFX1100-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1100-GISEL-NEXT: v_lshlrev_b32_e32 v1, 5, v1
; GFX1100-GISEL-NEXT: v_ldexp_f32 v0, -|v0|, v1
+; GFX1100-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_3) | instid1(VALU_DEP_2)
; GFX1100-GISEL-NEXT: v_log_f32_e32 v0, v0
; GFX1100-GISEL-NEXT: s_waitcnt_depctr depctr_va_vdst(0)
; GFX1100-GISEL-NEXT: v_mul_f32_e32 v1, 0x3e9a209a, v0
; GFX1100-GISEL-NEXT: v_cmp_gt_f32_e64 vcc_lo, 0x7f800000, |v0|
-; GFX1100-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1100-GISEL-NEXT: v_fma_f32 v2, 0x3e9a209a, v0, -v1
-; GFX1100-GISEL-NEXT: v_fmac_f32_e32 v2, 0x3284fbcf, v0
; GFX1100-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1100-GISEL-NEXT: v_fmac_f32_e32 v2, 0x3284fbcf, v0
; GFX1100-GISEL-NEXT: v_add_f32_e32 v1, v1, v2
+; GFX1100-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX1100-GISEL-NEXT: v_cndmask_b32_e32 v0, v0, v1, vcc_lo
; GFX1100-GISEL-NEXT: v_cndmask_b32_e64 v1, 0, 0x411a209b, s0
-; GFX1100-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1100-GISEL-NEXT: v_sub_f32_e32 v0, v0, v1
; GFX1100-GISEL-NEXT: s_setpc_b64 s[30:31]
;
@@ -3575,23 +3581,22 @@ define float @v_log10_fneg_f32(float %in) {
; GFX1100-GISEL: ; %bb.0:
; GFX1100-GISEL-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX1100-GISEL-NEXT: v_cmp_gt_f32_e64 s0, 0x800000, -v0
-; GFX1100-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1100-GISEL-NEXT: v_cndmask_b32_e64 v1, 0, 1, s0
-; GFX1100-GISEL-NEXT: v_lshlrev_b32_e32 v1, 5, v1
; GFX1100-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1100-GISEL-NEXT: v_lshlrev_b32_e32 v1, 5, v1
; GFX1100-GISEL-NEXT: v_ldexp_f32 v0, -v0, v1
+; GFX1100-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_3) | instid1(VALU_DEP_2)
; GFX1100-GISEL-NEXT: v_log_f32_e32 v0, v0
; GFX1100-GISEL-NEXT: s_waitcnt_depctr depctr_va_vdst(0)
; GFX1100-GISEL-NEXT: v_mul_f32_e32 v1, 0x3e9a209a, v0
; GFX1100-GISEL-NEXT: v_cmp_gt_f32_e64 vcc_lo, 0x7f800000, |v0|
-; GFX1100-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1100-GISEL-NEXT: v_fma_f32 v2, 0x3e9a209a, v0, -v1
-; GFX1100-GISEL-NEXT: v_fmac_f32_e32 v2, 0x3284fbcf, v0
; GFX1100-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1100-GISEL-NEXT: v_fmac_f32_e32 v2, 0x3284fbcf, v0
; GFX1100-GISEL-NEXT: v_add_f32_e32 v1, v1, v2
+; GFX1100-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX1100-GISEL-NEXT: v_cndmask_b32_e32 v0, v0, v1, vcc_lo
; GFX1100-GISEL-NEXT: v_cndmask_b32_e64 v1, 0, 0x411a209b, s0
-; GFX1100-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1100-GISEL-NEXT: v_sub_f32_e32 v0, v0, v1
; GFX1100-GISEL-NEXT: s_setpc_b64 s[30:31]
;
@@ -4317,11 +4322,10 @@ define float @v_fabs_log10_f32_afn(float %in) {
; GFX1100-SDAG: ; %bb.0:
; GFX1100-SDAG-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX1100-SDAG-NEXT: v_cmp_gt_f32_e64 s0, 0x800000, |v0|
-; GFX1100-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX1100-SDAG-NEXT: v_cndmask_b32_e64 v2, 0, 32, s0
; GFX1100-SDAG-NEXT: v_cndmask_b32_e64 v1, 0, 0xc11a209b, s0
+; GFX1100-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1100-SDAG-NEXT: v_ldexp_f32 v0, |v0|, v2
-; GFX1100-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1100-SDAG-NEXT: v_log_f32_e32 v0, v0
; GFX1100-SDAG-NEXT: s_waitcnt_depctr depctr_va_vdst(0)
; GFX1100-SDAG-NEXT: v_fmamk_f32 v0, v0, 0x3e9a209b, v1
@@ -4331,11 +4335,11 @@ define float @v_fabs_log10_f32_afn(float %in) {
; GFX1100-GISEL: ; %bb.0:
; GFX1100-GISEL-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX1100-GISEL-NEXT: v_cmp_gt_f32_e64 s0, 0x800000, |v0|
-; GFX1100-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1100-GISEL-NEXT: v_cndmask_b32_e64 v1, 0, 1, s0
-; GFX1100-GISEL-NEXT: v_lshlrev_b32_e32 v1, 5, v1
; GFX1100-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1100-GISEL-NEXT: v_lshlrev_b32_e32 v1, 5, v1
; GFX1100-GISEL-NEXT: v_ldexp_f32 v0, |v0|, v1
+; GFX1100-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1100-GISEL-NEXT: v_log_f32_e32 v1, v0
; GFX1100-GISEL-NEXT: v_cndmask_b32_e64 v0, 0, 0xc11a209b, s0
; GFX1100-GISEL-NEXT: s_waitcnt_depctr depctr_va_vdst(0)
diff --git a/llvm/test/CodeGen/AMDGPU/llvm.log2.ll b/llvm/test/CodeGen/AMDGPU/llvm.log2.ll
index fc4ac9040e7e03..3c78ce3286448e 100644
--- a/llvm/test/CodeGen/AMDGPU/llvm.log2.ll
+++ b/llvm/test/CodeGen/AMDGPU/llvm.log2.ll
@@ -145,12 +145,12 @@ define amdgpu_kernel void @s_log2_f32(ptr addrspace(1) %out, float %in) {
; GFX1100-SDAG-NEXT: v_mov_b32_e32 v2, 0
; GFX1100-SDAG-NEXT: s_waitcnt lgkmcnt(0)
; GFX1100-SDAG-NEXT: v_cmp_gt_f32_e64 s0, 0x800000, s2
-; GFX1100-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_4) | instid1(VALU_DEP_1)
; GFX1100-SDAG-NEXT: v_cndmask_b32_e64 v0, 0, 0x42000000, s0
; GFX1100-SDAG-NEXT: s_and_b32 s0, s0, exec_lo
; GFX1100-SDAG-NEXT: s_cselect_b32 s3, 32, 0
; GFX1100-SDAG-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
; GFX1100-SDAG-NEXT: v_ldexp_f32 v1, s2, s3
+; GFX1100-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1100-SDAG-NEXT: v_log_f32_e32 v1, v1
; GFX1100-SDAG-NEXT: s_waitcnt_depctr depctr_va_vdst(0)
; GFX1100-SDAG-NEXT: v_sub_f32_e32 v0, v1, v0
@@ -165,6 +165,7 @@ define amdgpu_kernel void @s_log2_f32(ptr addrspace(1) %out, float %in) {
; GFX1100-GISEL-NEXT: s_waitcnt lgkmcnt(0)
; GFX1100-GISEL-NEXT: v_cmp_gt_f32_e64 s1, 0x800000, s0
; GFX1100-GISEL-NEXT: s_cmp_lg_u32 s1, 0
+; GFX1100-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1100-GISEL-NEXT: s_cselect_b32 s1, 1, 0
; GFX1100-GISEL-NEXT: s_cselect_b32 s2, 1, 0
; GFX1100-GISEL-NEXT: s_lshl_b32 s1, s1, 5
@@ -409,7 +410,6 @@ define amdgpu_kernel void @s_log2_v2f32(ptr addrspace(1) %out, <2 x float> %in)
; GFX1100-SDAG-NEXT: s_waitcnt lgkmcnt(0)
; GFX1100-SDAG-NEXT: v_cmp_gt_f32_e64 s4, 0x800000, s3
; GFX1100-SDAG-NEXT: v_cmp_gt_f32_e64 s5, 0x800000, s2
-; GFX1100-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX1100-SDAG-NEXT: v_cndmask_b32_e64 v0, 0, 0x42000000, s4
; GFX1100-SDAG-NEXT: s_and_b32 s4, s4, exec_lo
; GFX1100-SDAG-NEXT: v_cndmask_b32_e64 v2, 0, 0x42000000, s5
@@ -435,6 +435,7 @@ define amdgpu_kernel void @s_log2_v2f32(ptr addrspace(1) %out, <2 x float> %in)
; GFX1100-GISEL-NEXT: v_cmp_gt_f32_e64 s4, 0x800000, s2
; GFX1100-GISEL-NEXT: v_cmp_gt_f32_e64 s5, 0x800000, s3
; GFX1100-GISEL-NEXT: s_cmp_lg_u32 s4, 0
+; GFX1100-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1100-GISEL-NEXT: s_cselect_b32 s6, 1, 0
; GFX1100-GISEL-NEXT: s_cselect_b32 s4, 1, 0
; GFX1100-GISEL-NEXT: s_lshl_b32 s6, s6, 5
@@ -442,14 +443,15 @@ define amdgpu_kernel void @s_log2_v2f32(ptr addrspace(1) %out, <2 x float> %in)
; GFX1100-GISEL-NEXT: v_ldexp_f32 v0, s2, s6
; GFX1100-GISEL-NEXT: s_cselect_b32 s4, 0x42000000, 0
; GFX1100-GISEL-NEXT: s_cmp_lg_u32 s5, 0
+; GFX1100-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1100-GISEL-NEXT: s_cselect_b32 s7, 1, 0
; GFX1100-GISEL-NEXT: s_cselect_b32 s5, 1, 0
; GFX1100-GISEL-NEXT: s_lshl_b32 s7, s7, 5
; GFX1100-GISEL-NEXT: v_log_f32_e32 v0, v0
; GFX1100-GISEL-NEXT: v_ldexp_f32 v1, s3, s7
; GFX1100-GISEL-NEXT: s_cmp_lg_u32 s5, 0
+; GFX1100-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1100-GISEL-NEXT: s_cselect_b32 s2, 0x42000000, 0
-; GFX1100-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1100-GISEL-NEXT: v_log_f32_e32 v1, v1
; GFX1100-GISEL-NEXT: s_waitcnt_depctr depctr_va_vdst(0)
; GFX1100-GISEL-NEXT: v_dual_subrev_f32 v0, s4, v0 :: v_dual_subrev_f32 v1, s2, v1
@@ -769,7 +771,6 @@ define amdgpu_kernel void @s_log2_v3f32(ptr addrspace(1) %out, <3 x float> %in)
; GFX1100-SDAG-NEXT: v_cmp_gt_f32_e64 s3, 0x800000, s2
; GFX1100-SDAG-NEXT: v_cmp_gt_f32_e64 s6, 0x800000, s1
; GFX1100-SDAG-NEXT: v_cmp_gt_f32_e64 s7, 0x800000, s0
-; GFX1100-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX1100-SDAG-NEXT: v_cndmask_b32_e64 v0, 0, 0x42000000, s3
; GFX1100-SDAG-NEXT: s_and_b32 s3, s3, exec_lo
; GFX1100-SDAG-NEXT: v_cndmask_b32_e64 v1, 0, 0x42000000, s6
@@ -803,6 +804,7 @@ define amdgpu_kernel void @s_log2_v3f32(ptr addrspace(1) %out, <3 x float> %in)
; GFX1100-GISEL-NEXT: v_cmp_gt_f32_e64 s6, 0x800000, s1
; GFX1100-GISEL-NEXT: v_cmp_gt_f32_e64 s7, 0x800000, s2
; GFX1100-GISEL-NEXT: s_cmp_lg_u32 s3, 0
+; GFX1100-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1100-GISEL-NEXT: s_cselect_b32 s8, 1, 0
; GFX1100-GISEL-NEXT: s_cselect_b32 s3, 1, 0
; GFX1100-GISEL-NEXT: s_lshl_b32 s8, s8, 5
@@ -810,6 +812,7 @@ define amdgpu_kernel void @s_log2_v3f32(ptr addrspace(1) %out, <3 x float> %in)
; GFX1100-GISEL-NEXT: v_ldexp_f32 v0, s0, s8
; GFX1100-GISEL-NEXT: s_cselect_b32 s3, 0x42000000, 0
; GFX1100-GISEL-NEXT: s_cmp_lg_u32 s6, 0
+; GFX1100-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1100-GISEL-NEXT: s_cselect_b32 s0, 1, 0
; GFX1100-GISEL-NEXT: s_cselect_b32 s6, 1, 0
; GFX1100-GISEL-NEXT: s_lshl_b32 s0, s0, 5
@@ -825,8 +828,8 @@ define amdgpu_kernel void @s_log2_v3f32(ptr addrspace(1) %out, <3 x float> %in)
; GFX1100-GISEL-NEXT: v_ldexp_f32 v2, s2, s8
; GFX1100-GISEL-NEXT: v_log_f32_e32 v1, v1
; GFX1100-GISEL-NEXT: s_cmp_lg_u32 s7, 0
+; GFX1100-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1100-GISEL-NEXT: s_cselect_b32 s2, 0x42000000, 0
-; GFX1100-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1100-GISEL-NEXT: v_log_f32_e32 v2, v2
; GFX1100-GISEL-NEXT: s_waitcnt_depctr depctr_va_vdst(0)
; GFX1100-GISEL-NEXT: v_dual_subrev_f32 v0, s3, v0 :: v_dual_subrev_f32 v1, s6, v1
@@ -1227,7 +1230,6 @@ define amdgpu_kernel void @s_log2_v4f32(ptr addrspace(1) %out, <4 x float> %in)
; GFX1100-SDAG-NEXT: v_cmp_gt_f32_e64 s7, 0x800000, s2
; GFX1100-SDAG-NEXT: v_cmp_gt_f32_e64 s8, 0x800000, s1
; GFX1100-SDAG-NEXT: v_cmp_gt_f32_e64 s9, 0x800000, s0
-; GFX1100-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_4)
; GFX1100-SDAG-NEXT: v_cndmask_b32_e64 v0, 0, 0x42000000, s6
; GFX1100-SDAG-NEXT: s_and_b32 s6, s6, exec_lo
; GFX1100-SDAG-NEXT: v_cndmask_b32_e64 v1, 0, 0x42000000, s7
@@ -1265,6 +1267,7 @@ define amdgpu_kernel void @s_log2_v4f32(ptr addrspace(1) %out, <4 x float> %in)
; GFX1100-GISEL-NEXT: v_cmp_gt_f32_e64 s7, 0x800000, s1
; GFX1100-GISEL-NEXT: v_cmp_gt_f32_e64 s9, 0x800000, s2
; GFX1100-GISEL-NEXT: s_cmp_lg_u32 s6, 0
+; GFX1100-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1100-GISEL-NEXT: s_cselect_b32 s8, 1, 0
; GFX1100-GISEL-NEXT: s_cselect_b32 s6, 1, 0
; GFX1100-GISEL-NEXT: s_lshl_b32 s8, s8, 5
@@ -1613,11 +1616,10 @@ define float @v_log2_fabs_f32(float %in) {
; GFX1100-SDAG: ; %bb.0:
; GFX1100-SDAG-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX1100-SDAG-NEXT: v_cmp_gt_f32_e64 s0, 0x800000, |v0|
-; GFX1100-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX1100-SDAG-NEXT: v_cndmask_b32_e64 v2, 0, 32, s0
; GFX1100-SDAG-NEXT: v_cndmask_b32_e64 v1, 0, 0x42000000, s0
+; GFX1100-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1100-SDAG-NEXT: v_ldexp_f32 v0, |v0|, v2
-; GFX1100-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1100-SDAG-NEXT: v_log_f32_e32 v0, v0
; GFX1100-SDAG-NEXT: s_waitcnt_depctr depctr_va_vdst(0)
; GFX1100-SDAG-NEXT: v_sub_f32_e32 v0, v0, v1
@@ -1627,12 +1629,12 @@ define float @v_log2_fabs_f32(float %in) {
; GFX1100-GISEL: ; %bb.0:
; GFX1100-GISEL-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX1100-GISEL-NEXT: v_cmp_gt_f32_e64 s0, 0x800000, |v0|
-; GFX1100-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1100-GISEL-NEXT: v_cndmask_b32_e64 v1, 0, 1, s0
+; GFX1100-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1100-GISEL-NEXT: v_lshlrev_b32_e32 v1, 5, v1
-; GFX1100-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX1100-GISEL-NEXT: v_ldexp_f32 v0, |v0|, v1
; GFX1100-GISEL-NEXT: v_cndmask_b32_e64 v1, 0, 0x42000000, s0
+; GFX1100-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX1100-GISEL-NEXT: v_log_f32_e32 v0, v0
; GFX1100-GISEL-NEXT: s_waitcnt_depctr depctr_va_vdst(0)
; GFX1100-GISEL-NEXT: v_sub_f32_e32 v0, v0, v1
@@ -1738,11 +1740,10 @@ define float @v_log2_fneg_fabs_f32(float %in) {
; GFX1100-SDAG: ; %bb.0:
; GFX1100-SDAG-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX1100-SDAG-NEXT: v_cmp_lt_f32_e64 s0, 0x80800000, |v0|
-; GFX1100-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX1100-SDAG-NEXT: v_cndmask_b32_e64 v2, 0, 32, s0
; GFX1100-SDAG-NEXT: v_cndmask_b32_e64 v1, 0, 0x42000000, s0
+; GFX1100-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1100-SDAG-NEXT: v_ldexp_f32 v0, -|v0|, v2
-; GFX1100-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1100-SDAG-NEXT: v_log_f32_e32 v0, v0
; GFX1100-SDAG-NEXT: s_waitcnt_depctr depctr_va_vdst(0)
; GFX1100-SDAG-NEXT: v_sub_f32_e32 v0, v0, v1
@@ -1752,12 +1753,12 @@ define float @v_log2_fneg_fabs_f32(float %in) {
; GFX1100-GISEL: ; %bb.0:
; GFX1100-GISEL-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX1100-GISEL-NEXT: v_cmp_gt_f32_e64 s0, 0x800000, -|v0|
-; GFX1100-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1100-GISEL-NEXT: v_cndmask_b32_e64 v1, 0, 1, s0
+; GFX1100-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1100-GISEL-NEXT: v_lshlrev_b32_e32 v1, 5, v1
-; GFX1100-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX1100-GISEL-NEXT: v_ldexp_f32 v0, -|v0|, v1
; GFX1100-GISEL-NEXT: v_cndmask_b32_e64 v1, 0, 0x42000000, s0
+; GFX1100-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX1100-GISEL-NEXT: v_log_f32_e32 v0, v0
; GFX1100-GISEL-NEXT: s_waitcnt_depctr depctr_va_vdst(0)
; GFX1100-GISEL-NEXT: v_sub_f32_e32 v0, v0, v1
@@ -1877,12 +1878,12 @@ define float @v_log2_fneg_f32(float %in) {
; GFX1100-GISEL: ; %bb.0:
; GFX1100-GISEL-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX1100-GISEL-NEXT: v_cmp_gt_f32_e64 s0, 0x800000, -v0
-; GFX1100-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1100-GISEL-NEXT: v_cndmask_b32_e64 v1, 0, 1, s0
+; GFX1100-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1100-GISEL-NEXT: v_lshlrev_b32_e32 v1, 5, v1
-; GFX1100-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX1100-GISEL-NEXT: v_ldexp_f32 v0, -v0, v1
; GFX1100-GISEL-NEXT: v_cndmask_b32_e64 v1, 0, 0x42000000, s0
+; GFX1100-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX1100-GISEL-NEXT: v_log_f32_e32 v0, v0
; GFX1100-GISEL-NEXT: s_waitcnt_depctr depctr_va_vdst(0)
; GFX1100-GISEL-NEXT: v_sub_f32_e32 v0, v0, v1
@@ -2629,11 +2630,10 @@ define float @v_fabs_log2_f32_afn(float %in) {
; GFX1100-SDAG: ; %bb.0:
; GFX1100-SDAG-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX1100-SDAG-NEXT: v_cmp_gt_f32_e64 s0, 0x800000, |v0|
-; GFX1100-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX1100-SDAG-NEXT: v_cndmask_b32_e64 v2, 0, 32, s0
; GFX1100-SDAG-NEXT: v_cndmask_b32_e64 v1, 0, 0x42000000, s0
+; GFX1100-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1100-SDAG-NEXT: v_ldexp_f32 v0, |v0|, v2
-; GFX1100-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1100-SDAG-NEXT: v_log_f32_e32 v0, v0
; GFX1100-SDAG-NEXT: s_waitcnt_depctr depctr_va_vdst(0)
; GFX1100-SDAG-NEXT: v_sub_f32_e32 v0, v0, v1
@@ -2643,12 +2643,12 @@ define float @v_fabs_log2_f32_afn(float %in) {
; GFX1100-GISEL: ; %bb.0:
; GFX1100-GISEL-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX1100-GISEL-NEXT: v_cmp_gt_f32_e64 s0, 0x800000, |v0|
-; GFX1100-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1100-GISEL-NEXT: v_cndmask_b32_e64 v1, 0, 1, s0
+; GFX1100-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1100-GISEL-NEXT: v_lshlrev_b32_e32 v1, 5, v1
-; GFX1100-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX1100-GISEL-NEXT: v_ldexp_f32 v0, |v0|, v1
; GFX1100-GISEL-NEXT: v_cndmask_b32_e64 v1, 0, 0x42000000, s0
+; GFX1100-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX1100-GISEL-NEXT: v_log_f32_e32 v0, v0
; GFX1100-GISEL-NEXT: s_waitcnt_depctr depctr_va_vdst(0)
; GFX1100-GISEL-NEXT: v_sub_f32_e32 v0, v0, v1
diff --git a/llvm/test/CodeGen/AMDGPU/llvm.maximum.f16.ll b/llvm/test/CodeGen/AMDGPU/llvm.maximum.f16.ll
index 50430cac9ee3ba..4c69b163a337d8 100644
--- a/llvm/test/CodeGen/AMDGPU/llvm.maximum.f16.ll
+++ b/llvm/test/CodeGen/AMDGPU/llvm.maximum.f16.ll
@@ -1195,8 +1195,8 @@ define void @s_maximum_v2f16(<2 x half> inreg %src0, <2 x half> inreg %src1) {
; GFX11-TRUE16-NEXT: v_cmp_o_f16_e64 s0, s0, s1
; GFX11-TRUE16-NEXT: v_cmp_o_f16_e64 s1, s3, s2
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v1, 16, v0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-TRUE16-NEXT: v_cndmask_b16 v0.l, 0x7e00, v0.l, s0
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX11-TRUE16-NEXT: v_cndmask_b16 v0.h, 0x7e00, v1.l, s1
; GFX11-TRUE16-NEXT: ;;#ASMSTART
; GFX11-TRUE16-NEXT: ; use v0
@@ -1346,10 +1346,9 @@ define <3 x half> @v_maximum_v3f16(<3 x half> %src0, <3 x half> %src1) {
; GFX11-TRUE16-NEXT: v_pk_max_f16 v1, v1, v3
; GFX11-TRUE16-NEXT: v_cmp_o_f16_e64 s1, v5.l, v4.l
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v2, 16, v6
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
; GFX11-TRUE16-NEXT: v_cndmask_b16 v0.l, 0x7e00, v6.l, s0
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_3)
; GFX11-TRUE16-NEXT: v_cndmask_b16 v1.l, 0x7e00, v1.l, vcc_lo
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX11-TRUE16-NEXT: v_cndmask_b16 v0.h, 0x7e00, v2.l, s1
; GFX11-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -1563,10 +1562,9 @@ define <3 x half> @v_maximum_v3f16__nsz(<3 x half> %src0, <3 x half> %src1) {
; GFX11-TRUE16-NEXT: v_pk_max_f16 v1, v1, v3
; GFX11-TRUE16-NEXT: v_cmp_o_f16_e64 s1, v5.l, v4.l
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v2, 16, v6
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
; GFX11-TRUE16-NEXT: v_cndmask_b16 v0.l, 0x7e00, v6.l, s0
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_3)
; GFX11-TRUE16-NEXT: v_cndmask_b16 v1.l, 0x7e00, v1.l, vcc_lo
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX11-TRUE16-NEXT: v_cndmask_b16 v0.h, 0x7e00, v2.l, s1
; GFX11-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
diff --git a/llvm/test/CodeGen/AMDGPU/llvm.maximum.f64.ll b/llvm/test/CodeGen/AMDGPU/llvm.maximum.f64.ll
index 3ed09b1589fb0d..80d7f398543dbb 100644
--- a/llvm/test/CodeGen/AMDGPU/llvm.maximum.f64.ll
+++ b/llvm/test/CodeGen/AMDGPU/llvm.maximum.f64.ll
@@ -530,7 +530,7 @@ define void @s_maximum_f64(double inreg %src0, double inreg %src1) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: v_max_f64 v[0:1], s[0:1], s[2:3]
; GFX11-NEXT: v_cmp_u_f64_e64 s0, s[0:1], s[2:3]
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_3)
; GFX11-NEXT: v_cndmask_b32_e64 v1, v1, 0x7ff80000, s0
; GFX11-NEXT: v_cndmask_b32_e64 v0, v0, 0, s0
; GFX11-NEXT: ;;#ASMSTART
@@ -643,7 +643,7 @@ define <2 x double> @v_maximum_v2f64(<2 x double> %src0, <2 x double> %src1) #0
; GFX11-NEXT: v_cmp_u_f64_e32 vcc_lo, v[0:1], v[4:5]
; GFX11-NEXT: v_max_f64 v[4:5], v[2:3], v[6:7]
; GFX11-NEXT: v_cmp_u_f64_e64 s0, v[2:3], v[6:7]
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(SKIP_1) | instid1(VALU_DEP_3)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(SKIP_1) | instid1(VALU_DEP_4)
; GFX11-NEXT: v_cndmask_b32_e64 v0, v8, 0, vcc_lo
; GFX11-NEXT: v_cndmask_b32_e64 v1, v9, 0x7ff80000, vcc_lo
; GFX11-NEXT: v_cndmask_b32_e64 v2, v4, 0, s0
@@ -807,7 +807,7 @@ define <2 x double> @v_maximum_v2f64__nsz(<2 x double> %src0, <2 x double> %src1
; GFX11-NEXT: v_cmp_u_f64_e32 vcc_lo, v[0:1], v[4:5]
; GFX11-NEXT: v_max_f64 v[4:5], v[2:3], v[6:7]
; GFX11-NEXT: v_cmp_u_f64_e64 s0, v[2:3], v[6:7]
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(SKIP_1) | instid1(VALU_DEP_3)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(SKIP_1) | instid1(VALU_DEP_4)
; GFX11-NEXT: v_cndmask_b32_e64 v0, v8, 0, vcc_lo
; GFX11-NEXT: v_cndmask_b32_e64 v1, v9, 0x7ff80000, vcc_lo
; GFX11-NEXT: v_cndmask_b32_e64 v2, v4, 0, s0
@@ -999,7 +999,7 @@ define void @s_maximum_v2f64(<2 x double> inreg %src0, <2 x double> inreg %src1)
; GFX11-NEXT: v_cmp_u_f64_e64 s2, s[2:3], s[18:19]
; GFX11-NEXT: v_max_f64 v[4:5], s[0:1], s[16:17]
; GFX11-NEXT: v_cmp_u_f64_e64 s0, s[0:1], s[16:17]
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_1) | instid1(VALU_DEP_3)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(SKIP_1) | instid1(VALU_DEP_4)
; GFX11-NEXT: v_cndmask_b32_e64 v3, v1, 0x7ff80000, s2
; GFX11-NEXT: v_cndmask_b32_e64 v2, v0, 0, s2
; GFX11-NEXT: v_cndmask_b32_e64 v1, v5, 0x7ff80000, s0
@@ -2132,7 +2132,7 @@ define <8 x double> @v_maximum_v8f64(<8 x double> %src0, <8 x double> %src1) #0
; GFX11-NEXT: s_waitcnt vmcnt(0)
; GFX11-NEXT: v_max_f64 v[28:29], v[14:15], v[30:31]
; GFX11-NEXT: v_cmp_u_f64_e64 s6, v[14:15], v[30:31]
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_3)
; GFX11-NEXT: v_cndmask_b32_e64 v14, v28, 0, s6
; GFX11-NEXT: v_cndmask_b32_e64 v15, v29, 0x7ff80000, s6
; GFX11-NEXT: s_setpc_b64 s[30:31]
diff --git a/llvm/test/CodeGen/AMDGPU/llvm.minimum.f16.ll b/llvm/test/CodeGen/AMDGPU/llvm.minimum.f16.ll
index 963d4fc819a2a5..aa21d3ebf9f5a2 100644
--- a/llvm/test/CodeGen/AMDGPU/llvm.minimum.f16.ll
+++ b/llvm/test/CodeGen/AMDGPU/llvm.minimum.f16.ll
@@ -1005,8 +1005,8 @@ define void @s_minimum_v2f16(<2 x half> inreg %src0, <2 x half> inreg %src1) {
; GFX11-TRUE16-NEXT: v_cmp_o_f16_e64 s0, s0, s1
; GFX11-TRUE16-NEXT: v_cmp_o_f16_e64 s1, s3, s2
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v1, 16, v0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-TRUE16-NEXT: v_cndmask_b16 v0.l, 0x7e00, v0.l, s0
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX11-TRUE16-NEXT: v_cndmask_b16 v0.h, 0x7e00, v1.l, s1
; GFX11-TRUE16-NEXT: ;;#ASMSTART
; GFX11-TRUE16-NEXT: ; use v0
@@ -1128,10 +1128,9 @@ define <3 x half> @v_minimum_v3f16(<3 x half> %src0, <3 x half> %src1) {
; GFX11-TRUE16-NEXT: v_pk_min_f16 v1, v1, v3
; GFX11-TRUE16-NEXT: v_cmp_o_f16_e64 s1, v5.l, v4.l
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v2, 16, v6
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
; GFX11-TRUE16-NEXT: v_cndmask_b16 v0.l, 0x7e00, v6.l, s0
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_3)
; GFX11-TRUE16-NEXT: v_cndmask_b16 v1.l, 0x7e00, v1.l, vcc_lo
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX11-TRUE16-NEXT: v_cndmask_b16 v0.h, 0x7e00, v2.l, s1
; GFX11-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -1296,10 +1295,9 @@ define <3 x half> @v_minimum_v3f16__nsz(<3 x half> %src0, <3 x half> %src1) {
; GFX11-TRUE16-NEXT: v_pk_min_f16 v1, v1, v3
; GFX11-TRUE16-NEXT: v_cmp_o_f16_e64 s1, v5.l, v4.l
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v2, 16, v6
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
; GFX11-TRUE16-NEXT: v_cndmask_b16 v0.l, 0x7e00, v6.l, s0
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_3)
; GFX11-TRUE16-NEXT: v_cndmask_b16 v1.l, 0x7e00, v1.l, vcc_lo
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX11-TRUE16-NEXT: v_cndmask_b16 v0.h, 0x7e00, v2.l, s1
; GFX11-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
diff --git a/llvm/test/CodeGen/AMDGPU/llvm.minimum.f64.ll b/llvm/test/CodeGen/AMDGPU/llvm.minimum.f64.ll
index 2f18ad05ad3e23..778b68bfdf54c7 100644
--- a/llvm/test/CodeGen/AMDGPU/llvm.minimum.f64.ll
+++ b/llvm/test/CodeGen/AMDGPU/llvm.minimum.f64.ll
@@ -530,7 +530,7 @@ define void @s_minimum_f64(double inreg %src0, double inreg %src1) #0 {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: v_min_f64 v[0:1], s[0:1], s[2:3]
; GFX11-NEXT: v_cmp_u_f64_e64 s0, s[0:1], s[2:3]
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_3)
; GFX11-NEXT: v_cndmask_b32_e64 v1, v1, 0x7ff80000, s0
; GFX11-NEXT: v_cndmask_b32_e64 v0, v0, 0, s0
; GFX11-NEXT: ;;#ASMSTART
@@ -643,7 +643,7 @@ define <2 x double> @v_minimum_v2f64(<2 x double> %src0, <2 x double> %src1) #0
; GFX11-NEXT: v_cmp_u_f64_e32 vcc_lo, v[0:1], v[4:5]
; GFX11-NEXT: v_min_f64 v[4:5], v[2:3], v[6:7]
; GFX11-NEXT: v_cmp_u_f64_e64 s0, v[2:3], v[6:7]
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(SKIP_1) | instid1(VALU_DEP_3)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(SKIP_1) | instid1(VALU_DEP_4)
; GFX11-NEXT: v_cndmask_b32_e64 v0, v8, 0, vcc_lo
; GFX11-NEXT: v_cndmask_b32_e64 v1, v9, 0x7ff80000, vcc_lo
; GFX11-NEXT: v_cndmask_b32_e64 v2, v4, 0, s0
@@ -807,7 +807,7 @@ define <2 x double> @v_minimum_v2f64__nsz(<2 x double> %src0, <2 x double> %src1
; GFX11-NEXT: v_cmp_u_f64_e32 vcc_lo, v[0:1], v[4:5]
; GFX11-NEXT: v_min_f64 v[4:5], v[2:3], v[6:7]
; GFX11-NEXT: v_cmp_u_f64_e64 s0, v[2:3], v[6:7]
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(SKIP_1) | instid1(VALU_DEP_3)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(SKIP_1) | instid1(VALU_DEP_4)
; GFX11-NEXT: v_cndmask_b32_e64 v0, v8, 0, vcc_lo
; GFX11-NEXT: v_cndmask_b32_e64 v1, v9, 0x7ff80000, vcc_lo
; GFX11-NEXT: v_cndmask_b32_e64 v2, v4, 0, s0
@@ -999,7 +999,7 @@ define void @s_minimum_v2f64(<2 x double> inreg %src0, <2 x double> inreg %src1)
; GFX11-NEXT: v_cmp_u_f64_e64 s2, s[2:3], s[18:19]
; GFX11-NEXT: v_min_f64 v[4:5], s[0:1], s[16:17]
; GFX11-NEXT: v_cmp_u_f64_e64 s0, s[0:1], s[16:17]
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_1) | instid1(VALU_DEP_3)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(SKIP_1) | instid1(VALU_DEP_4)
; GFX11-NEXT: v_cndmask_b32_e64 v3, v1, 0x7ff80000, s2
; GFX11-NEXT: v_cndmask_b32_e64 v2, v0, 0, s2
; GFX11-NEXT: v_cndmask_b32_e64 v1, v5, 0x7ff80000, s0
@@ -2132,7 +2132,7 @@ define <8 x double> @v_minimum_v8f64(<8 x double> %src0, <8 x double> %src1) #0
; GFX11-NEXT: s_waitcnt vmcnt(0)
; GFX11-NEXT: v_min_f64 v[28:29], v[14:15], v[30:31]
; GFX11-NEXT: v_cmp_u_f64_e64 s6, v[14:15], v[30:31]
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_3)
; GFX11-NEXT: v_cndmask_b32_e64 v14, v28, 0, s6
; GFX11-NEXT: v_cndmask_b32_e64 v15, v29, 0x7ff80000, s6
; GFX11-NEXT: s_setpc_b64 s[30:31]
diff --git a/llvm/test/CodeGen/AMDGPU/llvm.mulo.ll b/llvm/test/CodeGen/AMDGPU/llvm.mulo.ll
index de9e2e77a1f0bc..ecd706cdc83cbc 100644
--- a/llvm/test/CodeGen/AMDGPU/llvm.mulo.ll
+++ b/llvm/test/CodeGen/AMDGPU/llvm.mulo.ll
@@ -78,18 +78,18 @@ define { i64, i1 } @umulo_i64_v_v(i64 %x, i64 %y) {
; GFX11-NEXT: v_mad_u64_u32 v[0:1], null, v4, v2, 0
; GFX11-NEXT: v_mad_u64_u32 v[8:9], null, v5, v2, 0
; GFX11-NEXT: v_mad_u64_u32 v[10:11], null, v5, v3, 0
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_1) | instid1(VALU_DEP_4)
; GFX11-NEXT: v_add_co_u32 v2, vcc_lo, v1, v6
; GFX11-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v7, vcc_lo
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_3)
; GFX11-NEXT: v_add3_u32 v1, v1, v6, v8
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_3)
; GFX11-NEXT: v_add_co_u32 v2, vcc_lo, v2, v8
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX11-NEXT: v_add_co_ci_u32_e32 v2, vcc_lo, v3, v9, vcc_lo
; GFX11-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v11, vcc_lo
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-NEXT: v_add_co_u32 v2, vcc_lo, v2, v10
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_cmp_ne_u64_e32 vcc_lo, 0, v[2:3]
; GFX11-NEXT: v_cndmask_b32_e64 v2, 0, 1, vcc_lo
; GFX11-NEXT: s_setpc_b64 s[30:31]
@@ -242,33 +242,33 @@ define { i64, i1 } @smulo_i64_v_v(i64 %x, i64 %y) {
; GFX11-NEXT: v_mad_u64_u32 v[0:1], null, v4, v2, 0
; GFX11-NEXT: v_mad_u64_u32 v[8:9], null, v5, v2, 0
; GFX11-NEXT: v_mad_i64_i32 v[10:11], null, v5, v3, 0
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_1) | instid1(VALU_DEP_4)
; GFX11-NEXT: v_add_co_u32 v12, vcc_lo, v1, v6
; GFX11-NEXT: v_add_co_ci_u32_e64 v7, null, 0, v7, vcc_lo
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_3)
; GFX11-NEXT: v_add3_u32 v1, v1, v6, v8
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_3)
; GFX11-NEXT: v_add_co_u32 v12, vcc_lo, v12, v8
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX11-NEXT: v_add_co_ci_u32_e32 v7, vcc_lo, v7, v9, vcc_lo
; GFX11-NEXT: v_add_co_ci_u32_e64 v9, null, 0, v11, vcc_lo
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-NEXT: v_add_co_u32 v7, vcc_lo, v7, v10
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-NEXT: v_add_co_ci_u32_e64 v9, null, 0, v9, vcc_lo
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-NEXT: v_sub_co_u32 v2, vcc_lo, v7, v2
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_3)
; GFX11-NEXT: v_subrev_co_ci_u32_e64 v10, null, 0, v9, vcc_lo
; GFX11-NEXT: v_cmp_gt_i32_e32 vcc_lo, 0, v5
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_3)
; GFX11-NEXT: v_cndmask_b32_e32 v6, v7, v2, vcc_lo
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_1) | instid1(VALU_DEP_3)
; GFX11-NEXT: v_cndmask_b32_e32 v5, v9, v10, vcc_lo
; GFX11-NEXT: v_ashrrev_i32_e32 v2, 31, v1
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_3)
; GFX11-NEXT: v_sub_co_u32 v4, vcc_lo, v6, v4
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_4)
; GFX11-NEXT: v_subrev_co_ci_u32_e64 v7, null, 0, v5, vcc_lo
; GFX11-NEXT: v_cmp_gt_i32_e32 vcc_lo, 0, v3
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_3)
; GFX11-NEXT: v_mov_b32_e32 v3, v2
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_dual_cndmask_b32 v4, v6, v4 :: v_dual_cndmask_b32 v5, v5, v7
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_cmp_ne_u64_e32 vcc_lo, v[4:5], v[2:3]
; GFX11-NEXT: v_cndmask_b32_e64 v2, 0, 1, vcc_lo
; GFX11-NEXT: s_setpc_b64 s[30:31]
@@ -443,9 +443,9 @@ define amdgpu_kernel void @umulo_i64_s(i64 %x, i64 %y) {
; GFX11-NEXT: s_mul_i32 s0, s0, s2
; GFX11-NEXT: s_add_i32 s1, s1, s6
; GFX11-NEXT: s_cmp_lg_u64 s[4:5], 0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_cselect_b32 s0, 0, s0
; GFX11-NEXT: s_cselect_b32 s1, 0, s1
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: v_dual_mov_b32 v0, s0 :: v_dual_mov_b32 v1, s1
; GFX11-NEXT: global_store_b64 v[0:1], v[0:1], off
; GFX11-NEXT: s_endpgm
@@ -468,10 +468,11 @@ define amdgpu_kernel void @umulo_i64_s(i64 %x, i64 %y) {
; GFX12-NEXT: s_add_co_ci_u32 s9, s11, 0
; GFX12-NEXT: s_mul_u64 s[0:1], s[0:1], s[2:3]
; GFX12-NEXT: s_add_nc_u64 s[4:5], s[4:5], s[8:9]
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_lg_u64 s[4:5], 0
; GFX12-NEXT: s_cselect_b32 s0, 0, s0
; GFX12-NEXT: s_cselect_b32 s1, 0, s1
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: v_dual_mov_b32 v0, s0 :: v_dual_mov_b32 v1, s1
; GFX12-NEXT: global_store_b64 v[0:1], v[0:1], off
; GFX12-NEXT: s_endpgm
@@ -637,6 +638,7 @@ define amdgpu_kernel void @smulo_i64_s(i64 %x, i64 %y) {
; GFX11-NEXT: s_sub_u32 s9, s4, s2
; GFX11-NEXT: s_subb_u32 s10, s5, 0
; GFX11-NEXT: s_cmp_lt_i32 s1, 0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cselect_b32 s1, s9, s4
; GFX11-NEXT: s_cselect_b32 s4, s10, s5
; GFX11-NEXT: s_sub_u32 s9, s1, s0
@@ -652,9 +654,9 @@ define amdgpu_kernel void @smulo_i64_s(i64 %x, i64 %y) {
; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_mov_b32 s7, s6
; GFX11-NEXT: s_cmp_lg_u64 s[4:5], s[6:7]
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_cselect_b32 s0, 0, s0
; GFX11-NEXT: s_cselect_b32 s1, 0, s1
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: v_dual_mov_b32 v0, s0 :: v_dual_mov_b32 v1, s1
; GFX11-NEXT: global_store_b64 v[0:1], v[0:1], off
; GFX11-NEXT: s_endpgm
@@ -692,9 +694,9 @@ define amdgpu_kernel void @smulo_i64_s(i64 %x, i64 %y) {
; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_mov_b32 s5, s4
; GFX12-NEXT: s_cmp_lg_u64 s[2:3], s[4:5]
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s0, 0, s0
; GFX12-NEXT: s_cselect_b32 s1, 0, s1
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: v_dual_mov_b32 v0, s0 :: v_dual_mov_b32 v1, s1
; GFX12-NEXT: global_store_b64 v[0:1], v[0:1], off
; GFX12-NEXT: s_endpgm
diff --git a/llvm/test/CodeGen/AMDGPU/llvm.round.ll b/llvm/test/CodeGen/AMDGPU/llvm.round.ll
index c2776a0439996a..15b868f0d97caf 100644
--- a/llvm/test/CodeGen/AMDGPU/llvm.round.ll
+++ b/llvm/test/CodeGen/AMDGPU/llvm.round.ll
@@ -71,12 +71,11 @@ define amdgpu_kernel void @round_f32(ptr addrspace(1) %out, float %x) #0 {
; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_sub_f32_e32 v1, s2, v0
; GFX11-NEXT: v_cmp_ge_f32_e64 s3, |v1|, 0.5
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_cndmask_b32_e64 v1, 0, 1.0, s3
; GFX11-NEXT: s_mov_b32 s3, 0x31016000
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_bfi_b32 v1, 0x7fffffff, v1, s2
; GFX11-NEXT: s_mov_b32 s2, -1
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_add_f32_e32 v0, v0, v1
; GFX11-NEXT: buffer_store_b32 v0, off, s[0:3], 0
; GFX11-NEXT: s_endpgm
@@ -193,13 +192,12 @@ define amdgpu_kernel void @round_v2f32(ptr addrspace(1) %out, <2 x float> %in) #
; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-NEXT: v_sub_f32_e32 v1, s3, v0
; GFX11-NEXT: v_sub_f32_e32 v3, s2, v2
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_3)
; GFX11-NEXT: v_cmp_ge_f32_e64 s4, |v1|, 0.5
; GFX11-NEXT: v_cndmask_b32_e64 v1, 0, 1.0, s4
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-NEXT: v_cmp_ge_f32_e64 s4, |v3|, 0.5
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_2) | instid1(VALU_DEP_2)
; GFX11-NEXT: v_bfi_b32 v1, 0x7fffffff, v1, s3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX11-NEXT: v_cndmask_b32_e64 v3, 0, 1.0, s4
; GFX11-NEXT: s_mov_b32 s3, 0x31016000
; GFX11-NEXT: v_add_f32_e32 v1, v0, v1
@@ -368,21 +366,19 @@ define amdgpu_kernel void @round_v4f32(ptr addrspace(1) %out, <4 x float> %in) #
; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-NEXT: v_dual_sub_f32 v2, s3, v0 :: v_dual_sub_f32 v3, s2, v1
; GFX11-NEXT: v_dual_sub_f32 v6, s1, v4 :: v_dual_sub_f32 v7, s0, v5
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_4)
; GFX11-NEXT: v_cmp_ge_f32_e64 s6, |v2|, 0.5
; GFX11-NEXT: v_cndmask_b32_e64 v2, 0, 1.0, s6
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-NEXT: v_cmp_ge_f32_e64 s6, |v3|, 0.5
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_2) | instid1(VALU_DEP_2)
; GFX11-NEXT: v_bfi_b32 v2, 0x7fffffff, v2, s3
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX11-NEXT: v_cndmask_b32_e64 v3, 0, 1.0, s6
; GFX11-NEXT: v_cmp_ge_f32_e64 s6, |v6|, 0.5
; GFX11-NEXT: v_bfi_b32 v8, 0x7fffffff, v3, s2
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_3)
; GFX11-NEXT: v_cndmask_b32_e64 v6, 0, 1.0, s6
; GFX11-NEXT: v_cmp_ge_f32_e64 s6, |v7|, 0.5
-; GFX11-NEXT: v_dual_add_f32 v3, v0, v2 :: v_dual_add_f32 v2, v1, v8
; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX11-NEXT: v_dual_add_f32 v3, v0, v2 :: v_dual_add_f32 v2, v1, v8
; GFX11-NEXT: v_bfi_b32 v6, 0x7fffffff, v6, s1
; GFX11-NEXT: v_cndmask_b32_e64 v7, 0, 1.0, s6
; GFX11-NEXT: s_mov_b32 s6, -1
@@ -660,11 +656,10 @@ define amdgpu_kernel void @round_v8f32(ptr addrspace(1) %out, <8 x float> %in) #
; GFX11-NEXT: v_trunc_f32_e32 v10, s12
; GFX11-NEXT: v_cndmask_b32_e64 v2, 0, 1.0, s2
; GFX11-NEXT: v_cmp_ge_f32_e64 s2, |v3|, 0.5
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_2) | instid1(VALU_DEP_2)
; GFX11-NEXT: v_bfi_b32 v2, 0x7fffffff, v2, s11
; GFX11-NEXT: v_cndmask_b32_e64 v3, 0, 1.0, s2
; GFX11-NEXT: v_cmp_ge_f32_e64 s2, |v7|, 0.5
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-NEXT: v_bfi_b32 v16, 0x7fffffff, v3, s10
; GFX11-NEXT: v_cndmask_b32_e64 v7, 0, 1.0, s2
; GFX11-NEXT: v_cmp_ge_f32_e64 s2, |v11|, 0.5
@@ -672,34 +667,32 @@ define amdgpu_kernel void @round_v8f32(ptr addrspace(1) %out, <8 x float> %in) #
; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
; GFX11-NEXT: v_dual_add_f32 v3, v0, v2 :: v_dual_add_f32 v2, v1, v16
; GFX11-NEXT: v_bfi_b32 v7, 0x7fffffff, v7, s9
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(SKIP_1) | instid1(VALU_DEP_3)
; GFX11-NEXT: v_cndmask_b32_e64 v11, 0, 1.0, s2
; GFX11-NEXT: v_cmp_ge_f32_e64 s2, |v12|, 0.5
-; GFX11-NEXT: v_add_f32_e32 v1, v4, v7
; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX11-NEXT: v_add_f32_e32 v1, v4, v7
; GFX11-NEXT: v_bfi_b32 v11, 0x7fffffff, v11, s8
; GFX11-NEXT: v_cndmask_b32_e64 v12, 0, 1.0, s2
; GFX11-NEXT: v_cmp_ge_f32_e64 s2, |v13|, 0.5
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_2) | instid1(VALU_DEP_3)
; GFX11-NEXT: v_bfi_b32 v12, 0x7fffffff, v12, s15
; GFX11-NEXT: v_cndmask_b32_e64 v13, 0, 1.0, s2
; GFX11-NEXT: v_cmp_ge_f32_e64 s2, |v14|, 0.5
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_3)
; GFX11-NEXT: v_add_f32_e32 v7, v5, v12
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_2) | instid1(VALU_DEP_3)
; GFX11-NEXT: v_bfi_b32 v13, 0x7fffffff, v13, s14
; GFX11-NEXT: v_sub_f32_e32 v15, s12, v10
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_3)
; GFX11-NEXT: v_cndmask_b32_e64 v14, 0, 1.0, s2
; GFX11-NEXT: v_add_f32_e32 v6, v6, v13
; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_3)
; GFX11-NEXT: v_cmp_ge_f32_e64 s2, |v15|, 0.5
; GFX11-NEXT: v_bfi_b32 v0, 0x7fffffff, v14, s13
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-NEXT: v_cndmask_b32_e64 v15, 0, 1.0, s2
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX11-NEXT: v_dual_add_f32 v5, v9, v0 :: v_dual_add_f32 v0, v8, v11
; GFX11-NEXT: s_mov_b32 s2, -1
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_bfi_b32 v4, 0x7fffffff, v15, s12
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_add_f32_e32 v4, v10, v4
; GFX11-NEXT: s_clause 0x1
; GFX11-NEXT: buffer_store_b128 v[4:7], off, s[0:3], 0 offset:16
@@ -761,6 +754,8 @@ define amdgpu_kernel void @round_v8f32(ptr addrspace(1) %out, <8 x float> %in) #
; R600-NEXT: ADD T0.X, T3.W, PV.W,
; R600-NEXT: LSHR * T1.X, KC0[2].Y, literal.x,
; R600-NEXT: 2(2.802597e-45), 0(0.000000e+00)
+; R600-NEXT: ADD_INT * T2.X, PS, literal.x,
+; R600-NEXT: 4(5.605194e-45), 0(0.000000e+00)
%result = call <8 x float> @llvm.round.v8f32(<8 x float> %in) #1
store <8 x float> %result, ptr addrspace(1) %out
ret void
@@ -837,11 +832,10 @@ define amdgpu_kernel void @round_f16(ptr addrspace(1) %out, i32 %x.arg) #0 {
; GFX11-TRUE16-NEXT: v_sub_f16_e32 v0.h, s2, v0.l
; GFX11-TRUE16-NEXT: s_mov_b32 s2, -1
; GFX11-TRUE16-NEXT: v_cmp_ge_f16_e64 s3, |v0.h|, 0.5
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_cndmask_b16 v2.l, 0, 0x3c00, s3
; GFX11-TRUE16-NEXT: s_mov_b32 s3, 0x31016000
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_bfi_b32 v1, 0x7fff, v2, v1
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_add_f16_e32 v0.l, v0.l, v1.l
; GFX11-TRUE16-NEXT: buffer_store_b16 v0, off, s[0:3], 0
; GFX11-TRUE16-NEXT: s_endpgm
@@ -856,12 +850,11 @@ define amdgpu_kernel void @round_f16(ptr addrspace(1) %out, i32 %x.arg) #0 {
; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_sub_f16_e32 v1, s2, v0
; GFX11-FAKE16-NEXT: v_cmp_ge_f16_e64 s3, |v1|, 0.5
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v1, 0, 0x3c00, s3
; GFX11-FAKE16-NEXT: s_mov_b32 s3, 0x31016000
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_bfi_b32 v1, 0x7fff, v1, s2
; GFX11-FAKE16-NEXT: s_mov_b32 s2, -1
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_add_f16_e32 v0, v0, v1
; GFX11-FAKE16-NEXT: buffer_store_b16 v0, off, s[0:3], 0
; GFX11-FAKE16-NEXT: s_endpgm
@@ -1029,13 +1022,12 @@ define amdgpu_kernel void @round_v2f16(ptr addrspace(1) %out, i32 %in.arg) #0 {
; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-FAKE16-NEXT: v_sub_f16_e32 v3, s2, v1
; GFX11-FAKE16-NEXT: v_sub_f16_e32 v2, s3, v0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_4)
; GFX11-FAKE16-NEXT: v_cmp_ge_f16_e64 s4, |v2|, 0.5
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v2, 0, 0x3c00, s4
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-FAKE16-NEXT: v_cmp_ge_f16_e64 s4, |v3|, 0.5
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_2) | instid1(VALU_DEP_2)
; GFX11-FAKE16-NEXT: v_bfi_b32 v2, 0x7fff, v2, s3
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v3, 0, 0x3c00, s4
; GFX11-FAKE16-NEXT: s_mov_b32 s3, 0x31016000
; GFX11-FAKE16-NEXT: v_add_f16_e32 v0, v0, v2
diff --git a/llvm/test/CodeGen/AMDGPU/llvm.sponentry.ll b/llvm/test/CodeGen/AMDGPU/llvm.sponentry.ll
index 6f5f67401e2652..c1e6239ee8f986 100644
--- a/llvm/test/CodeGen/AMDGPU/llvm.sponentry.ll
+++ b/llvm/test/CodeGen/AMDGPU/llvm.sponentry.ll
@@ -12,6 +12,7 @@ define amdgpu_cs ptr addrspace(5) @sponentry_cs_dvgpr_16(i32 %val) #0 {
; CHECK-NEXT: s_getreg_b32 s33, hwreg(HW_REG_WAVE_HW_ID2, 8, 2)
; CHECK-NEXT: s_getreg_b32 s0, hwreg(HW_REG_WAVE_HW_ID2, 8, 2)
; CHECK-NEXT: s_cmp_lg_u32 0, s33
+; CHECK-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; CHECK-NEXT: s_cmovk_i32 s33, 0x1c0
; CHECK-NEXT: s_cmp_lg_u32 0, s0
; CHECK-NEXT: scratch_store_b32 off, v0, s33 scope:SCOPE_SYS
@@ -32,6 +33,7 @@ define amdgpu_cs ptr addrspace(5) @sponentry_cs_dvgpr_32(i32 %val) #1 {
; CHECK-NEXT: s_getreg_b32 s33, hwreg(HW_REG_WAVE_HW_ID2, 8, 2)
; CHECK-NEXT: s_getreg_b32 s0, hwreg(HW_REG_WAVE_HW_ID2, 8, 2)
; CHECK-NEXT: s_cmp_lg_u32 0, s33
+; CHECK-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; CHECK-NEXT: s_cmovk_i32 s33, 0x380
; CHECK-NEXT: s_cmp_lg_u32 0, s0
; CHECK-NEXT: scratch_store_b32 off, v0, s33 scope:SCOPE_SYS
@@ -69,19 +71,21 @@ define amdgpu_cs ptr addrspace(5) @sponentry_cs_dvgpr_control_flow(i32 %val, ptr
; CHECK-NEXT: s_getreg_b32 s33, hwreg(HW_REG_WAVE_HW_ID2, 8, 2)
; CHECK-NEXT: s_mov_b32 s0, exec_lo
; CHECK-NEXT: s_cmp_lg_u32 0, s33
+; CHECK-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; CHECK-NEXT: s_cmovk_i32 s33, 0x1c0
; CHECK-NEXT: scratch_store_b32 off, v0, s33 scope:SCOPE_SYS
; CHECK-NEXT: s_wait_storecnt 0x0
; CHECK-NEXT: v_cmpx_gt_i32_e32 0x43, v0
; CHECK-NEXT: ; %bb.1: ; %if.then
; CHECK-NEXT: s_getreg_b32 s1, hwreg(HW_REG_WAVE_HW_ID2, 8, 2)
-; CHECK-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
+; CHECK-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; CHECK-NEXT: s_cmp_lg_u32 0, s1
; CHECK-NEXT: s_cmovk_i32 s1, 0x1c0
+; CHECK-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; CHECK-NEXT: v_mov_b32_e32 v1, s1
; CHECK-NEXT: ; %bb.2: ; %if.end
; CHECK-NEXT: s_or_b32 exec_lo, exec_lo, s0
-; CHECK-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; CHECK-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; CHECK-NEXT: v_readfirstlane_b32 s0, v1
; CHECK-NEXT: s_wait_alu depctr_va_sdst(0)
; CHECK-NEXT: ; return to shader part epilog
@@ -120,6 +124,7 @@ define amdgpu_cs ptr addrspace(5) @sponentry_cs_dvgpr_calls(i32 %val) #0 {
; DAGISEL-NEXT: s_wait_storecnt 0x0
; DAGISEL-NEXT: s_wait_alu depctr_sa_sdst(0)
; DAGISEL-NEXT: s_cmp_lg_u32 0, s0
+; DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; DAGISEL-NEXT: s_cmovk_i32 s0, 0x1c0
; DAGISEL-NEXT: s_wait_alu depctr_sa_sdst(0)
; DAGISEL-NEXT: ; return to shader part epilog
@@ -139,6 +144,7 @@ define amdgpu_cs ptr addrspace(5) @sponentry_cs_dvgpr_calls(i32 %val) #0 {
; GISEL-NEXT: s_wait_storecnt 0x0
; GISEL-NEXT: s_wait_alu depctr_sa_sdst(0)
; GISEL-NEXT: s_cmp_lg_u32 0, s0
+; GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GISEL-NEXT: s_cmovk_i32 s0, 0x1c0
; GISEL-NEXT: s_wait_alu depctr_sa_sdst(0)
; GISEL-NEXT: ; return to shader part epilog
@@ -157,6 +163,7 @@ define amdgpu_cs ptr addrspace(5) @sponentry_cs_dvgpr_realign(i32 %val) #0 {
; CHECK-NEXT: s_getreg_b32 s33, hwreg(HW_REG_WAVE_HW_ID2, 8, 2)
; CHECK-NEXT: s_getreg_b32 s0, hwreg(HW_REG_WAVE_HW_ID2, 8, 2)
; CHECK-NEXT: s_cmp_lg_u32 0, s33
+; CHECK-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; CHECK-NEXT: s_cmovk_i32 s33, 0x200
; CHECK-NEXT: s_cmp_lg_u32 0, s0
; CHECK-NEXT: scratch_store_b32 off, v0, s33 scope:SCOPE_SYS
@@ -189,8 +196,9 @@ define amdgpu_gfx ptr addrspace(5) @sponentry_gfx(i32 %val, ptr addrspace(5) %pt
; DAGISEL-NEXT: v_mov_b32_e32 v1, s32
; DAGISEL-NEXT: ; %bb.2: ; %if.end
; DAGISEL-NEXT: s_wait_alu depctr_sa_sdst(0)
+; DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2)
; DAGISEL-NEXT: s_or_b32 exec_lo, exec_lo, s0
-; DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; DAGISEL-NEXT: v_mov_b32_e32 v0, v1
; DAGISEL-NEXT: s_setpc_b64 s[30:31]
;
@@ -213,6 +221,7 @@ define amdgpu_gfx ptr addrspace(5) @sponentry_gfx(i32 %val, ptr addrspace(5) %pt
; GISEL-NEXT: s_wait_alu depctr_sa_sdst(0)
; GISEL-NEXT: v_mov_b32_e32 v0, s1
; GISEL-NEXT: ; %bb.2: ; %if.end
+; GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GISEL-NEXT: s_or_b32 exec_lo, exec_lo, s0
; GISEL-NEXT: s_setpc_b64 s[30:31]
entry:
@@ -325,23 +334,23 @@ define amdgpu_gfx ptr addrspace(5) @sponentry_gfx_dyn_alloc(i32 %val) #0 {
; DAGISEL-NEXT: scratch_store_b32 off, v2, s33 offset:8
; DAGISEL-NEXT: s_wait_alu depctr_sa_sdst(0)
; DAGISEL-NEXT: s_mov_b32 exec_lo, s0
+; DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; DAGISEL-NEXT: v_lshl_add_u32 v3, v0, 2, 15
; DAGISEL-NEXT: s_add_co_i32 s32, s32, 16
-; DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_2) | instid1(VALU_DEP_1)
; DAGISEL-NEXT: v_and_b32_e32 v3, -16, v3
; DAGISEL-NEXT: s_or_saveexec_b32 s0, -1
; DAGISEL-NEXT: s_wait_alu depctr_sa_sdst(0)
-; DAGISEL-NEXT: v_cndmask_b32_e64 v1, 0, v3, s0
; DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; DAGISEL-NEXT: v_cndmask_b32_e64 v1, 0, v3, s0
; DAGISEL-NEXT: v_max_u32_dpp v1, v1, v1 row_shr:1 row_mask:0xf bank_mask:0xf
-; DAGISEL-NEXT: v_max_u32_dpp v1, v1, v1 row_shr:2 row_mask:0xf bank_mask:0xf
; DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; DAGISEL-NEXT: v_max_u32_dpp v1, v1, v1 row_shr:2 row_mask:0xf bank_mask:0xf
; DAGISEL-NEXT: v_max_u32_dpp v1, v1, v1 row_shr:4 row_mask:0xf bank_mask:0xf
+; DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_3) | instid1(VALU_DEP_1)
; DAGISEL-NEXT: v_max_u32_dpp v1, v1, v1 row_shr:8 row_mask:0xf bank_mask:0xf
; DAGISEL-NEXT: ds_swizzle_b32 v2, v1 offset:swizzle(BROADCAST,32,15)
; DAGISEL-NEXT: s_wait_dscnt 0x0
; DAGISEL-NEXT: v_max_u32_e32 v1, v1, v2
-; DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; DAGISEL-NEXT: v_readlane_b32 s1, v1, 31
; DAGISEL-NEXT: s_mov_b32 exec_lo, s0
; DAGISEL-NEXT: s_mov_b32 s0, s32
@@ -379,25 +388,25 @@ define amdgpu_gfx ptr addrspace(5) @sponentry_gfx_dyn_alloc(i32 %val) #0 {
; GISEL-NEXT: scratch_store_b32 off, v2, s33 offset:8
; GISEL-NEXT: s_wait_alu depctr_sa_sdst(0)
; GISEL-NEXT: s_mov_b32 exec_lo, s0
+; GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_3) | instid1(VALU_DEP_1)
; GISEL-NEXT: v_lshl_add_u32 v3, v0, 2, 15
; GISEL-NEXT: s_add_co_i32 s32, s32, 16
; GISEL-NEXT: s_wait_alu depctr_sa_sdst(0)
; GISEL-NEXT: s_mov_b32 s0, s32
-; GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_2) | instid1(VALU_DEP_1)
; GISEL-NEXT: v_and_b32_e32 v3, -16, v3
; GISEL-NEXT: s_or_saveexec_b32 s1, -1
; GISEL-NEXT: s_wait_alu depctr_sa_sdst(0)
-; GISEL-NEXT: v_cndmask_b32_e64 v1, 0, v3, s1
; GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GISEL-NEXT: v_cndmask_b32_e64 v1, 0, v3, s1
; GISEL-NEXT: v_max_u32_dpp v1, v1, v1 row_shr:1 row_mask:0xf bank_mask:0xf
-; GISEL-NEXT: v_max_u32_dpp v1, v1, v1 row_shr:2 row_mask:0xf bank_mask:0xf
; GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GISEL-NEXT: v_max_u32_dpp v1, v1, v1 row_shr:2 row_mask:0xf bank_mask:0xf
; GISEL-NEXT: v_max_u32_dpp v1, v1, v1 row_shr:4 row_mask:0xf bank_mask:0xf
+; GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_3) | instid1(VALU_DEP_1)
; GISEL-NEXT: v_max_u32_dpp v1, v1, v1 row_shr:8 row_mask:0xf bank_mask:0xf
; GISEL-NEXT: ds_swizzle_b32 v2, v1 offset:swizzle(BROADCAST,32,15)
; GISEL-NEXT: s_wait_dscnt 0x0
; GISEL-NEXT: v_max_u32_e32 v1, v1, v2
-; GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GISEL-NEXT: v_readlane_b32 s2, v1, 31
; GISEL-NEXT: s_mov_b32 exec_lo, s1
; GISEL-NEXT: s_add_co_u32 s32, s0, s2
@@ -442,6 +451,7 @@ define amdgpu_cs_chain void @sponentry_cs_chain(i32 %val, ptr addrspace(5) %ptr)
; DAGISEL-NEXT: v_mov_b32_e32 v0, s32
; DAGISEL-NEXT: ; %bb.2: ; %if.end
; DAGISEL-NEXT: s_wait_alu depctr_sa_sdst(0)
+; DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2)
; DAGISEL-NEXT: s_or_b32 exec_lo, exec_lo, s0
; DAGISEL-NEXT: scratch_store_b32 v9, v0, off scope:SCOPE_SYS
; DAGISEL-NEXT: s_wait_storecnt 0x0
@@ -466,6 +476,7 @@ define amdgpu_cs_chain void @sponentry_cs_chain(i32 %val, ptr addrspace(5) %ptr)
; GISEL-NEXT: s_wait_alu depctr_sa_sdst(0)
; GISEL-NEXT: v_mov_b32_e32 v0, s1
; GISEL-NEXT: ; %bb.2: ; %if.end
+; GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GISEL-NEXT: s_or_b32 exec_lo, exec_lo, s0
; GISEL-NEXT: scratch_store_b32 v9, v0, off scope:SCOPE_SYS
; GISEL-NEXT: s_wait_storecnt 0x0
@@ -591,6 +602,7 @@ define amdgpu_cs_chain void @sponentry_cs_chain_dyn_alloc(i32 %val) #0 {
; GISEL-NEXT: v_max_u32_e32 v0, v0, v1
; GISEL-NEXT: v_readlane_b32 s2, v0, 31
; GISEL-NEXT: s_mov_b32 exec_lo, s1
+; GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GISEL-NEXT: v_mov_b32_e32 v2, s33
; GISEL-NEXT: s_wait_storecnt 0x0
; GISEL-NEXT: scratch_store_b32 off, v8, s0 scope:SCOPE_SYS
diff --git a/llvm/test/CodeGen/AMDGPU/load-atomic-flat.ll b/llvm/test/CodeGen/AMDGPU/load-atomic-flat.ll
index 96e71dc51eeacd..50b90d4b000846 100644
--- a/llvm/test/CodeGen/AMDGPU/load-atomic-flat.ll
+++ b/llvm/test/CodeGen/AMDGPU/load-atomic-flat.ll
@@ -352,7 +352,6 @@ define amdgpu_cs void @atomic_load_f32x2_monotonic_agent_offset_min(ptr addrspac
; GFX11-LABEL: atomic_load_f32x2_monotonic_agent_offset_min:
; GFX11: ; %bb.0:
; GFX11-NEXT: v_add_co_u32 v0, vcc_lo, 0xfffff000, v0
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, -1, v1, vcc_lo
; GFX11-NEXT: flat_load_b64 v[0:1], v[0:1] glc
; GFX11-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
@@ -444,11 +443,11 @@ define amdgpu_cs void @atomic_load_i16x2_monotonic_agent_offset_min(ptr addrspac
; GFX11-SDAG-LABEL: atomic_load_i16x2_monotonic_agent_offset_min:
; GFX11-SDAG: ; %bb.0:
; GFX11-SDAG-NEXT: v_add_co_u32 v0, vcc_lo, 0xfffff000, v0
-; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_3) | instid1(VALU_DEP_1)
; GFX11-SDAG-NEXT: v_add_co_ci_u32_e64 v1, null, -1, v1, vcc_lo
; GFX11-SDAG-NEXT: flat_load_b32 v0, v[0:1] glc
; GFX11-SDAG-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
; GFX11-SDAG-NEXT: v_lshrrev_b32_e32 v1, 16, v0
+; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-SDAG-NEXT: v_add_nc_u16 v0.l, v0.l, v1.l
; GFX11-SDAG-NEXT: global_store_b16 v[2:3], v0, off
; GFX11-SDAG-NEXT: s_endpgm
@@ -456,7 +455,6 @@ define amdgpu_cs void @atomic_load_i16x2_monotonic_agent_offset_min(ptr addrspac
; GFX11-GISEL-LABEL: atomic_load_i16x2_monotonic_agent_offset_min:
; GFX11-GISEL: ; %bb.0:
; GFX11-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, 0xfffff000, v0
-; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, -1, v1, vcc_lo
; GFX11-GISEL-NEXT: flat_load_b32 v0, v[0:1] glc
; GFX11-GISEL-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
diff --git a/llvm/test/CodeGen/AMDGPU/load-constant-always-uniform.ll b/llvm/test/CodeGen/AMDGPU/load-constant-always-uniform.ll
index d31670af469bed..df7e08d7f23c60 100644
--- a/llvm/test/CodeGen/AMDGPU/load-constant-always-uniform.ll
+++ b/llvm/test/CodeGen/AMDGPU/load-constant-always-uniform.ll
@@ -9,7 +9,7 @@ define amdgpu_cs void @test_uniform_load_b96(ptr addrspace(1) %ptr, i32 %arg) "a
; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_lshlrev_b64 v[2:3], 2, v[2:3]
; GFX11-NEXT: v_add_co_u32 v2, vcc_lo, v0, v2
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX11-NEXT: v_add_co_ci_u32_e64 v3, null, v1, v3, vcc_lo
; GFX11-NEXT: global_load_b96 v[2:4], v[2:3], off
; GFX11-NEXT: s_waitcnt vmcnt(0)
@@ -23,7 +23,7 @@ define amdgpu_cs void @test_uniform_load_b96(ptr addrspace(1) %ptr, i32 %arg) "a
; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX12-NEXT: v_lshlrev_b64_e32 v[2:3], 2, v[2:3]
; GFX12-NEXT: v_add_co_u32 v2, vcc_lo, v0, v2
-; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX12-NEXT: v_add_co_ci_u32_e64 v3, null, v1, v3, vcc_lo
; GFX12-NEXT: global_load_b96 v[2:4], v[2:3], off
; GFX12-NEXT: s_wait_loadcnt 0x0
diff --git a/llvm/test/CodeGen/AMDGPU/load-saddr-offset-imm.ll b/llvm/test/CodeGen/AMDGPU/load-saddr-offset-imm.ll
index c9c1894fe9e1d4..0f623673db07de 100644
--- a/llvm/test/CodeGen/AMDGPU/load-saddr-offset-imm.ll
+++ b/llvm/test/CodeGen/AMDGPU/load-saddr-offset-imm.ll
@@ -62,7 +62,7 @@ define amdgpu_ps <2 x float> @global_load_scale_add_foldable_nowrap(ptr addrspac
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
; GFX1250-GISEL-NEXT: v_lshlrev_b32_e32 v0, 3, v0
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_3)
; GFX1250-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: global_load_b64 v[0:1], v[0:1], off offset:128
@@ -107,7 +107,7 @@ define amdgpu_ps <2 x float> @global_load_scale_add_unfoldable(ptr addrspace(1)
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[2:3], s[2:3]
; GFX1250-GISEL-NEXT: v_lshl_add_u32 v0, v0, 3, 0x80
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_3)
; GFX1250-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, v2, v0
; GFX1250-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v3, vcc_lo
; GFX1250-GISEL-NEXT: global_load_b64 v[0:1], v[0:1], off
diff --git a/llvm/test/CodeGen/AMDGPU/local-atomicrmw-fadd.ll b/llvm/test/CodeGen/AMDGPU/local-atomicrmw-fadd.ll
index ac8526dbce15ca..928ef1a5666f12 100644
--- a/llvm/test/CodeGen/AMDGPU/local-atomicrmw-fadd.ll
+++ b/llvm/test/CodeGen/AMDGPU/local-atomicrmw-fadd.ll
@@ -508,6 +508,7 @@ define double @local_atomic_fadd_ret_f64(ptr addrspace(3) %ptr) nounwind {
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB4_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -539,7 +540,7 @@ define double @local_atomic_fadd_ret_f64(ptr addrspace(3) %ptr) nounwind {
; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: v_cmp_eq_u64_e32 vcc_lo, v[0:1], v[3:4]
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB4_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -700,6 +701,7 @@ define double @local_atomic_fadd_ret_f64__offset(ptr addrspace(3) %ptr) nounwind
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB5_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -731,7 +733,7 @@ define double @local_atomic_fadd_ret_f64__offset(ptr addrspace(3) %ptr) nounwind
; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: v_cmp_eq_u64_e32 vcc_lo, v[0:1], v[3:4]
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB5_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -893,6 +895,7 @@ define void @local_atomic_fadd_noret_f64(ptr addrspace(3) %ptr) nounwind {
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB6_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -922,7 +925,7 @@ define void @local_atomic_fadd_noret_f64(ptr addrspace(3) %ptr) nounwind {
; GFX11-NEXT: v_cmp_eq_u64_e32 vcc_lo, v[3:4], v[1:2]
; GFX11-NEXT: v_dual_mov_b32 v1, v3 :: v_dual_mov_b32 v2, v4
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB6_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -1076,6 +1079,7 @@ define void @local_atomic_fadd_noret_f64__offset(ptr addrspace(3) %ptr) nounwind
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB7_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -1105,7 +1109,7 @@ define void @local_atomic_fadd_noret_f64__offset(ptr addrspace(3) %ptr) nounwind
; GFX11-NEXT: v_cmp_eq_u64_e32 vcc_lo, v[3:4], v[1:2]
; GFX11-NEXT: v_dual_mov_b32 v1, v3 :: v_dual_mov_b32 v2, v4
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB7_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -1279,9 +1283,11 @@ define half @local_atomic_fadd_ret_f16(ptr addrspace(3) %ptr) nounwind {
; GFX12-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_cbranch_execnz .LBB8_1
; GFX12-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: v_lshrrev_b32_e32 v0, v0, v2
; GFX12-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -1322,9 +1328,11 @@ define half @local_atomic_fadd_ret_f16(ptr addrspace(3) %ptr) nounwind {
; GFX12-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: s_cbranch_execnz .LBB8_1
; GFX12-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: v_lshrrev_b32_e32 v0, v0, v2
; GFX12-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -1387,11 +1395,12 @@ define half @local_atomic_fadd_ret_f16(ptr addrspace(3) %ptr) nounwind {
; GFX11-TRUE16-NEXT: buffer_gl0_inv
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v4
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB8_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v0, v0, v2
; GFX11-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -1424,11 +1433,12 @@ define half @local_atomic_fadd_ret_f16(ptr addrspace(3) %ptr) nounwind {
; GFX11-FAKE16-NEXT: buffer_gl0_inv
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v4
; GFX11-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB8_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v0, v0, v2
; GFX11-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -1664,9 +1674,11 @@ define half @local_atomic_fadd_ret_f16__offset(ptr addrspace(3) %ptr) nounwind {
; GFX12-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_cbranch_execnz .LBB9_1
; GFX12-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: v_lshrrev_b32_e32 v0, v1, v3
; GFX12-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -1709,9 +1721,11 @@ define half @local_atomic_fadd_ret_f16__offset(ptr addrspace(3) %ptr) nounwind {
; GFX12-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: s_cbranch_execnz .LBB9_1
; GFX12-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: v_lshrrev_b32_e32 v0, v1, v3
; GFX12-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -1777,11 +1791,12 @@ define half @local_atomic_fadd_ret_f16__offset(ptr addrspace(3) %ptr) nounwind {
; GFX11-TRUE16-NEXT: buffer_gl0_inv
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB9_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v0, v1, v3
; GFX11-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -1816,11 +1831,12 @@ define half @local_atomic_fadd_ret_f16__offset(ptr addrspace(3) %ptr) nounwind {
; GFX11-FAKE16-NEXT: buffer_gl0_inv
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB9_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v0, v1, v3
; GFX11-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -2061,6 +2077,7 @@ define void @local_atomic_fadd_noret_f16(ptr addrspace(3) %ptr) nounwind {
; GFX12-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_cbranch_execnz .LBB10_1
; GFX12-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -2103,6 +2120,7 @@ define void @local_atomic_fadd_noret_f16(ptr addrspace(3) %ptr) nounwind {
; GFX12-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: s_cbranch_execnz .LBB10_1
; GFX12-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -2166,7 +2184,7 @@ define void @local_atomic_fadd_noret_f16(ptr addrspace(3) %ptr) nounwind {
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v4, v2
; GFX11-TRUE16-NEXT: v_mov_b32_e32 v2, v4
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB10_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -2202,7 +2220,7 @@ define void @local_atomic_fadd_noret_f16(ptr addrspace(3) %ptr) nounwind {
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v4, v2
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v2, v4
; GFX11-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB10_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -2434,6 +2452,7 @@ define void @local_atomic_fadd_noret_f16__offset(ptr addrspace(3) %ptr) nounwind
; GFX12-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_cbranch_execnz .LBB11_1
; GFX12-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -2477,6 +2496,7 @@ define void @local_atomic_fadd_noret_f16__offset(ptr addrspace(3) %ptr) nounwind
; GFX12-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: s_cbranch_execnz .LBB11_1
; GFX12-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -2542,7 +2562,7 @@ define void @local_atomic_fadd_noret_f16__offset(ptr addrspace(3) %ptr) nounwind
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v4, v3
; GFX11-TRUE16-NEXT: v_mov_b32_e32 v3, v4
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB11_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -2579,7 +2599,7 @@ define void @local_atomic_fadd_noret_f16__offset(ptr addrspace(3) %ptr) nounwind
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v4, v3
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v3, v4
; GFX11-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB11_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -2808,9 +2828,11 @@ define half @local_atomic_fadd_ret_f16__offset__align4(ptr addrspace(3) %ptr) no
; GFX12-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_cbranch_execnz .LBB12_1
; GFX12-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: v_mov_b16_e32 v0.l, v1.l
; GFX12-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -2842,9 +2864,11 @@ define half @local_atomic_fadd_ret_f16__offset__align4(ptr addrspace(3) %ptr) no
; GFX12-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: s_cbranch_execnz .LBB12_1
; GFX12-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: v_mov_b32_e32 v0, v1
; GFX12-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -2891,11 +2915,12 @@ define half @local_atomic_fadd_ret_f16__offset__align4(ptr addrspace(3) %ptr) no
; GFX11-TRUE16-NEXT: buffer_gl0_inv
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v1, v2
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB12_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v0.l, v1.l
; GFX11-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -2919,11 +2944,12 @@ define half @local_atomic_fadd_ret_f16__offset__align4(ptr addrspace(3) %ptr) no
; GFX11-FAKE16-NEXT: buffer_gl0_inv
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v1, v2
; GFX11-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB12_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v0, v1
; GFX11-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -3108,6 +3134,7 @@ define void @local_atomic_fadd_noret_f16__offset__align4(ptr addrspace(3) %ptr)
; GFX12-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_cbranch_execnz .LBB13_1
; GFX12-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -3140,6 +3167,7 @@ define void @local_atomic_fadd_noret_f16__offset__align4(ptr addrspace(3) %ptr)
; GFX12-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: s_cbranch_execnz .LBB13_1
; GFX12-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -3186,7 +3214,7 @@ define void @local_atomic_fadd_noret_f16__offset__align4(ptr addrspace(3) %ptr)
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v1
; GFX11-TRUE16-NEXT: v_mov_b32_e32 v1, v2
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB13_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -3212,7 +3240,7 @@ define void @local_atomic_fadd_noret_f16__offset__align4(ptr addrspace(3) %ptr)
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v1
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v1, v2
; GFX11-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB13_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -3417,9 +3445,11 @@ define bfloat @local_atomic_fadd_ret_bf16(ptr addrspace(3) %ptr) nounwind {
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB14_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: v_lshrrev_b32_e32 v0, v0, v3
; GFX12-NEXT: s_setpc_b64 s[30:31]
;
@@ -3499,11 +3529,12 @@ define bfloat @local_atomic_fadd_ret_bf16(ptr addrspace(3) %ptr) nounwind {
; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB14_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: v_lshrrev_b32_e32 v0, v0, v3
; GFX11-NEXT: s_setpc_b64 s[30:31]
;
@@ -3771,9 +3802,11 @@ define bfloat @local_atomic_fadd_ret_bf16__offset(ptr addrspace(3) %ptr) nounwin
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB15_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: v_lshrrev_b32_e32 v0, v1, v3
; GFX12-NEXT: s_setpc_b64 s[30:31]
;
@@ -3856,11 +3889,12 @@ define bfloat @local_atomic_fadd_ret_bf16__offset(ptr addrspace(3) %ptr) nounwin
; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB15_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: v_lshrrev_b32_e32 v0, v1, v3
; GFX11-NEXT: s_setpc_b64 s[30:31]
;
@@ -4133,6 +4167,7 @@ define void @local_atomic_fadd_noret_bf16(ptr addrspace(3) %ptr) nounwind {
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB16_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -4213,7 +4248,7 @@ define void @local_atomic_fadd_noret_bf16(ptr addrspace(3) %ptr) nounwind {
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v4, v2
; GFX11-NEXT: v_mov_b32_e32 v2, v4
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB16_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -4477,6 +4512,7 @@ define void @local_atomic_fadd_noret_bf16__offset(ptr addrspace(3) %ptr) nounwin
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB17_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -4559,7 +4595,7 @@ define void @local_atomic_fadd_noret_bf16__offset(ptr addrspace(3) %ptr) nounwin
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v4, v3
; GFX11-NEXT: v_mov_b32_e32 v3, v4
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB17_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -4820,9 +4856,11 @@ define bfloat @local_atomic_fadd_ret_bf16__offset__align4(ptr addrspace(3) %ptr)
; GFX12-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_cbranch_execnz .LBB18_1
; GFX12-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: v_mov_b16_e32 v0.l, v1.l
; GFX12-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -4863,9 +4901,11 @@ define bfloat @local_atomic_fadd_ret_bf16__offset__align4(ptr addrspace(3) %ptr)
; GFX12-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: s_cbranch_execnz .LBB18_1
; GFX12-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: v_mov_b32_e32 v0, v1
; GFX12-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -4930,11 +4970,12 @@ define bfloat @local_atomic_fadd_ret_bf16__offset__align4(ptr addrspace(3) %ptr)
; GFX11-TRUE16-NEXT: buffer_gl0_inv
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v1, v2
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB18_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v0.l, v1.l
; GFX11-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -4967,11 +5008,12 @@ define bfloat @local_atomic_fadd_ret_bf16__offset__align4(ptr addrspace(3) %ptr)
; GFX11-FAKE16-NEXT: buffer_gl0_inv
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v1, v2
; GFX11-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB18_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v0, v1
; GFX11-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -5194,6 +5236,7 @@ define void @local_atomic_fadd_noret_bf16__offset__align4(ptr addrspace(3) %ptr)
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB19_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -5258,7 +5301,7 @@ define void @local_atomic_fadd_noret_bf16__offset__align4(ptr addrspace(3) %ptr)
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v1
; GFX11-NEXT: v_mov_b32_e32 v1, v2
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB19_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -5483,11 +5526,12 @@ define <2 x half> @local_atomic_fadd_ret_v2f16(ptr addrspace(3) %ptr, <2 x half>
; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v3
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB20_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: v_mov_b32_e32 v0, v2
; GFX11-NEXT: s_setpc_b64 s[30:31]
;
@@ -5698,11 +5742,12 @@ define <2 x half> @local_atomic_fadd_ret_v2f16__offset(ptr addrspace(3) %ptr, <2
; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v3
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB21_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: v_mov_b32_e32 v0, v2
; GFX11-NEXT: s_setpc_b64 s[30:31]
;
@@ -5914,7 +5959,7 @@ define void @local_atomic_fadd_noret_v2f16(ptr addrspace(3) %ptr, <2 x half> %va
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v2
; GFX11-NEXT: v_mov_b32_e32 v2, v3
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB22_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -6116,7 +6161,7 @@ define void @local_atomic_fadd_noret_v2f16__offset(ptr addrspace(3) %ptr, <2 x h
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v2
; GFX11-NEXT: v_mov_b32_e32 v2, v3
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB23_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -6351,12 +6396,13 @@ define <2 x bfloat> @local_atomic_fadd_ret_v2bf16(ptr addrspace(3) %ptr, <2 x bf
; GFX11-TRUE16-NEXT: buffer_gl0_inv
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v4
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB24_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-TRUE16-NEXT: s_set_inst_prefetch_distance 0x2
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_mov_b32_e32 v0, v2
; GFX11-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -6398,12 +6444,13 @@ define <2 x bfloat> @local_atomic_fadd_ret_v2bf16(ptr addrspace(3) %ptr, <2 x bf
; GFX11-FAKE16-NEXT: buffer_gl0_inv
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v4
; GFX11-FAKE16-NEXT: s_or_b32 s1, vcc_lo, s1
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB24_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-FAKE16-NEXT: s_set_inst_prefetch_distance 0x2
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s1
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v0, v2
; GFX11-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -6717,12 +6764,13 @@ define <2 x bfloat> @local_atomic_fadd_ret_v2bf16__offset(ptr addrspace(3) %ptr,
; GFX11-TRUE16-NEXT: buffer_gl0_inv
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v4
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB25_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-TRUE16-NEXT: s_set_inst_prefetch_distance 0x2
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_mov_b32_e32 v0, v2
; GFX11-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -6764,12 +6812,13 @@ define <2 x bfloat> @local_atomic_fadd_ret_v2bf16__offset(ptr addrspace(3) %ptr,
; GFX11-FAKE16-NEXT: buffer_gl0_inv
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v4
; GFX11-FAKE16-NEXT: s_or_b32 s1, vcc_lo, s1
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB25_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-FAKE16-NEXT: s_set_inst_prefetch_distance 0x2
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s1
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v0, v2
; GFX11-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -7084,7 +7133,7 @@ define void @local_atomic_fadd_noret_v2bf16(ptr addrspace(3) %ptr, <2 x bfloat>
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v4, v3
; GFX11-TRUE16-NEXT: v_mov_b32_e32 v3, v4
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB26_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -7117,7 +7166,7 @@ define void @local_atomic_fadd_noret_v2bf16(ptr addrspace(3) %ptr, <2 x bfloat>
; GFX11-FAKE16-NEXT: v_add3_u32 v7, v7, v5, 0x7fff
; GFX11-FAKE16-NEXT: v_add3_u32 v6, v6, v4, 0x7fff
; GFX11-FAKE16-NEXT: v_cmp_u_f32_e64 s0, v4, v4
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_cndmask_b32_e32 v5, v7, v9, vcc_lo
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v4, v6, v8, s0
; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
@@ -7129,7 +7178,7 @@ define void @local_atomic_fadd_noret_v2bf16(ptr addrspace(3) %ptr, <2 x bfloat>
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v4, v3
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v3, v4
; GFX11-FAKE16-NEXT: s_or_b32 s1, vcc_lo, s1
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB26_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -7433,7 +7482,7 @@ define void @local_atomic_fadd_noret_v2bf16__ofset(ptr addrspace(3) %ptr, <2 x b
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v4, v3
; GFX11-TRUE16-NEXT: v_mov_b32_e32 v3, v4
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB27_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -7466,7 +7515,7 @@ define void @local_atomic_fadd_noret_v2bf16__ofset(ptr addrspace(3) %ptr, <2 x b
; GFX11-FAKE16-NEXT: v_add3_u32 v7, v7, v5, 0x7fff
; GFX11-FAKE16-NEXT: v_add3_u32 v6, v6, v4, 0x7fff
; GFX11-FAKE16-NEXT: v_cmp_u_f32_e64 s0, v4, v4
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_cndmask_b32_e32 v5, v7, v9, vcc_lo
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v4, v6, v8, s0
; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
@@ -7478,7 +7527,7 @@ define void @local_atomic_fadd_noret_v2bf16__ofset(ptr addrspace(3) %ptr, <2 x b
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v4, v3
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v3, v4
; GFX11-FAKE16-NEXT: s_or_b32 s1, vcc_lo, s1
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB27_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -7760,6 +7809,7 @@ define amdgpu_kernel void @local_ds_fadd(ptr addrspace(1) %out, ptr addrspace(3)
; GFX12-NEXT: v_mbcnt_lo_u32_b32 v2, s7, 0
; GFX12-NEXT: s_mov_b32 s6, exec_lo
; GFX12-NEXT: v_cmpx_eq_u32_e32 0, v2
+; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-NEXT: s_cbranch_execz .LBB28_4
; GFX12-NEXT: ; %bb.3:
; GFX12-NEXT: s_bcnt1_i32_b32 s0, s7
@@ -7775,13 +7825,13 @@ define amdgpu_kernel void @local_ds_fadd(ptr addrspace(1) %out, ptr addrspace(3)
; GFX12-NEXT: .LBB28_4:
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s6
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(VALU_DEP_1)
; GFX12-NEXT: v_cvt_f32_ubyte0_e32 v0, v0
; GFX12-NEXT: s_mov_b32 s1, exec_lo
; GFX12-NEXT: s_brev_b32 s0, 1
-; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX12-NEXT: v_mul_f32_e32 v0, 0x42280000, v0
+; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX12-NEXT: v_add_f32_e32 v0, s3, v0
-; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-NEXT: v_cndmask_b32_e64 v1, v0, s3, vcc_lo
; GFX12-NEXT: ; implicit-def: $vgpr0
; GFX12-NEXT: .LBB28_5: ; %ComputeLoop
@@ -7938,6 +7988,7 @@ define amdgpu_kernel void @local_ds_fadd(ptr addrspace(1) %out, ptr addrspace(3)
; GFX11-NEXT: v_mbcnt_lo_u32_b32 v2, s7, 0
; GFX11-NEXT: s_mov_b32 s6, exec_lo
; GFX11-NEXT: v_cmpx_eq_u32_e32 0, v2
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_cbranch_execz .LBB28_4
; GFX11-NEXT: ; %bb.3:
; GFX11-NEXT: s_bcnt1_i32_b32 s0, s7
@@ -7951,13 +8002,13 @@ define amdgpu_kernel void @local_ds_fadd(ptr addrspace(1) %out, ptr addrspace(3)
; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: .LBB28_4:
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s6
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(VALU_DEP_2)
; GFX11-NEXT: v_cvt_f32_ubyte0_e32 v0, v0
; GFX11-NEXT: v_bfrev_b32_e32 v1, 1
; GFX11-NEXT: s_mov_b32 s0, exec_lo
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_mul_f32_e32 v0, 0x42280000, v0
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_f32_e32 v0, s3, v0
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_cndmask_b32_e64 v2, v0, s3, vcc_lo
; GFX11-NEXT: ; implicit-def: $vgpr0
; GFX11-NEXT: .LBB28_5: ; %ComputeLoop
@@ -8618,7 +8669,7 @@ define amdgpu_kernel void @local_ds_fadd_one_as(ptr addrspace(1) %out, ptr addrs
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: v_mbcnt_lo_u32_b32 v2, s7, 0
; GFX12-NEXT: s_mov_b32 s6, exec_lo
-; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX12-NEXT: v_cmpx_eq_u32_e32 0, v2
; GFX12-NEXT: s_cbranch_execz .LBB29_4
; GFX12-NEXT: ; %bb.3:
@@ -8632,13 +8683,13 @@ define amdgpu_kernel void @local_ds_fadd_one_as(ptr addrspace(1) %out, ptr addrs
; GFX12-NEXT: .LBB29_4:
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s6
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(VALU_DEP_1)
; GFX12-NEXT: v_cvt_f32_ubyte0_e32 v0, v0
; GFX12-NEXT: s_mov_b32 s1, exec_lo
; GFX12-NEXT: s_brev_b32 s0, 1
-; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX12-NEXT: v_mul_f32_e32 v0, 0x42280000, v0
+; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX12-NEXT: v_add_f32_e32 v0, s3, v0
-; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-NEXT: v_cndmask_b32_e64 v1, v0, s3, vcc_lo
; GFX12-NEXT: ; implicit-def: $vgpr0
; GFX12-NEXT: .LBB29_5: ; %ComputeLoop
@@ -8788,6 +8839,7 @@ define amdgpu_kernel void @local_ds_fadd_one_as(ptr addrspace(1) %out, ptr addrs
; GFX11-NEXT: v_mbcnt_lo_u32_b32 v2, s7, 0
; GFX11-NEXT: s_mov_b32 s6, exec_lo
; GFX11-NEXT: v_cmpx_eq_u32_e32 0, v2
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_cbranch_execz .LBB29_4
; GFX11-NEXT: ; %bb.3:
; GFX11-NEXT: s_bcnt1_i32_b32 s0, s7
@@ -8799,13 +8851,13 @@ define amdgpu_kernel void @local_ds_fadd_one_as(ptr addrspace(1) %out, ptr addrs
; GFX11-NEXT: ds_add_f32 v2, v1
; GFX11-NEXT: .LBB29_4:
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s6
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(VALU_DEP_2)
; GFX11-NEXT: v_cvt_f32_ubyte0_e32 v0, v0
; GFX11-NEXT: v_bfrev_b32_e32 v1, 1
; GFX11-NEXT: s_mov_b32 s0, exec_lo
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_mul_f32_e32 v0, 0x42280000, v0
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_f32_e32 v0, s3, v0
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_cndmask_b32_e64 v2, v0, s3, vcc_lo
; GFX11-NEXT: ; implicit-def: $vgpr0
; GFX11-NEXT: .LBB29_5: ; %ComputeLoop
diff --git a/llvm/test/CodeGen/AMDGPU/local-atomicrmw-fmax.ll b/llvm/test/CodeGen/AMDGPU/local-atomicrmw-fmax.ll
index 369b06bb29697e..8587377c0fc2dd 100644
--- a/llvm/test/CodeGen/AMDGPU/local-atomicrmw-fmax.ll
+++ b/llvm/test/CodeGen/AMDGPU/local-atomicrmw-fmax.ll
@@ -806,9 +806,11 @@ define half @local_atomic_fmax_ret_f16(ptr addrspace(3) %ptr) nounwind {
; GFX12-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_cbranch_execnz .LBB8_1
; GFX12-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: v_lshrrev_b32_e32 v0, v0, v3
; GFX12-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -850,9 +852,11 @@ define half @local_atomic_fmax_ret_f16(ptr addrspace(3) %ptr) nounwind {
; GFX12-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: s_cbranch_execnz .LBB8_1
; GFX12-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: v_lshrrev_b32_e32 v0, v0, v3
; GFX12-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -917,11 +921,12 @@ define half @local_atomic_fmax_ret_f16(ptr addrspace(3) %ptr) nounwind {
; GFX11-TRUE16-NEXT: buffer_gl0_inv
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB8_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v0, v0, v3
; GFX11-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -955,11 +960,12 @@ define half @local_atomic_fmax_ret_f16(ptr addrspace(3) %ptr) nounwind {
; GFX11-FAKE16-NEXT: buffer_gl0_inv
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB8_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v0, v0, v3
; GFX11-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -1200,9 +1206,11 @@ define half @local_atomic_fmax_ret_f16__offset(ptr addrspace(3) %ptr) nounwind {
; GFX12-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_cbranch_execnz .LBB9_1
; GFX12-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: v_lshrrev_b32_e32 v0, v1, v3
; GFX12-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -1246,9 +1254,11 @@ define half @local_atomic_fmax_ret_f16__offset(ptr addrspace(3) %ptr) nounwind {
; GFX12-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: s_cbranch_execnz .LBB9_1
; GFX12-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: v_lshrrev_b32_e32 v0, v1, v3
; GFX12-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -1316,11 +1326,12 @@ define half @local_atomic_fmax_ret_f16__offset(ptr addrspace(3) %ptr) nounwind {
; GFX11-TRUE16-NEXT: buffer_gl0_inv
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB9_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v0, v1, v3
; GFX11-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -1356,11 +1367,12 @@ define half @local_atomic_fmax_ret_f16__offset(ptr addrspace(3) %ptr) nounwind {
; GFX11-FAKE16-NEXT: buffer_gl0_inv
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB9_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v0, v1, v3
; GFX11-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -1606,6 +1618,7 @@ define void @local_atomic_fmax_noret_f16(ptr addrspace(3) %ptr) nounwind {
; GFX12-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_cbranch_execnz .LBB10_1
; GFX12-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -1649,6 +1662,7 @@ define void @local_atomic_fmax_noret_f16(ptr addrspace(3) %ptr) nounwind {
; GFX12-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: s_cbranch_execnz .LBB10_1
; GFX12-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -1714,7 +1728,7 @@ define void @local_atomic_fmax_noret_f16(ptr addrspace(3) %ptr) nounwind {
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v4, v2
; GFX11-TRUE16-NEXT: v_mov_b32_e32 v2, v4
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB10_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -1751,7 +1765,7 @@ define void @local_atomic_fmax_noret_f16(ptr addrspace(3) %ptr) nounwind {
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v4, v2
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v2, v4
; GFX11-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB10_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -1989,6 +2003,7 @@ define void @local_atomic_fmax_noret_f16__offset(ptr addrspace(3) %ptr) nounwind
; GFX12-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_cbranch_execnz .LBB11_1
; GFX12-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -2034,6 +2049,7 @@ define void @local_atomic_fmax_noret_f16__offset(ptr addrspace(3) %ptr) nounwind
; GFX12-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: s_cbranch_execnz .LBB11_1
; GFX12-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -2102,7 +2118,7 @@ define void @local_atomic_fmax_noret_f16__offset(ptr addrspace(3) %ptr) nounwind
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v4, v3
; GFX11-TRUE16-NEXT: v_mov_b32_e32 v3, v4
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB11_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -2141,7 +2157,7 @@ define void @local_atomic_fmax_noret_f16__offset(ptr addrspace(3) %ptr) nounwind
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v4, v3
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v3, v4
; GFX11-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB11_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -2375,9 +2391,11 @@ define half @local_atomic_fmax_ret_f16__offset__align4(ptr addrspace(3) %ptr) no
; GFX12-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_cbranch_execnz .LBB12_1
; GFX12-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: v_mov_b16_e32 v0.l, v1.l
; GFX12-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -2410,9 +2428,11 @@ define half @local_atomic_fmax_ret_f16__offset__align4(ptr addrspace(3) %ptr) no
; GFX12-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: s_cbranch_execnz .LBB12_1
; GFX12-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: v_mov_b32_e32 v0, v1
; GFX12-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -2461,11 +2481,12 @@ define half @local_atomic_fmax_ret_f16__offset__align4(ptr addrspace(3) %ptr) no
; GFX11-TRUE16-NEXT: buffer_gl0_inv
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v1, v2
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB12_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v0.l, v1.l
; GFX11-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -2490,11 +2511,12 @@ define half @local_atomic_fmax_ret_f16__offset__align4(ptr addrspace(3) %ptr) no
; GFX11-FAKE16-NEXT: buffer_gl0_inv
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v1, v2
; GFX11-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB12_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v0, v1
; GFX11-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -2685,6 +2707,7 @@ define void @local_atomic_fmax_noret_f16__offset__align4(ptr addrspace(3) %ptr)
; GFX12-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_cbranch_execnz .LBB13_1
; GFX12-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -2719,6 +2742,7 @@ define void @local_atomic_fmax_noret_f16__offset__align4(ptr addrspace(3) %ptr)
; GFX12-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: s_cbranch_execnz .LBB13_1
; GFX12-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -2768,7 +2792,7 @@ define void @local_atomic_fmax_noret_f16__offset__align4(ptr addrspace(3) %ptr)
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v1
; GFX11-TRUE16-NEXT: v_mov_b32_e32 v1, v2
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB13_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -2796,7 +2820,7 @@ define void @local_atomic_fmax_noret_f16__offset__align4(ptr addrspace(3) %ptr)
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v1
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v1, v2
; GFX11-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB13_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -3005,9 +3029,11 @@ define bfloat @local_atomic_fmax_ret_bf16(ptr addrspace(3) %ptr) nounwind {
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB14_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: v_lshrrev_b32_e32 v0, v0, v3
; GFX12-NEXT: s_setpc_b64 s[30:31]
;
@@ -3087,11 +3113,12 @@ define bfloat @local_atomic_fmax_ret_bf16(ptr addrspace(3) %ptr) nounwind {
; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB14_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: v_lshrrev_b32_e32 v0, v0, v3
; GFX11-NEXT: s_setpc_b64 s[30:31]
;
@@ -3361,9 +3388,11 @@ define bfloat @local_atomic_fmax_ret_bf16__offset(ptr addrspace(3) %ptr) nounwin
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB15_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: v_lshrrev_b32_e32 v0, v1, v3
; GFX12-NEXT: s_setpc_b64 s[30:31]
;
@@ -3446,11 +3475,12 @@ define bfloat @local_atomic_fmax_ret_bf16__offset(ptr addrspace(3) %ptr) nounwin
; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB15_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: v_lshrrev_b32_e32 v0, v1, v3
; GFX11-NEXT: s_setpc_b64 s[30:31]
;
@@ -3725,6 +3755,7 @@ define void @local_atomic_fmax_noret_bf16(ptr addrspace(3) %ptr) nounwind {
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB16_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -3805,7 +3836,7 @@ define void @local_atomic_fmax_noret_bf16(ptr addrspace(3) %ptr) nounwind {
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v4, v2
; GFX11-NEXT: v_mov_b32_e32 v2, v4
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB16_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -4071,6 +4102,7 @@ define void @local_atomic_fmax_noret_bf16__offset(ptr addrspace(3) %ptr) nounwin
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB17_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -4153,7 +4185,7 @@ define void @local_atomic_fmax_noret_bf16__offset(ptr addrspace(3) %ptr) nounwin
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v4, v3
; GFX11-NEXT: v_mov_b32_e32 v3, v4
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB17_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -4416,9 +4448,11 @@ define bfloat @local_atomic_fmax_ret_bf16__offset__align4(ptr addrspace(3) %ptr)
; GFX12-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_cbranch_execnz .LBB18_1
; GFX12-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: v_mov_b16_e32 v0.l, v1.l
; GFX12-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -4459,9 +4493,11 @@ define bfloat @local_atomic_fmax_ret_bf16__offset__align4(ptr addrspace(3) %ptr)
; GFX12-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: s_cbranch_execnz .LBB18_1
; GFX12-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: v_mov_b32_e32 v0, v1
; GFX12-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -4526,11 +4562,12 @@ define bfloat @local_atomic_fmax_ret_bf16__offset__align4(ptr addrspace(3) %ptr)
; GFX11-TRUE16-NEXT: buffer_gl0_inv
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v1, v2
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB18_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v0.l, v1.l
; GFX11-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -4563,11 +4600,12 @@ define bfloat @local_atomic_fmax_ret_bf16__offset__align4(ptr addrspace(3) %ptr)
; GFX11-FAKE16-NEXT: buffer_gl0_inv
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v1, v2
; GFX11-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB18_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v0, v1
; GFX11-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -4792,6 +4830,7 @@ define void @local_atomic_fmax_noret_bf16__offset__align4(ptr addrspace(3) %ptr)
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB19_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -4856,7 +4895,7 @@ define void @local_atomic_fmax_noret_bf16__offset__align4(ptr addrspace(3) %ptr)
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v1
; GFX11-NEXT: v_mov_b32_e32 v1, v2
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB19_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -5073,9 +5112,11 @@ define <2 x half> @local_atomic_fmax_ret_v2f16(ptr addrspace(3) %ptr, <2 x half>
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB20_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: v_mov_b32_e32 v0, v2
; GFX12-NEXT: s_setpc_b64 s[30:31]
;
@@ -5122,11 +5163,12 @@ define <2 x half> @local_atomic_fmax_ret_v2f16(ptr addrspace(3) %ptr, <2 x half>
; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v3
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB20_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: v_mov_b32_e32 v0, v2
; GFX11-NEXT: s_setpc_b64 s[30:31]
;
@@ -5337,9 +5379,11 @@ define <2 x half> @local_atomic_fmax_ret_v2f16__offset(ptr addrspace(3) %ptr, <2
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB21_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: v_mov_b32_e32 v0, v2
; GFX12-NEXT: s_setpc_b64 s[30:31]
;
@@ -5386,11 +5430,12 @@ define <2 x half> @local_atomic_fmax_ret_v2f16__offset(ptr addrspace(3) %ptr, <2
; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v3
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB21_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: v_mov_b32_e32 v0, v2
; GFX11-NEXT: s_setpc_b64 s[30:31]
;
@@ -5603,6 +5648,7 @@ define void @local_atomic_fmax_noret_v2f16(ptr addrspace(3) %ptr, <2 x half> %va
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB22_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -5650,7 +5696,7 @@ define void @local_atomic_fmax_noret_v2f16(ptr addrspace(3) %ptr, <2 x half> %va
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v2
; GFX11-NEXT: v_mov_b32_e32 v2, v3
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB22_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -5853,6 +5899,7 @@ define void @local_atomic_fmax_noret_v2f16__offset(ptr addrspace(3) %ptr, <2 x h
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB23_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -5900,7 +5947,7 @@ define void @local_atomic_fmax_noret_v2f16__offset(ptr addrspace(3) %ptr, <2 x h
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v2
; GFX11-NEXT: v_mov_b32_e32 v2, v3
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB23_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -6133,9 +6180,11 @@ define <2 x bfloat> @local_atomic_fmax_ret_v2bf16(ptr addrspace(3) %ptr, <2 x bf
; GFX12-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_cbranch_execnz .LBB24_1
; GFX12-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: v_mov_b32_e32 v0, v2
; GFX12-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -6185,9 +6234,11 @@ define <2 x bfloat> @local_atomic_fmax_ret_v2bf16(ptr addrspace(3) %ptr, <2 x bf
; GFX12-FAKE16-NEXT: s_or_b32 s1, vcc_lo, s1
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: s_cbranch_execnz .LBB24_1
; GFX12-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s1
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: v_mov_b32_e32 v0, v2
; GFX12-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -6273,12 +6324,13 @@ define <2 x bfloat> @local_atomic_fmax_ret_v2bf16(ptr addrspace(3) %ptr, <2 x bf
; GFX11-TRUE16-NEXT: buffer_gl0_inv
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v4
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB24_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-TRUE16-NEXT: s_set_inst_prefetch_distance 0x2
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_mov_b32_e32 v0, v2
; GFX11-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -6320,12 +6372,13 @@ define <2 x bfloat> @local_atomic_fmax_ret_v2bf16(ptr addrspace(3) %ptr, <2 x bf
; GFX11-FAKE16-NEXT: buffer_gl0_inv
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v4
; GFX11-FAKE16-NEXT: s_or_b32 s1, vcc_lo, s1
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB24_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-FAKE16-NEXT: s_set_inst_prefetch_distance 0x2
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s1
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v0, v2
; GFX11-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -6627,9 +6680,11 @@ define <2 x bfloat> @local_atomic_fmax_ret_v2bf16__offset(ptr addrspace(3) %ptr,
; GFX12-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_cbranch_execnz .LBB25_1
; GFX12-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: v_mov_b32_e32 v0, v2
; GFX12-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -6679,9 +6734,11 @@ define <2 x bfloat> @local_atomic_fmax_ret_v2bf16__offset(ptr addrspace(3) %ptr,
; GFX12-FAKE16-NEXT: s_or_b32 s1, vcc_lo, s1
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: s_cbranch_execnz .LBB25_1
; GFX12-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s1
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: v_mov_b32_e32 v0, v2
; GFX12-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -6767,12 +6824,13 @@ define <2 x bfloat> @local_atomic_fmax_ret_v2bf16__offset(ptr addrspace(3) %ptr,
; GFX11-TRUE16-NEXT: buffer_gl0_inv
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v4
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB25_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-TRUE16-NEXT: s_set_inst_prefetch_distance 0x2
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_mov_b32_e32 v0, v2
; GFX11-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -6814,12 +6872,13 @@ define <2 x bfloat> @local_atomic_fmax_ret_v2bf16__offset(ptr addrspace(3) %ptr,
; GFX11-FAKE16-NEXT: buffer_gl0_inv
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v4
; GFX11-FAKE16-NEXT: s_or_b32 s1, vcc_lo, s1
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB25_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-FAKE16-NEXT: s_set_inst_prefetch_distance 0x2
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s1
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v0, v2
; GFX11-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -7121,6 +7180,7 @@ define void @local_atomic_fmax_noret_v2bf16(ptr addrspace(3) %ptr, <2 x bfloat>
; GFX12-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_cbranch_execnz .LBB26_1
; GFX12-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -7154,11 +7214,10 @@ define void @local_atomic_fmax_noret_v2bf16(ptr addrspace(3) %ptr, <2 x bfloat>
; GFX12-FAKE16-NEXT: v_add3_u32 v6, v6, v4, 0x7fff
; GFX12-FAKE16-NEXT: v_cmp_u_f32_e64 s0, v4, v4
; GFX12-FAKE16-NEXT: s_wait_alu depctr_va_vcc(0)
-; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_1) | instid1(VALU_DEP_2)
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_2) | instid1(VALU_DEP_1)
; GFX12-FAKE16-NEXT: v_cndmask_b32_e32 v5, v7, v9, vcc_lo
; GFX12-FAKE16-NEXT: s_wait_alu depctr_va_sdst(0)
; GFX12-FAKE16-NEXT: v_cndmask_b32_e64 v4, v6, v8, s0
-; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-FAKE16-NEXT: v_perm_b32 v4, v5, v4, 0x7060302
; GFX12-FAKE16-NEXT: s_wait_loadcnt 0x0
; GFX12-FAKE16-NEXT: s_wait_storecnt 0x0
@@ -7171,6 +7230,7 @@ define void @local_atomic_fmax_noret_v2bf16(ptr addrspace(3) %ptr, <2 x bfloat>
; GFX12-FAKE16-NEXT: s_or_b32 s1, vcc_lo, s1
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: s_cbranch_execnz .LBB26_1
; GFX12-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s1
@@ -7255,7 +7315,7 @@ define void @local_atomic_fmax_noret_v2bf16(ptr addrspace(3) %ptr, <2 x bfloat>
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v4, v3
; GFX11-TRUE16-NEXT: v_mov_b32_e32 v3, v4
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB26_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -7288,7 +7348,7 @@ define void @local_atomic_fmax_noret_v2bf16(ptr addrspace(3) %ptr, <2 x bfloat>
; GFX11-FAKE16-NEXT: v_add3_u32 v7, v7, v5, 0x7fff
; GFX11-FAKE16-NEXT: v_add3_u32 v6, v6, v4, 0x7fff
; GFX11-FAKE16-NEXT: v_cmp_u_f32_e64 s0, v4, v4
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_cndmask_b32_e32 v5, v7, v9, vcc_lo
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v4, v6, v8, s0
; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
@@ -7300,7 +7360,7 @@ define void @local_atomic_fmax_noret_v2bf16(ptr addrspace(3) %ptr, <2 x bfloat>
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v4, v3
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v3, v4
; GFX11-FAKE16-NEXT: s_or_b32 s1, vcc_lo, s1
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB26_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -7591,6 +7651,7 @@ define void @local_atomic_fmax_noret_v2bf16__ofset(ptr addrspace(3) %ptr, <2 x b
; GFX12-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_cbranch_execnz .LBB27_1
; GFX12-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -7624,11 +7685,10 @@ define void @local_atomic_fmax_noret_v2bf16__ofset(ptr addrspace(3) %ptr, <2 x b
; GFX12-FAKE16-NEXT: v_add3_u32 v6, v6, v4, 0x7fff
; GFX12-FAKE16-NEXT: v_cmp_u_f32_e64 s0, v4, v4
; GFX12-FAKE16-NEXT: s_wait_alu depctr_va_vcc(0)
-; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_1) | instid1(VALU_DEP_2)
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_2) | instid1(VALU_DEP_1)
; GFX12-FAKE16-NEXT: v_cndmask_b32_e32 v5, v7, v9, vcc_lo
; GFX12-FAKE16-NEXT: s_wait_alu depctr_va_sdst(0)
; GFX12-FAKE16-NEXT: v_cndmask_b32_e64 v4, v6, v8, s0
-; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-FAKE16-NEXT: v_perm_b32 v4, v5, v4, 0x7060302
; GFX12-FAKE16-NEXT: s_wait_loadcnt 0x0
; GFX12-FAKE16-NEXT: s_wait_storecnt 0x0
@@ -7641,6 +7701,7 @@ define void @local_atomic_fmax_noret_v2bf16__ofset(ptr addrspace(3) %ptr, <2 x b
; GFX12-FAKE16-NEXT: s_or_b32 s1, vcc_lo, s1
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: s_cbranch_execnz .LBB27_1
; GFX12-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s1
@@ -7725,7 +7786,7 @@ define void @local_atomic_fmax_noret_v2bf16__ofset(ptr addrspace(3) %ptr, <2 x b
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v4, v3
; GFX11-TRUE16-NEXT: v_mov_b32_e32 v3, v4
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB27_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -7758,7 +7819,7 @@ define void @local_atomic_fmax_noret_v2bf16__ofset(ptr addrspace(3) %ptr, <2 x b
; GFX11-FAKE16-NEXT: v_add3_u32 v7, v7, v5, 0x7fff
; GFX11-FAKE16-NEXT: v_add3_u32 v6, v6, v4, 0x7fff
; GFX11-FAKE16-NEXT: v_cmp_u_f32_e64 s0, v4, v4
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_cndmask_b32_e32 v5, v7, v9, vcc_lo
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v4, v6, v8, s0
; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
@@ -7770,7 +7831,7 @@ define void @local_atomic_fmax_noret_v2bf16__ofset(ptr addrspace(3) %ptr, <2 x b
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v4, v3
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v3, v4
; GFX11-FAKE16-NEXT: s_or_b32 s1, vcc_lo, s1
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB27_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
diff --git a/llvm/test/CodeGen/AMDGPU/local-atomicrmw-fmin.ll b/llvm/test/CodeGen/AMDGPU/local-atomicrmw-fmin.ll
index 88e1469580618d..c900a56fe8fa53 100644
--- a/llvm/test/CodeGen/AMDGPU/local-atomicrmw-fmin.ll
+++ b/llvm/test/CodeGen/AMDGPU/local-atomicrmw-fmin.ll
@@ -806,9 +806,11 @@ define half @local_atomic_fmin_ret_f16(ptr addrspace(3) %ptr) nounwind {
; GFX12-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_cbranch_execnz .LBB8_1
; GFX12-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: v_lshrrev_b32_e32 v0, v0, v3
; GFX12-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -850,9 +852,11 @@ define half @local_atomic_fmin_ret_f16(ptr addrspace(3) %ptr) nounwind {
; GFX12-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: s_cbranch_execnz .LBB8_1
; GFX12-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: v_lshrrev_b32_e32 v0, v0, v3
; GFX12-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -917,11 +921,12 @@ define half @local_atomic_fmin_ret_f16(ptr addrspace(3) %ptr) nounwind {
; GFX11-TRUE16-NEXT: buffer_gl0_inv
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB8_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v0, v0, v3
; GFX11-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -955,11 +960,12 @@ define half @local_atomic_fmin_ret_f16(ptr addrspace(3) %ptr) nounwind {
; GFX11-FAKE16-NEXT: buffer_gl0_inv
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB8_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v0, v0, v3
; GFX11-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -1200,9 +1206,11 @@ define half @local_atomic_fmin_ret_f16__offset(ptr addrspace(3) %ptr) nounwind {
; GFX12-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_cbranch_execnz .LBB9_1
; GFX12-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: v_lshrrev_b32_e32 v0, v1, v3
; GFX12-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -1246,9 +1254,11 @@ define half @local_atomic_fmin_ret_f16__offset(ptr addrspace(3) %ptr) nounwind {
; GFX12-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: s_cbranch_execnz .LBB9_1
; GFX12-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: v_lshrrev_b32_e32 v0, v1, v3
; GFX12-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -1316,11 +1326,12 @@ define half @local_atomic_fmin_ret_f16__offset(ptr addrspace(3) %ptr) nounwind {
; GFX11-TRUE16-NEXT: buffer_gl0_inv
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB9_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v0, v1, v3
; GFX11-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -1356,11 +1367,12 @@ define half @local_atomic_fmin_ret_f16__offset(ptr addrspace(3) %ptr) nounwind {
; GFX11-FAKE16-NEXT: buffer_gl0_inv
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB9_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v0, v1, v3
; GFX11-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -1606,6 +1618,7 @@ define void @local_atomic_fmin_noret_f16(ptr addrspace(3) %ptr) nounwind {
; GFX12-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_cbranch_execnz .LBB10_1
; GFX12-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -1649,6 +1662,7 @@ define void @local_atomic_fmin_noret_f16(ptr addrspace(3) %ptr) nounwind {
; GFX12-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: s_cbranch_execnz .LBB10_1
; GFX12-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -1714,7 +1728,7 @@ define void @local_atomic_fmin_noret_f16(ptr addrspace(3) %ptr) nounwind {
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v4, v2
; GFX11-TRUE16-NEXT: v_mov_b32_e32 v2, v4
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB10_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -1751,7 +1765,7 @@ define void @local_atomic_fmin_noret_f16(ptr addrspace(3) %ptr) nounwind {
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v4, v2
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v2, v4
; GFX11-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB10_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -1989,6 +2003,7 @@ define void @local_atomic_fmin_noret_f16__offset(ptr addrspace(3) %ptr) nounwind
; GFX12-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_cbranch_execnz .LBB11_1
; GFX12-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -2034,6 +2049,7 @@ define void @local_atomic_fmin_noret_f16__offset(ptr addrspace(3) %ptr) nounwind
; GFX12-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: s_cbranch_execnz .LBB11_1
; GFX12-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -2102,7 +2118,7 @@ define void @local_atomic_fmin_noret_f16__offset(ptr addrspace(3) %ptr) nounwind
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v4, v3
; GFX11-TRUE16-NEXT: v_mov_b32_e32 v3, v4
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB11_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -2141,7 +2157,7 @@ define void @local_atomic_fmin_noret_f16__offset(ptr addrspace(3) %ptr) nounwind
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v4, v3
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v3, v4
; GFX11-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB11_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -2375,9 +2391,11 @@ define half @local_atomic_fmin_ret_f16__offset__align4(ptr addrspace(3) %ptr) no
; GFX12-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_cbranch_execnz .LBB12_1
; GFX12-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: v_mov_b16_e32 v0.l, v1.l
; GFX12-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -2410,9 +2428,11 @@ define half @local_atomic_fmin_ret_f16__offset__align4(ptr addrspace(3) %ptr) no
; GFX12-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: s_cbranch_execnz .LBB12_1
; GFX12-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: v_mov_b32_e32 v0, v1
; GFX12-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -2461,11 +2481,12 @@ define half @local_atomic_fmin_ret_f16__offset__align4(ptr addrspace(3) %ptr) no
; GFX11-TRUE16-NEXT: buffer_gl0_inv
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v1, v2
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB12_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v0.l, v1.l
; GFX11-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -2490,11 +2511,12 @@ define half @local_atomic_fmin_ret_f16__offset__align4(ptr addrspace(3) %ptr) no
; GFX11-FAKE16-NEXT: buffer_gl0_inv
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v1, v2
; GFX11-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB12_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v0, v1
; GFX11-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -2685,6 +2707,7 @@ define void @local_atomic_fmin_noret_f16__offset__align4(ptr addrspace(3) %ptr)
; GFX12-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_cbranch_execnz .LBB13_1
; GFX12-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -2719,6 +2742,7 @@ define void @local_atomic_fmin_noret_f16__offset__align4(ptr addrspace(3) %ptr)
; GFX12-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: s_cbranch_execnz .LBB13_1
; GFX12-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -2768,7 +2792,7 @@ define void @local_atomic_fmin_noret_f16__offset__align4(ptr addrspace(3) %ptr)
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v1
; GFX11-TRUE16-NEXT: v_mov_b32_e32 v1, v2
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB13_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -2796,7 +2820,7 @@ define void @local_atomic_fmin_noret_f16__offset__align4(ptr addrspace(3) %ptr)
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v1
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v1, v2
; GFX11-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB13_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -3005,9 +3029,11 @@ define bfloat @local_atomic_fmin_ret_bf16(ptr addrspace(3) %ptr) nounwind {
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB14_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: v_lshrrev_b32_e32 v0, v0, v3
; GFX12-NEXT: s_setpc_b64 s[30:31]
;
@@ -3087,11 +3113,12 @@ define bfloat @local_atomic_fmin_ret_bf16(ptr addrspace(3) %ptr) nounwind {
; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB14_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: v_lshrrev_b32_e32 v0, v0, v3
; GFX11-NEXT: s_setpc_b64 s[30:31]
;
@@ -3361,9 +3388,11 @@ define bfloat @local_atomic_fmin_ret_bf16__offset(ptr addrspace(3) %ptr) nounwin
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB15_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: v_lshrrev_b32_e32 v0, v1, v3
; GFX12-NEXT: s_setpc_b64 s[30:31]
;
@@ -3446,11 +3475,12 @@ define bfloat @local_atomic_fmin_ret_bf16__offset(ptr addrspace(3) %ptr) nounwin
; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB15_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: v_lshrrev_b32_e32 v0, v1, v3
; GFX11-NEXT: s_setpc_b64 s[30:31]
;
@@ -3725,6 +3755,7 @@ define void @local_atomic_fmin_noret_bf16(ptr addrspace(3) %ptr) nounwind {
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB16_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -3805,7 +3836,7 @@ define void @local_atomic_fmin_noret_bf16(ptr addrspace(3) %ptr) nounwind {
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v4, v2
; GFX11-NEXT: v_mov_b32_e32 v2, v4
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB16_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -4071,6 +4102,7 @@ define void @local_atomic_fmin_noret_bf16__offset(ptr addrspace(3) %ptr) nounwin
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB17_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -4153,7 +4185,7 @@ define void @local_atomic_fmin_noret_bf16__offset(ptr addrspace(3) %ptr) nounwin
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v4, v3
; GFX11-NEXT: v_mov_b32_e32 v3, v4
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB17_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -4416,9 +4448,11 @@ define bfloat @local_atomic_fmin_ret_bf16__offset__align4(ptr addrspace(3) %ptr)
; GFX12-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_cbranch_execnz .LBB18_1
; GFX12-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: v_mov_b16_e32 v0.l, v1.l
; GFX12-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -4459,9 +4493,11 @@ define bfloat @local_atomic_fmin_ret_bf16__offset__align4(ptr addrspace(3) %ptr)
; GFX12-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: s_cbranch_execnz .LBB18_1
; GFX12-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: v_mov_b32_e32 v0, v1
; GFX12-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -4526,11 +4562,12 @@ define bfloat @local_atomic_fmin_ret_bf16__offset__align4(ptr addrspace(3) %ptr)
; GFX11-TRUE16-NEXT: buffer_gl0_inv
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v1, v2
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB18_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v0.l, v1.l
; GFX11-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -4563,11 +4600,12 @@ define bfloat @local_atomic_fmin_ret_bf16__offset__align4(ptr addrspace(3) %ptr)
; GFX11-FAKE16-NEXT: buffer_gl0_inv
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v1, v2
; GFX11-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB18_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v0, v1
; GFX11-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -4792,6 +4830,7 @@ define void @local_atomic_fmin_noret_bf16__offset__align4(ptr addrspace(3) %ptr)
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB19_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -4856,7 +4895,7 @@ define void @local_atomic_fmin_noret_bf16__offset__align4(ptr addrspace(3) %ptr)
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v1
; GFX11-NEXT: v_mov_b32_e32 v1, v2
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB19_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -5073,9 +5112,11 @@ define <2 x half> @local_atomic_fmin_ret_v2f16(ptr addrspace(3) %ptr, <2 x half>
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB20_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: v_mov_b32_e32 v0, v2
; GFX12-NEXT: s_setpc_b64 s[30:31]
;
@@ -5122,11 +5163,12 @@ define <2 x half> @local_atomic_fmin_ret_v2f16(ptr addrspace(3) %ptr, <2 x half>
; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v3
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB20_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: v_mov_b32_e32 v0, v2
; GFX11-NEXT: s_setpc_b64 s[30:31]
;
@@ -5337,9 +5379,11 @@ define <2 x half> @local_atomic_fmin_ret_v2f16__offset(ptr addrspace(3) %ptr, <2
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB21_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: v_mov_b32_e32 v0, v2
; GFX12-NEXT: s_setpc_b64 s[30:31]
;
@@ -5386,11 +5430,12 @@ define <2 x half> @local_atomic_fmin_ret_v2f16__offset(ptr addrspace(3) %ptr, <2
; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v3
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB21_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: v_mov_b32_e32 v0, v2
; GFX11-NEXT: s_setpc_b64 s[30:31]
;
@@ -5603,6 +5648,7 @@ define void @local_atomic_fmin_noret_v2f16(ptr addrspace(3) %ptr, <2 x half> %va
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB22_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -5650,7 +5696,7 @@ define void @local_atomic_fmin_noret_v2f16(ptr addrspace(3) %ptr, <2 x half> %va
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v2
; GFX11-NEXT: v_mov_b32_e32 v2, v3
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB22_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -5853,6 +5899,7 @@ define void @local_atomic_fmin_noret_v2f16__offset(ptr addrspace(3) %ptr, <2 x h
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB23_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -5900,7 +5947,7 @@ define void @local_atomic_fmin_noret_v2f16__offset(ptr addrspace(3) %ptr, <2 x h
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v2
; GFX11-NEXT: v_mov_b32_e32 v2, v3
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB23_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -6133,9 +6180,11 @@ define <2 x bfloat> @local_atomic_fmin_ret_v2bf16(ptr addrspace(3) %ptr, <2 x bf
; GFX12-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_cbranch_execnz .LBB24_1
; GFX12-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: v_mov_b32_e32 v0, v2
; GFX12-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -6185,9 +6234,11 @@ define <2 x bfloat> @local_atomic_fmin_ret_v2bf16(ptr addrspace(3) %ptr, <2 x bf
; GFX12-FAKE16-NEXT: s_or_b32 s1, vcc_lo, s1
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: s_cbranch_execnz .LBB24_1
; GFX12-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s1
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: v_mov_b32_e32 v0, v2
; GFX12-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -6273,12 +6324,13 @@ define <2 x bfloat> @local_atomic_fmin_ret_v2bf16(ptr addrspace(3) %ptr, <2 x bf
; GFX11-TRUE16-NEXT: buffer_gl0_inv
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v4
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB24_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-TRUE16-NEXT: s_set_inst_prefetch_distance 0x2
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_mov_b32_e32 v0, v2
; GFX11-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -6320,12 +6372,13 @@ define <2 x bfloat> @local_atomic_fmin_ret_v2bf16(ptr addrspace(3) %ptr, <2 x bf
; GFX11-FAKE16-NEXT: buffer_gl0_inv
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v4
; GFX11-FAKE16-NEXT: s_or_b32 s1, vcc_lo, s1
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB24_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-FAKE16-NEXT: s_set_inst_prefetch_distance 0x2
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s1
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v0, v2
; GFX11-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -6627,9 +6680,11 @@ define <2 x bfloat> @local_atomic_fmin_ret_v2bf16__offset(ptr addrspace(3) %ptr,
; GFX12-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_cbranch_execnz .LBB25_1
; GFX12-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: v_mov_b32_e32 v0, v2
; GFX12-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -6679,9 +6734,11 @@ define <2 x bfloat> @local_atomic_fmin_ret_v2bf16__offset(ptr addrspace(3) %ptr,
; GFX12-FAKE16-NEXT: s_or_b32 s1, vcc_lo, s1
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: s_cbranch_execnz .LBB25_1
; GFX12-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s1
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: v_mov_b32_e32 v0, v2
; GFX12-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -6767,12 +6824,13 @@ define <2 x bfloat> @local_atomic_fmin_ret_v2bf16__offset(ptr addrspace(3) %ptr,
; GFX11-TRUE16-NEXT: buffer_gl0_inv
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v4
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB25_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-TRUE16-NEXT: s_set_inst_prefetch_distance 0x2
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_mov_b32_e32 v0, v2
; GFX11-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -6814,12 +6872,13 @@ define <2 x bfloat> @local_atomic_fmin_ret_v2bf16__offset(ptr addrspace(3) %ptr,
; GFX11-FAKE16-NEXT: buffer_gl0_inv
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v4
; GFX11-FAKE16-NEXT: s_or_b32 s1, vcc_lo, s1
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB25_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-FAKE16-NEXT: s_set_inst_prefetch_distance 0x2
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s1
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v0, v2
; GFX11-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -7121,6 +7180,7 @@ define void @local_atomic_fmin_noret_v2bf16(ptr addrspace(3) %ptr, <2 x bfloat>
; GFX12-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_cbranch_execnz .LBB26_1
; GFX12-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -7154,11 +7214,10 @@ define void @local_atomic_fmin_noret_v2bf16(ptr addrspace(3) %ptr, <2 x bfloat>
; GFX12-FAKE16-NEXT: v_add3_u32 v6, v6, v4, 0x7fff
; GFX12-FAKE16-NEXT: v_cmp_u_f32_e64 s0, v4, v4
; GFX12-FAKE16-NEXT: s_wait_alu depctr_va_vcc(0)
-; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_1) | instid1(VALU_DEP_2)
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_2) | instid1(VALU_DEP_1)
; GFX12-FAKE16-NEXT: v_cndmask_b32_e32 v5, v7, v9, vcc_lo
; GFX12-FAKE16-NEXT: s_wait_alu depctr_va_sdst(0)
; GFX12-FAKE16-NEXT: v_cndmask_b32_e64 v4, v6, v8, s0
-; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-FAKE16-NEXT: v_perm_b32 v4, v5, v4, 0x7060302
; GFX12-FAKE16-NEXT: s_wait_loadcnt 0x0
; GFX12-FAKE16-NEXT: s_wait_storecnt 0x0
@@ -7171,6 +7230,7 @@ define void @local_atomic_fmin_noret_v2bf16(ptr addrspace(3) %ptr, <2 x bfloat>
; GFX12-FAKE16-NEXT: s_or_b32 s1, vcc_lo, s1
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: s_cbranch_execnz .LBB26_1
; GFX12-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s1
@@ -7255,7 +7315,7 @@ define void @local_atomic_fmin_noret_v2bf16(ptr addrspace(3) %ptr, <2 x bfloat>
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v4, v3
; GFX11-TRUE16-NEXT: v_mov_b32_e32 v3, v4
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB26_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -7288,7 +7348,7 @@ define void @local_atomic_fmin_noret_v2bf16(ptr addrspace(3) %ptr, <2 x bfloat>
; GFX11-FAKE16-NEXT: v_add3_u32 v7, v7, v5, 0x7fff
; GFX11-FAKE16-NEXT: v_add3_u32 v6, v6, v4, 0x7fff
; GFX11-FAKE16-NEXT: v_cmp_u_f32_e64 s0, v4, v4
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_cndmask_b32_e32 v5, v7, v9, vcc_lo
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v4, v6, v8, s0
; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
@@ -7300,7 +7360,7 @@ define void @local_atomic_fmin_noret_v2bf16(ptr addrspace(3) %ptr, <2 x bfloat>
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v4, v3
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v3, v4
; GFX11-FAKE16-NEXT: s_or_b32 s1, vcc_lo, s1
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB26_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -7591,6 +7651,7 @@ define void @local_atomic_fmin_noret_v2bf16__ofset(ptr addrspace(3) %ptr, <2 x b
; GFX12-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_cbranch_execnz .LBB27_1
; GFX12-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -7624,11 +7685,10 @@ define void @local_atomic_fmin_noret_v2bf16__ofset(ptr addrspace(3) %ptr, <2 x b
; GFX12-FAKE16-NEXT: v_add3_u32 v6, v6, v4, 0x7fff
; GFX12-FAKE16-NEXT: v_cmp_u_f32_e64 s0, v4, v4
; GFX12-FAKE16-NEXT: s_wait_alu depctr_va_vcc(0)
-; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_1) | instid1(VALU_DEP_2)
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_2) | instid1(VALU_DEP_1)
; GFX12-FAKE16-NEXT: v_cndmask_b32_e32 v5, v7, v9, vcc_lo
; GFX12-FAKE16-NEXT: s_wait_alu depctr_va_sdst(0)
; GFX12-FAKE16-NEXT: v_cndmask_b32_e64 v4, v6, v8, s0
-; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-FAKE16-NEXT: v_perm_b32 v4, v5, v4, 0x7060302
; GFX12-FAKE16-NEXT: s_wait_loadcnt 0x0
; GFX12-FAKE16-NEXT: s_wait_storecnt 0x0
@@ -7641,6 +7701,7 @@ define void @local_atomic_fmin_noret_v2bf16__ofset(ptr addrspace(3) %ptr, <2 x b
; GFX12-FAKE16-NEXT: s_or_b32 s1, vcc_lo, s1
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: s_cbranch_execnz .LBB27_1
; GFX12-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s1
@@ -7725,7 +7786,7 @@ define void @local_atomic_fmin_noret_v2bf16__ofset(ptr addrspace(3) %ptr, <2 x b
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v4, v3
; GFX11-TRUE16-NEXT: v_mov_b32_e32 v3, v4
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB27_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -7758,7 +7819,7 @@ define void @local_atomic_fmin_noret_v2bf16__ofset(ptr addrspace(3) %ptr, <2 x b
; GFX11-FAKE16-NEXT: v_add3_u32 v7, v7, v5, 0x7fff
; GFX11-FAKE16-NEXT: v_add3_u32 v6, v6, v4, 0x7fff
; GFX11-FAKE16-NEXT: v_cmp_u_f32_e64 s0, v4, v4
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_cndmask_b32_e32 v5, v7, v9, vcc_lo
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v4, v6, v8, s0
; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
@@ -7770,7 +7831,7 @@ define void @local_atomic_fmin_noret_v2bf16__ofset(ptr addrspace(3) %ptr, <2 x b
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v4, v3
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v3, v4
; GFX11-FAKE16-NEXT: s_or_b32 s1, vcc_lo, s1
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB27_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
diff --git a/llvm/test/CodeGen/AMDGPU/local-atomicrmw-fsub.ll b/llvm/test/CodeGen/AMDGPU/local-atomicrmw-fsub.ll
index 515a544b1d0d1b..b48cb37d7f5d21 100644
--- a/llvm/test/CodeGen/AMDGPU/local-atomicrmw-fsub.ll
+++ b/llvm/test/CodeGen/AMDGPU/local-atomicrmw-fsub.ll
@@ -41,9 +41,11 @@ define float @local_atomic_fsub_ret_f32(ptr addrspace(3) %ptr) nounwind {
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB0_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: v_mov_b32_e32 v0, v1
; GFX12-NEXT: s_setpc_b64 s[30:31]
;
@@ -85,11 +87,12 @@ define float @local_atomic_fsub_ret_f32(ptr addrspace(3) %ptr) nounwind {
; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v1, v2
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB0_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: v_mov_b32_e32 v0, v1
; GFX11-NEXT: s_setpc_b64 s[30:31]
;
@@ -253,9 +256,11 @@ define float @local_atomic_fsub_ret_f32__offset(ptr addrspace(3) %ptr) nounwind
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB1_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: v_mov_b32_e32 v0, v1
; GFX12-NEXT: s_setpc_b64 s[30:31]
;
@@ -297,11 +302,12 @@ define float @local_atomic_fsub_ret_f32__offset(ptr addrspace(3) %ptr) nounwind
; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v1, v2
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB1_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: v_mov_b32_e32 v0, v1
; GFX11-NEXT: s_setpc_b64 s[30:31]
;
@@ -467,6 +473,7 @@ define void @local_atomic_fsub_noret_f32(ptr addrspace(3) %ptr) nounwind {
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB2_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -508,7 +515,7 @@ define void @local_atomic_fsub_noret_f32(ptr addrspace(3) %ptr) nounwind {
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v1
; GFX11-NEXT: v_mov_b32_e32 v1, v2
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB2_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -668,6 +675,7 @@ define void @local_atomic_fsub_noret_f32__offset(ptr addrspace(3) %ptr) nounwind
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB3_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -709,7 +717,7 @@ define void @local_atomic_fsub_noret_f32__offset(ptr addrspace(3) %ptr) nounwind
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v1
; GFX11-NEXT: v_mov_b32_e32 v1, v2
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB3_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -878,6 +886,7 @@ define double @local_atomic_fsub_ret_f64(ptr addrspace(3) %ptr) nounwind {
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB4_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -922,7 +931,7 @@ define double @local_atomic_fsub_ret_f64(ptr addrspace(3) %ptr) nounwind {
; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: v_cmp_eq_u64_e32 vcc_lo, v[0:1], v[3:4]
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB4_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -1095,6 +1104,7 @@ define double @local_atomic_fsub_ret_f64__offset(ptr addrspace(3) %ptr) nounwind
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB5_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -1139,7 +1149,7 @@ define double @local_atomic_fsub_ret_f64__offset(ptr addrspace(3) %ptr) nounwind
; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: v_cmp_eq_u64_e32 vcc_lo, v[0:1], v[3:4]
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB5_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -1313,6 +1323,7 @@ define void @local_atomic_fsub_noret_f64(ptr addrspace(3) %ptr) nounwind {
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB6_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -1354,7 +1365,7 @@ define void @local_atomic_fsub_noret_f64(ptr addrspace(3) %ptr) nounwind {
; GFX11-NEXT: v_cmp_eq_u64_e32 vcc_lo, v[3:4], v[1:2]
; GFX11-NEXT: v_dual_mov_b32 v1, v3 :: v_dual_mov_b32 v2, v4
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB6_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -1519,6 +1530,7 @@ define void @local_atomic_fsub_noret_f64__offset(ptr addrspace(3) %ptr) nounwind
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB7_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -1560,7 +1572,7 @@ define void @local_atomic_fsub_noret_f64__offset(ptr addrspace(3) %ptr) nounwind
; GFX11-NEXT: v_cmp_eq_u64_e32 vcc_lo, v[3:4], v[1:2]
; GFX11-NEXT: v_dual_mov_b32 v1, v3 :: v_dual_mov_b32 v2, v4
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB7_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -1745,9 +1757,11 @@ define half @local_atomic_fsub_ret_f16(ptr addrspace(3) %ptr) nounwind {
; GFX12-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_cbranch_execnz .LBB8_1
; GFX12-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: v_lshrrev_b32_e32 v0, v0, v2
; GFX12-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -1788,9 +1802,11 @@ define half @local_atomic_fsub_ret_f16(ptr addrspace(3) %ptr) nounwind {
; GFX12-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: s_cbranch_execnz .LBB8_1
; GFX12-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: v_lshrrev_b32_e32 v0, v0, v2
; GFX12-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -1853,11 +1869,12 @@ define half @local_atomic_fsub_ret_f16(ptr addrspace(3) %ptr) nounwind {
; GFX11-TRUE16-NEXT: buffer_gl0_inv
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v4
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB8_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v0, v0, v2
; GFX11-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -1890,11 +1907,12 @@ define half @local_atomic_fsub_ret_f16(ptr addrspace(3) %ptr) nounwind {
; GFX11-FAKE16-NEXT: buffer_gl0_inv
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v4
; GFX11-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB8_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v0, v0, v2
; GFX11-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -2130,9 +2148,11 @@ define half @local_atomic_fsub_ret_f16__offset(ptr addrspace(3) %ptr) nounwind {
; GFX12-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_cbranch_execnz .LBB9_1
; GFX12-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: v_lshrrev_b32_e32 v0, v1, v3
; GFX12-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -2175,9 +2195,11 @@ define half @local_atomic_fsub_ret_f16__offset(ptr addrspace(3) %ptr) nounwind {
; GFX12-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: s_cbranch_execnz .LBB9_1
; GFX12-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: v_lshrrev_b32_e32 v0, v1, v3
; GFX12-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -2243,11 +2265,12 @@ define half @local_atomic_fsub_ret_f16__offset(ptr addrspace(3) %ptr) nounwind {
; GFX11-TRUE16-NEXT: buffer_gl0_inv
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB9_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v0, v1, v3
; GFX11-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -2282,11 +2305,12 @@ define half @local_atomic_fsub_ret_f16__offset(ptr addrspace(3) %ptr) nounwind {
; GFX11-FAKE16-NEXT: buffer_gl0_inv
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB9_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v0, v1, v3
; GFX11-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -2527,6 +2551,7 @@ define void @local_atomic_fsub_noret_f16(ptr addrspace(3) %ptr) nounwind {
; GFX12-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_cbranch_execnz .LBB10_1
; GFX12-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -2569,6 +2594,7 @@ define void @local_atomic_fsub_noret_f16(ptr addrspace(3) %ptr) nounwind {
; GFX12-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: s_cbranch_execnz .LBB10_1
; GFX12-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -2632,7 +2658,7 @@ define void @local_atomic_fsub_noret_f16(ptr addrspace(3) %ptr) nounwind {
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v4, v2
; GFX11-TRUE16-NEXT: v_mov_b32_e32 v2, v4
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB10_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -2668,7 +2694,7 @@ define void @local_atomic_fsub_noret_f16(ptr addrspace(3) %ptr) nounwind {
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v4, v2
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v2, v4
; GFX11-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB10_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -2900,6 +2926,7 @@ define void @local_atomic_fsub_noret_f16__offset(ptr addrspace(3) %ptr) nounwind
; GFX12-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_cbranch_execnz .LBB11_1
; GFX12-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -2943,6 +2970,7 @@ define void @local_atomic_fsub_noret_f16__offset(ptr addrspace(3) %ptr) nounwind
; GFX12-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: s_cbranch_execnz .LBB11_1
; GFX12-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -3008,7 +3036,7 @@ define void @local_atomic_fsub_noret_f16__offset(ptr addrspace(3) %ptr) nounwind
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v4, v3
; GFX11-TRUE16-NEXT: v_mov_b32_e32 v3, v4
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB11_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -3045,7 +3073,7 @@ define void @local_atomic_fsub_noret_f16__offset(ptr addrspace(3) %ptr) nounwind
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v4, v3
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v3, v4
; GFX11-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB11_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -3274,9 +3302,11 @@ define half @local_atomic_fsub_ret_f16__offset__align4(ptr addrspace(3) %ptr) no
; GFX12-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_cbranch_execnz .LBB12_1
; GFX12-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: v_mov_b16_e32 v0.l, v1.l
; GFX12-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -3308,9 +3338,11 @@ define half @local_atomic_fsub_ret_f16__offset__align4(ptr addrspace(3) %ptr) no
; GFX12-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: s_cbranch_execnz .LBB12_1
; GFX12-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: v_mov_b32_e32 v0, v1
; GFX12-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -3357,11 +3389,12 @@ define half @local_atomic_fsub_ret_f16__offset__align4(ptr addrspace(3) %ptr) no
; GFX11-TRUE16-NEXT: buffer_gl0_inv
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v1, v2
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB12_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v0.l, v1.l
; GFX11-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -3385,11 +3418,12 @@ define half @local_atomic_fsub_ret_f16__offset__align4(ptr addrspace(3) %ptr) no
; GFX11-FAKE16-NEXT: buffer_gl0_inv
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v1, v2
; GFX11-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB12_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v0, v1
; GFX11-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -3574,6 +3608,7 @@ define void @local_atomic_fsub_noret_f16__offset__align4(ptr addrspace(3) %ptr)
; GFX12-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_cbranch_execnz .LBB13_1
; GFX12-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -3606,6 +3641,7 @@ define void @local_atomic_fsub_noret_f16__offset__align4(ptr addrspace(3) %ptr)
; GFX12-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: s_cbranch_execnz .LBB13_1
; GFX12-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -3652,7 +3688,7 @@ define void @local_atomic_fsub_noret_f16__offset__align4(ptr addrspace(3) %ptr)
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v1
; GFX11-TRUE16-NEXT: v_mov_b32_e32 v1, v2
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB13_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -3678,7 +3714,7 @@ define void @local_atomic_fsub_noret_f16__offset__align4(ptr addrspace(3) %ptr)
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v1
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v1, v2
; GFX11-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB13_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -3883,9 +3919,11 @@ define bfloat @local_atomic_fsub_ret_bf16(ptr addrspace(3) %ptr) nounwind {
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB14_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: v_lshrrev_b32_e32 v0, v0, v3
; GFX12-NEXT: s_setpc_b64 s[30:31]
;
@@ -3965,11 +4003,12 @@ define bfloat @local_atomic_fsub_ret_bf16(ptr addrspace(3) %ptr) nounwind {
; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB14_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: v_lshrrev_b32_e32 v0, v0, v3
; GFX11-NEXT: s_setpc_b64 s[30:31]
;
@@ -4237,9 +4276,11 @@ define bfloat @local_atomic_fsub_ret_bf16__offset(ptr addrspace(3) %ptr) nounwin
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB15_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: v_lshrrev_b32_e32 v0, v1, v3
; GFX12-NEXT: s_setpc_b64 s[30:31]
;
@@ -4322,11 +4363,12 @@ define bfloat @local_atomic_fsub_ret_bf16__offset(ptr addrspace(3) %ptr) nounwin
; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v4
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB15_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: v_lshrrev_b32_e32 v0, v1, v3
; GFX11-NEXT: s_setpc_b64 s[30:31]
;
@@ -4599,6 +4641,7 @@ define void @local_atomic_fsub_noret_bf16(ptr addrspace(3) %ptr) nounwind {
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB16_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -4679,7 +4722,7 @@ define void @local_atomic_fsub_noret_bf16(ptr addrspace(3) %ptr) nounwind {
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v4, v2
; GFX11-NEXT: v_mov_b32_e32 v2, v4
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB16_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -4943,6 +4986,7 @@ define void @local_atomic_fsub_noret_bf16__offset(ptr addrspace(3) %ptr) nounwin
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB17_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -5025,7 +5069,7 @@ define void @local_atomic_fsub_noret_bf16__offset(ptr addrspace(3) %ptr) nounwin
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v4, v3
; GFX11-NEXT: v_mov_b32_e32 v3, v4
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB17_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -5286,9 +5330,11 @@ define bfloat @local_atomic_fsub_ret_bf16__offset__align4(ptr addrspace(3) %ptr)
; GFX12-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_cbranch_execnz .LBB18_1
; GFX12-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: v_mov_b16_e32 v0.l, v1.l
; GFX12-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -5329,9 +5375,11 @@ define bfloat @local_atomic_fsub_ret_bf16__offset__align4(ptr addrspace(3) %ptr)
; GFX12-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: s_cbranch_execnz .LBB18_1
; GFX12-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: v_mov_b32_e32 v0, v1
; GFX12-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -5396,11 +5444,12 @@ define bfloat @local_atomic_fsub_ret_bf16__offset__align4(ptr addrspace(3) %ptr)
; GFX11-TRUE16-NEXT: buffer_gl0_inv
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v1, v2
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB18_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v0.l, v1.l
; GFX11-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -5433,11 +5482,12 @@ define bfloat @local_atomic_fsub_ret_bf16__offset__align4(ptr addrspace(3) %ptr)
; GFX11-FAKE16-NEXT: buffer_gl0_inv
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v1, v2
; GFX11-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB18_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v0, v1
; GFX11-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -5660,6 +5710,7 @@ define void @local_atomic_fsub_noret_bf16__offset__align4(ptr addrspace(3) %ptr)
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB19_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -5724,7 +5775,7 @@ define void @local_atomic_fsub_noret_bf16__offset__align4(ptr addrspace(3) %ptr)
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v1
; GFX11-NEXT: v_mov_b32_e32 v1, v2
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB19_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -5937,9 +5988,11 @@ define <2 x half> @local_atomic_fsub_ret_v2f16(ptr addrspace(3) %ptr, <2 x half>
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB20_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: v_mov_b32_e32 v0, v2
; GFX12-NEXT: s_setpc_b64 s[30:31]
;
@@ -5981,11 +6034,12 @@ define <2 x half> @local_atomic_fsub_ret_v2f16(ptr addrspace(3) %ptr, <2 x half>
; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v3
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB20_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: v_mov_b32_e32 v0, v2
; GFX11-NEXT: s_setpc_b64 s[30:31]
;
@@ -6184,9 +6238,11 @@ define <2 x half> @local_atomic_fsub_ret_v2f16__offset(ptr addrspace(3) %ptr, <2
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB21_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: v_mov_b32_e32 v0, v2
; GFX12-NEXT: s_setpc_b64 s[30:31]
;
@@ -6228,11 +6284,12 @@ define <2 x half> @local_atomic_fsub_ret_v2f16__offset(ptr addrspace(3) %ptr, <2
; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v3
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB21_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: v_mov_b32_e32 v0, v2
; GFX11-NEXT: s_setpc_b64 s[30:31]
;
@@ -6432,6 +6489,7 @@ define void @local_atomic_fsub_noret_v2f16(ptr addrspace(3) %ptr, <2 x half> %va
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB22_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -6473,7 +6531,7 @@ define void @local_atomic_fsub_noret_v2f16(ptr addrspace(3) %ptr, <2 x half> %va
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v2
; GFX11-NEXT: v_mov_b32_e32 v2, v3
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB22_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -6663,6 +6721,7 @@ define void @local_atomic_fsub_noret_v2f16__offset(ptr addrspace(3) %ptr, <2 x h
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB23_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -6704,7 +6763,7 @@ define void @local_atomic_fsub_noret_v2f16__offset(ptr addrspace(3) %ptr, <2 x h
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v3, v2
; GFX11-NEXT: v_mov_b32_e32 v2, v3
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB23_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -6927,9 +6986,11 @@ define <2 x bfloat> @local_atomic_fsub_ret_v2bf16(ptr addrspace(3) %ptr, <2 x bf
; GFX12-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_cbranch_execnz .LBB24_1
; GFX12-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: v_mov_b32_e32 v0, v2
; GFX12-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -6979,9 +7040,11 @@ define <2 x bfloat> @local_atomic_fsub_ret_v2bf16(ptr addrspace(3) %ptr, <2 x bf
; GFX12-FAKE16-NEXT: s_or_b32 s1, vcc_lo, s1
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: s_cbranch_execnz .LBB24_1
; GFX12-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s1
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: v_mov_b32_e32 v0, v2
; GFX12-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -7067,12 +7130,13 @@ define <2 x bfloat> @local_atomic_fsub_ret_v2bf16(ptr addrspace(3) %ptr, <2 x bf
; GFX11-TRUE16-NEXT: buffer_gl0_inv
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v4
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB24_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-TRUE16-NEXT: s_set_inst_prefetch_distance 0x2
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_mov_b32_e32 v0, v2
; GFX11-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -7114,12 +7178,13 @@ define <2 x bfloat> @local_atomic_fsub_ret_v2bf16(ptr addrspace(3) %ptr, <2 x bf
; GFX11-FAKE16-NEXT: buffer_gl0_inv
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v4
; GFX11-FAKE16-NEXT: s_or_b32 s1, vcc_lo, s1
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB24_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-FAKE16-NEXT: s_set_inst_prefetch_distance 0x2
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s1
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v0, v2
; GFX11-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -7421,9 +7486,11 @@ define <2 x bfloat> @local_atomic_fsub_ret_v2bf16__offset(ptr addrspace(3) %ptr,
; GFX12-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_cbranch_execnz .LBB25_1
; GFX12-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: v_mov_b32_e32 v0, v2
; GFX12-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -7473,9 +7540,11 @@ define <2 x bfloat> @local_atomic_fsub_ret_v2bf16__offset(ptr addrspace(3) %ptr,
; GFX12-FAKE16-NEXT: s_or_b32 s1, vcc_lo, s1
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: s_cbranch_execnz .LBB25_1
; GFX12-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s1
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: v_mov_b32_e32 v0, v2
; GFX12-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -7561,12 +7630,13 @@ define <2 x bfloat> @local_atomic_fsub_ret_v2bf16__offset(ptr addrspace(3) %ptr,
; GFX11-TRUE16-NEXT: buffer_gl0_inv
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v4
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB25_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-TRUE16-NEXT: s_set_inst_prefetch_distance 0x2
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_mov_b32_e32 v0, v2
; GFX11-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -7608,12 +7678,13 @@ define <2 x bfloat> @local_atomic_fsub_ret_v2bf16__offset(ptr addrspace(3) %ptr,
; GFX11-FAKE16-NEXT: buffer_gl0_inv
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v4
; GFX11-FAKE16-NEXT: s_or_b32 s1, vcc_lo, s1
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB25_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-FAKE16-NEXT: s_set_inst_prefetch_distance 0x2
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s1
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v0, v2
; GFX11-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -7915,6 +7986,7 @@ define void @local_atomic_fsub_noret_v2bf16(ptr addrspace(3) %ptr, <2 x bfloat>
; GFX12-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_cbranch_execnz .LBB26_1
; GFX12-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -7948,11 +8020,10 @@ define void @local_atomic_fsub_noret_v2bf16(ptr addrspace(3) %ptr, <2 x bfloat>
; GFX12-FAKE16-NEXT: v_add3_u32 v6, v6, v4, 0x7fff
; GFX12-FAKE16-NEXT: v_cmp_u_f32_e64 s0, v4, v4
; GFX12-FAKE16-NEXT: s_wait_alu depctr_va_vcc(0)
-; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_1) | instid1(VALU_DEP_2)
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_2) | instid1(VALU_DEP_1)
; GFX12-FAKE16-NEXT: v_cndmask_b32_e32 v5, v7, v9, vcc_lo
; GFX12-FAKE16-NEXT: s_wait_alu depctr_va_sdst(0)
; GFX12-FAKE16-NEXT: v_cndmask_b32_e64 v4, v6, v8, s0
-; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-FAKE16-NEXT: v_perm_b32 v4, v5, v4, 0x7060302
; GFX12-FAKE16-NEXT: s_wait_loadcnt 0x0
; GFX12-FAKE16-NEXT: s_wait_storecnt 0x0
@@ -7965,6 +8036,7 @@ define void @local_atomic_fsub_noret_v2bf16(ptr addrspace(3) %ptr, <2 x bfloat>
; GFX12-FAKE16-NEXT: s_or_b32 s1, vcc_lo, s1
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: s_cbranch_execnz .LBB26_1
; GFX12-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s1
@@ -8049,7 +8121,7 @@ define void @local_atomic_fsub_noret_v2bf16(ptr addrspace(3) %ptr, <2 x bfloat>
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v4, v3
; GFX11-TRUE16-NEXT: v_mov_b32_e32 v3, v4
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB26_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -8082,7 +8154,7 @@ define void @local_atomic_fsub_noret_v2bf16(ptr addrspace(3) %ptr, <2 x bfloat>
; GFX11-FAKE16-NEXT: v_add3_u32 v7, v7, v5, 0x7fff
; GFX11-FAKE16-NEXT: v_add3_u32 v6, v6, v4, 0x7fff
; GFX11-FAKE16-NEXT: v_cmp_u_f32_e64 s0, v4, v4
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_cndmask_b32_e32 v5, v7, v9, vcc_lo
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v4, v6, v8, s0
; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
@@ -8094,7 +8166,7 @@ define void @local_atomic_fsub_noret_v2bf16(ptr addrspace(3) %ptr, <2 x bfloat>
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v4, v3
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v3, v4
; GFX11-FAKE16-NEXT: s_or_b32 s1, vcc_lo, s1
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB26_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -8385,6 +8457,7 @@ define void @local_atomic_fsub_noret_v2bf16__ofset(ptr addrspace(3) %ptr, <2 x b
; GFX12-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_cbranch_execnz .LBB27_1
; GFX12-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -8418,11 +8491,10 @@ define void @local_atomic_fsub_noret_v2bf16__ofset(ptr addrspace(3) %ptr, <2 x b
; GFX12-FAKE16-NEXT: v_add3_u32 v6, v6, v4, 0x7fff
; GFX12-FAKE16-NEXT: v_cmp_u_f32_e64 s0, v4, v4
; GFX12-FAKE16-NEXT: s_wait_alu depctr_va_vcc(0)
-; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_1) | instid1(VALU_DEP_2)
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_2) | instid1(VALU_DEP_1)
; GFX12-FAKE16-NEXT: v_cndmask_b32_e32 v5, v7, v9, vcc_lo
; GFX12-FAKE16-NEXT: s_wait_alu depctr_va_sdst(0)
; GFX12-FAKE16-NEXT: v_cndmask_b32_e64 v4, v6, v8, s0
-; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-FAKE16-NEXT: v_perm_b32 v4, v5, v4, 0x7060302
; GFX12-FAKE16-NEXT: s_wait_loadcnt 0x0
; GFX12-FAKE16-NEXT: s_wait_storecnt 0x0
@@ -8435,6 +8507,7 @@ define void @local_atomic_fsub_noret_v2bf16__ofset(ptr addrspace(3) %ptr, <2 x b
; GFX12-FAKE16-NEXT: s_or_b32 s1, vcc_lo, s1
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: s_cbranch_execnz .LBB27_1
; GFX12-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s1
@@ -8519,7 +8592,7 @@ define void @local_atomic_fsub_noret_v2bf16__ofset(ptr addrspace(3) %ptr, <2 x b
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v4, v3
; GFX11-TRUE16-NEXT: v_mov_b32_e32 v3, v4
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB27_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -8552,7 +8625,7 @@ define void @local_atomic_fsub_noret_v2bf16__ofset(ptr addrspace(3) %ptr, <2 x b
; GFX11-FAKE16-NEXT: v_add3_u32 v7, v7, v5, 0x7fff
; GFX11-FAKE16-NEXT: v_add3_u32 v6, v6, v4, 0x7fff
; GFX11-FAKE16-NEXT: v_cmp_u_f32_e64 s0, v4, v4
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_cndmask_b32_e32 v5, v7, v9, vcc_lo
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v4, v6, v8, s0
; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
@@ -8564,7 +8637,7 @@ define void @local_atomic_fsub_noret_v2bf16__ofset(ptr addrspace(3) %ptr, <2 x b
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, v4, v3
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v3, v4
; GFX11-FAKE16-NEXT: s_or_b32 s1, vcc_lo, s1
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB27_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %atomicrmw.end
@@ -8840,9 +8913,11 @@ define float @local_atomic_fsub_ret_f32__amdgpu_ignore_denormal_mode(ptr addrspa
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB28_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: v_mov_b32_e32 v0, v1
; GFX12-NEXT: s_setpc_b64 s[30:31]
;
@@ -8884,11 +8959,12 @@ define float @local_atomic_fsub_ret_f32__amdgpu_ignore_denormal_mode(ptr addrspa
; GFX11-NEXT: buffer_gl0_inv
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v1, v2
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB28_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX11-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: v_mov_b32_e32 v0, v1
; GFX11-NEXT: s_setpc_b64 s[30:31]
;
@@ -9051,6 +9127,7 @@ define void @local_atomic_fsub_noret_f32__amdgpu_ignore_denormal_mode(ptr addrsp
; GFX12-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_execnz .LBB29_1
; GFX12-NEXT: ; %bb.2: ; %atomicrmw.end
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -9092,7 +9169,7 @@ define void @local_atomic_fsub_noret_f32__amdgpu_ignore_denormal_mode(ptr addrsp
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc_lo, v2, v1
; GFX11-NEXT: v_mov_b32_e32 v1, v2
; GFX11-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-NEXT: s_cbranch_execnz .LBB29_1
; GFX11-NEXT: ; %bb.2: ; %atomicrmw.end
diff --git a/llvm/test/CodeGen/AMDGPU/loop-prefetch-data.ll b/llvm/test/CodeGen/AMDGPU/loop-prefetch-data.ll
index 4ef9df249c41cd..8b2dbd2a7c0f32 100644
--- a/llvm/test/CodeGen/AMDGPU/loop-prefetch-data.ll
+++ b/llvm/test/CodeGen/AMDGPU/loop-prefetch-data.ll
@@ -10,6 +10,7 @@ define amdgpu_kernel void @copy_flat(ptr nocapture %d, ptr nocapture readonly %s
; GFX12-NEXT: s_load_b32 s6, s[4:5], 0x34
; GFX12-NEXT: s_wait_kmcnt 0x0
; GFX12-NEXT: s_cmp_eq_u32 s6, 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_scc1 .LBB0_3
; GFX12-NEXT: ; %bb.1: ; %for.body.preheader
; GFX12-NEXT: s_load_b128 s[0:3], s[4:5], 0x24
@@ -40,6 +41,7 @@ define amdgpu_kernel void @copy_flat(ptr nocapture %d, ptr nocapture readonly %s
; GFX12-SPREFETCH-NEXT: s_load_b32 s6, s[4:5], 0x34
; GFX12-SPREFETCH-NEXT: s_wait_kmcnt 0x0
; GFX12-SPREFETCH-NEXT: s_cmp_eq_u32 s6, 0
+; GFX12-SPREFETCH-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-SPREFETCH-NEXT: s_cbranch_scc1 .LBB0_3
; GFX12-SPREFETCH-NEXT: ; %bb.1: ; %for.body.preheader
; GFX12-SPREFETCH-NEXT: s_load_b128 s[0:3], s[4:5], 0x24
@@ -71,6 +73,7 @@ define amdgpu_kernel void @copy_flat(ptr nocapture %d, ptr nocapture readonly %s
; GFX12ES2-SPREFETCH-NEXT: s_load_b32 s6, s[4:5], 0x34
; GFX12ES2-SPREFETCH-NEXT: s_wait_kmcnt 0x0
; GFX12ES2-SPREFETCH-NEXT: s_cmp_eq_u32 s6, 0
+; GFX12ES2-SPREFETCH-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12ES2-SPREFETCH-NEXT: s_cbranch_scc1 .LBB0_3
; GFX12ES2-SPREFETCH-NEXT: ; %bb.1: ; %for.body.preheader
; GFX12ES2-SPREFETCH-NEXT: s_load_b128 s[0:3], s[4:5], 0x24
@@ -108,6 +111,7 @@ define amdgpu_kernel void @copy_flat(ptr nocapture %d, ptr nocapture readonly %s
; GFX1250-NEXT: s_load_b32 s6, s[4:5], 0x34 nv
; GFX1250-NEXT: s_wait_kmcnt 0x0
; GFX1250-NEXT: s_cmp_eq_u32 s6, 0
+; GFX1250-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1250-NEXT: s_cbranch_scc1 .LBB0_3
; GFX1250-NEXT: ; %bb.1: ; %for.body.preheader
; GFX1250-NEXT: s_load_b128 s[0:3], s[4:5], 0x24 nv
@@ -154,6 +158,7 @@ define amdgpu_kernel void @copy_global(ptr addrspace(1) nocapture %d, ptr addrsp
; GFX12-NEXT: s_load_b32 s6, s[4:5], 0x34
; GFX12-NEXT: s_wait_kmcnt 0x0
; GFX12-NEXT: s_cmp_eq_u32 s6, 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_scc1 .LBB1_3
; GFX12-NEXT: ; %bb.1: ; %for.body.preheader
; GFX12-NEXT: s_load_b128 s[0:3], s[4:5], 0x24
@@ -178,6 +183,7 @@ define amdgpu_kernel void @copy_global(ptr addrspace(1) nocapture %d, ptr addrsp
; GFX12-SPREFETCH-NEXT: s_load_b32 s6, s[4:5], 0x34
; GFX12-SPREFETCH-NEXT: s_wait_kmcnt 0x0
; GFX12-SPREFETCH-NEXT: s_cmp_eq_u32 s6, 0
+; GFX12-SPREFETCH-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-SPREFETCH-NEXT: s_cbranch_scc1 .LBB1_3
; GFX12-SPREFETCH-NEXT: ; %bb.1: ; %for.body.preheader
; GFX12-SPREFETCH-NEXT: s_load_b128 s[0:3], s[4:5], 0x24
@@ -204,6 +210,7 @@ define amdgpu_kernel void @copy_global(ptr addrspace(1) nocapture %d, ptr addrsp
; GFX12ES2-SPREFETCH-NEXT: s_load_b32 s6, s[4:5], 0x34
; GFX12ES2-SPREFETCH-NEXT: s_wait_kmcnt 0x0
; GFX12ES2-SPREFETCH-NEXT: s_cmp_eq_u32 s6, 0
+; GFX12ES2-SPREFETCH-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12ES2-SPREFETCH-NEXT: s_cbranch_scc1 .LBB1_3
; GFX12ES2-SPREFETCH-NEXT: ; %bb.1: ; %for.body.preheader
; GFX12ES2-SPREFETCH-NEXT: s_load_b128 s[0:3], s[4:5], 0x24
@@ -234,6 +241,7 @@ define amdgpu_kernel void @copy_global(ptr addrspace(1) nocapture %d, ptr addrsp
; GFX1250-NEXT: s_load_b32 s6, s[4:5], 0x34 nv
; GFX1250-NEXT: s_wait_kmcnt 0x0
; GFX1250-NEXT: s_cmp_eq_u32 s6, 0
+; GFX1250-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1250-NEXT: s_cbranch_scc1 .LBB1_3
; GFX1250-NEXT: ; %bb.1: ; %for.body.preheader
; GFX1250-NEXT: s_load_b128 s[0:3], s[4:5], 0x24 nv
@@ -280,6 +288,7 @@ define amdgpu_kernel void @copy_constant(ptr addrspace(1) nocapture %d, ptr addr
; GFX12-NEXT: s_load_b32 s6, s[4:5], 0x34
; GFX12-NEXT: s_wait_kmcnt 0x0
; GFX12-NEXT: s_cmp_eq_u32 s6, 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_scc1 .LBB2_3
; GFX12-NEXT: ; %bb.1: ; %for.body.preheader
; GFX12-NEXT: s_load_b128 s[0:3], s[4:5], 0x24
@@ -305,6 +314,7 @@ define amdgpu_kernel void @copy_constant(ptr addrspace(1) nocapture %d, ptr addr
; GFX12-SPREFETCH-NEXT: s_load_b32 s6, s[4:5], 0x34
; GFX12-SPREFETCH-NEXT: s_wait_kmcnt 0x0
; GFX12-SPREFETCH-NEXT: s_cmp_eq_u32 s6, 0
+; GFX12-SPREFETCH-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-SPREFETCH-NEXT: s_cbranch_scc1 .LBB2_3
; GFX12-SPREFETCH-NEXT: ; %bb.1: ; %for.body.preheader
; GFX12-SPREFETCH-NEXT: s_load_b128 s[0:3], s[4:5], 0x24
@@ -332,6 +342,7 @@ define amdgpu_kernel void @copy_constant(ptr addrspace(1) nocapture %d, ptr addr
; GFX12ES2-SPREFETCH-NEXT: s_load_b32 s6, s[4:5], 0x34
; GFX12ES2-SPREFETCH-NEXT: s_wait_kmcnt 0x0
; GFX12ES2-SPREFETCH-NEXT: s_cmp_eq_u32 s6, 0
+; GFX12ES2-SPREFETCH-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12ES2-SPREFETCH-NEXT: s_cbranch_scc1 .LBB2_3
; GFX12ES2-SPREFETCH-NEXT: ; %bb.1: ; %for.body.preheader
; GFX12ES2-SPREFETCH-NEXT: s_load_b128 s[0:3], s[4:5], 0x24
@@ -364,6 +375,7 @@ define amdgpu_kernel void @copy_constant(ptr addrspace(1) nocapture %d, ptr addr
; GFX1250-NEXT: s_load_b32 s6, s[4:5], 0x34 nv
; GFX1250-NEXT: s_wait_kmcnt 0x0
; GFX1250-NEXT: s_cmp_eq_u32 s6, 0
+; GFX1250-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1250-NEXT: s_cbranch_scc1 .LBB2_3
; GFX1250-NEXT: ; %bb.1: ; %for.body.preheader
; GFX1250-NEXT: s_load_b128 s[0:3], s[4:5], 0x24 nv
@@ -411,6 +423,7 @@ define amdgpu_kernel void @copy_local(ptr addrspace(3) nocapture %d, ptr addrspa
; GFX12-NEXT: s_load_b96 s[0:2], s[4:5], 0x24
; GFX12-NEXT: s_wait_kmcnt 0x0
; GFX12-NEXT: s_cmp_eq_u32 s2, 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_scc1 .LBB3_2
; GFX12-NEXT: .LBB3_1: ; %for.body
; GFX12-NEXT: ; =>This Inner Loop Header: Depth=1
@@ -436,6 +449,7 @@ define amdgpu_kernel void @copy_local(ptr addrspace(3) nocapture %d, ptr addrspa
; GFX12-SPREFETCH-NEXT: s_load_b96 s[0:2], s[4:5], 0x24
; GFX12-SPREFETCH-NEXT: s_wait_kmcnt 0x0
; GFX12-SPREFETCH-NEXT: s_cmp_eq_u32 s2, 0
+; GFX12-SPREFETCH-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-SPREFETCH-NEXT: s_cbranch_scc1 .LBB3_2
; GFX12-SPREFETCH-NEXT: .LBB3_1: ; %for.body
; GFX12-SPREFETCH-NEXT: ; =>This Inner Loop Header: Depth=1
@@ -462,6 +476,7 @@ define amdgpu_kernel void @copy_local(ptr addrspace(3) nocapture %d, ptr addrspa
; GFX12ES2-SPREFETCH-NEXT: s_load_b96 s[0:2], s[4:5], 0x24
; GFX12ES2-SPREFETCH-NEXT: s_wait_kmcnt 0x0
; GFX12ES2-SPREFETCH-NEXT: s_cmp_eq_u32 s2, 0
+; GFX12ES2-SPREFETCH-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12ES2-SPREFETCH-NEXT: s_cbranch_scc1 .LBB3_2
; GFX12ES2-SPREFETCH-NEXT: .LBB3_1: ; %for.body
; GFX12ES2-SPREFETCH-NEXT: ; =>This Inner Loop Header: Depth=1
@@ -494,6 +509,7 @@ define amdgpu_kernel void @copy_local(ptr addrspace(3) nocapture %d, ptr addrspa
; GFX1250-NEXT: s_load_b96 s[0:2], s[4:5], 0x24 nv
; GFX1250-NEXT: s_wait_kmcnt 0x0
; GFX1250-NEXT: s_cmp_eq_u32 s2, 0
+; GFX1250-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1250-NEXT: s_cbranch_scc1 .LBB3_2
; GFX1250-NEXT: .LBB3_1: ; %for.body
; GFX1250-NEXT: ; =>This Inner Loop Header: Depth=1
@@ -536,6 +552,7 @@ define amdgpu_kernel void @copy_flat_divergent(ptr nocapture %d, ptr nocapture r
; GFX12-NEXT: s_load_b32 s0, s[4:5], 0x34
; GFX12-NEXT: s_wait_kmcnt 0x0
; GFX12-NEXT: s_cmp_eq_u32 s0, 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_scc1 .LBB4_3
; GFX12-NEXT: ; %bb.1: ; %for.body.preheader
; GFX12-NEXT: s_load_b128 s[4:7], s[4:5], 0x24
@@ -544,9 +561,9 @@ define amdgpu_kernel void @copy_flat_divergent(ptr nocapture %d, ptr nocapture r
; GFX12-NEXT: v_lshlrev_b32_e32 v0, 4, v0
; GFX12-NEXT: s_wait_kmcnt 0x0
; GFX12-NEXT: v_add_co_u32 v2, s1, s6, v0
-; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_3)
; GFX12-NEXT: v_add_co_ci_u32_e64 v3, null, s7, 0, s1
; GFX12-NEXT: v_add_co_u32 v0, s1, s4, v0
+; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX12-NEXT: v_add_co_u32 v2, vcc_lo, 0xb0, v2
; GFX12-NEXT: s_wait_alu depctr_va_sdst(0)
; GFX12-NEXT: v_add_co_ci_u32_e64 v1, null, s5, 0, s1
@@ -578,6 +595,7 @@ define amdgpu_kernel void @copy_flat_divergent(ptr nocapture %d, ptr nocapture r
; GFX12-SPREFETCH-NEXT: s_load_b32 s0, s[4:5], 0x34
; GFX12-SPREFETCH-NEXT: s_wait_kmcnt 0x0
; GFX12-SPREFETCH-NEXT: s_cmp_eq_u32 s0, 0
+; GFX12-SPREFETCH-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-SPREFETCH-NEXT: s_cbranch_scc1 .LBB4_3
; GFX12-SPREFETCH-NEXT: ; %bb.1: ; %for.body.preheader
; GFX12-SPREFETCH-NEXT: s_load_b128 s[4:7], s[4:5], 0x24
@@ -586,9 +604,9 @@ define amdgpu_kernel void @copy_flat_divergent(ptr nocapture %d, ptr nocapture r
; GFX12-SPREFETCH-NEXT: v_lshlrev_b32_e32 v0, 4, v0
; GFX12-SPREFETCH-NEXT: s_wait_kmcnt 0x0
; GFX12-SPREFETCH-NEXT: v_add_co_u32 v2, s1, s6, v0
-; GFX12-SPREFETCH-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_3)
; GFX12-SPREFETCH-NEXT: v_add_co_ci_u32_e64 v3, null, s7, 0, s1
; GFX12-SPREFETCH-NEXT: v_add_co_u32 v0, s1, s4, v0
+; GFX12-SPREFETCH-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX12-SPREFETCH-NEXT: v_add_co_u32 v2, vcc_lo, 0xb0, v2
; GFX12-SPREFETCH-NEXT: s_wait_alu depctr_va_sdst(0)
; GFX12-SPREFETCH-NEXT: v_add_co_ci_u32_e64 v1, null, s5, 0, s1
@@ -621,6 +639,7 @@ define amdgpu_kernel void @copy_flat_divergent(ptr nocapture %d, ptr nocapture r
; GFX12ES2-SPREFETCH-NEXT: s_load_b32 s0, s[4:5], 0x34
; GFX12ES2-SPREFETCH-NEXT: s_wait_kmcnt 0x0
; GFX12ES2-SPREFETCH-NEXT: s_cmp_eq_u32 s0, 0
+; GFX12ES2-SPREFETCH-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12ES2-SPREFETCH-NEXT: s_cbranch_scc1 .LBB4_3
; GFX12ES2-SPREFETCH-NEXT: ; %bb.1: ; %for.body.preheader
; GFX12ES2-SPREFETCH-NEXT: s_load_b128 s[4:7], s[4:5], 0x24
@@ -629,9 +648,9 @@ define amdgpu_kernel void @copy_flat_divergent(ptr nocapture %d, ptr nocapture r
; GFX12ES2-SPREFETCH-NEXT: v_lshlrev_b32_e32 v0, 4, v0
; GFX12ES2-SPREFETCH-NEXT: s_wait_kmcnt 0x0
; GFX12ES2-SPREFETCH-NEXT: v_add_co_u32 v2, s1, s6, v0
-; GFX12ES2-SPREFETCH-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_3)
; GFX12ES2-SPREFETCH-NEXT: v_add_co_ci_u32_e64 v3, null, s7, 0, s1
; GFX12ES2-SPREFETCH-NEXT: v_add_co_u32 v0, s1, s4, v0
+; GFX12ES2-SPREFETCH-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX12ES2-SPREFETCH-NEXT: v_add_co_u32 v2, vcc_lo, 0xb0, v2
; GFX12ES2-SPREFETCH-NEXT: s_wait_alu depctr_va_sdst(0)
; GFX12ES2-SPREFETCH-NEXT: v_add_co_ci_u32_e64 v1, null, s5, 0, s1
@@ -669,6 +688,7 @@ define amdgpu_kernel void @copy_flat_divergent(ptr nocapture %d, ptr nocapture r
; GFX1250-NEXT: s_load_b32 s2, s[4:5], 0x34 nv
; GFX1250-NEXT: s_wait_kmcnt 0x0
; GFX1250-NEXT: s_cmp_eq_u32 s2, 0
+; GFX1250-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1250-NEXT: s_cbranch_scc1 .LBB4_3
; GFX1250-NEXT: ; %bb.1: ; %for.body.preheader
; GFX1250-NEXT: s_load_b128 s[8:11], s[4:5], 0x24 nv
@@ -726,6 +746,7 @@ define amdgpu_kernel void @copy_global_divergent(ptr addrspace(1) nocapture %d,
; GFX12-NEXT: s_load_b32 s0, s[4:5], 0x34
; GFX12-NEXT: s_wait_kmcnt 0x0
; GFX12-NEXT: s_cmp_eq_u32 s0, 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_scc1 .LBB5_3
; GFX12-NEXT: ; %bb.1: ; %for.body.preheader
; GFX12-NEXT: s_load_b128 s[4:7], s[4:5], 0x24
@@ -734,9 +755,9 @@ define amdgpu_kernel void @copy_global_divergent(ptr addrspace(1) nocapture %d,
; GFX12-NEXT: v_lshlrev_b32_e32 v0, 4, v0
; GFX12-NEXT: s_wait_kmcnt 0x0
; GFX12-NEXT: v_add_co_u32 v2, s1, s6, v0
-; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_3)
; GFX12-NEXT: v_add_co_ci_u32_e64 v3, null, s7, 0, s1
; GFX12-NEXT: v_add_co_u32 v0, s1, s4, v0
+; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX12-NEXT: v_add_co_u32 v2, vcc_lo, 0xb0, v2
; GFX12-NEXT: s_wait_alu depctr_va_sdst(0)
; GFX12-NEXT: v_add_co_ci_u32_e64 v1, null, s5, 0, s1
@@ -764,6 +785,7 @@ define amdgpu_kernel void @copy_global_divergent(ptr addrspace(1) nocapture %d,
; GFX12-SPREFETCH-NEXT: s_load_b32 s0, s[4:5], 0x34
; GFX12-SPREFETCH-NEXT: s_wait_kmcnt 0x0
; GFX12-SPREFETCH-NEXT: s_cmp_eq_u32 s0, 0
+; GFX12-SPREFETCH-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-SPREFETCH-NEXT: s_cbranch_scc1 .LBB5_3
; GFX12-SPREFETCH-NEXT: ; %bb.1: ; %for.body.preheader
; GFX12-SPREFETCH-NEXT: s_load_b128 s[4:7], s[4:5], 0x24
@@ -772,9 +794,9 @@ define amdgpu_kernel void @copy_global_divergent(ptr addrspace(1) nocapture %d,
; GFX12-SPREFETCH-NEXT: v_lshlrev_b32_e32 v0, 4, v0
; GFX12-SPREFETCH-NEXT: s_wait_kmcnt 0x0
; GFX12-SPREFETCH-NEXT: v_add_co_u32 v2, s1, s6, v0
-; GFX12-SPREFETCH-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_3)
; GFX12-SPREFETCH-NEXT: v_add_co_ci_u32_e64 v3, null, s7, 0, s1
; GFX12-SPREFETCH-NEXT: v_add_co_u32 v0, s1, s4, v0
+; GFX12-SPREFETCH-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX12-SPREFETCH-NEXT: v_add_co_u32 v2, vcc_lo, 0xb0, v2
; GFX12-SPREFETCH-NEXT: s_wait_alu depctr_va_sdst(0)
; GFX12-SPREFETCH-NEXT: v_add_co_ci_u32_e64 v1, null, s5, 0, s1
@@ -803,6 +825,7 @@ define amdgpu_kernel void @copy_global_divergent(ptr addrspace(1) nocapture %d,
; GFX12ES2-SPREFETCH-NEXT: s_load_b32 s0, s[4:5], 0x34
; GFX12ES2-SPREFETCH-NEXT: s_wait_kmcnt 0x0
; GFX12ES2-SPREFETCH-NEXT: s_cmp_eq_u32 s0, 0
+; GFX12ES2-SPREFETCH-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12ES2-SPREFETCH-NEXT: s_cbranch_scc1 .LBB5_3
; GFX12ES2-SPREFETCH-NEXT: ; %bb.1: ; %for.body.preheader
; GFX12ES2-SPREFETCH-NEXT: s_load_b128 s[4:7], s[4:5], 0x24
@@ -811,9 +834,9 @@ define amdgpu_kernel void @copy_global_divergent(ptr addrspace(1) nocapture %d,
; GFX12ES2-SPREFETCH-NEXT: v_lshlrev_b32_e32 v0, 4, v0
; GFX12ES2-SPREFETCH-NEXT: s_wait_kmcnt 0x0
; GFX12ES2-SPREFETCH-NEXT: v_add_co_u32 v2, s1, s6, v0
-; GFX12ES2-SPREFETCH-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_3)
; GFX12ES2-SPREFETCH-NEXT: v_add_co_ci_u32_e64 v3, null, s7, 0, s1
; GFX12ES2-SPREFETCH-NEXT: v_add_co_u32 v0, s1, s4, v0
+; GFX12ES2-SPREFETCH-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX12ES2-SPREFETCH-NEXT: v_add_co_u32 v2, vcc_lo, 0xb0, v2
; GFX12ES2-SPREFETCH-NEXT: s_wait_alu depctr_va_sdst(0)
; GFX12ES2-SPREFETCH-NEXT: v_add_co_ci_u32_e64 v1, null, s5, 0, s1
@@ -848,6 +871,7 @@ define amdgpu_kernel void @copy_global_divergent(ptr addrspace(1) nocapture %d,
; GFX1250-NEXT: s_load_b32 s0, s[4:5], 0x34 nv
; GFX1250-NEXT: s_wait_kmcnt 0x0
; GFX1250-NEXT: s_cmp_eq_u32 s0, 0
+; GFX1250-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1250-NEXT: s_cbranch_scc1 .LBB5_3
; GFX1250-NEXT: ; %bb.1: ; %for.body.preheader
; GFX1250-NEXT: s_load_b128 s[8:11], s[4:5], 0x24 nv
diff --git a/llvm/test/CodeGen/AMDGPU/lower-work-group-id-intrinsics-opt.ll b/llvm/test/CodeGen/AMDGPU/lower-work-group-id-intrinsics-opt.ll
index 63d02e09d611e8..0fb445d6c5ae48 100644
--- a/llvm/test/CodeGen/AMDGPU/lower-work-group-id-intrinsics-opt.ll
+++ b/llvm/test/CodeGen/AMDGPU/lower-work-group-id-intrinsics-opt.ll
@@ -12,10 +12,11 @@ define void @test_workgroup_id_x_non_kernel(ptr addrspace(1) %out) {
; GFX1250-SDAG-NEXT: s_add_co_i32 s0, s0, 1
; GFX1250-SDAG-NEXT: s_getreg_b32 s2, hwreg(HW_REG_IB_STS2, 6, 4)
; GFX1250-SDAG-NEXT: s_mul_i32 s0, ttmp9, s0
-; GFX1250-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
+; GFX1250-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX1250-SDAG-NEXT: s_add_co_i32 s1, s1, s0
; GFX1250-SDAG-NEXT: s_cmp_eq_u32 s2, 0
; GFX1250-SDAG-NEXT: s_cselect_b32 s0, ttmp9, s1
+; GFX1250-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1250-SDAG-NEXT: v_mov_b32_e32 v2, s0
; GFX1250-SDAG-NEXT: global_store_b32 v[0:1], v2, off
; GFX1250-SDAG-NEXT: s_set_pc_i64 s[30:31]
@@ -29,10 +30,11 @@ define void @test_workgroup_id_x_non_kernel(ptr addrspace(1) %out) {
; GFX1250-GISEL-NEXT: s_add_co_i32 s0, s0, 1
; GFX1250-GISEL-NEXT: s_getreg_b32 s2, hwreg(HW_REG_IB_STS2, 6, 4)
; GFX1250-GISEL-NEXT: s_mul_i32 s0, ttmp9, s0
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX1250-GISEL-NEXT: s_add_co_i32 s1, s1, s0
; GFX1250-GISEL-NEXT: s_cmp_eq_u32 s2, 0
; GFX1250-GISEL-NEXT: s_cselect_b32 s0, ttmp9, s1
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1250-GISEL-NEXT: v_mov_b32_e32 v2, s0
; GFX1250-GISEL-NEXT: global_store_b32 v[0:1], v2, off
; GFX1250-GISEL-NEXT: s_set_pc_i64 s[30:31]
@@ -137,8 +139,8 @@ define void @test_workgroup_id_y_non_kernel(ptr addrspace(1) %out) {
; GFX1250-SDAG-NEXT: s_getreg_b32 s3, hwreg(HW_REG_IB_STS2, 6, 4)
; GFX1250-SDAG-NEXT: s_add_co_i32 s2, s2, s0
; GFX1250-SDAG-NEXT: s_cmp_eq_u32 s3, 0
+; GFX1250-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1250-SDAG-NEXT: s_cselect_b32 s0, s1, s2
-; GFX1250-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1250-SDAG-NEXT: v_mov_b32_e32 v2, s0
; GFX1250-SDAG-NEXT: global_store_b32 v[0:1], v2, off
; GFX1250-SDAG-NEXT: s_set_pc_i64 s[30:31]
@@ -155,8 +157,8 @@ define void @test_workgroup_id_y_non_kernel(ptr addrspace(1) %out) {
; GFX1250-GISEL-NEXT: s_getreg_b32 s3, hwreg(HW_REG_IB_STS2, 6, 4)
; GFX1250-GISEL-NEXT: s_add_co_i32 s2, s2, s0
; GFX1250-GISEL-NEXT: s_cmp_eq_u32 s3, 0
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1250-GISEL-NEXT: s_cselect_b32 s0, s1, s2
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1250-GISEL-NEXT: v_mov_b32_e32 v2, s0
; GFX1250-GISEL-NEXT: global_store_b32 v[0:1], v2, off
; GFX1250-GISEL-NEXT: s_set_pc_i64 s[30:31]
@@ -264,8 +266,8 @@ define void @test_workgroup_id_z_non_kernel(ptr addrspace(1) %out) {
; GFX1250-SDAG-NEXT: s_getreg_b32 s3, hwreg(HW_REG_IB_STS2, 6, 4)
; GFX1250-SDAG-NEXT: s_add_co_i32 s2, s2, s0
; GFX1250-SDAG-NEXT: s_cmp_eq_u32 s3, 0
+; GFX1250-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1250-SDAG-NEXT: s_cselect_b32 s0, s1, s2
-; GFX1250-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1250-SDAG-NEXT: v_mov_b32_e32 v2, s0
; GFX1250-SDAG-NEXT: global_store_b32 v[0:1], v2, off
; GFX1250-SDAG-NEXT: s_set_pc_i64 s[30:31]
@@ -282,8 +284,8 @@ define void @test_workgroup_id_z_non_kernel(ptr addrspace(1) %out) {
; GFX1250-GISEL-NEXT: s_getreg_b32 s3, hwreg(HW_REG_IB_STS2, 6, 4)
; GFX1250-GISEL-NEXT: s_add_co_i32 s2, s2, s0
; GFX1250-GISEL-NEXT: s_cmp_eq_u32 s3, 0
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1250-GISEL-NEXT: s_cselect_b32 s0, s1, s2
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1250-GISEL-NEXT: v_mov_b32_e32 v2, s0
; GFX1250-GISEL-NEXT: global_store_b32 v[0:1], v2, off
; GFX1250-GISEL-NEXT: s_set_pc_i64 s[30:31]
diff --git a/llvm/test/CodeGen/AMDGPU/lower-work-group-id-intrinsics.ll b/llvm/test/CodeGen/AMDGPU/lower-work-group-id-intrinsics.ll
index a64937abebb415..f38880ebd95ca1 100644
--- a/llvm/test/CodeGen/AMDGPU/lower-work-group-id-intrinsics.ll
+++ b/llvm/test/CodeGen/AMDGPU/lower-work-group-id-intrinsics.ll
@@ -72,10 +72,10 @@ define amdgpu_cs void @_amdgpu_cs_main() {
; GFX1250-SDAG-NEXT: s_getreg_b32 s6, hwreg(HW_REG_IB_STS2, 6, 4)
; GFX1250-SDAG-NEXT: s_add_co_i32 s4, s4, s0
; GFX1250-SDAG-NEXT: s_cmp_eq_u32 s6, 0
+; GFX1250-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
; GFX1250-SDAG-NEXT: s_cselect_b32 s0, s5, s4
; GFX1250-SDAG-NEXT: s_cselect_b32 s1, ttmp9, s1
; GFX1250-SDAG-NEXT: s_cselect_b32 s2, s3, s2
-; GFX1250-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1250-SDAG-NEXT: v_dual_mov_b32 v0, s1 :: v_dual_mov_b32 v1, s2
; GFX1250-SDAG-NEXT: v_mov_b32_e32 v2, s0
; GFX1250-SDAG-NEXT: buffer_store_b96 v[0:2], off, s[0:3], null
@@ -92,7 +92,7 @@ define amdgpu_cs void @_amdgpu_cs_main() {
; GFX1250-GISEL-NEXT: s_add_co_i32 s0, s0, 1
; GFX1250-GISEL-NEXT: s_getreg_b32 s2, hwreg(HW_REG_IB_STS2, 6, 4)
; GFX1250-GISEL-NEXT: s_mul_i32 s0, ttmp9, s0
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX1250-GISEL-NEXT: s_add_co_i32 s1, s1, s0
; GFX1250-GISEL-NEXT: s_cmp_eq_u32 s2, 0
; GFX1250-GISEL-NEXT: s_cselect_b32 s0, ttmp9, s1
@@ -101,7 +101,7 @@ define amdgpu_cs void @_amdgpu_cs_main() {
; GFX1250-GISEL-NEXT: s_add_co_i32 s1, s1, 1
; GFX1250-GISEL-NEXT: s_bfe_u32 s4, ttmp6, 0x40004
; GFX1250-GISEL-NEXT: s_mul_i32 s1, s3, s1
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX1250-GISEL-NEXT: s_add_co_i32 s4, s4, s1
; GFX1250-GISEL-NEXT: s_cmp_eq_u32 s2, 0
; GFX1250-GISEL-NEXT: s_cselect_b32 s1, s3, s4
@@ -110,7 +110,7 @@ define amdgpu_cs void @_amdgpu_cs_main() {
; GFX1250-GISEL-NEXT: s_add_co_i32 s3, s3, 1
; GFX1250-GISEL-NEXT: s_bfe_u32 s5, ttmp6, 0x40008
; GFX1250-GISEL-NEXT: s_mul_i32 s3, s4, s3
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX1250-GISEL-NEXT: s_add_co_i32 s5, s5, s3
; GFX1250-GISEL-NEXT: s_cmp_eq_u32 s2, 0
; GFX1250-GISEL-NEXT: s_cselect_b32 s2, s4, s5
@@ -369,6 +369,7 @@ define amdgpu_cs void @caller() {
; GFX1250-SDAG-NEXT: s_mov_b32 s32, 0
; GFX1250-SDAG-NEXT: s_add_co_i32 s1, s1, s0
; GFX1250-SDAG-NEXT: s_cmp_eq_u32 s2, 0
+; GFX1250-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1250-SDAG-NEXT: s_cselect_b32 s2, ttmp9, s1
; GFX1250-SDAG-NEXT: s_mov_b64 s[0:1], callee at abs64
; GFX1250-SDAG-NEXT: v_mov_b32_e32 v0, s2
@@ -389,6 +390,7 @@ define amdgpu_cs void @caller() {
; GFX1250-GISEL-NEXT: s_mov_b32 s32, 0
; GFX1250-GISEL-NEXT: s_add_co_i32 s1, s1, s0
; GFX1250-GISEL-NEXT: s_cmp_eq_u32 s2, 0
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1250-GISEL-NEXT: s_cselect_b32 s2, ttmp9, s1
; GFX1250-GISEL-NEXT: s_mov_b64 s[0:1], callee at abs64
; GFX1250-GISEL-NEXT: v_mov_b32_e32 v0, s2
diff --git a/llvm/test/CodeGen/AMDGPU/lrint.ll b/llvm/test/CodeGen/AMDGPU/lrint.ll
index 61820afd9b7742..95ebd6610524e0 100644
--- a/llvm/test/CodeGen/AMDGPU/lrint.ll
+++ b/llvm/test/CodeGen/AMDGPU/lrint.ll
@@ -172,7 +172,7 @@ define i64 @intrinsic_lrint_i64_f32(float %arg) {
; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-SDAG-NEXT: v_xor_b32_e32 v1, v1, v3
; GFX11-SDAG-NEXT: v_xor_b32_e32 v0, v0, v3
-; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_3)
; GFX11-SDAG-NEXT: v_sub_co_u32 v0, vcc_lo, v0, v3
; GFX11-SDAG-NEXT: v_sub_co_ci_u32_e64 v1, null, v1, v3, vcc_lo
; GFX11-SDAG-NEXT: s_setpc_b64 s[30:31]
@@ -195,7 +195,7 @@ define i64 @intrinsic_lrint_i64_f32(float %arg) {
; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-GISEL-NEXT: v_xor_b32_e32 v1, v1, v3
; GFX11-GISEL-NEXT: v_sub_co_u32 v0, vcc_lo, v0, v3
-; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX11-GISEL-NEXT: v_sub_co_ci_u32_e64 v1, null, v1, v3, vcc_lo
; GFX11-GISEL-NEXT: s_setpc_b64 s[30:31]
entry:
@@ -375,7 +375,7 @@ define i64 @intrinsic_llrint_i64_f32(float %arg) {
; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-SDAG-NEXT: v_xor_b32_e32 v1, v1, v3
; GFX11-SDAG-NEXT: v_xor_b32_e32 v0, v0, v3
-; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_3)
; GFX11-SDAG-NEXT: v_sub_co_u32 v0, vcc_lo, v0, v3
; GFX11-SDAG-NEXT: v_sub_co_ci_u32_e64 v1, null, v1, v3, vcc_lo
; GFX11-SDAG-NEXT: s_setpc_b64 s[30:31]
@@ -398,7 +398,7 @@ define i64 @intrinsic_llrint_i64_f32(float %arg) {
; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-GISEL-NEXT: v_xor_b32_e32 v1, v1, v3
; GFX11-GISEL-NEXT: v_sub_co_u32 v0, vcc_lo, v0, v3
-; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX11-GISEL-NEXT: v_sub_co_ci_u32_e64 v1, null, v1, v3, vcc_lo
; GFX11-GISEL-NEXT: s_setpc_b64 s[30:31]
entry:
@@ -779,10 +779,9 @@ define <2 x i64> @intrinsic_lrint_v2i64_v2f32(<2 x float> %arg) {
; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
; GFX11-SDAG-NEXT: v_xor_b32_e32 v1, v1, v5
; GFX11-SDAG-NEXT: v_xor_b32_e32 v4, v0, v6
-; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_3)
; GFX11-SDAG-NEXT: v_sub_co_u32 v0, vcc_lo, v1, v5
; GFX11-SDAG-NEXT: v_sub_co_ci_u32_e64 v1, null, v2, v5, vcc_lo
-; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-SDAG-NEXT: v_sub_co_u32 v2, vcc_lo, v4, v6
; GFX11-SDAG-NEXT: v_sub_co_ci_u32_e64 v3, null, v3, v6, vcc_lo
; GFX11-SDAG-NEXT: s_setpc_b64 s[30:31]
@@ -817,10 +816,10 @@ define <2 x i64> @intrinsic_lrint_v2i64_v2f32(<2 x float> %arg) {
; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
; GFX11-GISEL-NEXT: v_xor_b32_e32 v5, v0, v3
; GFX11-GISEL-NEXT: v_xor_b32_e32 v4, v4, v3
-; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
; GFX11-GISEL-NEXT: v_sub_co_u32 v0, vcc_lo, v1, v6
; GFX11-GISEL-NEXT: v_sub_co_ci_u32_e64 v1, null, v2, v6, vcc_lo
-; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
; GFX11-GISEL-NEXT: v_sub_co_u32 v2, vcc_lo, v5, v3
; GFX11-GISEL-NEXT: v_sub_co_ci_u32_e64 v3, null, v4, v3, vcc_lo
; GFX11-GISEL-NEXT: s_setpc_b64 s[30:31]
diff --git a/llvm/test/CodeGen/AMDGPU/lround.ll b/llvm/test/CodeGen/AMDGPU/lround.ll
index 887735af85b7d2..a5745172e4232f 100644
--- a/llvm/test/CodeGen/AMDGPU/lround.ll
+++ b/llvm/test/CodeGen/AMDGPU/lround.ll
@@ -68,11 +68,11 @@ define i32 @intrinsic_lround_i32_f32(float %arg) {
; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-SDAG-NEXT: v_sub_f32_e32 v2, v0, v1
; GFX11-SDAG-NEXT: v_cmp_ge_f32_e64 s0, |v2|, 0.5
-; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-SDAG-NEXT: v_cndmask_b32_e64 v2, 0, 1.0, s0
-; GFX11-SDAG-NEXT: v_bfi_b32 v0, 0x7fffffff, v2, v0
; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX11-SDAG-NEXT: v_bfi_b32 v0, 0x7fffffff, v2, v0
; GFX11-SDAG-NEXT: v_add_f32_e32 v0, v1, v0
+; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-SDAG-NEXT: v_cvt_i32_f32_e32 v0, v0
; GFX11-SDAG-NEXT: s_setpc_b64 s[30:31]
;
@@ -84,11 +84,11 @@ define i32 @intrinsic_lround_i32_f32(float %arg) {
; GFX11-GISEL-NEXT: v_sub_f32_e32 v2, v0, v1
; GFX11-GISEL-NEXT: v_and_b32_e32 v0, 0x80000000, v0
; GFX11-GISEL-NEXT: v_cmp_ge_f32_e64 s0, |v2|, 0.5
-; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-GISEL-NEXT: v_cndmask_b32_e64 v2, 0, 1.0, s0
-; GFX11-GISEL-NEXT: v_and_or_b32 v0, 0x7fffffff, v2, v0
; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX11-GISEL-NEXT: v_and_or_b32 v0, 0x7fffffff, v2, v0
; GFX11-GISEL-NEXT: v_add_f32_e32 v0, v1, v0
+; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-GISEL-NEXT: v_cvt_i32_f32_e32 v0, v0
; GFX11-GISEL-NEXT: s_setpc_b64 s[30:31]
entry:
@@ -162,12 +162,12 @@ define i32 @intrinsic_lround_i32_f64(double %arg) {
; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-SDAG-NEXT: v_add_f64 v[4:5], v[0:1], -v[2:3]
; GFX11-SDAG-NEXT: v_cmp_ge_f64_e64 s0, |v[4:5]|, 0.5
-; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-SDAG-NEXT: v_cndmask_b32_e64 v0, 0, 0x3ff00000, s0
+; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-SDAG-NEXT: v_bfi_b32 v1, 0x7fffffff, v0, v1
; GFX11-SDAG-NEXT: v_mov_b32_e32 v0, 0
-; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-SDAG-NEXT: v_add_f64 v[0:1], v[2:3], v[0:1]
+; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-SDAG-NEXT: v_cvt_i32_f64_e32 v0, v[0:1]
; GFX11-SDAG-NEXT: s_setpc_b64 s[30:31]
;
@@ -179,11 +179,11 @@ define i32 @intrinsic_lround_i32_f64(double %arg) {
; GFX11-GISEL-NEXT: v_add_f64 v[4:5], v[0:1], -v[2:3]
; GFX11-GISEL-NEXT: v_dual_mov_b32 v0, 0 :: v_dual_and_b32 v1, 0x80000000, v1
; GFX11-GISEL-NEXT: v_cmp_ge_f64_e64 s0, |v[4:5]|, 0.5
-; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-GISEL-NEXT: v_cndmask_b32_e64 v4, 0, 0x3ff00000, s0
-; GFX11-GISEL-NEXT: v_and_or_b32 v1, 0x7fffffff, v4, v1
; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX11-GISEL-NEXT: v_and_or_b32 v1, 0x7fffffff, v4, v1
; GFX11-GISEL-NEXT: v_add_f64 v[0:1], v[2:3], v[0:1]
+; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-GISEL-NEXT: v_cvt_i32_f64_e32 v0, v[0:1]
; GFX11-GISEL-NEXT: s_setpc_b64 s[30:31]
entry:
@@ -295,25 +295,25 @@ define i64 @intrinsic_lround_i64_f32(float %arg) {
; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-SDAG-NEXT: v_sub_f32_e32 v2, v0, v1
; GFX11-SDAG-NEXT: v_cmp_ge_f32_e64 s0, |v2|, 0.5
-; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-SDAG-NEXT: v_cndmask_b32_e64 v2, 0, 1.0, s0
-; GFX11-SDAG-NEXT: v_bfi_b32 v0, 0x7fffffff, v2, v0
; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX11-SDAG-NEXT: v_bfi_b32 v0, 0x7fffffff, v2, v0
; GFX11-SDAG-NEXT: v_add_f32_e32 v0, v1, v0
+; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-SDAG-NEXT: v_trunc_f32_e32 v0, v0
-; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX11-SDAG-NEXT: v_mul_f32_e64 v1, 0x2f800000, |v0|
; GFX11-SDAG-NEXT: v_ashrrev_i32_e32 v3, 31, v0
+; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-SDAG-NEXT: v_floor_f32_e32 v1, v1
-; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX11-SDAG-NEXT: v_fma_f32 v2, 0xcf800000, v1, |v0|
; GFX11-SDAG-NEXT: v_cvt_u32_f32_e32 v1, v1
-; GFX11-SDAG-NEXT: v_cvt_u32_f32_e32 v0, v2
; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX11-SDAG-NEXT: v_cvt_u32_f32_e32 v0, v2
; GFX11-SDAG-NEXT: v_xor_b32_e32 v1, v1, v3
+; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-SDAG-NEXT: v_xor_b32_e32 v0, v0, v3
-; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-SDAG-NEXT: v_sub_co_u32 v0, vcc_lo, v0, v3
+; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX11-SDAG-NEXT: v_sub_co_ci_u32_e64 v1, null, v1, v3, vcc_lo
; GFX11-SDAG-NEXT: s_setpc_b64 s[30:31]
;
@@ -325,25 +325,25 @@ define i64 @intrinsic_lround_i64_f32(float %arg) {
; GFX11-GISEL-NEXT: v_sub_f32_e32 v2, v0, v1
; GFX11-GISEL-NEXT: v_and_b32_e32 v0, 0x80000000, v0
; GFX11-GISEL-NEXT: v_cmp_ge_f32_e64 s0, |v2|, 0.5
-; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-GISEL-NEXT: v_cndmask_b32_e64 v2, 0, 1.0, s0
-; GFX11-GISEL-NEXT: v_and_or_b32 v0, 0x7fffffff, v2, v0
; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX11-GISEL-NEXT: v_and_or_b32 v0, 0x7fffffff, v2, v0
; GFX11-GISEL-NEXT: v_add_f32_e32 v0, v1, v0
+; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX11-GISEL-NEXT: v_trunc_f32_e32 v1, v0
; GFX11-GISEL-NEXT: v_ashrrev_i32_e32 v3, 31, v0
-; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-GISEL-NEXT: v_mul_f32_e64 v2, 0x2f800000, |v1|
-; GFX11-GISEL-NEXT: v_floor_f32_e32 v2, v2
; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX11-GISEL-NEXT: v_floor_f32_e32 v2, v2
; GFX11-GISEL-NEXT: v_fma_f32 v1, 0xcf800000, v2, |v1|
+; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX11-GISEL-NEXT: v_cvt_u32_f32_e32 v0, v1
; GFX11-GISEL-NEXT: v_cvt_u32_f32_e32 v1, v2
-; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-GISEL-NEXT: v_xor_b32_e32 v0, v0, v3
+; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-GISEL-NEXT: v_xor_b32_e32 v1, v1, v3
-; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-GISEL-NEXT: v_sub_co_u32 v0, vcc_lo, v0, v3
+; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX11-GISEL-NEXT: v_sub_co_ci_u32_e64 v1, null, v1, v3, vcc_lo
; GFX11-GISEL-NEXT: s_setpc_b64 s[30:31]
entry:
@@ -445,17 +445,17 @@ define i64 @intrinsic_lround_i64_f64(double %arg) {
; GFX11-SDAG-NEXT: v_add_f64 v[4:5], v[0:1], -v[2:3]
; GFX11-SDAG-NEXT: v_mov_b32_e32 v0, 0
; GFX11-SDAG-NEXT: v_cmp_ge_f64_e64 s0, |v[4:5]|, 0.5
-; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-SDAG-NEXT: v_cndmask_b32_e64 v4, 0, 0x3ff00000, s0
-; GFX11-SDAG-NEXT: v_bfi_b32 v1, 0x7fffffff, v4, v1
; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX11-SDAG-NEXT: v_bfi_b32 v1, 0x7fffffff, v4, v1
; GFX11-SDAG-NEXT: v_add_f64 v[0:1], v[2:3], v[0:1]
-; GFX11-SDAG-NEXT: v_trunc_f64_e32 v[0:1], v[0:1]
; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX11-SDAG-NEXT: v_trunc_f64_e32 v[0:1], v[0:1]
; GFX11-SDAG-NEXT: v_ldexp_f64 v[2:3], v[0:1], 0xffffffe0
-; GFX11-SDAG-NEXT: v_floor_f64_e32 v[2:3], v[2:3]
; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX11-SDAG-NEXT: v_floor_f64_e32 v[2:3], v[2:3]
; GFX11-SDAG-NEXT: v_fma_f64 v[0:1], 0xc1f00000, v[2:3], v[0:1]
+; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-SDAG-NEXT: v_cvt_u32_f64_e32 v0, v[0:1]
; GFX11-SDAG-NEXT: v_cvt_i32_f64_e32 v1, v[2:3]
; GFX11-SDAG-NEXT: s_setpc_b64 s[30:31]
@@ -468,17 +468,17 @@ define i64 @intrinsic_lround_i64_f64(double %arg) {
; GFX11-GISEL-NEXT: v_add_f64 v[4:5], v[0:1], -v[2:3]
; GFX11-GISEL-NEXT: v_dual_mov_b32 v0, 0 :: v_dual_and_b32 v1, 0x80000000, v1
; GFX11-GISEL-NEXT: v_cmp_ge_f64_e64 s0, |v[4:5]|, 0.5
-; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-GISEL-NEXT: v_cndmask_b32_e64 v4, 0, 0x3ff00000, s0
-; GFX11-GISEL-NEXT: v_and_or_b32 v1, 0x7fffffff, v4, v1
; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX11-GISEL-NEXT: v_and_or_b32 v1, 0x7fffffff, v4, v1
; GFX11-GISEL-NEXT: v_add_f64 v[0:1], v[2:3], v[0:1]
-; GFX11-GISEL-NEXT: v_trunc_f64_e32 v[0:1], v[0:1]
; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX11-GISEL-NEXT: v_trunc_f64_e32 v[0:1], v[0:1]
; GFX11-GISEL-NEXT: v_mul_f64 v[2:3], 0x3df00000, v[0:1]
-; GFX11-GISEL-NEXT: v_floor_f64_e32 v[2:3], v[2:3]
; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX11-GISEL-NEXT: v_floor_f64_e32 v[2:3], v[2:3]
; GFX11-GISEL-NEXT: v_fma_f64 v[0:1], 0xc1f00000, v[2:3], v[0:1]
+; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-GISEL-NEXT: v_cvt_u32_f64_e32 v0, v[0:1]
; GFX11-GISEL-NEXT: v_cvt_i32_f64_e32 v1, v[2:3]
; GFX11-GISEL-NEXT: s_setpc_b64 s[30:31]
@@ -591,25 +591,25 @@ define i64 @intrinsic_llround_i64_f32(float %arg) {
; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-SDAG-NEXT: v_sub_f32_e32 v2, v0, v1
; GFX11-SDAG-NEXT: v_cmp_ge_f32_e64 s0, |v2|, 0.5
-; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-SDAG-NEXT: v_cndmask_b32_e64 v2, 0, 1.0, s0
-; GFX11-SDAG-NEXT: v_bfi_b32 v0, 0x7fffffff, v2, v0
; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX11-SDAG-NEXT: v_bfi_b32 v0, 0x7fffffff, v2, v0
; GFX11-SDAG-NEXT: v_add_f32_e32 v0, v1, v0
+; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-SDAG-NEXT: v_trunc_f32_e32 v0, v0
-; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX11-SDAG-NEXT: v_mul_f32_e64 v1, 0x2f800000, |v0|
; GFX11-SDAG-NEXT: v_ashrrev_i32_e32 v3, 31, v0
+; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-SDAG-NEXT: v_floor_f32_e32 v1, v1
-; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX11-SDAG-NEXT: v_fma_f32 v2, 0xcf800000, v1, |v0|
; GFX11-SDAG-NEXT: v_cvt_u32_f32_e32 v1, v1
-; GFX11-SDAG-NEXT: v_cvt_u32_f32_e32 v0, v2
; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX11-SDAG-NEXT: v_cvt_u32_f32_e32 v0, v2
; GFX11-SDAG-NEXT: v_xor_b32_e32 v1, v1, v3
+; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-SDAG-NEXT: v_xor_b32_e32 v0, v0, v3
-; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-SDAG-NEXT: v_sub_co_u32 v0, vcc_lo, v0, v3
+; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX11-SDAG-NEXT: v_sub_co_ci_u32_e64 v1, null, v1, v3, vcc_lo
; GFX11-SDAG-NEXT: s_setpc_b64 s[30:31]
;
@@ -621,25 +621,25 @@ define i64 @intrinsic_llround_i64_f32(float %arg) {
; GFX11-GISEL-NEXT: v_sub_f32_e32 v2, v0, v1
; GFX11-GISEL-NEXT: v_and_b32_e32 v0, 0x80000000, v0
; GFX11-GISEL-NEXT: v_cmp_ge_f32_e64 s0, |v2|, 0.5
-; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-GISEL-NEXT: v_cndmask_b32_e64 v2, 0, 1.0, s0
-; GFX11-GISEL-NEXT: v_and_or_b32 v0, 0x7fffffff, v2, v0
; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX11-GISEL-NEXT: v_and_or_b32 v0, 0x7fffffff, v2, v0
; GFX11-GISEL-NEXT: v_add_f32_e32 v0, v1, v0
+; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX11-GISEL-NEXT: v_trunc_f32_e32 v1, v0
; GFX11-GISEL-NEXT: v_ashrrev_i32_e32 v3, 31, v0
-; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-GISEL-NEXT: v_mul_f32_e64 v2, 0x2f800000, |v1|
-; GFX11-GISEL-NEXT: v_floor_f32_e32 v2, v2
; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX11-GISEL-NEXT: v_floor_f32_e32 v2, v2
; GFX11-GISEL-NEXT: v_fma_f32 v1, 0xcf800000, v2, |v1|
+; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX11-GISEL-NEXT: v_cvt_u32_f32_e32 v0, v1
; GFX11-GISEL-NEXT: v_cvt_u32_f32_e32 v1, v2
-; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-GISEL-NEXT: v_xor_b32_e32 v0, v0, v3
+; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-GISEL-NEXT: v_xor_b32_e32 v1, v1, v3
-; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-GISEL-NEXT: v_sub_co_u32 v0, vcc_lo, v0, v3
+; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX11-GISEL-NEXT: v_sub_co_ci_u32_e64 v1, null, v1, v3, vcc_lo
; GFX11-GISEL-NEXT: s_setpc_b64 s[30:31]
entry:
@@ -741,17 +741,17 @@ define i64 @intrinsic_llround_i64_f64(double %arg) {
; GFX11-SDAG-NEXT: v_add_f64 v[4:5], v[0:1], -v[2:3]
; GFX11-SDAG-NEXT: v_mov_b32_e32 v0, 0
; GFX11-SDAG-NEXT: v_cmp_ge_f64_e64 s0, |v[4:5]|, 0.5
-; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-SDAG-NEXT: v_cndmask_b32_e64 v4, 0, 0x3ff00000, s0
-; GFX11-SDAG-NEXT: v_bfi_b32 v1, 0x7fffffff, v4, v1
; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX11-SDAG-NEXT: v_bfi_b32 v1, 0x7fffffff, v4, v1
; GFX11-SDAG-NEXT: v_add_f64 v[0:1], v[2:3], v[0:1]
-; GFX11-SDAG-NEXT: v_trunc_f64_e32 v[0:1], v[0:1]
; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX11-SDAG-NEXT: v_trunc_f64_e32 v[0:1], v[0:1]
; GFX11-SDAG-NEXT: v_ldexp_f64 v[2:3], v[0:1], 0xffffffe0
-; GFX11-SDAG-NEXT: v_floor_f64_e32 v[2:3], v[2:3]
; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX11-SDAG-NEXT: v_floor_f64_e32 v[2:3], v[2:3]
; GFX11-SDAG-NEXT: v_fma_f64 v[0:1], 0xc1f00000, v[2:3], v[0:1]
+; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-SDAG-NEXT: v_cvt_u32_f64_e32 v0, v[0:1]
; GFX11-SDAG-NEXT: v_cvt_i32_f64_e32 v1, v[2:3]
; GFX11-SDAG-NEXT: s_setpc_b64 s[30:31]
@@ -764,17 +764,17 @@ define i64 @intrinsic_llround_i64_f64(double %arg) {
; GFX11-GISEL-NEXT: v_add_f64 v[4:5], v[0:1], -v[2:3]
; GFX11-GISEL-NEXT: v_dual_mov_b32 v0, 0 :: v_dual_and_b32 v1, 0x80000000, v1
; GFX11-GISEL-NEXT: v_cmp_ge_f64_e64 s0, |v[4:5]|, 0.5
-; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-GISEL-NEXT: v_cndmask_b32_e64 v4, 0, 0x3ff00000, s0
-; GFX11-GISEL-NEXT: v_and_or_b32 v1, 0x7fffffff, v4, v1
; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX11-GISEL-NEXT: v_and_or_b32 v1, 0x7fffffff, v4, v1
; GFX11-GISEL-NEXT: v_add_f64 v[0:1], v[2:3], v[0:1]
-; GFX11-GISEL-NEXT: v_trunc_f64_e32 v[0:1], v[0:1]
; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX11-GISEL-NEXT: v_trunc_f64_e32 v[0:1], v[0:1]
; GFX11-GISEL-NEXT: v_mul_f64 v[2:3], 0x3df00000, v[0:1]
-; GFX11-GISEL-NEXT: v_floor_f64_e32 v[2:3], v[2:3]
; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX11-GISEL-NEXT: v_floor_f64_e32 v[2:3], v[2:3]
; GFX11-GISEL-NEXT: v_fma_f64 v[0:1], 0xc1f00000, v[2:3], v[0:1]
+; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-GISEL-NEXT: v_cvt_u32_f64_e32 v0, v[0:1]
; GFX11-GISEL-NEXT: v_cvt_i32_f64_e32 v1, v[2:3]
; GFX11-GISEL-NEXT: s_setpc_b64 s[30:31]
@@ -842,11 +842,11 @@ define half @intrinsic_fround_half(half %arg) {
; GFX11-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-SDAG-TRUE16-NEXT: v_trunc_f16_e32 v1.h, v1.l
; GFX11-SDAG-TRUE16-NEXT: v_sub_f16_e32 v1.l, v1.l, v1.h
-; GFX11-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX11-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-SDAG-TRUE16-NEXT: v_cmp_ge_f16_e64 s0, |v1.l|, 0.5
; GFX11-SDAG-TRUE16-NEXT: v_cndmask_b16 v1.l, 0, 0x3c00, s0
-; GFX11-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-SDAG-TRUE16-NEXT: v_bfi_b32 v0, 0x7fff, v1, v0
+; GFX11-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-SDAG-TRUE16-NEXT: v_add_f16_e32 v0.l, v1.h, v0.l
; GFX11-SDAG-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -857,10 +857,9 @@ define half @intrinsic_fround_half(half %arg) {
; GFX11-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-SDAG-FAKE16-NEXT: v_sub_f16_e32 v2, v0, v1
; GFX11-SDAG-FAKE16-NEXT: v_cmp_ge_f16_e64 s0, |v2|, 0.5
-; GFX11-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-SDAG-FAKE16-NEXT: v_cndmask_b32_e64 v2, 0, 0x3c00, s0
+; GFX11-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-SDAG-FAKE16-NEXT: v_bfi_b32 v0, 0x7fff, v2, v0
-; GFX11-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-SDAG-FAKE16-NEXT: v_add_f16_e32 v0, v1, v0
; GFX11-SDAG-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -872,11 +871,11 @@ define half @intrinsic_fround_half(half %arg) {
; GFX11-GISEL-TRUE16-NEXT: v_sub_f16_e32 v1.l, v0.l, v0.h
; GFX11-GISEL-TRUE16-NEXT: v_and_b16 v0.l, 0x8000, v0.l
; GFX11-GISEL-TRUE16-NEXT: v_cmp_ge_f16_e64 s0, |v1.l|, 0.5
-; GFX11-GISEL-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-GISEL-TRUE16-NEXT: v_cndmask_b16 v1.l, 0, 0x3c00, s0
-; GFX11-GISEL-TRUE16-NEXT: v_and_b16 v1.l, 0x7fff, v1.l
; GFX11-GISEL-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX11-GISEL-TRUE16-NEXT: v_and_b16 v1.l, 0x7fff, v1.l
; GFX11-GISEL-TRUE16-NEXT: v_or_b16 v0.l, v1.l, v0.l
+; GFX11-GISEL-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-GISEL-TRUE16-NEXT: v_add_f16_e32 v0.l, v0.h, v0.l
; GFX11-GISEL-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -888,11 +887,11 @@ define half @intrinsic_fround_half(half %arg) {
; GFX11-GISEL-FAKE16-NEXT: v_sub_f16_e32 v2, v0, v1
; GFX11-GISEL-FAKE16-NEXT: v_and_b32_e32 v0, 0xffff8000, v0
; GFX11-GISEL-FAKE16-NEXT: v_cmp_ge_f16_e64 s0, |v2|, 0.5
-; GFX11-GISEL-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-GISEL-FAKE16-NEXT: v_cndmask_b32_e64 v2, 0, 0x3c00, s0
-; GFX11-GISEL-FAKE16-NEXT: v_and_b32_e32 v2, 0x7fff, v2
; GFX11-GISEL-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX11-GISEL-FAKE16-NEXT: v_and_b32_e32 v2, 0x7fff, v2
; GFX11-GISEL-FAKE16-NEXT: v_or_b32_e32 v0, v2, v0
+; GFX11-GISEL-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-GISEL-FAKE16-NEXT: v_add_f16_e32 v0, v1, v0
; GFX11-GISEL-FAKE16-NEXT: s_setpc_b64 s[30:31]
entry:
@@ -967,14 +966,14 @@ define i32 @intrinsic_lround_i32_f16(half %arg) {
; GFX11-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-SDAG-TRUE16-NEXT: v_trunc_f16_e32 v1.h, v1.l
; GFX11-SDAG-TRUE16-NEXT: v_sub_f16_e32 v1.l, v1.l, v1.h
-; GFX11-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX11-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-SDAG-TRUE16-NEXT: v_cmp_ge_f16_e64 s0, |v1.l|, 0.5
; GFX11-SDAG-TRUE16-NEXT: v_cndmask_b16 v1.l, 0, 0x3c00, s0
-; GFX11-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-SDAG-TRUE16-NEXT: v_bfi_b32 v0, 0x7fff, v1, v0
-; GFX11-SDAG-TRUE16-NEXT: v_add_f16_e32 v0.l, v1.h, v0.l
; GFX11-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX11-SDAG-TRUE16-NEXT: v_add_f16_e32 v0.l, v1.h, v0.l
; GFX11-SDAG-TRUE16-NEXT: v_cvt_f32_f16_e32 v0, v0.l
+; GFX11-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-SDAG-TRUE16-NEXT: v_cvt_i32_f32_e32 v0, v0
; GFX11-SDAG-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -985,13 +984,12 @@ define i32 @intrinsic_lround_i32_f16(half %arg) {
; GFX11-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-SDAG-FAKE16-NEXT: v_sub_f16_e32 v2, v0, v1
; GFX11-SDAG-FAKE16-NEXT: v_cmp_ge_f16_e64 s0, |v2|, 0.5
-; GFX11-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-SDAG-FAKE16-NEXT: v_cndmask_b32_e64 v2, 0, 0x3c00, s0
-; GFX11-SDAG-FAKE16-NEXT: v_bfi_b32 v0, 0x7fff, v2, v0
; GFX11-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX11-SDAG-FAKE16-NEXT: v_bfi_b32 v0, 0x7fff, v2, v0
; GFX11-SDAG-FAKE16-NEXT: v_add_f16_e32 v0, v1, v0
+; GFX11-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-SDAG-FAKE16-NEXT: v_cvt_f32_f16_e32 v0, v0
-; GFX11-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-SDAG-FAKE16-NEXT: v_cvt_i32_f32_e32 v0, v0
; GFX11-SDAG-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -1003,14 +1001,14 @@ define i32 @intrinsic_lround_i32_f16(half %arg) {
; GFX11-GISEL-TRUE16-NEXT: v_sub_f16_e32 v1.l, v0.l, v0.h
; GFX11-GISEL-TRUE16-NEXT: v_and_b16 v0.l, 0x8000, v0.l
; GFX11-GISEL-TRUE16-NEXT: v_cmp_ge_f16_e64 s0, |v1.l|, 0.5
-; GFX11-GISEL-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-GISEL-TRUE16-NEXT: v_cndmask_b16 v1.l, 0, 0x3c00, s0
-; GFX11-GISEL-TRUE16-NEXT: v_and_b16 v1.l, 0x7fff, v1.l
; GFX11-GISEL-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX11-GISEL-TRUE16-NEXT: v_and_b16 v1.l, 0x7fff, v1.l
; GFX11-GISEL-TRUE16-NEXT: v_or_b16 v0.l, v1.l, v0.l
-; GFX11-GISEL-TRUE16-NEXT: v_add_f16_e32 v0.l, v0.h, v0.l
; GFX11-GISEL-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX11-GISEL-TRUE16-NEXT: v_add_f16_e32 v0.l, v0.h, v0.l
; GFX11-GISEL-TRUE16-NEXT: v_cvt_f32_f16_e32 v0, v0.l
+; GFX11-GISEL-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-GISEL-TRUE16-NEXT: v_cvt_i32_f32_e32 v0, v0
; GFX11-GISEL-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -1022,14 +1020,14 @@ define i32 @intrinsic_lround_i32_f16(half %arg) {
; GFX11-GISEL-FAKE16-NEXT: v_sub_f16_e32 v2, v0, v1
; GFX11-GISEL-FAKE16-NEXT: v_and_b32_e32 v0, 0xffff8000, v0
; GFX11-GISEL-FAKE16-NEXT: v_cmp_ge_f16_e64 s0, |v2|, 0.5
-; GFX11-GISEL-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-GISEL-FAKE16-NEXT: v_cndmask_b32_e64 v2, 0, 0x3c00, s0
-; GFX11-GISEL-FAKE16-NEXT: v_and_b32_e32 v2, 0x7fff, v2
; GFX11-GISEL-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX11-GISEL-FAKE16-NEXT: v_and_b32_e32 v2, 0x7fff, v2
; GFX11-GISEL-FAKE16-NEXT: v_or_b32_e32 v0, v2, v0
-; GFX11-GISEL-FAKE16-NEXT: v_add_f16_e32 v0, v1, v0
; GFX11-GISEL-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX11-GISEL-FAKE16-NEXT: v_add_f16_e32 v0, v1, v0
; GFX11-GISEL-FAKE16-NEXT: v_cvt_f32_f16_e32 v0, v0
+; GFX11-GISEL-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-GISEL-FAKE16-NEXT: v_cvt_i32_f32_e32 v0, v0
; GFX11-GISEL-FAKE16-NEXT: s_setpc_b64 s[30:31]
entry:
@@ -1128,10 +1126,9 @@ define <2 x i32> @intrinsic_lround_v2i32_v2f32(<2 x float> %arg) {
; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-SDAG-NEXT: v_dual_sub_f32 v4, v0, v2 :: v_dual_sub_f32 v5, v1, v3
; GFX11-SDAG-NEXT: v_cmp_ge_f32_e64 s0, |v4|, 0.5
-; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_3)
; GFX11-SDAG-NEXT: v_cndmask_b32_e64 v4, 0, 1.0, s0
+; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-SDAG-NEXT: v_cmp_ge_f32_e64 s0, |v5|, 0.5
-; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-SDAG-NEXT: v_bfi_b32 v0, 0x7fffffff, v4, v0
; GFX11-SDAG-NEXT: v_cndmask_b32_e64 v5, 0, 1.0, s0
; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
@@ -1151,10 +1148,9 @@ define <2 x i32> @intrinsic_lround_v2i32_v2f32(<2 x float> %arg) {
; GFX11-GISEL-NEXT: v_dual_sub_f32 v4, v0, v2 :: v_dual_sub_f32 v5, v1, v3
; GFX11-GISEL-NEXT: v_and_b32_e32 v1, 0x80000000, v1
; GFX11-GISEL-NEXT: v_cmp_ge_f32_e64 s0, |v4|, 0.5
-; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_4)
; GFX11-GISEL-NEXT: v_cndmask_b32_e64 v4, 0, 1.0, s0
+; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-GISEL-NEXT: v_cmp_ge_f32_e64 s0, |v5|, 0.5
-; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-GISEL-NEXT: v_cndmask_b32_e64 v5, 0, 1.0, s0
; GFX11-GISEL-NEXT: v_and_or_b32 v1, 0x7fffffff, v5, v1
; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
@@ -1346,10 +1342,9 @@ define <2 x i64> @intrinsic_lround_v2i64_v2f32(<2 x float> %arg) {
; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-SDAG-NEXT: v_dual_sub_f32 v4, v0, v2 :: v_dual_sub_f32 v5, v1, v3
; GFX11-SDAG-NEXT: v_cmp_ge_f32_e64 s0, |v4|, 0.5
-; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_3)
; GFX11-SDAG-NEXT: v_cndmask_b32_e64 v4, 0, 1.0, s0
+; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-SDAG-NEXT: v_cmp_ge_f32_e64 s0, |v5|, 0.5
-; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-SDAG-NEXT: v_bfi_b32 v0, 0x7fffffff, v4, v0
; GFX11-SDAG-NEXT: v_cndmask_b32_e64 v5, 0, 1.0, s0
; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
@@ -1380,10 +1375,9 @@ define <2 x i64> @intrinsic_lround_v2i64_v2f32(<2 x float> %arg) {
; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_3)
; GFX11-SDAG-NEXT: v_xor_b32_e32 v3, v3, v6
; GFX11-SDAG-NEXT: v_xor_b32_e32 v4, v0, v6
-; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_1) | instid1(VALU_DEP_3)
; GFX11-SDAG-NEXT: v_sub_co_u32 v0, vcc_lo, v1, v5
; GFX11-SDAG-NEXT: v_sub_co_ci_u32_e64 v1, null, v2, v5, vcc_lo
-; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-SDAG-NEXT: v_sub_co_u32 v2, vcc_lo, v4, v6
; GFX11-SDAG-NEXT: v_sub_co_ci_u32_e64 v3, null, v3, v6, vcc_lo
; GFX11-SDAG-NEXT: s_setpc_b64 s[30:31]
@@ -1397,10 +1391,9 @@ define <2 x i64> @intrinsic_lround_v2i64_v2f32(<2 x float> %arg) {
; GFX11-GISEL-NEXT: v_dual_sub_f32 v4, v0, v2 :: v_dual_sub_f32 v5, v1, v3
; GFX11-GISEL-NEXT: v_and_b32_e32 v1, 0x80000000, v1
; GFX11-GISEL-NEXT: v_cmp_ge_f32_e64 s0, |v4|, 0.5
-; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_4)
; GFX11-GISEL-NEXT: v_cndmask_b32_e64 v4, 0, 1.0, s0
+; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-GISEL-NEXT: v_cmp_ge_f32_e64 s0, |v5|, 0.5
-; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-GISEL-NEXT: v_cndmask_b32_e64 v5, 0, 1.0, s0
; GFX11-GISEL-NEXT: v_and_or_b32 v1, 0x7fffffff, v5, v1
; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
@@ -1432,11 +1425,11 @@ define <2 x i64> @intrinsic_lround_v2i64_v2f32(<2 x float> %arg) {
; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
; GFX11-GISEL-NEXT: v_xor_b32_e32 v2, v2, v6
; GFX11-GISEL-NEXT: v_xor_b32_e32 v4, v4, v3
-; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_3)
; GFX11-GISEL-NEXT: v_sub_co_u32 v0, vcc_lo, v1, v6
; GFX11-GISEL-NEXT: v_sub_co_ci_u32_e64 v1, null, v2, v6, vcc_lo
; GFX11-GISEL-NEXT: v_sub_co_u32 v2, vcc_lo, v5, v3
-; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_4)
; GFX11-GISEL-NEXT: v_sub_co_ci_u32_e64 v3, null, v4, v3, vcc_lo
; GFX11-GISEL-NEXT: s_setpc_b64 s[30:31]
entry:
diff --git a/llvm/test/CodeGen/AMDGPU/mad_64_32.ll b/llvm/test/CodeGen/AMDGPU/mad_64_32.ll
index bbc7f6cce50b09..2a58de73060ae5 100644
--- a/llvm/test/CodeGen/AMDGPU/mad_64_32.ll
+++ b/llvm/test/CodeGen/AMDGPU/mad_64_32.ll
@@ -341,18 +341,17 @@ define i128 @mad_i64_i32_sextops_i32_i128(i32 %arg0, i32 %arg1, i128 %arg2) #0 {
; GFX1100-NEXT: v_mad_u64_u32 v[11:12], null, v0, v15, v[7:8]
; GFX1100-NEXT: v_mad_i64_i32 v[7:8], null, v1, v14, 0
; GFX1100-NEXT: v_add_co_u32 v9, s0, v10, v12
-; GFX1100-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_3)
; GFX1100-NEXT: v_add_co_ci_u32_e64 v10, null, 0, 0, s0
+; GFX1100-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1100-NEXT: v_mad_i64_i32 v[12:13], null, v15, v0, v[7:8]
-; GFX1100-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1100-NEXT: v_mad_u64_u32 v[0:1], null, v14, v15, v[9:10]
+; GFX1100-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1100-NEXT: v_add_co_u32 v7, vcc_lo, v0, v12
-; GFX1100-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_2) | instid1(VALU_DEP_4)
; GFX1100-NEXT: v_add_co_ci_u32_e64 v8, null, v1, v13, vcc_lo
; GFX1100-NEXT: v_add_co_u32 v0, vcc_lo, v6, v2
; GFX1100-NEXT: v_add_co_ci_u32_e32 v1, vcc_lo, v11, v3, vcc_lo
+; GFX1100-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
; GFX1100-NEXT: v_add_co_ci_u32_e32 v2, vcc_lo, v7, v4, vcc_lo
-; GFX1100-NEXT: s_delay_alu instid0(VALU_DEP_4)
; GFX1100-NEXT: v_add_co_ci_u32_e64 v3, null, v8, v5, vcc_lo
; GFX1100-NEXT: s_setpc_b64 s[30:31]
;
@@ -372,16 +371,16 @@ define i128 @mad_i64_i32_sextops_i32_i128(i32 %arg0, i32 %arg1, i128 %arg2) #0 {
; GFX1150-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1150-NEXT: v_mad_i64_i32 v[0:1], null, v14, v0, v[11:12]
; GFX1150-NEXT: v_add_co_u32 v8, s0, v10, v8
-; GFX1150-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1150-NEXT: v_add_co_ci_u32_e64 v9, null, 0, 0, s0
-; GFX1150-NEXT: v_mad_u64_u32 v[8:9], null, v13, v14, v[8:9]
; GFX1150-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1150-NEXT: v_mad_u64_u32 v[8:9], null, v13, v14, v[8:9]
; GFX1150-NEXT: v_add_co_u32 v8, vcc_lo, v8, v0
+; GFX1150-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_2) | instid1(VALU_DEP_4)
; GFX1150-NEXT: v_add_co_ci_u32_e64 v9, null, v9, v1, vcc_lo
; GFX1150-NEXT: v_add_co_u32 v0, vcc_lo, v6, v2
; GFX1150-NEXT: v_add_co_ci_u32_e32 v1, vcc_lo, v7, v3, vcc_lo
-; GFX1150-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
; GFX1150-NEXT: v_add_co_ci_u32_e32 v2, vcc_lo, v8, v4, vcc_lo
+; GFX1150-NEXT: s_delay_alu instid0(VALU_DEP_4)
; GFX1150-NEXT: v_add_co_ci_u32_e64 v3, null, v9, v5, vcc_lo
; GFX1150-NEXT: s_setpc_b64 s[30:31]
;
@@ -1283,11 +1282,10 @@ define i64 @mad_i64_i32_thrice(i32 %arg0, i32 %arg1, i64 %arg2, i64 %arg3, i64 %
; GFX1100: ; %bb.0:
; GFX1100-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX1100-NEXT: v_mad_i64_i32 v[8:9], null, v0, v1, 0
-; GFX1100-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1100-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1100-NEXT: v_add_co_u32 v0, vcc_lo, v8, v2
; GFX1100-NEXT: v_add_co_ci_u32_e64 v1, null, v9, v3, vcc_lo
; GFX1100-NEXT: v_add_co_u32 v2, vcc_lo, v8, v4
-; GFX1100-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX1100-NEXT: v_add_co_ci_u32_e64 v3, null, v9, v5, vcc_lo
; GFX1100-NEXT: v_add_co_u32 v4, vcc_lo, v8, v6
; GFX1100-NEXT: v_add_co_ci_u32_e64 v5, null, v9, v7, vcc_lo
@@ -1303,11 +1301,10 @@ define i64 @mad_i64_i32_thrice(i32 %arg0, i32 %arg1, i64 %arg2, i64 %arg3, i64 %
; GFX1150: ; %bb.0:
; GFX1150-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX1150-NEXT: v_mad_i64_i32 v[0:1], null, v0, v1, 0
-; GFX1150-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1150-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1150-NEXT: v_add_co_u32 v2, vcc_lo, v0, v2
; GFX1150-NEXT: v_add_co_ci_u32_e64 v3, null, v1, v3, vcc_lo
; GFX1150-NEXT: v_add_co_u32 v4, vcc_lo, v0, v4
-; GFX1150-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX1150-NEXT: v_add_co_ci_u32_e64 v5, null, v1, v5, vcc_lo
; GFX1150-NEXT: v_add_co_u32 v0, vcc_lo, v0, v6
; GFX1150-NEXT: v_add_co_ci_u32_e64 v1, null, v1, v7, vcc_lo
@@ -1409,7 +1406,7 @@ define i64 @mad_i64_i32_secondary_use(i32 %arg0, i32 %arg1, i64 %arg2) #0 {
; GFX1100: ; %bb.0:
; GFX1100-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX1100-NEXT: v_mad_i64_i32 v[4:5], null, v0, v1, 0
-; GFX1100-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1100-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1100-NEXT: v_add_co_u32 v0, vcc_lo, v4, v2
; GFX1100-NEXT: v_add_co_ci_u32_e64 v1, null, v5, v3, vcc_lo
; GFX1100-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
@@ -1421,7 +1418,7 @@ define i64 @mad_i64_i32_secondary_use(i32 %arg0, i32 %arg1, i64 %arg2) #0 {
; GFX1150: ; %bb.0:
; GFX1150-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX1150-NEXT: v_mad_i64_i32 v[0:1], null, v0, v1, 0
-; GFX1150-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1150-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1150-NEXT: v_add_co_u32 v2, vcc_lo, v0, v2
; GFX1150-NEXT: v_add_co_ci_u32_e64 v3, null, v1, v3, vcc_lo
; GFX1150-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
@@ -2049,10 +2046,10 @@ define i64 @lshr_mad_i64_negative_3(i64 %arg0) #0 {
; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_and_b32_e32 v2, 0xfffffc00, v2
; GFX11-NEXT: v_sub_co_u32 v0, vcc_lo, v0, v2
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-NEXT: v_sub_co_ci_u32_e64 v1, null, v1, v3, vcc_lo
; GFX11-NEXT: v_add_co_u32 v0, vcc_lo, v0, 1
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-NEXT: s_setpc_b64 s[30:31]
;
diff --git a/llvm/test/CodeGen/AMDGPU/madak.ll b/llvm/test/CodeGen/AMDGPU/madak.ll
index 211b93e9ace29c..d0ec47e4db747b 100644
--- a/llvm/test/CodeGen/AMDGPU/madak.ll
+++ b/llvm/test/CodeGen/AMDGPU/madak.ll
@@ -1408,6 +1408,7 @@ define amdgpu_kernel void @madak_constant_bus_violation(i32 %arg1, [8 x i32], fl
; GFX11-MAD-NEXT: s_load_b32 s0, s[4:5], 0x24
; GFX11-MAD-NEXT: s_waitcnt lgkmcnt(0)
; GFX11-MAD-NEXT: s_cmp_lg_u32 s0, 0
+; GFX11-MAD-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-MAD-NEXT: s_cbranch_scc1 .LBB9_2
; GFX11-MAD-NEXT: ; %bb.1: ; %bb3
; GFX11-MAD-NEXT: v_mov_b32_e32 v0, 0
@@ -1475,6 +1476,7 @@ define amdgpu_kernel void @madak_constant_bus_violation(i32 %arg1, [8 x i32], fl
; GFX11-FMA-NEXT: s_load_b32 s0, s[4:5], 0x24
; GFX11-FMA-NEXT: s_waitcnt lgkmcnt(0)
; GFX11-FMA-NEXT: s_cmp_lg_u32 s0, 0
+; GFX11-FMA-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FMA-NEXT: s_cbranch_scc1 .LBB9_2
; GFX11-FMA-NEXT: ; %bb.1: ; %bb3
; GFX11-FMA-NEXT: v_mov_b32_e32 v0, 0
diff --git a/llvm/test/CodeGen/AMDGPU/match-perm-extract-vector-elt-bug.ll b/llvm/test/CodeGen/AMDGPU/match-perm-extract-vector-elt-bug.ll
index 60c7a3a45ab575..3c6f327f312d12 100644
--- a/llvm/test/CodeGen/AMDGPU/match-perm-extract-vector-elt-bug.ll
+++ b/llvm/test/CodeGen/AMDGPU/match-perm-extract-vector-elt-bug.ll
@@ -72,7 +72,7 @@ define amdgpu_kernel void @test(ptr addrspace(1) %src, ptr addrspace(1) %dst) {
; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_ashrrev_i64 v[4:5], 28, v[0:1]
; GFX11-NEXT: v_add_co_u32 v0, vcc_lo, s0, v4
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, s1, v5, vcc_lo
; GFX11-NEXT: v_add_co_u32 v4, vcc_lo, s2, v4
; GFX11-NEXT: v_add_co_ci_u32_e64 v5, null, s3, v5, vcc_lo
diff --git a/llvm/test/CodeGen/AMDGPU/maximumnum.bf16.ll b/llvm/test/CodeGen/AMDGPU/maximumnum.bf16.ll
index bd2fa2a024fa66..3f8b47867a751a 100644
--- a/llvm/test/CodeGen/AMDGPU/maximumnum.bf16.ll
+++ b/llvm/test/CodeGen/AMDGPU/maximumnum.bf16.ll
@@ -152,7 +152,7 @@ define bfloat @v_maximumnum_bf16(bfloat %x, bfloat %y) #1 {
; GFX11-FAKE16-NEXT: v_cndmask_b32_e32 v1, v1, v0, vcc_lo
; GFX11-FAKE16-NEXT: v_cmp_eq_u16_e32 vcc_lo, 0, v0
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v2, 16, v1
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_cmp_eq_f32_e64 s0, 0, v2
; GFX11-FAKE16-NEXT: s_and_b32 vcc_lo, s0, vcc_lo
; GFX11-FAKE16-NEXT: v_cndmask_b32_e32 v0, v1, v0, vcc_lo
@@ -323,7 +323,7 @@ define bfloat @v_maximumnum_bf16_nnan(bfloat %x, bfloat %y) #1 {
; GFX11-FAKE16-NEXT: v_cndmask_b32_e32 v1, v1, v0, vcc_lo
; GFX11-FAKE16-NEXT: v_cmp_eq_u16_e32 vcc_lo, 0, v0
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v2, 16, v1
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_cmp_eq_f32_e64 s0, 0, v2
; GFX11-FAKE16-NEXT: s_and_b32 vcc_lo, s0, vcc_lo
; GFX11-FAKE16-NEXT: v_cndmask_b32_e32 v0, v1, v0, vcc_lo
@@ -565,24 +565,23 @@ define <2 x bfloat> @v_maximumnum_v2bf16(<2 x bfloat> %x, <2 x bfloat> %y) #1 {
; GFX11-TRUE16-NEXT: v_cmp_u_f32_e64 s1, v6, v6
; GFX11-TRUE16-NEXT: v_cmp_u_f32_e64 s2, v7, v7
; GFX11-TRUE16-NEXT: v_cndmask_b16 v2.l, v5.l, v3.l, vcc_lo
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-TRUE16-NEXT: v_cndmask_b16 v0.l, v0.l, v1.l, s0
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-TRUE16-NEXT: v_cndmask_b16 v3.l, v3.l, v2.l, s1
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_2) | instid1(VALU_DEP_4)
; GFX11-TRUE16-NEXT: v_cndmask_b16 v1.l, v1.l, v0.l, s2
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v4.l, v2.l
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v5.l, v0.l
-; GFX11-TRUE16-NEXT: v_mov_b16_e32 v6.l, v3.l
; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
+; GFX11-TRUE16-NEXT: v_mov_b16_e32 v6.l, v3.l
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v7.l, v1.l
-; GFX11-TRUE16-NEXT: v_lshlrev_b32_e32 v4, 16, v4
; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
+; GFX11-TRUE16-NEXT: v_lshlrev_b32_e32 v4, 16, v4
; GFX11-TRUE16-NEXT: v_lshlrev_b32_e32 v5, 16, v5
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
; GFX11-TRUE16-NEXT: v_lshlrev_b32_e32 v6, 16, v6
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-TRUE16-NEXT: v_lshlrev_b32_e32 v7, 16, v7
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-TRUE16-NEXT: v_cmp_gt_f32_e32 vcc_lo, v4, v6
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX11-TRUE16-NEXT: v_cmp_gt_f32_e64 s0, v5, v7
; GFX11-TRUE16-NEXT: v_cndmask_b16 v3.l, v3.l, v2.l, vcc_lo
; GFX11-TRUE16-NEXT: v_cndmask_b16 v1.l, v1.l, v0.l, s0
@@ -641,8 +640,8 @@ define <2 x bfloat> @v_maximumnum_v2bf16(<2 x bfloat> %x, <2 x bfloat> %y) #1 {
; GFX11-FAKE16-NEXT: v_cmp_eq_f32_e64 s2, 0, v5
; GFX11-FAKE16-NEXT: v_cndmask_b32_e32 v2, v3, v2, vcc_lo
; GFX11-FAKE16-NEXT: s_and_b32 vcc_lo, s2, s1
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_cndmask_b32_e32 v0, v1, v0, vcc_lo
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_perm_b32 v0, v2, v0, 0x5040100
; GFX11-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -688,17 +687,17 @@ define <2 x bfloat> @v_maximumnum_v2bf16(<2 x bfloat> %x, <2 x bfloat> %y) #1 {
; GFX12-TRUE16-NEXT: s_wait_alu depctr_va_vcc(0)
; GFX12-TRUE16-NEXT: v_cndmask_b16 v3.l, v3.l, v2.l, vcc_lo
; GFX12-TRUE16-NEXT: s_wait_alu depctr_va_sdst(0)
-; GFX12-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_3) | instid1(VALU_DEP_4)
; GFX12-TRUE16-NEXT: v_cndmask_b16 v1.l, v1.l, v0.l, s0
; GFX12-TRUE16-NEXT: v_cmp_eq_u16_e32 vcc_lo, 0, v2.l
; GFX12-TRUE16-NEXT: v_cmp_eq_u16_e64 s0, 0, v0.l
; GFX12-TRUE16-NEXT: v_mov_b16_e32 v4.l, v3.l
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX12-TRUE16-NEXT: v_mov_b16_e32 v5.l, v1.l
-; GFX12-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX12-TRUE16-NEXT: v_lshlrev_b32_e32 v4, 16, v4
-; GFX12-TRUE16-NEXT: v_lshlrev_b32_e32 v5, 16, v5
; GFX12-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX12-TRUE16-NEXT: v_lshlrev_b32_e32 v5, 16, v5
; GFX12-TRUE16-NEXT: v_cmp_eq_f32_e64 s1, 0, v4
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX12-TRUE16-NEXT: v_cmp_eq_f32_e64 s2, 0, v5
; GFX12-TRUE16-NEXT: s_and_b32 s1, s1, vcc_lo
; GFX12-TRUE16-NEXT: s_and_b32 s0, s2, s0
@@ -894,17 +893,16 @@ define <2 x bfloat> @v_maximumnum_v2bf16_nnan(<2 x bfloat> %x, <2 x bfloat> %y)
; GFX11-TRUE16-NEXT: v_cmp_gt_f32_e64 s0, v5, v4
; GFX11-TRUE16-NEXT: v_cndmask_b16 v2.l, v7.l, v6.l, vcc_lo
; GFX11-TRUE16-NEXT: v_cmp_eq_u16_e32 vcc_lo, 0, v0.l
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_1) | instid1(VALU_DEP_4)
; GFX11-TRUE16-NEXT: v_cndmask_b16 v1.l, v1.l, v0.l, s0
; GFX11-TRUE16-NEXT: v_cmp_eq_u16_e64 s0, 0, v6.l
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_3)
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v3.l, v2.l
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v4.l, v1.l
-; GFX11-TRUE16-NEXT: v_lshlrev_b32_e32 v3, 16, v3
; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX11-TRUE16-NEXT: v_lshlrev_b32_e32 v3, 16, v3
; GFX11-TRUE16-NEXT: v_lshlrev_b32_e32 v4, 16, v4
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-TRUE16-NEXT: v_cmp_eq_f32_e64 s1, 0, v3
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX11-TRUE16-NEXT: v_cmp_eq_f32_e64 s2, 0, v4
; GFX11-TRUE16-NEXT: s_and_b32 s0, s1, s0
; GFX11-TRUE16-NEXT: s_and_b32 s1, s2, vcc_lo
@@ -936,8 +934,8 @@ define <2 x bfloat> @v_maximumnum_v2bf16_nnan(<2 x bfloat> %x, <2 x bfloat> %y)
; GFX11-FAKE16-NEXT: v_cmp_eq_f32_e64 s2, 0, v4
; GFX11-FAKE16-NEXT: v_cndmask_b32_e32 v0, v1, v0, vcc_lo
; GFX11-FAKE16-NEXT: s_and_b32 vcc_lo, s2, s1
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_cndmask_b32_e32 v1, v2, v6, vcc_lo
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_perm_b32 v0, v1, v0, 0x5040100
; GFX11-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -1302,7 +1300,6 @@ define <3 x bfloat> @v_maximumnum_v3bf16(<3 x bfloat> %x, <3 x bfloat> %y) #1 {
; GFX11-TRUE16-NEXT: v_cmp_gt_f32_e64 s1, v11, v10
; GFX11-TRUE16-NEXT: v_cndmask_b16 v5.l, v5.l, v4.l, vcc_lo
; GFX11-TRUE16-NEXT: v_cmp_eq_u16_e32 vcc_lo, 0, v4.l
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_4)
; GFX11-TRUE16-NEXT: v_cndmask_b16 v3.l, v3.l, v1.l, s0
; GFX11-TRUE16-NEXT: v_cmp_eq_u16_e64 s0, 0, v1.l
; GFX11-TRUE16-NEXT: v_cndmask_b16 v2.l, v2.l, v0.l, s1
@@ -1370,19 +1367,20 @@ define <3 x bfloat> @v_maximumnum_v3bf16(<3 x bfloat> %x, <3 x bfloat> %y) #1 {
; GFX11-FAKE16-NEXT: v_cmp_gt_f32_e32 vcc_lo, v8, v7
; GFX11-FAKE16-NEXT: v_cndmask_b32_e32 v3, v3, v1, vcc_lo
; GFX11-FAKE16-NEXT: v_cmp_eq_f32_e32 vcc_lo, 0, v6
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_2) | instid1(VALU_DEP_2)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v6, 16, v3
; GFX11-FAKE16-NEXT: s_and_b32 vcc_lo, vcc_lo, s0
; GFX11-FAKE16-NEXT: v_dual_cndmask_b32 v4, v5, v4 :: v_dual_lshlrev_b32 v7, 16, v2
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-FAKE16-NEXT: v_cmp_eq_f32_e32 vcc_lo, 0, v6
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_cmp_eq_f32_e64 s1, 0, v7
; GFX11-FAKE16-NEXT: s_and_b32 s0, s1, s2
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v0, v2, v0, s0
; GFX11-FAKE16-NEXT: v_cmp_eq_u16_e64 s0, 0, v1
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX11-FAKE16-NEXT: v_perm_b32 v0, v4, v0, 0x5040100
; GFX11-FAKE16-NEXT: s_and_b32 vcc_lo, vcc_lo, s0
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_cndmask_b32_e32 v1, v3, v1, vcc_lo
; GFX11-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -1717,10 +1715,9 @@ define <3 x bfloat> @v_maximumnum_v3bf16_nnan(<3 x bfloat> %x, <3 x bfloat> %y)
; GFX11-TRUE16-NEXT: v_cmp_gt_f32_e64 s1, v9, v8
; GFX11-TRUE16-NEXT: v_cndmask_b16 v3.l, v3.l, v1.l, vcc_lo
; GFX11-TRUE16-NEXT: v_cmp_eq_u16_e32 vcc_lo, 0, v1.l
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(SKIP_1) | instid1(VALU_DEP_4)
; GFX11-TRUE16-NEXT: v_cndmask_b16 v4.l, v4.l, v5.l, s0
; GFX11-TRUE16-NEXT: v_cndmask_b16 v2.l, v2.l, v0.l, s1
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_4)
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v6.l, v3.l
; GFX11-TRUE16-NEXT: v_cmp_eq_u16_e64 s0, 0, v0.l
; GFX11-TRUE16-NEXT: v_cmp_eq_u16_e64 s1, 0, v5.l
@@ -1766,15 +1763,16 @@ define <3 x bfloat> @v_maximumnum_v3bf16_nnan(<3 x bfloat> %x, <3 x bfloat> %y)
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v6, 16, v5
; GFX11-FAKE16-NEXT: v_cndmask_b32_e32 v3, v3, v1, vcc_lo
; GFX11-FAKE16-NEXT: v_cmp_eq_f32_e32 vcc_lo, 0, v4
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_cmp_eq_f32_e64 s1, 0, v6
; GFX11-FAKE16-NEXT: s_and_b32 vcc_lo, vcc_lo, s0
; GFX11-FAKE16-NEXT: v_cndmask_b32_e32 v0, v2, v0, vcc_lo
; GFX11-FAKE16-NEXT: s_and_b32 vcc_lo, s1, s2
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX11-FAKE16-NEXT: v_dual_cndmask_b32 v2, v5, v9 :: v_dual_lshlrev_b32 v7, 16, v3
; GFX11-FAKE16-NEXT: v_cmp_eq_u16_e32 vcc_lo, 0, v1
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_cmp_eq_f32_e64 s3, 0, v7
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_perm_b32 v0, v2, v0, 0x5040100
; GFX11-FAKE16-NEXT: s_and_b32 vcc_lo, s3, vcc_lo
; GFX11-FAKE16-NEXT: v_cndmask_b32_e32 v1, v3, v1, vcc_lo
@@ -2332,28 +2330,27 @@ define <4 x bfloat> @v_maximumnum_v4bf16(<4 x bfloat> %x, <4 x bfloat> %y) #1 {
; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
; GFX11-FAKE16-NEXT: v_cmp_gt_f32_e64 s1, v13, v12
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v10, 16, v2
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_2) | instid1(VALU_DEP_4)
; GFX11-FAKE16-NEXT: v_cmp_gt_f32_e64 s0, v8, v9
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v3, v3, v1, s1
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_4)
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v7, v7, v6, s0
; GFX11-FAKE16-NEXT: v_cmp_gt_f32_e64 s0, v11, v10
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_2) | instid1(VALU_DEP_2)
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v8, 16, v7
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v2, v2, v0, s0
; GFX11-FAKE16-NEXT: v_cmp_eq_u16_e64 s0, 0, v4
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v9, 16, v2
; GFX11-FAKE16-NEXT: s_and_b32 vcc_lo, vcc_lo, s0
; GFX11-FAKE16-NEXT: v_cmp_eq_u16_e64 s0, 0, v6
; GFX11-FAKE16-NEXT: v_dual_cndmask_b32 v4, v5, v4 :: v_dual_lshlrev_b32 v5, 16, v3
; GFX11-FAKE16-NEXT: v_cmp_eq_f32_e32 vcc_lo, 0, v8
; GFX11-FAKE16-NEXT: v_cmp_eq_f32_e64 s1, 0, v9
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_cmp_eq_f32_e64 s3, 0, v5
; GFX11-FAKE16-NEXT: s_and_b32 vcc_lo, vcc_lo, s0
; GFX11-FAKE16-NEXT: v_cndmask_b32_e32 v5, v7, v6, vcc_lo
; GFX11-FAKE16-NEXT: s_and_b32 vcc_lo, s1, s2
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_cndmask_b32_e32 v0, v2, v0, vcc_lo
; GFX11-FAKE16-NEXT: s_and_b32 vcc_lo, s3, s4
; GFX11-FAKE16-NEXT: v_cndmask_b32_e32 v1, v3, v1, vcc_lo
@@ -2518,18 +2515,17 @@ define <4 x bfloat> @v_maximumnum_v4bf16(<4 x bfloat> %x, <4 x bfloat> %y) #1 {
; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_3)
; GFX12-FAKE16-NEXT: v_lshlrev_b32_e32 v10, 16, v2
; GFX12-FAKE16-NEXT: v_cmp_gt_f32_e64 s0, v8, v9
-; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_1) | instid1(VALU_DEP_2)
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_4) | instid1(VALU_DEP_3)
; GFX12-FAKE16-NEXT: v_cmp_gt_f32_e64 s1, v13, v12
; GFX12-FAKE16-NEXT: s_wait_alu depctr_va_sdst(0)
; GFX12-FAKE16-NEXT: v_cndmask_b32_e64 v7, v7, v6, s0
; GFX12-FAKE16-NEXT: v_cmp_gt_f32_e64 s0, v11, v10
-; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_3)
; GFX12-FAKE16-NEXT: v_cndmask_b32_e64 v3, v3, v1, s1
; GFX12-FAKE16-NEXT: v_lshlrev_b32_e32 v8, 16, v7
; GFX12-FAKE16-NEXT: s_wait_alu depctr_va_sdst(0)
-; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX12-FAKE16-NEXT: v_cndmask_b32_e64 v2, v2, v0, s0
; GFX12-FAKE16-NEXT: v_cmp_eq_u16_e64 s0, 0, v4
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX12-FAKE16-NEXT: v_lshlrev_b32_e32 v9, 16, v2
; GFX12-FAKE16-NEXT: s_and_b32 vcc_lo, vcc_lo, s0
; GFX12-FAKE16-NEXT: v_cmp_eq_u16_e64 s0, 0, v6
@@ -2833,12 +2829,11 @@ define <4 x bfloat> @v_maximumnum_v4bf16_nnan(<4 x bfloat> %x, <4 x bfloat> %y)
; GFX11-FAKE16-NEXT: v_cmp_gt_f32_e64 s0, v9, v8
; GFX11-FAKE16-NEXT: v_and_b32_e32 v7, 0xffff0000, v1
; GFX11-FAKE16-NEXT: v_cmp_eq_u16_e64 s2, 0, v13
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_2) | instid1(VALU_DEP_2)
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v2, v2, v0, s0
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v5, 16, v1
; GFX11-FAKE16-NEXT: v_cmp_gt_f32_e64 s0, v12, v11
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_4) | instid1(VALU_DEP_2)
; GFX11-FAKE16-NEXT: v_cmp_gt_f32_e32 vcc_lo, v5, v4
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_3) | instid1(VALU_DEP_2)
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v8, v14, v13, s0
; GFX11-FAKE16-NEXT: v_cmp_eq_u16_e64 s0, 0, v1
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v5, 16, v1
@@ -2848,10 +2843,10 @@ define <4 x bfloat> @v_maximumnum_v4bf16_nnan(<4 x bfloat> %x, <4 x bfloat> %y)
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v10, 16, v4
; GFX11-FAKE16-NEXT: v_cmp_eq_f32_e32 vcc_lo, 0, v10
; GFX11-FAKE16-NEXT: s_and_b32 vcc_lo, vcc_lo, s0
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_dual_cndmask_b32 v1, v4, v1 :: v_dual_and_b32 v6, 0xffff0000, v3
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v3, 16, v3
; GFX11-FAKE16-NEXT: v_cmp_eq_u16_e64 s0, 0, v0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_2) | instid1(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_cmp_gt_f32_e64 s1, v7, v6
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v6, 16, v2
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v7, 16, v8
@@ -2859,17 +2854,18 @@ define <4 x bfloat> @v_maximumnum_v4bf16_nnan(<4 x bfloat> %x, <4 x bfloat> %y)
; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_cmp_eq_f32_e32 vcc_lo, 0, v6
; GFX11-FAKE16-NEXT: v_cmp_eq_f32_e64 s1, 0, v7
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_3) | instid1(VALU_DEP_2)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v4, 16, v3
; GFX11-FAKE16-NEXT: s_and_b32 vcc_lo, vcc_lo, s0
; GFX11-FAKE16-NEXT: v_cndmask_b32_e32 v0, v2, v0, vcc_lo
; GFX11-FAKE16-NEXT: s_and_b32 vcc_lo, s1, s2
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_2) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_cmp_eq_f32_e64 s3, 0, v4
; GFX11-FAKE16-NEXT: v_cndmask_b32_e32 v2, v8, v13, vcc_lo
; GFX11-FAKE16-NEXT: s_and_b32 vcc_lo, s3, s4
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_perm_b32 v0, v2, v0, 0x5040100
; GFX11-FAKE16-NEXT: v_cndmask_b32_e32 v3, v3, v5, vcc_lo
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_perm_b32 v1, v3, v1, 0x5040100
; GFX11-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -2967,16 +2963,15 @@ define <4 x bfloat> @v_maximumnum_v4bf16_nnan(<4 x bfloat> %x, <4 x bfloat> %y)
; GFX12-FAKE16-NEXT: v_dual_cndmask_b32 v1, v4, v1 :: v_dual_and_b32 v6, 0xffff0000, v3
; GFX12-FAKE16-NEXT: v_lshrrev_b32_e32 v3, 16, v3
; GFX12-FAKE16-NEXT: v_cmp_eq_u16_e64 s0, 0, v0
-; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_3) | instid1(VALU_DEP_3)
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_4) | instid1(VALU_DEP_3)
; GFX12-FAKE16-NEXT: v_cmp_gt_f32_e64 s1, v7, v6
; GFX12-FAKE16-NEXT: v_lshlrev_b32_e32 v6, 16, v2
; GFX12-FAKE16-NEXT: v_lshlrev_b32_e32 v7, 16, v8
; GFX12-FAKE16-NEXT: s_wait_alu depctr_va_sdst(0)
; GFX12-FAKE16-NEXT: v_cndmask_b32_e64 v3, v3, v5, s1
-; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_3)
; GFX12-FAKE16-NEXT: v_cmp_eq_f32_e32 vcc_lo, 0, v6
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_3)
; GFX12-FAKE16-NEXT: v_cmp_eq_f32_e64 s1, 0, v7
-; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX12-FAKE16-NEXT: v_lshlrev_b32_e32 v4, 16, v3
; GFX12-FAKE16-NEXT: s_and_b32 vcc_lo, vcc_lo, s0
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
@@ -3633,7 +3628,7 @@ define <6 x bfloat> @v_maximumnum_v6bf16(<6 x bfloat> %x, <6 x bfloat> %y) #1 {
; GFX11-FAKE16-NEXT: v_cndmask_b32_e32 v9, v9, v8, vcc_lo
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v11, 16, v7
; GFX11-FAKE16-NEXT: v_cmp_gt_f32_e64 s1, v13, v14
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_4) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_cmp_eq_f32_e32 vcc_lo, 0, v11
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v11, v12, v10, s1
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v15, 16, v9
@@ -3652,40 +3647,37 @@ define <6 x bfloat> @v_maximumnum_v6bf16(<6 x bfloat> %x, <6 x bfloat> %y) #1 {
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v15, 16, v2
; GFX11-FAKE16-NEXT: v_cmp_eq_f32_e32 vcc_lo, 0, v7
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v7, 16, v0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(SKIP_2) | instid1(VALU_DEP_3)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_4)
; GFX11-FAKE16-NEXT: v_cmp_u_f32_e64 s0, v9, v9
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v9, 16, v4
; GFX11-FAKE16-NEXT: v_cmp_eq_u16_e64 s4, 0, v2
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v1, v1, v4, s0
; GFX11-FAKE16-NEXT: v_cmp_u_f32_e64 s0, v7, v7
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v7, 16, v5
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_2) | instid1(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v0, v0, v3, s0
; GFX11-FAKE16-NEXT: v_cmp_u_f32_e64 s0, v9, v9
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v9, 16, v1
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v13, 16, v0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_2) | instid1(VALU_DEP_2)
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v4, v4, v1, s0
; GFX11-FAKE16-NEXT: v_cmp_u_f32_e64 s0, v12, v12
; GFX11-FAKE16-NEXT: v_cmp_eq_u16_e64 s2, 0, v0
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v3, v3, v0, s0
; GFX11-FAKE16-NEXT: v_cmp_u_f32_e64 s0, v7, v7
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v7, 16, v4
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_1) | instid1(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v12, 16, v3
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v5, v5, v2, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-FAKE16-NEXT: v_cmp_gt_f32_e64 s0, v9, v7
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_2) | instid1(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v14, 16, v5
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v4, v4, v1, s0
; GFX11-FAKE16-NEXT: v_cmp_gt_f32_e64 s0, v13, v12
; GFX11-FAKE16-NEXT: v_cmp_gt_f32_e64 s1, v15, v14
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_3) | instid1(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v7, 16, v4
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v3, v3, v0, s0
; GFX11-FAKE16-NEXT: v_cmp_eq_u16_e64 s0, 0, v10
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v5, v5, v2, s1
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v9, 16, v3
; GFX11-FAKE16-NEXT: s_and_b32 vcc_lo, vcc_lo, s0
@@ -3696,6 +3688,7 @@ define <6 x bfloat> @v_maximumnum_v6bf16(<6 x bfloat> %x, <6 x bfloat> %y) #1 {
; GFX11-FAKE16-NEXT: v_cmp_eq_f32_e64 s1, 0, v9
; GFX11-FAKE16-NEXT: v_cmp_eq_f32_e64 s3, 0, v11
; GFX11-FAKE16-NEXT: s_and_b32 vcc_lo, vcc_lo, s0
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_cndmask_b32_e32 v1, v4, v1, vcc_lo
; GFX11-FAKE16-NEXT: s_and_b32 vcc_lo, s1, s2
; GFX11-FAKE16-NEXT: v_cndmask_b32_e32 v0, v3, v0, vcc_lo
@@ -3904,7 +3897,6 @@ define <6 x bfloat> @v_maximumnum_v6bf16(<6 x bfloat> %x, <6 x bfloat> %y) #1 {
; GFX12-FAKE16-NEXT: v_cmp_gt_f32_e64 s1, v13, v14
; GFX12-FAKE16-NEXT: v_cmp_eq_f32_e32 vcc_lo, 0, v11
; GFX12-FAKE16-NEXT: s_wait_alu depctr_va_sdst(0)
-; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX12-FAKE16-NEXT: v_cndmask_b32_e64 v11, v12, v10, s1
; GFX12-FAKE16-NEXT: v_lshlrev_b32_e32 v15, 16, v9
; GFX12-FAKE16-NEXT: v_lshlrev_b32_e32 v12, 16, v2
@@ -3923,7 +3915,7 @@ define <6 x bfloat> @v_maximumnum_v6bf16(<6 x bfloat> %x, <6 x bfloat> %y) #1 {
; GFX12-FAKE16-NEXT: v_lshlrev_b32_e32 v15, 16, v2
; GFX12-FAKE16-NEXT: v_cmp_eq_f32_e32 vcc_lo, 0, v7
; GFX12-FAKE16-NEXT: v_lshlrev_b32_e32 v7, 16, v0
-; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(SKIP_3) | instid1(VALU_DEP_3)
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_4)
; GFX12-FAKE16-NEXT: v_cmp_u_f32_e64 s0, v9, v9
; GFX12-FAKE16-NEXT: v_lshlrev_b32_e32 v9, 16, v4
; GFX12-FAKE16-NEXT: v_cmp_eq_u16_e64 s4, 0, v2
@@ -3932,13 +3924,12 @@ define <6 x bfloat> @v_maximumnum_v6bf16(<6 x bfloat> %x, <6 x bfloat> %y) #1 {
; GFX12-FAKE16-NEXT: v_cmp_u_f32_e64 s0, v7, v7
; GFX12-FAKE16-NEXT: v_lshlrev_b32_e32 v7, 16, v5
; GFX12-FAKE16-NEXT: s_wait_alu depctr_va_sdst(0)
-; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_2) | instid1(VALU_DEP_3)
; GFX12-FAKE16-NEXT: v_cndmask_b32_e64 v0, v0, v3, s0
; GFX12-FAKE16-NEXT: v_cmp_u_f32_e64 s0, v9, v9
; GFX12-FAKE16-NEXT: v_lshlrev_b32_e32 v9, 16, v1
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX12-FAKE16-NEXT: v_lshlrev_b32_e32 v13, 16, v0
; GFX12-FAKE16-NEXT: s_wait_alu depctr_va_sdst(0)
-; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_3) | instid1(VALU_DEP_2)
; GFX12-FAKE16-NEXT: v_cndmask_b32_e64 v4, v4, v1, s0
; GFX12-FAKE16-NEXT: v_cmp_u_f32_e64 s0, v12, v12
; GFX12-FAKE16-NEXT: v_cmp_eq_u16_e64 s2, 0, v0
@@ -3946,25 +3937,23 @@ define <6 x bfloat> @v_maximumnum_v6bf16(<6 x bfloat> %x, <6 x bfloat> %y) #1 {
; GFX12-FAKE16-NEXT: v_cndmask_b32_e64 v3, v3, v0, s0
; GFX12-FAKE16-NEXT: v_cmp_u_f32_e64 s0, v7, v7
; GFX12-FAKE16-NEXT: v_lshlrev_b32_e32 v7, 16, v4
-; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_1) | instid1(VALU_DEP_3)
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_2) | instid1(VALU_DEP_3)
; GFX12-FAKE16-NEXT: v_lshlrev_b32_e32 v12, 16, v3
; GFX12-FAKE16-NEXT: s_wait_alu depctr_va_sdst(0)
; GFX12-FAKE16-NEXT: v_cndmask_b32_e64 v5, v5, v2, s0
-; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX12-FAKE16-NEXT: v_cmp_gt_f32_e64 s0, v9, v7
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_3) | instid1(VALU_DEP_3)
; GFX12-FAKE16-NEXT: v_lshlrev_b32_e32 v14, 16, v5
; GFX12-FAKE16-NEXT: s_wait_alu depctr_va_sdst(0)
-; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_3)
; GFX12-FAKE16-NEXT: v_cndmask_b32_e64 v4, v4, v1, s0
; GFX12-FAKE16-NEXT: v_cmp_gt_f32_e64 s0, v13, v12
; GFX12-FAKE16-NEXT: v_cmp_gt_f32_e64 s1, v15, v14
-; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_1) | instid1(VALU_DEP_3)
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_4) | instid1(VALU_DEP_3)
; GFX12-FAKE16-NEXT: v_lshlrev_b32_e32 v7, 16, v4
; GFX12-FAKE16-NEXT: s_wait_alu depctr_va_sdst(0)
; GFX12-FAKE16-NEXT: v_cndmask_b32_e64 v3, v3, v0, s0
; GFX12-FAKE16-NEXT: v_cmp_eq_u16_e64 s0, 0, v10
; GFX12-FAKE16-NEXT: v_cndmask_b32_e64 v5, v5, v2, s1
-; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX12-FAKE16-NEXT: v_lshlrev_b32_e32 v9, 16, v3
; GFX12-FAKE16-NEXT: s_and_b32 vcc_lo, vcc_lo, s0
; GFX12-FAKE16-NEXT: v_cmp_eq_u16_e64 s0, 0, v1
@@ -4819,16 +4808,16 @@ define <8 x bfloat> @v_maximumnum_v8bf16(<8 x bfloat> %x, <8 x bfloat> %y) #1 {
; GFX11-FAKE16-NEXT: v_cmp_eq_u16_e32 vcc_lo, 0, v8
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v13, 16, v11
; GFX11-FAKE16-NEXT: s_and_b32 vcc_lo, s0, vcc_lo
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_4) | instid1(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_cndmask_b32_e32 v8, v9, v8, vcc_lo
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v9, 16, v1
; GFX11-FAKE16-NEXT: v_and_b32_e32 v12, 0xffff0000, v1
; GFX11-FAKE16-NEXT: v_cmp_eq_f32_e64 s1, 0, v13
; GFX11-FAKE16-NEXT: v_and_b32_e32 v13, 0xffff0000, v0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_1) | instid1(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_cmp_u_f32_e32 vcc_lo, v12, v12
; GFX11-FAKE16-NEXT: v_dual_cndmask_b32 v9, v9, v14 :: v_dual_and_b32 v12, 0xffff0000, v5
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-FAKE16-NEXT: v_cmp_u_f32_e32 vcc_lo, v13, v13
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_2) | instid1(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_cmp_u_f32_e64 s0, v12, v12
; GFX11-FAKE16-NEXT: v_cndmask_b32_e32 v13, v16, v15, vcc_lo
; GFX11-FAKE16-NEXT: v_and_b32_e32 v17, 0xffff0000, v4
@@ -4846,19 +4835,17 @@ define <8 x bfloat> @v_maximumnum_v8bf16(<8 x bfloat> %x, <8 x bfloat> %y) #1 {
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v19, 16, v14
; GFX11-FAKE16-NEXT: v_cmp_u_f32_e64 s0, v15, v15
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v15, 16, v7
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v3, v3, v7, s0
; GFX11-FAKE16-NEXT: v_cmp_gt_f32_e64 s0, v16, v17
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_4) | instid1(VALU_DEP_4)
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v17, 16, v3
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v12, v12, v9, s0
; GFX11-FAKE16-NEXT: v_cmp_gt_f32_e64 s0, v18, v19
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v14, v14, v13, s0
; GFX11-FAKE16-NEXT: v_cmp_u_f32_e64 s0, v15, v15
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v15, 16, v12
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_1) | instid1(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v11, 16, v14
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v7, v7, v3, s0
; GFX11-FAKE16-NEXT: v_cmp_eq_f32_e32 vcc_lo, 0, v15
; GFX11-FAKE16-NEXT: v_cmp_eq_u16_e64 s0, 0, v9
@@ -4866,6 +4853,7 @@ define <8 x bfloat> @v_maximumnum_v8bf16(<8 x bfloat> %x, <8 x bfloat> %y) #1 {
; GFX11-FAKE16-NEXT: v_cmp_eq_f32_e64 s1, 0, v11
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v11, 16, v2
; GFX11-FAKE16-NEXT: s_and_b32 vcc_lo, vcc_lo, s0
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_dual_cndmask_b32 v9, v12, v9 :: v_dual_lshlrev_b32 v16, 16, v7
; GFX11-FAKE16-NEXT: s_and_b32 vcc_lo, s1, s2
; GFX11-FAKE16-NEXT: v_cmp_eq_u16_e64 s0, 0, v3
@@ -4877,18 +4865,18 @@ define <8 x bfloat> @v_maximumnum_v8bf16(<8 x bfloat> %x, <8 x bfloat> %y) #1 {
; GFX11-FAKE16-NEXT: v_cndmask_b32_e32 v2, v2, v6, vcc_lo
; GFX11-FAKE16-NEXT: v_cmp_u_f32_e32 vcc_lo, v14, v14
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v14, 16, v4
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v7, v7, v3, s3
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(SKIP_4) | instid1(VALU_DEP_4)
; GFX11-FAKE16-NEXT: v_cmp_eq_u16_e64 s2, 0, v2
; GFX11-FAKE16-NEXT: v_cndmask_b32_e32 v1, v1, v5, vcc_lo
; GFX11-FAKE16-NEXT: v_cmp_u_f32_e32 vcc_lo, v11, v11
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v11, 16, v5
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v13, 16, v7
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(SKIP_3) | instid1(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v18, 16, v1
; GFX11-FAKE16-NEXT: v_cndmask_b32_e32 v0, v0, v4, vcc_lo
; GFX11-FAKE16-NEXT: v_cmp_u_f32_e32 vcc_lo, v15, v15
; GFX11-FAKE16-NEXT: v_cmp_eq_u16_e64 s6, 0, v1
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_cmp_eq_u16_e64 s4, 0, v0
; GFX11-FAKE16-NEXT: v_cndmask_b32_e32 v6, v6, v2, vcc_lo
; GFX11-FAKE16-NEXT: v_cmp_u_f32_e32 vcc_lo, v14, v14
@@ -5199,23 +5187,21 @@ define <8 x bfloat> @v_maximumnum_v8bf16(<8 x bfloat> %x, <8 x bfloat> %y) #1 {
; GFX12-FAKE16-NEXT: v_cmp_u_f32_e64 s0, v15, v15
; GFX12-FAKE16-NEXT: v_lshlrev_b32_e32 v15, 16, v7
; GFX12-FAKE16-NEXT: s_wait_alu depctr_va_sdst(0)
-; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX12-FAKE16-NEXT: v_cndmask_b32_e64 v3, v3, v7, s0
; GFX12-FAKE16-NEXT: v_lshlrev_b32_e32 v16, 16, v9
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_3)
; GFX12-FAKE16-NEXT: v_cmp_gt_f32_e64 s0, v16, v17
-; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX12-FAKE16-NEXT: v_lshlrev_b32_e32 v17, 16, v3
; GFX12-FAKE16-NEXT: s_wait_alu depctr_va_sdst(0)
; GFX12-FAKE16-NEXT: v_cndmask_b32_e64 v12, v12, v9, s0
; GFX12-FAKE16-NEXT: v_cmp_gt_f32_e64 s0, v18, v19
; GFX12-FAKE16-NEXT: s_wait_alu depctr_va_sdst(0)
-; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_2) | instid1(VALU_DEP_3)
; GFX12-FAKE16-NEXT: v_cndmask_b32_e64 v14, v14, v13, s0
; GFX12-FAKE16-NEXT: v_cmp_u_f32_e64 s0, v15, v15
; GFX12-FAKE16-NEXT: v_lshlrev_b32_e32 v15, 16, v12
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_2) | instid1(VALU_DEP_3)
; GFX12-FAKE16-NEXT: v_lshlrev_b32_e32 v11, 16, v14
; GFX12-FAKE16-NEXT: s_wait_alu depctr_va_sdst(0)
-; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_3)
; GFX12-FAKE16-NEXT: v_cndmask_b32_e64 v7, v7, v3, s0
; GFX12-FAKE16-NEXT: v_cmp_eq_f32_e32 vcc_lo, 0, v15
; GFX12-FAKE16-NEXT: v_cmp_eq_u16_e64 s0, 0, v9
@@ -6652,11 +6638,11 @@ define <16 x bfloat> @v_maximumnum_v16bf16(<16 x bfloat> %x, <16 x bfloat> %y) #
; GFX11-TRUE16-NEXT: v_cmp_gt_f32_e64 s0, v27, v29
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v26.l, v24.l
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v27.l, v25.l
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(SKIP_1) | instid1(VALU_DEP_4)
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v28.l, v18.l
; GFX11-TRUE16-NEXT: v_cndmask_b16 v23.l, v23.l, v19.l, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
; GFX11-TRUE16-NEXT: v_lshlrev_b32_e32 v26, 16, v26
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_4)
; GFX11-TRUE16-NEXT: v_lshlrev_b32_e32 v27, 16, v27
; GFX11-TRUE16-NEXT: s_and_b32 s0, s1, s2
; GFX11-TRUE16-NEXT: s_and_b32 s1, vcc_lo, s3
@@ -6892,20 +6878,19 @@ define <16 x bfloat> @v_maximumnum_v16bf16(<16 x bfloat> %x, <16 x bfloat> %y) #
; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v22, 16, v19
; GFX11-FAKE16-NEXT: v_cmp_u_f32_e64 s1, v20, v20
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_2) | instid1(VALU_DEP_3)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_4) | instid1(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_cmp_gt_f32_e64 s0, v21, v22
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v21, 16, v13
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v22, 16, v5
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v19, v19, v18, s0
; GFX11-FAKE16-NEXT: v_cmp_eq_u16_e64 s0, 0, v16
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_1) | instid1(VALU_DEP_4)
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v20, v22, v21, s1
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v22, 16, v12
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(SKIP_3) | instid1(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v23, 16, v19
; GFX11-FAKE16-NEXT: s_and_b32 vcc_lo, vcc_lo, s0
; GFX11-FAKE16-NEXT: v_cmp_u_f32_e64 s0, v24, v24
; GFX11-FAKE16-NEXT: v_cndmask_b32_e32 v16, v17, v16, vcc_lo
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_1) | instid1(VALU_DEP_4)
; GFX11-FAKE16-NEXT: v_cmp_eq_f32_e32 vcc_lo, 0, v23
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v23, 16, v4
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v17, v21, v20, s0
@@ -6913,10 +6898,10 @@ define <16 x bfloat> @v_maximumnum_v16bf16(<16 x bfloat> %x, <16 x bfloat> %y) #
; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v25, 16, v17
; GFX11-FAKE16-NEXT: v_cmp_u_f32_e64 s0, v21, v21
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_2) | instid1(VALU_DEP_2)
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v21, v23, v22, s0
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v24, 16, v20
; GFX11-FAKE16-NEXT: v_cmp_eq_u16_e64 s0, 0, v18
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX11-FAKE16-NEXT: v_cmp_gt_f32_e64 s1, v24, v25
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v24, 16, v11
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v25, 16, v3
@@ -6934,29 +6919,28 @@ define <16 x bfloat> @v_maximumnum_v16bf16(<16 x bfloat> %x, <16 x bfloat> %y) #
; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v27, 16, v22
; GFX11-FAKE16-NEXT: v_cmp_u_f32_e64 s1, v23, v23
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v23, v25, v24, s1
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_4)
; GFX11-FAKE16-NEXT: v_cndmask_b32_e32 v24, v24, v23, vcc_lo
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(SKIP_2) | instid1(VALU_DEP_4)
; GFX11-FAKE16-NEXT: v_cmp_gt_f32_e32 vcc_lo, v26, v27
; GFX11-FAKE16-NEXT: v_and_b32_e32 v27, 0xffff0000, v2
; GFX11-FAKE16-NEXT: v_cmp_eq_u16_e64 s2, 0, v23
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(SKIP_2) | instid1(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_dual_cndmask_b32 v22, v22, v21 :: v_dual_lshlrev_b32 v25, 16, v24
; GFX11-FAKE16-NEXT: v_cmp_eq_f32_e32 vcc_lo, 0, v19
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v19, 16, v23
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v26, 16, v22
; GFX11-FAKE16-NEXT: s_and_b32 vcc_lo, vcc_lo, s0
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX11-FAKE16-NEXT: v_cmp_gt_f32_e64 s1, v19, v25
; GFX11-FAKE16-NEXT: v_cndmask_b32_e32 v17, v17, v20, vcc_lo
; GFX11-FAKE16-NEXT: v_cmp_eq_u16_e64 s0, 0, v21
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_3) | instid1(VALU_DEP_4)
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v19, v24, v23, s1
; GFX11-FAKE16-NEXT: v_and_b32_e32 v24, 0xffff0000, v10
; GFX11-FAKE16-NEXT: v_cmp_u_f32_e64 s1, v27, v27
; GFX11-FAKE16-NEXT: v_and_b32_e32 v27, 0xffff0000, v1
-; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v20, 16, v19
; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
+; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v20, 16, v19
; GFX11-FAKE16-NEXT: v_cmp_u_f32_e32 vcc_lo, v24, v24
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v25, v29, v28, s1
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v29, 16, v1
@@ -6972,13 +6956,12 @@ define <16 x bfloat> @v_maximumnum_v16bf16(<16 x bfloat> %x, <16 x bfloat> %y) #
; GFX11-FAKE16-NEXT: v_cmp_eq_u16_e64 s0, 0, v25
; GFX11-FAKE16-NEXT: v_cndmask_b32_e32 v21, v22, v21, vcc_lo
; GFX11-FAKE16-NEXT: v_cmp_gt_f32_e64 s3, v20, v26
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_3) | instid1(VALU_DEP_4)
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v20, v24, v25, s3
; GFX11-FAKE16-NEXT: v_and_b32_e32 v24, 0xffff0000, v9
; GFX11-FAKE16-NEXT: v_cmp_u_f32_e64 s3, v27, v27
; GFX11-FAKE16-NEXT: v_and_b32_e32 v27, 0xffff0000, v0
-; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v22, 16, v20
; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
+; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v22, 16, v20
; GFX11-FAKE16-NEXT: v_cmp_u_f32_e32 vcc_lo, v24, v24
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v26, v29, v28, s3
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v29, 16, v0
@@ -6991,13 +6974,12 @@ define <16 x bfloat> @v_maximumnum_v16bf16(<16 x bfloat> %x, <16 x bfloat> %y) #
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v22, 16, v26
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v23, 16, v24
; GFX11-FAKE16-NEXT: s_and_b32 vcc_lo, vcc_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_4) | instid1(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_cmp_gt_f32_e64 s1, v22, v23
; GFX11-FAKE16-NEXT: v_dual_cndmask_b32 v20, v20, v25 :: v_dual_lshlrev_b32 v25, 16, v7
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v22, v24, v26, s1
; GFX11-FAKE16-NEXT: v_cmp_u_f32_e64 s1, v27, v27
; GFX11-FAKE16-NEXT: v_and_b32_e32 v24, 0xffff0000, v8
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v27, 16, v22
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v23, v29, v28, s1
; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_2)
@@ -7024,14 +7006,13 @@ define <16 x bfloat> @v_maximumnum_v16bf16(<16 x bfloat> %x, <16 x bfloat> %y) #
; GFX11-FAKE16-NEXT: v_cmp_u_f32_e64 s1, v28, v28
; GFX11-FAKE16-NEXT: v_cndmask_b32_e32 v22, v22, v26, vcc_lo
; GFX11-FAKE16-NEXT: v_cmp_eq_u16_e64 s0, 0, v23
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(SKIP_2) | instid1(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v28, 16, v24
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v6, v6, v14, s1
; GFX11-FAKE16-NEXT: v_cmp_gt_f32_e64 s1, v27, v25
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_cmp_eq_f32_e32 vcc_lo, 0, v28
-; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v27, 16, v6
; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3)
+; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v27, 16, v6
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v15, v15, v7, s1
; GFX11-FAKE16-NEXT: v_cmp_u_f32_e64 s1, v29, v29
; GFX11-FAKE16-NEXT: s_and_b32 vcc_lo, vcc_lo, s0
@@ -7042,18 +7023,17 @@ define <16 x bfloat> @v_maximumnum_v16bf16(<16 x bfloat> %x, <16 x bfloat> %y) #
; GFX11-FAKE16-NEXT: v_cmp_eq_f32_e64 s1, 0, v25
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v26, 16, v14
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v25, 16, v5
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_2)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_2) | instid1(VALU_DEP_4)
; GFX11-FAKE16-NEXT: v_cmp_gt_f32_e64 s3, v27, v26
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v26, 16, v13
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v14, v14, v6, s3
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_cmp_u_f32_e64 s3, v25, v25
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_1) | instid1(VALU_DEP_4)
; GFX11-FAKE16-NEXT: v_cmp_u_f32_e32 vcc_lo, v26, v26
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v25, 16, v3
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v24, 16, v14
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v5, v5, v13, s3
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_cndmask_b32_e32 v13, v13, v5, vcc_lo
; GFX11-FAKE16-NEXT: s_and_b32 vcc_lo, s1, s2
; GFX11-FAKE16-NEXT: v_dual_cndmask_b32 v7, v15, v7 :: v_dual_lshlrev_b32 v26, 16, v5
@@ -7064,17 +7044,15 @@ define <16 x bfloat> @v_maximumnum_v16bf16(<16 x bfloat> %x, <16 x bfloat> %y) #
; GFX11-FAKE16-NEXT: v_perm_b32 v7, v16, v7, 0x5040100
; GFX11-FAKE16-NEXT: v_cmp_u_f32_e64 s0, v15, v15
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v15, 16, v12
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_2) | instid1(VALU_DEP_2)
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v4, v4, v12, s0
; GFX11-FAKE16-NEXT: v_cmp_u_f32_e64 s0, v25, v25
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v25, 16, v11
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v3, v3, v11, s0
; GFX11-FAKE16-NEXT: v_cmp_gt_f32_e64 s0, v26, v24
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_2) | instid1(VALU_DEP_2)
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v27, 16, v3
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v13, v13, v5, s0
; GFX11-FAKE16-NEXT: v_cmp_u_f32_e64 s0, v15, v15
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v15, 16, v13
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v12, v12, v4, s0
; GFX11-FAKE16-NEXT: v_cmp_u_f32_e64 s0, v25, v25
@@ -7082,12 +7060,11 @@ define <16 x bfloat> @v_maximumnum_v16bf16(<16 x bfloat> %x, <16 x bfloat> %y) #
; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
; GFX11-FAKE16-NEXT: v_cmp_eq_f32_e64 s1, 0, v15
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v24, 16, v12
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(SKIP_2) | instid1(VALU_DEP_4)
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v11, v11, v3, s0
; GFX11-FAKE16-NEXT: v_cmp_eq_u16_e64 s0, 0, v6
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v15, 16, v2
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
; GFX11-FAKE16-NEXT: v_cmp_gt_f32_e64 s3, v25, v24
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_4)
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v26, 16, v11
; GFX11-FAKE16-NEXT: s_and_b32 vcc_lo, vcc_lo, s0
; GFX11-FAKE16-NEXT: v_cmp_eq_u16_e64 s0, 0, v4
@@ -7099,8 +7076,8 @@ define <16 x bfloat> @v_maximumnum_v16bf16(<16 x bfloat> %x, <16 x bfloat> %y) #
; GFX11-FAKE16-NEXT: v_dual_cndmask_b32 v5, v13, v5 :: v_dual_lshlrev_b32 v14, 16, v12
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v11, v11, v3, s3
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v15, 16, v8
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v2, v2, v10, s2
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_4)
; GFX11-FAKE16-NEXT: v_perm_b32 v5, v17, v5, 0x5040100
; GFX11-FAKE16-NEXT: v_cmp_eq_f32_e32 vcc_lo, 0, v14
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v13, 16, v11
@@ -7114,7 +7091,6 @@ define <16 x bfloat> @v_maximumnum_v16bf16(<16 x bfloat> %x, <16 x bfloat> %y) #
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v14, 16, v9
; GFX11-FAKE16-NEXT: v_cndmask_b32_e32 v4, v12, v4, vcc_lo
; GFX11-FAKE16-NEXT: v_perm_b32 v6, v18, v6, 0x5040100
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(SKIP_3) | instid1(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v1, v1, v9, s2
; GFX11-FAKE16-NEXT: v_cmp_u_f32_e64 s2, v13, v13
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v13, 16, v10
@@ -7122,31 +7098,28 @@ define <16 x bfloat> @v_maximumnum_v16bf16(<16 x bfloat> %x, <16 x bfloat> %y) #
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v0, v0, v8, s2
; GFX11-FAKE16-NEXT: v_cmp_u_f32_e64 s2, v14, v14
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v14, 16, v1
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v24, 16, v0
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v9, v9, v1, s2
; GFX11-FAKE16-NEXT: v_cmp_u_f32_e64 s2, v15, v15
; GFX11-FAKE16-NEXT: v_cmp_eq_u16_e64 s4, 0, v0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_2) | instid1(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v8, v8, v0, s2
; GFX11-FAKE16-NEXT: v_cmp_u_f32_e64 s2, v13, v13
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v13, 16, v9
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_1) | instid1(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v15, 16, v8
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v10, v10, v2, s2
; GFX11-FAKE16-NEXT: v_cmp_gt_f32_e64 s2, v14, v13
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_2) | instid1(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v25, 16, v10
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v9, v9, v1, s2
; GFX11-FAKE16-NEXT: v_cmp_gt_f32_e64 s2, v24, v15
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_cmp_gt_f32_e64 s3, v26, v25
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_3) | instid1(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v13, 16, v9
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_1) | instid1(VALU_DEP_4)
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v8, v8, v0, s2
; GFX11-FAKE16-NEXT: v_cmp_eq_u16_e64 s2, 0, v3
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v10, v10, v2, s3
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v14, 16, v8
; GFX11-FAKE16-NEXT: s_and_b32 s1, s1, s2
; GFX11-FAKE16-NEXT: v_cmp_eq_u16_e64 s2, 0, v1
@@ -7592,7 +7565,7 @@ define <16 x bfloat> @v_maximumnum_v16bf16(<16 x bfloat> %x, <16 x bfloat> %y) #
; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX12-FAKE16-NEXT: v_lshlrev_b32_e32 v22, 16, v19
; GFX12-FAKE16-NEXT: v_cmp_u_f32_e64 s1, v20, v20
-; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_3) | instid1(VALU_DEP_3)
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX12-FAKE16-NEXT: v_cmp_gt_f32_e64 s0, v21, v22
; GFX12-FAKE16-NEXT: v_lshrrev_b32_e32 v21, 16, v13
; GFX12-FAKE16-NEXT: v_lshrrev_b32_e32 v22, 16, v5
@@ -7616,10 +7589,10 @@ define <16 x bfloat> @v_maximumnum_v16bf16(<16 x bfloat> %x, <16 x bfloat> %y) #
; GFX12-FAKE16-NEXT: v_lshlrev_b32_e32 v25, 16, v17
; GFX12-FAKE16-NEXT: v_cmp_u_f32_e64 s0, v21, v21
; GFX12-FAKE16-NEXT: s_wait_alu depctr_va_sdst(0)
-; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_2) | instid1(VALU_DEP_2)
; GFX12-FAKE16-NEXT: v_cndmask_b32_e64 v21, v23, v22, s0
; GFX12-FAKE16-NEXT: v_lshlrev_b32_e32 v24, 16, v20
; GFX12-FAKE16-NEXT: v_cmp_eq_u16_e64 s0, 0, v18
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX12-FAKE16-NEXT: v_cmp_gt_f32_e64 s1, v24, v25
; GFX12-FAKE16-NEXT: v_lshrrev_b32_e32 v24, 16, v11
; GFX12-FAKE16-NEXT: v_lshrrev_b32_e32 v25, 16, v3
@@ -7640,9 +7613,9 @@ define <16 x bfloat> @v_maximumnum_v16bf16(<16 x bfloat> %x, <16 x bfloat> %y) #
; GFX12-FAKE16-NEXT: v_lshlrev_b32_e32 v27, 16, v22
; GFX12-FAKE16-NEXT: v_cmp_u_f32_e64 s1, v23, v23
; GFX12-FAKE16-NEXT: s_wait_alu depctr_va_sdst(0)
-; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX12-FAKE16-NEXT: v_cndmask_b32_e64 v23, v25, v24, s1
; GFX12-FAKE16-NEXT: s_wait_alu depctr_va_vcc(0)
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-FAKE16-NEXT: v_cndmask_b32_e32 v24, v24, v23, vcc_lo
; GFX12-FAKE16-NEXT: v_cmp_gt_f32_e32 vcc_lo, v26, v27
; GFX12-FAKE16-NEXT: v_and_b32_e32 v27, 0xffff0000, v2
@@ -7683,18 +7656,18 @@ define <16 x bfloat> @v_maximumnum_v16bf16(<16 x bfloat> %x, <16 x bfloat> %y) #
; GFX12-FAKE16-NEXT: v_cndmask_b32_e32 v21, v22, v21, vcc_lo
; GFX12-FAKE16-NEXT: v_cmp_gt_f32_e64 s3, v20, v26
; GFX12-FAKE16-NEXT: s_wait_alu depctr_va_sdst(0)
-; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_3) | instid1(VALU_DEP_4)
; GFX12-FAKE16-NEXT: v_cndmask_b32_e64 v20, v24, v25, s3
; GFX12-FAKE16-NEXT: v_and_b32_e32 v24, 0xffff0000, v9
; GFX12-FAKE16-NEXT: v_cmp_u_f32_e64 s3, v27, v27
; GFX12-FAKE16-NEXT: v_and_b32_e32 v27, 0xffff0000, v0
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
; GFX12-FAKE16-NEXT: v_lshlrev_b32_e32 v22, 16, v20
-; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(SKIP_4) | instid1(VALU_DEP_2)
; GFX12-FAKE16-NEXT: v_cmp_u_f32_e32 vcc_lo, v24, v24
; GFX12-FAKE16-NEXT: s_wait_alu depctr_va_sdst(0)
; GFX12-FAKE16-NEXT: v_cndmask_b32_e64 v26, v29, v28, s3
; GFX12-FAKE16-NEXT: v_lshrrev_b32_e32 v29, 16, v0
; GFX12-FAKE16-NEXT: s_wait_alu depctr_va_vcc(0)
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX12-FAKE16-NEXT: v_cndmask_b32_e32 v24, v28, v26, vcc_lo
; GFX12-FAKE16-NEXT: s_and_b32 vcc_lo, s1, s2
; GFX12-FAKE16-NEXT: v_lshrrev_b32_e32 v28, 16, v8
@@ -7704,7 +7677,7 @@ define <16 x bfloat> @v_maximumnum_v16bf16(<16 x bfloat> %x, <16 x bfloat> %y) #
; GFX12-FAKE16-NEXT: v_lshlrev_b32_e32 v22, 16, v26
; GFX12-FAKE16-NEXT: v_lshlrev_b32_e32 v23, 16, v24
; GFX12-FAKE16-NEXT: s_and_b32 vcc_lo, vcc_lo, s0
-; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_3) | instid1(VALU_DEP_2)
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-FAKE16-NEXT: v_cmp_gt_f32_e64 s1, v22, v23
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-FAKE16-NEXT: v_dual_cndmask_b32 v20, v20, v25 :: v_dual_lshlrev_b32 v25, 16, v7
@@ -7712,18 +7685,17 @@ define <16 x bfloat> @v_maximumnum_v16bf16(<16 x bfloat> %x, <16 x bfloat> %y) #
; GFX12-FAKE16-NEXT: v_cndmask_b32_e64 v22, v24, v26, s1
; GFX12-FAKE16-NEXT: v_cmp_u_f32_e64 s1, v27, v27
; GFX12-FAKE16-NEXT: v_and_b32_e32 v24, 0xffff0000, v8
-; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_1) | instid1(VALU_DEP_3)
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_2) | instid1(VALU_DEP_3)
; GFX12-FAKE16-NEXT: v_lshlrev_b32_e32 v27, 16, v22
; GFX12-FAKE16-NEXT: s_wait_alu depctr_va_sdst(0)
; GFX12-FAKE16-NEXT: v_cndmask_b32_e64 v23, v29, v28, s1
-; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX12-FAKE16-NEXT: v_cmp_u_f32_e32 vcc_lo, v24, v24
; GFX12-FAKE16-NEXT: s_wait_alu depctr_va_vcc(0)
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_3) | instid1(VALU_DEP_3)
; GFX12-FAKE16-NEXT: v_dual_cndmask_b32 v24, v28, v23 :: v_dual_lshlrev_b32 v29, 16, v14
; GFX12-FAKE16-NEXT: v_lshlrev_b32_e32 v28, 16, v15
; GFX12-FAKE16-NEXT: v_cmp_u_f32_e32 vcc_lo, v25, v25
; GFX12-FAKE16-NEXT: v_lshlrev_b32_e32 v25, 16, v23
-; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX12-FAKE16-NEXT: v_cmp_u_f32_e64 s0, v28, v28
; GFX12-FAKE16-NEXT: s_wait_alu depctr_va_vcc(0)
; GFX12-FAKE16-NEXT: v_cndmask_b32_e32 v7, v7, v15, vcc_lo
@@ -7751,7 +7723,6 @@ define <16 x bfloat> @v_maximumnum_v16bf16(<16 x bfloat> %x, <16 x bfloat> %y) #
; GFX12-FAKE16-NEXT: v_cmp_eq_f32_e32 vcc_lo, 0, v28
; GFX12-FAKE16-NEXT: v_lshlrev_b32_e32 v27, 16, v6
; GFX12-FAKE16-NEXT: s_wait_alu depctr_va_sdst(0)
-; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX12-FAKE16-NEXT: v_cndmask_b32_e64 v15, v15, v7, s1
; GFX12-FAKE16-NEXT: v_cmp_u_f32_e64 s1, v29, v29
; GFX12-FAKE16-NEXT: s_and_b32 vcc_lo, vcc_lo, s0
@@ -7763,20 +7734,19 @@ define <16 x bfloat> @v_maximumnum_v16bf16(<16 x bfloat> %x, <16 x bfloat> %y) #
; GFX12-FAKE16-NEXT: v_cmp_eq_f32_e64 s1, 0, v25
; GFX12-FAKE16-NEXT: v_lshlrev_b32_e32 v26, 16, v14
; GFX12-FAKE16-NEXT: v_lshlrev_b32_e32 v25, 16, v5
-; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_2) | instid1(VALU_DEP_2)
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_4) | instid1(VALU_DEP_3)
; GFX12-FAKE16-NEXT: v_cmp_gt_f32_e64 s3, v27, v26
; GFX12-FAKE16-NEXT: v_lshlrev_b32_e32 v26, 16, v13
; GFX12-FAKE16-NEXT: s_wait_alu depctr_va_sdst(0)
; GFX12-FAKE16-NEXT: v_cndmask_b32_e64 v14, v14, v6, s3
; GFX12-FAKE16-NEXT: v_cmp_u_f32_e64 s3, v25, v25
-; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_1) | instid1(VALU_DEP_4)
; GFX12-FAKE16-NEXT: v_cmp_u_f32_e32 vcc_lo, v26, v26
; GFX12-FAKE16-NEXT: v_lshlrev_b32_e32 v25, 16, v3
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(SKIP_3) | instid1(VALU_DEP_1)
; GFX12-FAKE16-NEXT: v_lshlrev_b32_e32 v24, 16, v14
; GFX12-FAKE16-NEXT: s_wait_alu depctr_va_sdst(0)
; GFX12-FAKE16-NEXT: v_cndmask_b32_e64 v5, v5, v13, s3
; GFX12-FAKE16-NEXT: s_wait_alu depctr_va_vcc(0)
-; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-FAKE16-NEXT: v_cndmask_b32_e32 v13, v13, v5, vcc_lo
; GFX12-FAKE16-NEXT: s_and_b32 vcc_lo, s1, s2
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
@@ -7789,19 +7759,17 @@ define <16 x bfloat> @v_maximumnum_v16bf16(<16 x bfloat> %x, <16 x bfloat> %y) #
; GFX12-FAKE16-NEXT: v_cmp_u_f32_e64 s0, v15, v15
; GFX12-FAKE16-NEXT: v_lshlrev_b32_e32 v15, 16, v12
; GFX12-FAKE16-NEXT: s_wait_alu depctr_va_sdst(0)
-; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_3) | instid1(VALU_DEP_2)
; GFX12-FAKE16-NEXT: v_cndmask_b32_e64 v4, v4, v12, s0
; GFX12-FAKE16-NEXT: v_cmp_u_f32_e64 s0, v25, v25
; GFX12-FAKE16-NEXT: v_lshlrev_b32_e32 v25, 16, v11
; GFX12-FAKE16-NEXT: s_wait_alu depctr_va_sdst(0)
; GFX12-FAKE16-NEXT: v_cndmask_b32_e64 v3, v3, v11, s0
; GFX12-FAKE16-NEXT: v_cmp_gt_f32_e64 s0, v26, v24
-; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_2)
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_3) | instid1(VALU_DEP_2)
; GFX12-FAKE16-NEXT: v_lshlrev_b32_e32 v27, 16, v3
; GFX12-FAKE16-NEXT: s_wait_alu depctr_va_sdst(0)
; GFX12-FAKE16-NEXT: v_cndmask_b32_e64 v13, v13, v5, s0
; GFX12-FAKE16-NEXT: v_cmp_u_f32_e64 s0, v15, v15
-; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX12-FAKE16-NEXT: v_lshlrev_b32_e32 v15, 16, v13
; GFX12-FAKE16-NEXT: s_wait_alu depctr_va_sdst(0)
; GFX12-FAKE16-NEXT: v_cndmask_b32_e64 v12, v12, v4, s0
@@ -7851,13 +7819,12 @@ define <16 x bfloat> @v_maximumnum_v16bf16(<16 x bfloat> %x, <16 x bfloat> %y) #
; GFX12-FAKE16-NEXT: v_lshlrev_b32_e32 v13, 16, v10
; GFX12-FAKE16-NEXT: v_perm_b32 v4, v21, v4, 0x5040100
; GFX12-FAKE16-NEXT: s_wait_alu depctr_va_sdst(0)
-; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_2) | instid1(VALU_DEP_3)
; GFX12-FAKE16-NEXT: v_cndmask_b32_e64 v0, v0, v8, s2
; GFX12-FAKE16-NEXT: v_cmp_u_f32_e64 s2, v14, v14
; GFX12-FAKE16-NEXT: v_lshlrev_b32_e32 v14, 16, v1
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX12-FAKE16-NEXT: v_lshlrev_b32_e32 v24, 16, v0
; GFX12-FAKE16-NEXT: s_wait_alu depctr_va_sdst(0)
-; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_3) | instid1(VALU_DEP_2)
; GFX12-FAKE16-NEXT: v_cndmask_b32_e64 v9, v9, v1, s2
; GFX12-FAKE16-NEXT: v_cmp_u_f32_e64 s2, v15, v15
; GFX12-FAKE16-NEXT: v_cmp_eq_u16_e64 s4, 0, v0
@@ -7865,25 +7832,23 @@ define <16 x bfloat> @v_maximumnum_v16bf16(<16 x bfloat> %x, <16 x bfloat> %y) #
; GFX12-FAKE16-NEXT: v_cndmask_b32_e64 v8, v8, v0, s2
; GFX12-FAKE16-NEXT: v_cmp_u_f32_e64 s2, v13, v13
; GFX12-FAKE16-NEXT: v_lshlrev_b32_e32 v13, 16, v9
-; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_1) | instid1(VALU_DEP_3)
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_2) | instid1(VALU_DEP_3)
; GFX12-FAKE16-NEXT: v_lshlrev_b32_e32 v15, 16, v8
; GFX12-FAKE16-NEXT: s_wait_alu depctr_va_sdst(0)
; GFX12-FAKE16-NEXT: v_cndmask_b32_e64 v10, v10, v2, s2
-; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX12-FAKE16-NEXT: v_cmp_gt_f32_e64 s2, v14, v13
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_3) | instid1(VALU_DEP_3)
; GFX12-FAKE16-NEXT: v_lshlrev_b32_e32 v25, 16, v10
; GFX12-FAKE16-NEXT: s_wait_alu depctr_va_sdst(0)
-; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_3)
; GFX12-FAKE16-NEXT: v_cndmask_b32_e64 v9, v9, v1, s2
; GFX12-FAKE16-NEXT: v_cmp_gt_f32_e64 s2, v24, v15
; GFX12-FAKE16-NEXT: v_cmp_gt_f32_e64 s3, v26, v25
-; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_1) | instid1(VALU_DEP_3)
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_4) | instid1(VALU_DEP_3)
; GFX12-FAKE16-NEXT: v_lshlrev_b32_e32 v13, 16, v9
; GFX12-FAKE16-NEXT: s_wait_alu depctr_va_sdst(0)
; GFX12-FAKE16-NEXT: v_cndmask_b32_e64 v8, v8, v0, s2
; GFX12-FAKE16-NEXT: v_cmp_eq_u16_e64 s2, 0, v3
; GFX12-FAKE16-NEXT: v_cndmask_b32_e64 v10, v10, v2, s3
-; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX12-FAKE16-NEXT: v_lshlrev_b32_e32 v14, 16, v8
; GFX12-FAKE16-NEXT: s_and_b32 s1, s1, s2
; GFX12-FAKE16-NEXT: v_cmp_eq_u16_e64 s2, 0, v1
@@ -10701,11 +10666,10 @@ define <32 x bfloat> @v_maximumnum_v32bf16(<32 x bfloat> %x, <32 x bfloat> %y) #
; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v35.l, v31.l
; GFX11-TRUE16-NEXT: v_cmp_gt_f32_e64 s0, v96, v97
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_3)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_2) | instid1(VALU_DEP_3)
; GFX11-TRUE16-NEXT: v_lshlrev_b32_e32 v39, 16, v35
; GFX11-TRUE16-NEXT: v_cndmask_b16 v35.h, v80.l, v50.l, s8
; GFX11-TRUE16-NEXT: v_cndmask_b16 v33.l, v87.l, v32.l, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX11-TRUE16-NEXT: v_cmp_gt_f32_e32 vcc_lo, v36, v39
; GFX11-TRUE16-NEXT: v_cmp_eq_u16_e64 s0, 0, v32.l
; GFX11-TRUE16-NEXT: v_cndmask_b16 v36.h, v81.l, v51.l, s9
@@ -11296,13 +11260,14 @@ define <32 x bfloat> @v_maximumnum_v32bf16(<32 x bfloat> %x, <32 x bfloat> %y) #
; GFX11-FAKE16-NEXT: v_and_b32_e32 v66, 0xffff0000, v31
; GFX11-FAKE16-NEXT: v_cndmask_b32_e32 v32, v33, v65, vcc_lo
; GFX11-FAKE16-NEXT: s_and_b32 vcc_lo, s24, s8
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_cndmask_b32_e32 v33, v54, v112, vcc_lo
; GFX11-FAKE16-NEXT: v_cmp_u_f32_e32 vcc_lo, v50, v50
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_3) | instid1(VALU_DEP_2)
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v67, 16, v32
; GFX11-FAKE16-NEXT: v_cmp_eq_u16_e64 s1, 0, v32
; GFX11-FAKE16-NEXT: v_cndmask_b32_e32 v31, v31, v15, vcc_lo
; GFX11-FAKE16-NEXT: v_cmp_u_f32_e32 vcc_lo, v66, v66
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v66, 16, v31
; GFX11-FAKE16-NEXT: v_dual_cndmask_b32 v50, v65, v32 :: v_dual_lshlrev_b32 v65, 16, v15
; GFX11-FAKE16-NEXT: s_and_b32 vcc_lo, s25, s9
@@ -11314,12 +11279,13 @@ define <32 x bfloat> @v_maximumnum_v32bf16(<32 x bfloat> %x, <32 x bfloat> %y) #
; GFX11-FAKE16-NEXT: v_cmp_gt_f32_e32 vcc_lo, v65, v66
; GFX11-FAKE16-NEXT: v_cndmask_b32_e32 v31, v31, v15, vcc_lo
; GFX11-FAKE16-NEXT: v_cmp_gt_f32_e32 vcc_lo, v67, v68
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_4) | instid1(VALU_DEP_2)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v66, 16, v31
; GFX11-FAKE16-NEXT: v_cndmask_b32_e32 v50, v50, v32, vcc_lo
; GFX11-FAKE16-NEXT: s_and_b32 vcc_lo, s27, s11
; GFX11-FAKE16-NEXT: v_cndmask_b32_e32 v65, v83, v128, vcc_lo
; GFX11-FAKE16-NEXT: s_and_b32 vcc_lo, s28, s12
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v67, 16, v50
; GFX11-FAKE16-NEXT: v_cndmask_b32_e32 v49, v49, v130, vcc_lo
; GFX11-FAKE16-NEXT: s_and_b32 vcc_lo, s29, s13
@@ -11361,7 +11327,7 @@ define <32 x bfloat> @v_maximumnum_v32bf16(<32 x bfloat> %x, <32 x bfloat> %y) #
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v29, 16, v9
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v32, 16, v26
; GFX11-FAKE16-NEXT: s_and_b32 vcc_lo, vcc_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_3) | instid1(VALU_DEP_4)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX11-FAKE16-NEXT: v_cmp_u_f32_e64 s1, v29, v29
; GFX11-FAKE16-NEXT: v_cndmask_b32_e32 v12, v28, v12, vcc_lo
; GFX11-FAKE16-NEXT: v_cmp_eq_u16_e64 s0, 0, v10
@@ -11369,19 +11335,18 @@ define <32 x bfloat> @v_maximumnum_v32bf16(<32 x bfloat> %x, <32 x bfloat> %y) #
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v9, v9, v25, s1
; GFX11-FAKE16-NEXT: v_cmp_eq_f32_e64 s1, 0, v31
; GFX11-FAKE16-NEXT: v_perm_b32 v12, v39, v12, 0x5040100
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(SKIP_2) | instid1(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v26, v26, v10, s2
; GFX11-FAKE16-NEXT: v_cmp_u_f32_e64 s2, v50, v50
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v31, 16, v9
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_2) | instid1(VALU_DEP_2)
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v28, 16, v26
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v25, v25, v9, s2
; GFX11-FAKE16-NEXT: v_cmp_eq_u16_e64 s2, 0, v11
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v29, 16, v25
; GFX11-FAKE16-NEXT: s_and_b32 vcc_lo, s1, s2
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_cndmask_b32_e32 v11, v27, v11, vcc_lo
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v27, 16, v8
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_3) | instid1(VALU_DEP_4)
; GFX11-FAKE16-NEXT: v_cmp_gt_f32_e64 s1, v31, v29
; GFX11-FAKE16-NEXT: v_cmp_eq_f32_e32 vcc_lo, 0, v28
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v29, 16, v22
@@ -11390,6 +11355,7 @@ define <32 x bfloat> @v_maximumnum_v32bf16(<32 x bfloat> %x, <32 x bfloat> %y) #
; GFX11-FAKE16-NEXT: v_cmp_u_f32_e64 s1, v27, v27
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v27, 16, v24
; GFX11-FAKE16-NEXT: s_and_b32 vcc_lo, vcc_lo, s0
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_cndmask_b32_e32 v10, v26, v10, vcc_lo
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v8, v8, v24, s1
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v26, 16, v7
@@ -11417,15 +11383,14 @@ define <32 x bfloat> @v_maximumnum_v32bf16(<32 x bfloat> %x, <32 x bfloat> %y) #
; GFX11-FAKE16-NEXT: v_cmp_u_f32_e64 s1, v27, v27
; GFX11-FAKE16-NEXT: v_cndmask_b32_e32 v9, v25, v9, vcc_lo
; GFX11-FAKE16-NEXT: v_cmp_eq_u16_e64 s0, 0, v8
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(SKIP_3) | instid1(VALU_DEP_4)
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v27, 16, v24
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v6, v6, v22, s1
; GFX11-FAKE16-NEXT: v_cmp_gt_f32_e64 s1, v28, v26
; GFX11-FAKE16-NEXT: v_perm_b32 v9, v52, v9, 0x5040100
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
; GFX11-FAKE16-NEXT: v_cmp_eq_f32_e32 vcc_lo, 0, v27
-; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v27, 16, v6
; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_4)
+; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v27, 16, v6
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v23, v23, v7, s1
; GFX11-FAKE16-NEXT: v_cmp_u_f32_e64 s1, v29, v29
; GFX11-FAKE16-NEXT: s_and_b32 vcc_lo, vcc_lo, s0
@@ -11440,15 +11405,14 @@ define <32 x bfloat> @v_maximumnum_v32bf16(<32 x bfloat> %x, <32 x bfloat> %y) #
; GFX11-FAKE16-NEXT: v_perm_b32 v8, v53, v8, 0x5040100
; GFX11-FAKE16-NEXT: v_cmp_gt_f32_e64 s3, v27, v26
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v26, 16, v21
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v22, v22, v6, s3
; GFX11-FAKE16-NEXT: v_cmp_u_f32_e64 s3, v25, v25
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_1) | instid1(VALU_DEP_4)
; GFX11-FAKE16-NEXT: v_cmp_u_f32_e32 vcc_lo, v26, v26
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v25, 16, v3
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v24, 16, v22
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v5, v5, v21, s3
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_cndmask_b32_e32 v21, v21, v5, vcc_lo
; GFX11-FAKE16-NEXT: s_and_b32 vcc_lo, s1, s2
; GFX11-FAKE16-NEXT: v_dual_cndmask_b32 v7, v23, v7 :: v_dual_lshlrev_b32 v26, 16, v5
@@ -11459,17 +11423,15 @@ define <32 x bfloat> @v_maximumnum_v32bf16(<32 x bfloat> %x, <32 x bfloat> %y) #
; GFX11-FAKE16-NEXT: v_perm_b32 v7, v55, v7, 0x5040100
; GFX11-FAKE16-NEXT: v_cmp_u_f32_e64 s0, v23, v23
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v23, 16, v20
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_2) | instid1(VALU_DEP_2)
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v4, v4, v20, s0
; GFX11-FAKE16-NEXT: v_cmp_u_f32_e64 s0, v25, v25
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v25, 16, v19
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v3, v3, v19, s0
; GFX11-FAKE16-NEXT: v_cmp_gt_f32_e64 s0, v26, v24
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_2) | instid1(VALU_DEP_2)
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v27, 16, v3
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v21, v21, v5, s0
; GFX11-FAKE16-NEXT: v_cmp_u_f32_e64 s0, v23, v23
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v23, 16, v21
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v20, v20, v4, s0
; GFX11-FAKE16-NEXT: v_cmp_u_f32_e64 s0, v25, v25
@@ -11477,12 +11439,11 @@ define <32 x bfloat> @v_maximumnum_v32bf16(<32 x bfloat> %x, <32 x bfloat> %y) #
; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
; GFX11-FAKE16-NEXT: v_cmp_eq_f32_e64 s1, 0, v23
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v24, 16, v20
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(SKIP_2) | instid1(VALU_DEP_4)
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v19, v19, v3, s0
; GFX11-FAKE16-NEXT: v_cmp_eq_u16_e64 s0, 0, v6
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v23, 16, v2
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
; GFX11-FAKE16-NEXT: v_cmp_gt_f32_e64 s3, v25, v24
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_4)
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v26, 16, v19
; GFX11-FAKE16-NEXT: s_and_b32 vcc_lo, vcc_lo, s0
; GFX11-FAKE16-NEXT: v_cmp_eq_u16_e64 s0, 0, v4
@@ -11494,8 +11455,8 @@ define <32 x bfloat> @v_maximumnum_v32bf16(<32 x bfloat> %x, <32 x bfloat> %y) #
; GFX11-FAKE16-NEXT: v_dual_cndmask_b32 v5, v21, v5 :: v_dual_lshlrev_b32 v22, 16, v20
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v19, v19, v3, s3
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v23, 16, v16
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v2, v2, v18, s2
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_4)
; GFX11-FAKE16-NEXT: v_perm_b32 v5, v33, v5, 0x5040100
; GFX11-FAKE16-NEXT: v_cmp_eq_f32_e32 vcc_lo, 0, v22
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v21, 16, v19
@@ -11509,7 +11470,6 @@ define <32 x bfloat> @v_maximumnum_v32bf16(<32 x bfloat> %x, <32 x bfloat> %y) #
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v22, 16, v17
; GFX11-FAKE16-NEXT: v_cndmask_b32_e32 v4, v20, v4, vcc_lo
; GFX11-FAKE16-NEXT: v_perm_b32 v6, v64, v6, 0x5040100
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(SKIP_3) | instid1(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v1, v1, v17, s2
; GFX11-FAKE16-NEXT: v_cmp_u_f32_e64 s2, v21, v21
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v21, 16, v18
@@ -11517,31 +11477,28 @@ define <32 x bfloat> @v_maximumnum_v32bf16(<32 x bfloat> %x, <32 x bfloat> %y) #
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v0, v0, v16, s2
; GFX11-FAKE16-NEXT: v_cmp_u_f32_e64 s2, v22, v22
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v22, 16, v1
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v24, 16, v0
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v17, v17, v1, s2
; GFX11-FAKE16-NEXT: v_cmp_u_f32_e64 s2, v23, v23
; GFX11-FAKE16-NEXT: v_cmp_eq_u16_e64 s4, 0, v0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_2) | instid1(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v16, v16, v0, s2
; GFX11-FAKE16-NEXT: v_cmp_u_f32_e64 s2, v21, v21
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v21, 16, v17
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_1) | instid1(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v23, 16, v16
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v18, v18, v2, s2
; GFX11-FAKE16-NEXT: v_cmp_gt_f32_e64 s2, v22, v21
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_2) | instid1(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v25, 16, v18
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v17, v17, v1, s2
; GFX11-FAKE16-NEXT: v_cmp_gt_f32_e64 s2, v24, v23
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_cmp_gt_f32_e64 s3, v26, v25
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_3) | instid1(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v21, 16, v17
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_1) | instid1(VALU_DEP_4)
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v16, v16, v0, s2
; GFX11-FAKE16-NEXT: v_cmp_eq_u16_e64 s2, 0, v3
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v18, v18, v2, s3
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v22, 16, v16
; GFX11-FAKE16-NEXT: s_and_b32 s1, s1, s2
; GFX11-FAKE16-NEXT: v_cmp_eq_u16_e64 s2, 0, v1
@@ -11930,8 +11887,8 @@ define <32 x bfloat> @v_maximumnum_v32bf16(<32 x bfloat> %x, <32 x bfloat> %y) #
; GFX12-TRUE16-NEXT: v_lshlrev_b32_e32 v39, 16, v35
; GFX12-TRUE16-NEXT: v_cndmask_b16 v35.h, v80.l, v50.l, s8
; GFX12-TRUE16-NEXT: s_wait_alu depctr_va_sdst(0)
-; GFX12-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_3)
; GFX12-TRUE16-NEXT: v_cndmask_b16 v33.l, v87.l, v32.l, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX12-TRUE16-NEXT: v_cmp_gt_f32_e32 vcc_lo, v36, v39
; GFX12-TRUE16-NEXT: v_cmp_eq_u16_e64 s0, 0, v32.l
; GFX12-TRUE16-NEXT: v_cndmask_b16 v36.h, v81.l, v51.l, s9
@@ -12705,12 +12662,11 @@ define <32 x bfloat> @v_maximumnum_v32bf16(<32 x bfloat> %x, <32 x bfloat> %y) #
; GFX12-FAKE16-NEXT: v_cndmask_b32_e64 v26, v26, v10, s2
; GFX12-FAKE16-NEXT: v_cmp_u_f32_e64 s2, v50, v50
; GFX12-FAKE16-NEXT: v_lshlrev_b32_e32 v31, 16, v9
-; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_1) | instid1(VALU_DEP_3)
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_3) | instid1(VALU_DEP_2)
; GFX12-FAKE16-NEXT: v_lshlrev_b32_e32 v28, 16, v26
; GFX12-FAKE16-NEXT: s_wait_alu depctr_va_sdst(0)
; GFX12-FAKE16-NEXT: v_cndmask_b32_e64 v25, v25, v9, s2
; GFX12-FAKE16-NEXT: v_cmp_eq_u16_e64 s2, 0, v11
-; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX12-FAKE16-NEXT: v_lshlrev_b32_e32 v29, 16, v25
; GFX12-FAKE16-NEXT: s_and_b32 vcc_lo, s1, s2
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
@@ -12781,16 +12737,16 @@ define <32 x bfloat> @v_maximumnum_v32bf16(<32 x bfloat> %x, <32 x bfloat> %y) #
; GFX12-FAKE16-NEXT: v_cmp_gt_f32_e64 s3, v27, v26
; GFX12-FAKE16-NEXT: v_lshlrev_b32_e32 v26, 16, v21
; GFX12-FAKE16-NEXT: s_wait_alu depctr_va_sdst(0)
-; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_3)
; GFX12-FAKE16-NEXT: v_cndmask_b32_e64 v22, v22, v6, s3
; GFX12-FAKE16-NEXT: v_cmp_u_f32_e64 s3, v25, v25
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_1) | instid1(VALU_DEP_4)
; GFX12-FAKE16-NEXT: v_cmp_u_f32_e32 vcc_lo, v26, v26
; GFX12-FAKE16-NEXT: v_lshlrev_b32_e32 v25, 16, v3
-; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(SKIP_3) | instid1(VALU_DEP_1)
; GFX12-FAKE16-NEXT: v_lshlrev_b32_e32 v24, 16, v22
; GFX12-FAKE16-NEXT: s_wait_alu depctr_va_sdst(0)
; GFX12-FAKE16-NEXT: v_cndmask_b32_e64 v5, v5, v21, s3
; GFX12-FAKE16-NEXT: s_wait_alu depctr_va_vcc(0)
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-FAKE16-NEXT: v_cndmask_b32_e32 v21, v21, v5, vcc_lo
; GFX12-FAKE16-NEXT: s_and_b32 vcc_lo, s1, s2
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
@@ -12803,19 +12759,17 @@ define <32 x bfloat> @v_maximumnum_v32bf16(<32 x bfloat> %x, <32 x bfloat> %y) #
; GFX12-FAKE16-NEXT: v_cmp_u_f32_e64 s0, v23, v23
; GFX12-FAKE16-NEXT: v_lshlrev_b32_e32 v23, 16, v20
; GFX12-FAKE16-NEXT: s_wait_alu depctr_va_sdst(0)
-; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_3) | instid1(VALU_DEP_2)
; GFX12-FAKE16-NEXT: v_cndmask_b32_e64 v4, v4, v20, s0
; GFX12-FAKE16-NEXT: v_cmp_u_f32_e64 s0, v25, v25
; GFX12-FAKE16-NEXT: v_lshlrev_b32_e32 v25, 16, v19
; GFX12-FAKE16-NEXT: s_wait_alu depctr_va_sdst(0)
; GFX12-FAKE16-NEXT: v_cndmask_b32_e64 v3, v3, v19, s0
; GFX12-FAKE16-NEXT: v_cmp_gt_f32_e64 s0, v26, v24
-; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_2)
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_3) | instid1(VALU_DEP_2)
; GFX12-FAKE16-NEXT: v_lshlrev_b32_e32 v27, 16, v3
; GFX12-FAKE16-NEXT: s_wait_alu depctr_va_sdst(0)
; GFX12-FAKE16-NEXT: v_cndmask_b32_e64 v21, v21, v5, s0
; GFX12-FAKE16-NEXT: v_cmp_u_f32_e64 s0, v23, v23
-; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX12-FAKE16-NEXT: v_lshlrev_b32_e32 v23, 16, v21
; GFX12-FAKE16-NEXT: s_wait_alu depctr_va_sdst(0)
; GFX12-FAKE16-NEXT: v_cndmask_b32_e64 v20, v20, v4, s0
@@ -12865,13 +12819,12 @@ define <32 x bfloat> @v_maximumnum_v32bf16(<32 x bfloat> %x, <32 x bfloat> %y) #
; GFX12-FAKE16-NEXT: v_lshlrev_b32_e32 v21, 16, v18
; GFX12-FAKE16-NEXT: v_perm_b32 v4, v54, v4, 0x5040100
; GFX12-FAKE16-NEXT: s_wait_alu depctr_va_sdst(0)
-; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_2) | instid1(VALU_DEP_3)
; GFX12-FAKE16-NEXT: v_cndmask_b32_e64 v0, v0, v16, s2
; GFX12-FAKE16-NEXT: v_cmp_u_f32_e64 s2, v22, v22
; GFX12-FAKE16-NEXT: v_lshlrev_b32_e32 v22, 16, v1
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX12-FAKE16-NEXT: v_lshlrev_b32_e32 v24, 16, v0
; GFX12-FAKE16-NEXT: s_wait_alu depctr_va_sdst(0)
-; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_3) | instid1(VALU_DEP_2)
; GFX12-FAKE16-NEXT: v_cndmask_b32_e64 v17, v17, v1, s2
; GFX12-FAKE16-NEXT: v_cmp_u_f32_e64 s2, v23, v23
; GFX12-FAKE16-NEXT: v_cmp_eq_u16_e64 s4, 0, v0
@@ -12879,25 +12832,23 @@ define <32 x bfloat> @v_maximumnum_v32bf16(<32 x bfloat> %x, <32 x bfloat> %y) #
; GFX12-FAKE16-NEXT: v_cndmask_b32_e64 v16, v16, v0, s2
; GFX12-FAKE16-NEXT: v_cmp_u_f32_e64 s2, v21, v21
; GFX12-FAKE16-NEXT: v_lshlrev_b32_e32 v21, 16, v17
-; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_1) | instid1(VALU_DEP_3)
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_2) | instid1(VALU_DEP_3)
; GFX12-FAKE16-NEXT: v_lshlrev_b32_e32 v23, 16, v16
; GFX12-FAKE16-NEXT: s_wait_alu depctr_va_sdst(0)
; GFX12-FAKE16-NEXT: v_cndmask_b32_e64 v18, v18, v2, s2
-; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX12-FAKE16-NEXT: v_cmp_gt_f32_e64 s2, v22, v21
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_3) | instid1(VALU_DEP_3)
; GFX12-FAKE16-NEXT: v_lshlrev_b32_e32 v25, 16, v18
; GFX12-FAKE16-NEXT: s_wait_alu depctr_va_sdst(0)
-; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_3)
; GFX12-FAKE16-NEXT: v_cndmask_b32_e64 v17, v17, v1, s2
; GFX12-FAKE16-NEXT: v_cmp_gt_f32_e64 s2, v24, v23
; GFX12-FAKE16-NEXT: v_cmp_gt_f32_e64 s3, v26, v25
-; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_1) | instid1(VALU_DEP_3)
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_4) | instid1(VALU_DEP_3)
; GFX12-FAKE16-NEXT: v_lshlrev_b32_e32 v21, 16, v17
; GFX12-FAKE16-NEXT: s_wait_alu depctr_va_sdst(0)
; GFX12-FAKE16-NEXT: v_cndmask_b32_e64 v16, v16, v0, s2
; GFX12-FAKE16-NEXT: v_cmp_eq_u16_e64 s2, 0, v3
; GFX12-FAKE16-NEXT: v_cndmask_b32_e64 v18, v18, v2, s3
-; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX12-FAKE16-NEXT: v_lshlrev_b32_e32 v22, 16, v16
; GFX12-FAKE16-NEXT: s_and_b32 s1, s1, s2
; GFX12-FAKE16-NEXT: v_cmp_eq_u16_e64 s2, 0, v1
@@ -13068,7 +13019,7 @@ define bfloat @v_maximumnum_bf16_no_ieee(bfloat %x, bfloat %y) #0 {
; GFX11-FAKE16-NEXT: v_cndmask_b32_e32 v1, v1, v0, vcc_lo
; GFX11-FAKE16-NEXT: v_cmp_eq_u16_e32 vcc_lo, 0, v0
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v2, 16, v1
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_cmp_eq_f32_e64 s0, 0, v2
; GFX11-FAKE16-NEXT: s_and_b32 vcc_lo, s0, vcc_lo
; GFX11-FAKE16-NEXT: v_cndmask_b32_e32 v0, v1, v0, vcc_lo
@@ -13328,24 +13279,23 @@ define <2 x bfloat> @v_maximumnum_v2bf16_no_ieee(<2 x bfloat> %x, <2 x bfloat> %
; GFX11-TRUE16-NEXT: v_cmp_u_f32_e64 s1, v6, v6
; GFX11-TRUE16-NEXT: v_cmp_u_f32_e64 s2, v7, v7
; GFX11-TRUE16-NEXT: v_cndmask_b16 v2.l, v5.l, v3.l, vcc_lo
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-TRUE16-NEXT: v_cndmask_b16 v0.l, v0.l, v1.l, s0
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-TRUE16-NEXT: v_cndmask_b16 v3.l, v3.l, v2.l, s1
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_2) | instid1(VALU_DEP_4)
; GFX11-TRUE16-NEXT: v_cndmask_b16 v1.l, v1.l, v0.l, s2
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v4.l, v2.l
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v5.l, v0.l
-; GFX11-TRUE16-NEXT: v_mov_b16_e32 v6.l, v3.l
; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
+; GFX11-TRUE16-NEXT: v_mov_b16_e32 v6.l, v3.l
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v7.l, v1.l
-; GFX11-TRUE16-NEXT: v_lshlrev_b32_e32 v4, 16, v4
; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
+; GFX11-TRUE16-NEXT: v_lshlrev_b32_e32 v4, 16, v4
; GFX11-TRUE16-NEXT: v_lshlrev_b32_e32 v5, 16, v5
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
; GFX11-TRUE16-NEXT: v_lshlrev_b32_e32 v6, 16, v6
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-TRUE16-NEXT: v_lshlrev_b32_e32 v7, 16, v7
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-TRUE16-NEXT: v_cmp_gt_f32_e32 vcc_lo, v4, v6
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX11-TRUE16-NEXT: v_cmp_gt_f32_e64 s0, v5, v7
; GFX11-TRUE16-NEXT: v_cndmask_b16 v3.l, v3.l, v2.l, vcc_lo
; GFX11-TRUE16-NEXT: v_cndmask_b16 v1.l, v1.l, v0.l, s0
@@ -13404,8 +13354,8 @@ define <2 x bfloat> @v_maximumnum_v2bf16_no_ieee(<2 x bfloat> %x, <2 x bfloat> %
; GFX11-FAKE16-NEXT: v_cmp_eq_f32_e64 s2, 0, v5
; GFX11-FAKE16-NEXT: v_cndmask_b32_e32 v2, v3, v2, vcc_lo
; GFX11-FAKE16-NEXT: s_and_b32 vcc_lo, s2, s1
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_cndmask_b32_e32 v0, v1, v0, vcc_lo
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_perm_b32 v0, v2, v0, 0x5040100
; GFX11-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -13451,17 +13401,17 @@ define <2 x bfloat> @v_maximumnum_v2bf16_no_ieee(<2 x bfloat> %x, <2 x bfloat> %
; GFX12-TRUE16-NEXT: s_wait_alu depctr_va_vcc(0)
; GFX12-TRUE16-NEXT: v_cndmask_b16 v3.l, v3.l, v2.l, vcc_lo
; GFX12-TRUE16-NEXT: s_wait_alu depctr_va_sdst(0)
-; GFX12-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_3) | instid1(VALU_DEP_4)
; GFX12-TRUE16-NEXT: v_cndmask_b16 v1.l, v1.l, v0.l, s0
; GFX12-TRUE16-NEXT: v_cmp_eq_u16_e32 vcc_lo, 0, v2.l
; GFX12-TRUE16-NEXT: v_cmp_eq_u16_e64 s0, 0, v0.l
; GFX12-TRUE16-NEXT: v_mov_b16_e32 v4.l, v3.l
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX12-TRUE16-NEXT: v_mov_b16_e32 v5.l, v1.l
-; GFX12-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX12-TRUE16-NEXT: v_lshlrev_b32_e32 v4, 16, v4
-; GFX12-TRUE16-NEXT: v_lshlrev_b32_e32 v5, 16, v5
; GFX12-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX12-TRUE16-NEXT: v_lshlrev_b32_e32 v5, 16, v5
; GFX12-TRUE16-NEXT: v_cmp_eq_f32_e64 s1, 0, v4
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX12-TRUE16-NEXT: v_cmp_eq_f32_e64 s2, 0, v5
; GFX12-TRUE16-NEXT: s_and_b32 s1, s1, vcc_lo
; GFX12-TRUE16-NEXT: s_and_b32 s0, s2, s0
@@ -13804,7 +13754,6 @@ define <3 x bfloat> @v_maximumnum_v3bf16_no_ieee(<3 x bfloat> %x, <3 x bfloat> %
; GFX11-TRUE16-NEXT: v_cmp_gt_f32_e64 s1, v11, v10
; GFX11-TRUE16-NEXT: v_cndmask_b16 v5.l, v5.l, v4.l, vcc_lo
; GFX11-TRUE16-NEXT: v_cmp_eq_u16_e32 vcc_lo, 0, v4.l
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_4)
; GFX11-TRUE16-NEXT: v_cndmask_b16 v3.l, v3.l, v1.l, s0
; GFX11-TRUE16-NEXT: v_cmp_eq_u16_e64 s0, 0, v1.l
; GFX11-TRUE16-NEXT: v_cndmask_b16 v2.l, v2.l, v0.l, s1
@@ -13872,19 +13821,20 @@ define <3 x bfloat> @v_maximumnum_v3bf16_no_ieee(<3 x bfloat> %x, <3 x bfloat> %
; GFX11-FAKE16-NEXT: v_cmp_gt_f32_e32 vcc_lo, v8, v7
; GFX11-FAKE16-NEXT: v_cndmask_b32_e32 v3, v3, v1, vcc_lo
; GFX11-FAKE16-NEXT: v_cmp_eq_f32_e32 vcc_lo, 0, v6
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_2) | instid1(VALU_DEP_2)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v6, 16, v3
; GFX11-FAKE16-NEXT: s_and_b32 vcc_lo, vcc_lo, s0
; GFX11-FAKE16-NEXT: v_dual_cndmask_b32 v4, v5, v4 :: v_dual_lshlrev_b32 v7, 16, v2
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-FAKE16-NEXT: v_cmp_eq_f32_e32 vcc_lo, 0, v6
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_cmp_eq_f32_e64 s1, 0, v7
; GFX11-FAKE16-NEXT: s_and_b32 s0, s1, s2
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v0, v2, v0, s0
; GFX11-FAKE16-NEXT: v_cmp_eq_u16_e64 s0, 0, v1
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX11-FAKE16-NEXT: v_perm_b32 v0, v4, v0, 0x5040100
; GFX11-FAKE16-NEXT: s_and_b32 vcc_lo, vcc_lo, s0
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_cndmask_b32_e32 v1, v3, v1, vcc_lo
; GFX11-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -14487,28 +14437,27 @@ define <4 x bfloat> @v_maximumnum_v4bf16_no_ieee(<4 x bfloat> %x, <4 x bfloat> %
; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
; GFX11-FAKE16-NEXT: v_cmp_gt_f32_e64 s1, v13, v12
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v10, 16, v2
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_2) | instid1(VALU_DEP_4)
; GFX11-FAKE16-NEXT: v_cmp_gt_f32_e64 s0, v8, v9
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v3, v3, v1, s1
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_4)
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v7, v7, v6, s0
; GFX11-FAKE16-NEXT: v_cmp_gt_f32_e64 s0, v11, v10
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_2) | instid1(VALU_DEP_2)
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v8, 16, v7
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v2, v2, v0, s0
; GFX11-FAKE16-NEXT: v_cmp_eq_u16_e64 s0, 0, v4
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v9, 16, v2
; GFX11-FAKE16-NEXT: s_and_b32 vcc_lo, vcc_lo, s0
; GFX11-FAKE16-NEXT: v_cmp_eq_u16_e64 s0, 0, v6
; GFX11-FAKE16-NEXT: v_dual_cndmask_b32 v4, v5, v4 :: v_dual_lshlrev_b32 v5, 16, v3
; GFX11-FAKE16-NEXT: v_cmp_eq_f32_e32 vcc_lo, 0, v8
; GFX11-FAKE16-NEXT: v_cmp_eq_f32_e64 s1, 0, v9
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_cmp_eq_f32_e64 s3, 0, v5
; GFX11-FAKE16-NEXT: s_and_b32 vcc_lo, vcc_lo, s0
; GFX11-FAKE16-NEXT: v_cndmask_b32_e32 v5, v7, v6, vcc_lo
; GFX11-FAKE16-NEXT: s_and_b32 vcc_lo, s1, s2
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_cndmask_b32_e32 v0, v2, v0, vcc_lo
; GFX11-FAKE16-NEXT: s_and_b32 vcc_lo, s3, s4
; GFX11-FAKE16-NEXT: v_cndmask_b32_e32 v1, v3, v1, vcc_lo
@@ -14673,18 +14622,17 @@ define <4 x bfloat> @v_maximumnum_v4bf16_no_ieee(<4 x bfloat> %x, <4 x bfloat> %
; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_3)
; GFX12-FAKE16-NEXT: v_lshlrev_b32_e32 v10, 16, v2
; GFX12-FAKE16-NEXT: v_cmp_gt_f32_e64 s0, v8, v9
-; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_1) | instid1(VALU_DEP_2)
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_4) | instid1(VALU_DEP_3)
; GFX12-FAKE16-NEXT: v_cmp_gt_f32_e64 s1, v13, v12
; GFX12-FAKE16-NEXT: s_wait_alu depctr_va_sdst(0)
; GFX12-FAKE16-NEXT: v_cndmask_b32_e64 v7, v7, v6, s0
; GFX12-FAKE16-NEXT: v_cmp_gt_f32_e64 s0, v11, v10
-; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_3)
; GFX12-FAKE16-NEXT: v_cndmask_b32_e64 v3, v3, v1, s1
; GFX12-FAKE16-NEXT: v_lshlrev_b32_e32 v8, 16, v7
; GFX12-FAKE16-NEXT: s_wait_alu depctr_va_sdst(0)
-; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX12-FAKE16-NEXT: v_cndmask_b32_e64 v2, v2, v0, s0
; GFX12-FAKE16-NEXT: v_cmp_eq_u16_e64 s0, 0, v4
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX12-FAKE16-NEXT: v_lshlrev_b32_e32 v9, 16, v2
; GFX12-FAKE16-NEXT: s_and_b32 vcc_lo, vcc_lo, s0
; GFX12-FAKE16-NEXT: v_cmp_eq_u16_e64 s0, 0, v6
diff --git a/llvm/test/CodeGen/AMDGPU/min.ll b/llvm/test/CodeGen/AMDGPU/min.ll
index 5289df5ae917a1..67ce6279bbc17a 100644
--- a/llvm/test/CodeGen/AMDGPU/min.ll
+++ b/llvm/test/CodeGen/AMDGPU/min.ll
@@ -3108,6 +3108,7 @@ define amdgpu_kernel void @v_test_umin_ult_i32_multi_use(ptr addrspace(1) %out0,
; GFX11-NEXT: s_load_b32 s5, s[6:7], 0x0
; GFX11-NEXT: s_waitcnt lgkmcnt(0)
; GFX11-NEXT: s_cmp_lt_u32 s4, s5
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cselect_b32 s4, s4, s5
; GFX11-NEXT: s_cselect_b32 s5, 1, 0
; GFX11-NEXT: v_mov_b32_e32 v1, s4
@@ -3130,6 +3131,7 @@ define amdgpu_kernel void @v_test_umin_ult_i32_multi_use(ptr addrspace(1) %out0,
; GFX1250-NEXT: s_load_b32 s1, s[14:15], 0x0
; GFX1250-NEXT: s_wait_kmcnt 0x0
; GFX1250-NEXT: s_cmp_lt_u32 s0, s1
+; GFX1250-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1250-NEXT: s_cselect_b32 s0, s0, s1
; GFX1250-NEXT: s_cselect_b32 s1, 1, 0
; GFX1250-NEXT: v_mov_b32_e32 v1, s0
diff --git a/llvm/test/CodeGen/AMDGPU/minimumnum.bf16.ll b/llvm/test/CodeGen/AMDGPU/minimumnum.bf16.ll
index d5d49a74815fc7..b8e8811b1f1af9 100644
--- a/llvm/test/CodeGen/AMDGPU/minimumnum.bf16.ll
+++ b/llvm/test/CodeGen/AMDGPU/minimumnum.bf16.ll
@@ -154,7 +154,7 @@ define bfloat @v_minimumnum_bf16(bfloat %x, bfloat %y) #1 {
; GFX11-FAKE16-NEXT: v_cndmask_b32_e32 v1, v1, v0, vcc_lo
; GFX11-FAKE16-NEXT: v_cmp_eq_u16_e32 vcc_lo, 0x8000, v0
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v2, 16, v1
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_cmp_eq_f32_e64 s0, 0, v2
; GFX11-FAKE16-NEXT: s_and_b32 vcc_lo, s0, vcc_lo
; GFX11-FAKE16-NEXT: v_cndmask_b32_e32 v0, v1, v0, vcc_lo
@@ -328,7 +328,7 @@ define bfloat @v_minimumnum_bf16_nnan(bfloat %x, bfloat %y) #1 {
; GFX11-FAKE16-NEXT: v_cndmask_b32_e32 v1, v1, v0, vcc_lo
; GFX11-FAKE16-NEXT: v_cmp_eq_u16_e32 vcc_lo, 0x8000, v0
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v2, 16, v1
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_cmp_eq_f32_e64 s0, 0, v2
; GFX11-FAKE16-NEXT: s_and_b32 vcc_lo, s0, vcc_lo
; GFX11-FAKE16-NEXT: v_cndmask_b32_e32 v0, v1, v0, vcc_lo
@@ -574,24 +574,23 @@ define <2 x bfloat> @v_minimumnum_v2bf16(<2 x bfloat> %x, <2 x bfloat> %y) #1 {
; GFX11-TRUE16-NEXT: v_cmp_u_f32_e64 s1, v6, v6
; GFX11-TRUE16-NEXT: v_cmp_u_f32_e64 s2, v7, v7
; GFX11-TRUE16-NEXT: v_cndmask_b16 v2.l, v5.l, v3.l, vcc_lo
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-TRUE16-NEXT: v_cndmask_b16 v0.l, v0.l, v1.l, s0
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-TRUE16-NEXT: v_cndmask_b16 v3.l, v3.l, v2.l, s1
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_2) | instid1(VALU_DEP_4)
; GFX11-TRUE16-NEXT: v_cndmask_b16 v1.l, v1.l, v0.l, s2
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v4.l, v2.l
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v5.l, v0.l
-; GFX11-TRUE16-NEXT: v_mov_b16_e32 v6.l, v3.l
; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
+; GFX11-TRUE16-NEXT: v_mov_b16_e32 v6.l, v3.l
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v7.l, v1.l
-; GFX11-TRUE16-NEXT: v_lshlrev_b32_e32 v4, 16, v4
; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
+; GFX11-TRUE16-NEXT: v_lshlrev_b32_e32 v4, 16, v4
; GFX11-TRUE16-NEXT: v_lshlrev_b32_e32 v5, 16, v5
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
; GFX11-TRUE16-NEXT: v_lshlrev_b32_e32 v6, 16, v6
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-TRUE16-NEXT: v_lshlrev_b32_e32 v7, 16, v7
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-TRUE16-NEXT: v_cmp_lt_f32_e32 vcc_lo, v4, v6
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX11-TRUE16-NEXT: v_cmp_lt_f32_e64 s0, v5, v7
; GFX11-TRUE16-NEXT: v_cndmask_b16 v3.l, v3.l, v2.l, vcc_lo
; GFX11-TRUE16-NEXT: v_cndmask_b16 v1.l, v1.l, v0.l, s0
@@ -650,8 +649,8 @@ define <2 x bfloat> @v_minimumnum_v2bf16(<2 x bfloat> %x, <2 x bfloat> %y) #1 {
; GFX11-FAKE16-NEXT: v_cmp_eq_f32_e64 s2, 0, v5
; GFX11-FAKE16-NEXT: v_cndmask_b32_e32 v2, v3, v2, vcc_lo
; GFX11-FAKE16-NEXT: s_and_b32 vcc_lo, s2, s1
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_cndmask_b32_e32 v0, v1, v0, vcc_lo
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_perm_b32 v0, v2, v0, 0x5040100
; GFX11-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -697,17 +696,17 @@ define <2 x bfloat> @v_minimumnum_v2bf16(<2 x bfloat> %x, <2 x bfloat> %y) #1 {
; GFX12-TRUE16-NEXT: s_wait_alu depctr_va_vcc(0)
; GFX12-TRUE16-NEXT: v_cndmask_b16 v3.l, v3.l, v2.l, vcc_lo
; GFX12-TRUE16-NEXT: s_wait_alu depctr_va_sdst(0)
-; GFX12-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_3) | instid1(VALU_DEP_4)
; GFX12-TRUE16-NEXT: v_cndmask_b16 v1.l, v1.l, v0.l, s0
; GFX12-TRUE16-NEXT: v_cmp_eq_u16_e32 vcc_lo, 0x8000, v2.l
; GFX12-TRUE16-NEXT: v_cmp_eq_u16_e64 s0, 0x8000, v0.l
; GFX12-TRUE16-NEXT: v_mov_b16_e32 v4.l, v3.l
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX12-TRUE16-NEXT: v_mov_b16_e32 v5.l, v1.l
-; GFX12-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX12-TRUE16-NEXT: v_lshlrev_b32_e32 v4, 16, v4
-; GFX12-TRUE16-NEXT: v_lshlrev_b32_e32 v5, 16, v5
; GFX12-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX12-TRUE16-NEXT: v_lshlrev_b32_e32 v5, 16, v5
; GFX12-TRUE16-NEXT: v_cmp_eq_f32_e64 s1, 0, v4
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX12-TRUE16-NEXT: v_cmp_eq_f32_e64 s2, 0, v5
; GFX12-TRUE16-NEXT: s_and_b32 s1, s1, vcc_lo
; GFX12-TRUE16-NEXT: s_and_b32 s0, s2, s0
@@ -906,17 +905,16 @@ define <2 x bfloat> @v_minimumnum_v2bf16_nnan(<2 x bfloat> %x, <2 x bfloat> %y)
; GFX11-TRUE16-NEXT: v_cmp_lt_f32_e64 s0, v5, v4
; GFX11-TRUE16-NEXT: v_cndmask_b16 v2.l, v7.l, v6.l, vcc_lo
; GFX11-TRUE16-NEXT: v_cmp_eq_u16_e32 vcc_lo, 0x8000, v0.l
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_1) | instid1(VALU_DEP_4)
; GFX11-TRUE16-NEXT: v_cndmask_b16 v1.l, v1.l, v0.l, s0
; GFX11-TRUE16-NEXT: v_cmp_eq_u16_e64 s0, 0x8000, v6.l
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_3)
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v3.l, v2.l
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v4.l, v1.l
-; GFX11-TRUE16-NEXT: v_lshlrev_b32_e32 v3, 16, v3
; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX11-TRUE16-NEXT: v_lshlrev_b32_e32 v3, 16, v3
; GFX11-TRUE16-NEXT: v_lshlrev_b32_e32 v4, 16, v4
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-TRUE16-NEXT: v_cmp_eq_f32_e64 s1, 0, v3
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX11-TRUE16-NEXT: v_cmp_eq_f32_e64 s2, 0, v4
; GFX11-TRUE16-NEXT: s_and_b32 s0, s1, s0
; GFX11-TRUE16-NEXT: s_and_b32 s1, s2, vcc_lo
@@ -948,8 +946,8 @@ define <2 x bfloat> @v_minimumnum_v2bf16_nnan(<2 x bfloat> %x, <2 x bfloat> %y)
; GFX11-FAKE16-NEXT: v_cmp_eq_f32_e64 s2, 0, v4
; GFX11-FAKE16-NEXT: v_cndmask_b32_e32 v0, v1, v0, vcc_lo
; GFX11-FAKE16-NEXT: s_and_b32 vcc_lo, s2, s1
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_cndmask_b32_e32 v1, v2, v6, vcc_lo
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_perm_b32 v0, v1, v0, 0x5040100
; GFX11-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -1317,7 +1315,6 @@ define <3 x bfloat> @v_minimumnum_v3bf16(<3 x bfloat> %x, <3 x bfloat> %y) #1 {
; GFX11-TRUE16-NEXT: v_cmp_lt_f32_e64 s1, v11, v10
; GFX11-TRUE16-NEXT: v_cndmask_b16 v5.l, v5.l, v4.l, vcc_lo
; GFX11-TRUE16-NEXT: v_cmp_eq_u16_e32 vcc_lo, 0x8000, v4.l
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_4)
; GFX11-TRUE16-NEXT: v_cndmask_b16 v3.l, v3.l, v1.l, s0
; GFX11-TRUE16-NEXT: v_cmp_eq_u16_e64 s0, 0x8000, v1.l
; GFX11-TRUE16-NEXT: v_cndmask_b16 v2.l, v2.l, v0.l, s1
@@ -1385,19 +1382,20 @@ define <3 x bfloat> @v_minimumnum_v3bf16(<3 x bfloat> %x, <3 x bfloat> %y) #1 {
; GFX11-FAKE16-NEXT: v_cmp_lt_f32_e32 vcc_lo, v8, v7
; GFX11-FAKE16-NEXT: v_cndmask_b32_e32 v3, v3, v1, vcc_lo
; GFX11-FAKE16-NEXT: v_cmp_eq_f32_e32 vcc_lo, 0, v6
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_2) | instid1(VALU_DEP_2)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v6, 16, v3
; GFX11-FAKE16-NEXT: s_and_b32 vcc_lo, vcc_lo, s0
; GFX11-FAKE16-NEXT: v_dual_cndmask_b32 v4, v5, v4 :: v_dual_lshlrev_b32 v7, 16, v2
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-FAKE16-NEXT: v_cmp_eq_f32_e32 vcc_lo, 0, v6
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_cmp_eq_f32_e64 s1, 0, v7
; GFX11-FAKE16-NEXT: s_and_b32 s0, s1, s2
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v0, v2, v0, s0
; GFX11-FAKE16-NEXT: v_cmp_eq_u16_e64 s0, 0x8000, v1
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX11-FAKE16-NEXT: v_perm_b32 v0, v4, v0, 0x5040100
; GFX11-FAKE16-NEXT: s_and_b32 vcc_lo, vcc_lo, s0
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_cndmask_b32_e32 v1, v3, v1, vcc_lo
; GFX11-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -1734,10 +1732,9 @@ define <3 x bfloat> @v_minimumnum_v3bf16_nnan(<3 x bfloat> %x, <3 x bfloat> %y)
; GFX11-TRUE16-NEXT: v_cmp_lt_f32_e64 s1, v9, v8
; GFX11-TRUE16-NEXT: v_cndmask_b16 v3.l, v3.l, v1.l, vcc_lo
; GFX11-TRUE16-NEXT: v_cmp_eq_u16_e32 vcc_lo, 0x8000, v1.l
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(SKIP_1) | instid1(VALU_DEP_4)
; GFX11-TRUE16-NEXT: v_cndmask_b16 v4.l, v4.l, v5.l, s0
; GFX11-TRUE16-NEXT: v_cndmask_b16 v2.l, v2.l, v0.l, s1
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_4)
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v6.l, v3.l
; GFX11-TRUE16-NEXT: v_cmp_eq_u16_e64 s0, 0x8000, v0.l
; GFX11-TRUE16-NEXT: v_cmp_eq_u16_e64 s1, 0x8000, v5.l
@@ -1783,15 +1780,16 @@ define <3 x bfloat> @v_minimumnum_v3bf16_nnan(<3 x bfloat> %x, <3 x bfloat> %y)
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v6, 16, v5
; GFX11-FAKE16-NEXT: v_cndmask_b32_e32 v3, v3, v1, vcc_lo
; GFX11-FAKE16-NEXT: v_cmp_eq_f32_e32 vcc_lo, 0, v4
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_cmp_eq_f32_e64 s1, 0, v6
; GFX11-FAKE16-NEXT: s_and_b32 vcc_lo, vcc_lo, s0
; GFX11-FAKE16-NEXT: v_cndmask_b32_e32 v0, v2, v0, vcc_lo
; GFX11-FAKE16-NEXT: s_and_b32 vcc_lo, s1, s2
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX11-FAKE16-NEXT: v_dual_cndmask_b32 v2, v5, v9 :: v_dual_lshlrev_b32 v7, 16, v3
; GFX11-FAKE16-NEXT: v_cmp_eq_u16_e32 vcc_lo, 0x8000, v1
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_cmp_eq_f32_e64 s3, 0, v7
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_perm_b32 v0, v2, v0, 0x5040100
; GFX11-FAKE16-NEXT: s_and_b32 vcc_lo, s3, vcc_lo
; GFX11-FAKE16-NEXT: v_cndmask_b32_e32 v1, v3, v1, vcc_lo
@@ -2352,28 +2350,27 @@ define <4 x bfloat> @v_minimumnum_v4bf16(<4 x bfloat> %x, <4 x bfloat> %y) #1 {
; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
; GFX11-FAKE16-NEXT: v_cmp_lt_f32_e64 s1, v13, v12
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v10, 16, v2
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_2) | instid1(VALU_DEP_4)
; GFX11-FAKE16-NEXT: v_cmp_lt_f32_e64 s0, v8, v9
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v3, v3, v1, s1
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_4)
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v7, v7, v6, s0
; GFX11-FAKE16-NEXT: v_cmp_lt_f32_e64 s0, v11, v10
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_2) | instid1(VALU_DEP_2)
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v8, 16, v7
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v2, v2, v0, s0
; GFX11-FAKE16-NEXT: v_cmp_eq_u16_e64 s0, 0x8000, v4
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v9, 16, v2
; GFX11-FAKE16-NEXT: s_and_b32 vcc_lo, vcc_lo, s0
; GFX11-FAKE16-NEXT: v_cmp_eq_u16_e64 s0, 0x8000, v6
; GFX11-FAKE16-NEXT: v_dual_cndmask_b32 v4, v5, v4 :: v_dual_lshlrev_b32 v5, 16, v3
; GFX11-FAKE16-NEXT: v_cmp_eq_f32_e32 vcc_lo, 0, v8
; GFX11-FAKE16-NEXT: v_cmp_eq_f32_e64 s1, 0, v9
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_cmp_eq_f32_e64 s3, 0, v5
; GFX11-FAKE16-NEXT: s_and_b32 vcc_lo, vcc_lo, s0
; GFX11-FAKE16-NEXT: v_cndmask_b32_e32 v5, v7, v6, vcc_lo
; GFX11-FAKE16-NEXT: s_and_b32 vcc_lo, s1, s2
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_cndmask_b32_e32 v0, v2, v0, vcc_lo
; GFX11-FAKE16-NEXT: s_and_b32 vcc_lo, s3, s4
; GFX11-FAKE16-NEXT: v_cndmask_b32_e32 v1, v3, v1, vcc_lo
@@ -2538,18 +2535,17 @@ define <4 x bfloat> @v_minimumnum_v4bf16(<4 x bfloat> %x, <4 x bfloat> %y) #1 {
; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_3)
; GFX12-FAKE16-NEXT: v_lshlrev_b32_e32 v10, 16, v2
; GFX12-FAKE16-NEXT: v_cmp_lt_f32_e64 s0, v8, v9
-; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_1) | instid1(VALU_DEP_2)
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_4) | instid1(VALU_DEP_3)
; GFX12-FAKE16-NEXT: v_cmp_lt_f32_e64 s1, v13, v12
; GFX12-FAKE16-NEXT: s_wait_alu depctr_va_sdst(0)
; GFX12-FAKE16-NEXT: v_cndmask_b32_e64 v7, v7, v6, s0
; GFX12-FAKE16-NEXT: v_cmp_lt_f32_e64 s0, v11, v10
-; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_3)
; GFX12-FAKE16-NEXT: v_cndmask_b32_e64 v3, v3, v1, s1
; GFX12-FAKE16-NEXT: v_lshlrev_b32_e32 v8, 16, v7
; GFX12-FAKE16-NEXT: s_wait_alu depctr_va_sdst(0)
-; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX12-FAKE16-NEXT: v_cndmask_b32_e64 v2, v2, v0, s0
; GFX12-FAKE16-NEXT: v_cmp_eq_u16_e64 s0, 0x8000, v4
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX12-FAKE16-NEXT: v_lshlrev_b32_e32 v9, 16, v2
; GFX12-FAKE16-NEXT: s_and_b32 vcc_lo, vcc_lo, s0
; GFX12-FAKE16-NEXT: v_cmp_eq_u16_e64 s0, 0x8000, v6
@@ -2856,12 +2852,11 @@ define <4 x bfloat> @v_minimumnum_v4bf16_nnan(<4 x bfloat> %x, <4 x bfloat> %y)
; GFX11-FAKE16-NEXT: v_cmp_lt_f32_e64 s0, v9, v8
; GFX11-FAKE16-NEXT: v_and_b32_e32 v7, 0xffff0000, v1
; GFX11-FAKE16-NEXT: v_cmp_eq_u16_e64 s2, 0x8000, v13
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_2) | instid1(VALU_DEP_2)
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v2, v2, v0, s0
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v5, 16, v1
; GFX11-FAKE16-NEXT: v_cmp_lt_f32_e64 s0, v12, v11
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_4) | instid1(VALU_DEP_2)
; GFX11-FAKE16-NEXT: v_cmp_lt_f32_e32 vcc_lo, v5, v4
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_3) | instid1(VALU_DEP_2)
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v8, v14, v13, s0
; GFX11-FAKE16-NEXT: v_cmp_eq_u16_e64 s0, 0x8000, v1
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v5, 16, v1
@@ -2871,10 +2866,10 @@ define <4 x bfloat> @v_minimumnum_v4bf16_nnan(<4 x bfloat> %x, <4 x bfloat> %y)
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v10, 16, v4
; GFX11-FAKE16-NEXT: v_cmp_eq_f32_e32 vcc_lo, 0, v10
; GFX11-FAKE16-NEXT: s_and_b32 vcc_lo, vcc_lo, s0
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_dual_cndmask_b32 v1, v4, v1 :: v_dual_and_b32 v6, 0xffff0000, v3
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v3, 16, v3
; GFX11-FAKE16-NEXT: v_cmp_eq_u16_e64 s0, 0x8000, v0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_2) | instid1(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_cmp_lt_f32_e64 s1, v7, v6
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v6, 16, v2
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v7, 16, v8
@@ -2882,17 +2877,18 @@ define <4 x bfloat> @v_minimumnum_v4bf16_nnan(<4 x bfloat> %x, <4 x bfloat> %y)
; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_cmp_eq_f32_e32 vcc_lo, 0, v6
; GFX11-FAKE16-NEXT: v_cmp_eq_f32_e64 s1, 0, v7
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_3) | instid1(VALU_DEP_2)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v4, 16, v3
; GFX11-FAKE16-NEXT: s_and_b32 vcc_lo, vcc_lo, s0
; GFX11-FAKE16-NEXT: v_cndmask_b32_e32 v0, v2, v0, vcc_lo
; GFX11-FAKE16-NEXT: s_and_b32 vcc_lo, s1, s2
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_2) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_cmp_eq_f32_e64 s3, 0, v4
; GFX11-FAKE16-NEXT: v_cndmask_b32_e32 v2, v8, v13, vcc_lo
; GFX11-FAKE16-NEXT: s_and_b32 vcc_lo, s3, s4
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_perm_b32 v0, v2, v0, 0x5040100
; GFX11-FAKE16-NEXT: v_cndmask_b32_e32 v3, v3, v5, vcc_lo
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_perm_b32 v1, v3, v1, 0x5040100
; GFX11-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -2990,16 +2986,15 @@ define <4 x bfloat> @v_minimumnum_v4bf16_nnan(<4 x bfloat> %x, <4 x bfloat> %y)
; GFX12-FAKE16-NEXT: v_dual_cndmask_b32 v1, v4, v1 :: v_dual_and_b32 v6, 0xffff0000, v3
; GFX12-FAKE16-NEXT: v_lshrrev_b32_e32 v3, 16, v3
; GFX12-FAKE16-NEXT: v_cmp_eq_u16_e64 s0, 0x8000, v0
-; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_3) | instid1(VALU_DEP_3)
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_4) | instid1(VALU_DEP_3)
; GFX12-FAKE16-NEXT: v_cmp_lt_f32_e64 s1, v7, v6
; GFX12-FAKE16-NEXT: v_lshlrev_b32_e32 v6, 16, v2
; GFX12-FAKE16-NEXT: v_lshlrev_b32_e32 v7, 16, v8
; GFX12-FAKE16-NEXT: s_wait_alu depctr_va_sdst(0)
; GFX12-FAKE16-NEXT: v_cndmask_b32_e64 v3, v3, v5, s1
-; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_3)
; GFX12-FAKE16-NEXT: v_cmp_eq_f32_e32 vcc_lo, 0, v6
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_3)
; GFX12-FAKE16-NEXT: v_cmp_eq_f32_e64 s1, 0, v7
-; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX12-FAKE16-NEXT: v_lshlrev_b32_e32 v4, 16, v3
; GFX12-FAKE16-NEXT: s_and_b32 vcc_lo, vcc_lo, s0
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
@@ -3659,7 +3654,7 @@ define <6 x bfloat> @v_minimumnum_v6bf16(<6 x bfloat> %x, <6 x bfloat> %y) #1 {
; GFX11-FAKE16-NEXT: v_cndmask_b32_e32 v9, v9, v8, vcc_lo
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v11, 16, v7
; GFX11-FAKE16-NEXT: v_cmp_lt_f32_e64 s1, v13, v14
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_4) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_cmp_eq_f32_e32 vcc_lo, 0, v11
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v11, v12, v10, s1
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v15, 16, v9
@@ -3678,40 +3673,37 @@ define <6 x bfloat> @v_minimumnum_v6bf16(<6 x bfloat> %x, <6 x bfloat> %y) #1 {
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v15, 16, v2
; GFX11-FAKE16-NEXT: v_cmp_eq_f32_e32 vcc_lo, 0, v7
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v7, 16, v0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(SKIP_2) | instid1(VALU_DEP_3)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_4)
; GFX11-FAKE16-NEXT: v_cmp_u_f32_e64 s0, v9, v9
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v9, 16, v4
; GFX11-FAKE16-NEXT: v_cmp_eq_u16_e64 s4, 0x8000, v2
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v1, v1, v4, s0
; GFX11-FAKE16-NEXT: v_cmp_u_f32_e64 s0, v7, v7
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v7, 16, v5
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_2) | instid1(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v0, v0, v3, s0
; GFX11-FAKE16-NEXT: v_cmp_u_f32_e64 s0, v9, v9
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v9, 16, v1
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v13, 16, v0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_2) | instid1(VALU_DEP_2)
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v4, v4, v1, s0
; GFX11-FAKE16-NEXT: v_cmp_u_f32_e64 s0, v12, v12
; GFX11-FAKE16-NEXT: v_cmp_eq_u16_e64 s2, 0x8000, v0
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v3, v3, v0, s0
; GFX11-FAKE16-NEXT: v_cmp_u_f32_e64 s0, v7, v7
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v7, 16, v4
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_1) | instid1(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v12, 16, v3
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v5, v5, v2, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-FAKE16-NEXT: v_cmp_lt_f32_e64 s0, v9, v7
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_2) | instid1(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v14, 16, v5
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v4, v4, v1, s0
; GFX11-FAKE16-NEXT: v_cmp_lt_f32_e64 s0, v13, v12
; GFX11-FAKE16-NEXT: v_cmp_lt_f32_e64 s1, v15, v14
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_3) | instid1(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v7, 16, v4
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v3, v3, v0, s0
; GFX11-FAKE16-NEXT: v_cmp_eq_u16_e64 s0, 0x8000, v10
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v5, v5, v2, s1
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v9, 16, v3
; GFX11-FAKE16-NEXT: s_and_b32 vcc_lo, vcc_lo, s0
@@ -3722,6 +3714,7 @@ define <6 x bfloat> @v_minimumnum_v6bf16(<6 x bfloat> %x, <6 x bfloat> %y) #1 {
; GFX11-FAKE16-NEXT: v_cmp_eq_f32_e64 s1, 0, v9
; GFX11-FAKE16-NEXT: v_cmp_eq_f32_e64 s3, 0, v11
; GFX11-FAKE16-NEXT: s_and_b32 vcc_lo, vcc_lo, s0
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_cndmask_b32_e32 v1, v4, v1, vcc_lo
; GFX11-FAKE16-NEXT: s_and_b32 vcc_lo, s1, s2
; GFX11-FAKE16-NEXT: v_cndmask_b32_e32 v0, v3, v0, vcc_lo
@@ -3930,7 +3923,6 @@ define <6 x bfloat> @v_minimumnum_v6bf16(<6 x bfloat> %x, <6 x bfloat> %y) #1 {
; GFX12-FAKE16-NEXT: v_cmp_lt_f32_e64 s1, v13, v14
; GFX12-FAKE16-NEXT: v_cmp_eq_f32_e32 vcc_lo, 0, v11
; GFX12-FAKE16-NEXT: s_wait_alu depctr_va_sdst(0)
-; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX12-FAKE16-NEXT: v_cndmask_b32_e64 v11, v12, v10, s1
; GFX12-FAKE16-NEXT: v_lshlrev_b32_e32 v15, 16, v9
; GFX12-FAKE16-NEXT: v_lshlrev_b32_e32 v12, 16, v2
@@ -3949,7 +3941,7 @@ define <6 x bfloat> @v_minimumnum_v6bf16(<6 x bfloat> %x, <6 x bfloat> %y) #1 {
; GFX12-FAKE16-NEXT: v_lshlrev_b32_e32 v15, 16, v2
; GFX12-FAKE16-NEXT: v_cmp_eq_f32_e32 vcc_lo, 0, v7
; GFX12-FAKE16-NEXT: v_lshlrev_b32_e32 v7, 16, v0
-; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(SKIP_3) | instid1(VALU_DEP_3)
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_4)
; GFX12-FAKE16-NEXT: v_cmp_u_f32_e64 s0, v9, v9
; GFX12-FAKE16-NEXT: v_lshlrev_b32_e32 v9, 16, v4
; GFX12-FAKE16-NEXT: v_cmp_eq_u16_e64 s4, 0x8000, v2
@@ -3958,13 +3950,12 @@ define <6 x bfloat> @v_minimumnum_v6bf16(<6 x bfloat> %x, <6 x bfloat> %y) #1 {
; GFX12-FAKE16-NEXT: v_cmp_u_f32_e64 s0, v7, v7
; GFX12-FAKE16-NEXT: v_lshlrev_b32_e32 v7, 16, v5
; GFX12-FAKE16-NEXT: s_wait_alu depctr_va_sdst(0)
-; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_2) | instid1(VALU_DEP_3)
; GFX12-FAKE16-NEXT: v_cndmask_b32_e64 v0, v0, v3, s0
; GFX12-FAKE16-NEXT: v_cmp_u_f32_e64 s0, v9, v9
; GFX12-FAKE16-NEXT: v_lshlrev_b32_e32 v9, 16, v1
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX12-FAKE16-NEXT: v_lshlrev_b32_e32 v13, 16, v0
; GFX12-FAKE16-NEXT: s_wait_alu depctr_va_sdst(0)
-; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_3) | instid1(VALU_DEP_2)
; GFX12-FAKE16-NEXT: v_cndmask_b32_e64 v4, v4, v1, s0
; GFX12-FAKE16-NEXT: v_cmp_u_f32_e64 s0, v12, v12
; GFX12-FAKE16-NEXT: v_cmp_eq_u16_e64 s2, 0x8000, v0
@@ -3972,25 +3963,23 @@ define <6 x bfloat> @v_minimumnum_v6bf16(<6 x bfloat> %x, <6 x bfloat> %y) #1 {
; GFX12-FAKE16-NEXT: v_cndmask_b32_e64 v3, v3, v0, s0
; GFX12-FAKE16-NEXT: v_cmp_u_f32_e64 s0, v7, v7
; GFX12-FAKE16-NEXT: v_lshlrev_b32_e32 v7, 16, v4
-; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_1) | instid1(VALU_DEP_3)
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_2) | instid1(VALU_DEP_3)
; GFX12-FAKE16-NEXT: v_lshlrev_b32_e32 v12, 16, v3
; GFX12-FAKE16-NEXT: s_wait_alu depctr_va_sdst(0)
; GFX12-FAKE16-NEXT: v_cndmask_b32_e64 v5, v5, v2, s0
-; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX12-FAKE16-NEXT: v_cmp_lt_f32_e64 s0, v9, v7
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_3) | instid1(VALU_DEP_3)
; GFX12-FAKE16-NEXT: v_lshlrev_b32_e32 v14, 16, v5
; GFX12-FAKE16-NEXT: s_wait_alu depctr_va_sdst(0)
-; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_3)
; GFX12-FAKE16-NEXT: v_cndmask_b32_e64 v4, v4, v1, s0
; GFX12-FAKE16-NEXT: v_cmp_lt_f32_e64 s0, v13, v12
; GFX12-FAKE16-NEXT: v_cmp_lt_f32_e64 s1, v15, v14
-; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_1) | instid1(VALU_DEP_3)
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_4) | instid1(VALU_DEP_3)
; GFX12-FAKE16-NEXT: v_lshlrev_b32_e32 v7, 16, v4
; GFX12-FAKE16-NEXT: s_wait_alu depctr_va_sdst(0)
; GFX12-FAKE16-NEXT: v_cndmask_b32_e64 v3, v3, v0, s0
; GFX12-FAKE16-NEXT: v_cmp_eq_u16_e64 s0, 0x8000, v10
; GFX12-FAKE16-NEXT: v_cndmask_b32_e64 v5, v5, v2, s1
-; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX12-FAKE16-NEXT: v_lshlrev_b32_e32 v9, 16, v3
; GFX12-FAKE16-NEXT: s_and_b32 vcc_lo, vcc_lo, s0
; GFX12-FAKE16-NEXT: v_cmp_eq_u16_e64 s0, 0x8000, v1
@@ -4847,16 +4836,16 @@ define <8 x bfloat> @v_minimumnum_v8bf16(<8 x bfloat> %x, <8 x bfloat> %y) #1 {
; GFX11-FAKE16-NEXT: v_cmp_eq_u16_e32 vcc_lo, 0x8000, v8
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v13, 16, v11
; GFX11-FAKE16-NEXT: s_and_b32 vcc_lo, s0, vcc_lo
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_4) | instid1(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_cndmask_b32_e32 v8, v9, v8, vcc_lo
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v9, 16, v1
; GFX11-FAKE16-NEXT: v_and_b32_e32 v12, 0xffff0000, v1
; GFX11-FAKE16-NEXT: v_cmp_eq_f32_e64 s1, 0, v13
; GFX11-FAKE16-NEXT: v_and_b32_e32 v13, 0xffff0000, v0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_1) | instid1(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_cmp_u_f32_e32 vcc_lo, v12, v12
; GFX11-FAKE16-NEXT: v_dual_cndmask_b32 v9, v9, v14 :: v_dual_and_b32 v12, 0xffff0000, v5
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-FAKE16-NEXT: v_cmp_u_f32_e32 vcc_lo, v13, v13
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_2) | instid1(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_cmp_u_f32_e64 s0, v12, v12
; GFX11-FAKE16-NEXT: v_cndmask_b32_e32 v13, v16, v15, vcc_lo
; GFX11-FAKE16-NEXT: v_and_b32_e32 v17, 0xffff0000, v4
@@ -4874,19 +4863,17 @@ define <8 x bfloat> @v_minimumnum_v8bf16(<8 x bfloat> %x, <8 x bfloat> %y) #1 {
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v19, 16, v14
; GFX11-FAKE16-NEXT: v_cmp_u_f32_e64 s0, v15, v15
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v15, 16, v7
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v3, v3, v7, s0
; GFX11-FAKE16-NEXT: v_cmp_lt_f32_e64 s0, v16, v17
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_4) | instid1(VALU_DEP_4)
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v17, 16, v3
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v12, v12, v9, s0
; GFX11-FAKE16-NEXT: v_cmp_lt_f32_e64 s0, v18, v19
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v14, v14, v13, s0
; GFX11-FAKE16-NEXT: v_cmp_u_f32_e64 s0, v15, v15
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v15, 16, v12
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_1) | instid1(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v11, 16, v14
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v7, v7, v3, s0
; GFX11-FAKE16-NEXT: v_cmp_eq_f32_e32 vcc_lo, 0, v15
; GFX11-FAKE16-NEXT: v_cmp_eq_u16_e64 s0, 0x8000, v9
@@ -4894,6 +4881,7 @@ define <8 x bfloat> @v_minimumnum_v8bf16(<8 x bfloat> %x, <8 x bfloat> %y) #1 {
; GFX11-FAKE16-NEXT: v_cmp_eq_f32_e64 s1, 0, v11
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v11, 16, v2
; GFX11-FAKE16-NEXT: s_and_b32 vcc_lo, vcc_lo, s0
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_dual_cndmask_b32 v9, v12, v9 :: v_dual_lshlrev_b32 v16, 16, v7
; GFX11-FAKE16-NEXT: s_and_b32 vcc_lo, s1, s2
; GFX11-FAKE16-NEXT: v_cmp_eq_u16_e64 s0, 0x8000, v3
@@ -4905,18 +4893,18 @@ define <8 x bfloat> @v_minimumnum_v8bf16(<8 x bfloat> %x, <8 x bfloat> %y) #1 {
; GFX11-FAKE16-NEXT: v_cndmask_b32_e32 v2, v2, v6, vcc_lo
; GFX11-FAKE16-NEXT: v_cmp_u_f32_e32 vcc_lo, v14, v14
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v14, 16, v4
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v7, v7, v3, s3
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(SKIP_4) | instid1(VALU_DEP_4)
; GFX11-FAKE16-NEXT: v_cmp_eq_u16_e64 s2, 0x8000, v2
; GFX11-FAKE16-NEXT: v_cndmask_b32_e32 v1, v1, v5, vcc_lo
; GFX11-FAKE16-NEXT: v_cmp_u_f32_e32 vcc_lo, v11, v11
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v11, 16, v5
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v13, 16, v7
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(SKIP_3) | instid1(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v18, 16, v1
; GFX11-FAKE16-NEXT: v_cndmask_b32_e32 v0, v0, v4, vcc_lo
; GFX11-FAKE16-NEXT: v_cmp_u_f32_e32 vcc_lo, v15, v15
; GFX11-FAKE16-NEXT: v_cmp_eq_u16_e64 s6, 0x8000, v1
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_cmp_eq_u16_e64 s4, 0x8000, v0
; GFX11-FAKE16-NEXT: v_cndmask_b32_e32 v6, v6, v2, vcc_lo
; GFX11-FAKE16-NEXT: v_cmp_u_f32_e32 vcc_lo, v14, v14
@@ -5227,23 +5215,21 @@ define <8 x bfloat> @v_minimumnum_v8bf16(<8 x bfloat> %x, <8 x bfloat> %y) #1 {
; GFX12-FAKE16-NEXT: v_cmp_u_f32_e64 s0, v15, v15
; GFX12-FAKE16-NEXT: v_lshlrev_b32_e32 v15, 16, v7
; GFX12-FAKE16-NEXT: s_wait_alu depctr_va_sdst(0)
-; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX12-FAKE16-NEXT: v_cndmask_b32_e64 v3, v3, v7, s0
; GFX12-FAKE16-NEXT: v_lshlrev_b32_e32 v16, 16, v9
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_3)
; GFX12-FAKE16-NEXT: v_cmp_lt_f32_e64 s0, v16, v17
-; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX12-FAKE16-NEXT: v_lshlrev_b32_e32 v17, 16, v3
; GFX12-FAKE16-NEXT: s_wait_alu depctr_va_sdst(0)
; GFX12-FAKE16-NEXT: v_cndmask_b32_e64 v12, v12, v9, s0
; GFX12-FAKE16-NEXT: v_cmp_lt_f32_e64 s0, v18, v19
; GFX12-FAKE16-NEXT: s_wait_alu depctr_va_sdst(0)
-; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_2) | instid1(VALU_DEP_3)
; GFX12-FAKE16-NEXT: v_cndmask_b32_e64 v14, v14, v13, s0
; GFX12-FAKE16-NEXT: v_cmp_u_f32_e64 s0, v15, v15
; GFX12-FAKE16-NEXT: v_lshlrev_b32_e32 v15, 16, v12
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_2) | instid1(VALU_DEP_3)
; GFX12-FAKE16-NEXT: v_lshlrev_b32_e32 v11, 16, v14
; GFX12-FAKE16-NEXT: s_wait_alu depctr_va_sdst(0)
-; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_3)
; GFX12-FAKE16-NEXT: v_cndmask_b32_e64 v7, v7, v3, s0
; GFX12-FAKE16-NEXT: v_cmp_eq_f32_e32 vcc_lo, 0, v15
; GFX12-FAKE16-NEXT: v_cmp_eq_u16_e64 s0, 0x8000, v9
@@ -6682,11 +6668,11 @@ define <16 x bfloat> @v_minimumnum_v16bf16(<16 x bfloat> %x, <16 x bfloat> %y) #
; GFX11-TRUE16-NEXT: v_cmp_lt_f32_e64 s0, v27, v29
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v26.l, v24.l
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v27.l, v25.l
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(SKIP_1) | instid1(VALU_DEP_4)
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v28.l, v18.l
; GFX11-TRUE16-NEXT: v_cndmask_b16 v23.l, v23.l, v19.l, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
; GFX11-TRUE16-NEXT: v_lshlrev_b32_e32 v26, 16, v26
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_4)
; GFX11-TRUE16-NEXT: v_lshlrev_b32_e32 v27, 16, v27
; GFX11-TRUE16-NEXT: s_and_b32 s0, s1, s2
; GFX11-TRUE16-NEXT: s_and_b32 s1, vcc_lo, s3
@@ -6922,20 +6908,19 @@ define <16 x bfloat> @v_minimumnum_v16bf16(<16 x bfloat> %x, <16 x bfloat> %y) #
; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v22, 16, v19
; GFX11-FAKE16-NEXT: v_cmp_u_f32_e64 s1, v20, v20
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_2) | instid1(VALU_DEP_3)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_4) | instid1(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_cmp_lt_f32_e64 s0, v21, v22
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v21, 16, v13
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v22, 16, v5
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v19, v19, v18, s0
; GFX11-FAKE16-NEXT: v_cmp_eq_u16_e64 s0, 0x8000, v16
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_1) | instid1(VALU_DEP_4)
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v20, v22, v21, s1
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v22, 16, v12
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(SKIP_3) | instid1(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v23, 16, v19
; GFX11-FAKE16-NEXT: s_and_b32 vcc_lo, vcc_lo, s0
; GFX11-FAKE16-NEXT: v_cmp_u_f32_e64 s0, v24, v24
; GFX11-FAKE16-NEXT: v_cndmask_b32_e32 v16, v17, v16, vcc_lo
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_1) | instid1(VALU_DEP_4)
; GFX11-FAKE16-NEXT: v_cmp_eq_f32_e32 vcc_lo, 0, v23
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v23, 16, v4
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v17, v21, v20, s0
@@ -6943,10 +6928,10 @@ define <16 x bfloat> @v_minimumnum_v16bf16(<16 x bfloat> %x, <16 x bfloat> %y) #
; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v25, 16, v17
; GFX11-FAKE16-NEXT: v_cmp_u_f32_e64 s0, v21, v21
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_2) | instid1(VALU_DEP_2)
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v21, v23, v22, s0
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v24, 16, v20
; GFX11-FAKE16-NEXT: v_cmp_eq_u16_e64 s0, 0x8000, v18
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX11-FAKE16-NEXT: v_cmp_lt_f32_e64 s1, v24, v25
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v24, 16, v11
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v25, 16, v3
@@ -6964,29 +6949,28 @@ define <16 x bfloat> @v_minimumnum_v16bf16(<16 x bfloat> %x, <16 x bfloat> %y) #
; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v27, 16, v22
; GFX11-FAKE16-NEXT: v_cmp_u_f32_e64 s1, v23, v23
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v23, v25, v24, s1
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_4)
; GFX11-FAKE16-NEXT: v_cndmask_b32_e32 v24, v24, v23, vcc_lo
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(SKIP_2) | instid1(VALU_DEP_4)
; GFX11-FAKE16-NEXT: v_cmp_lt_f32_e32 vcc_lo, v26, v27
; GFX11-FAKE16-NEXT: v_and_b32_e32 v27, 0xffff0000, v2
; GFX11-FAKE16-NEXT: v_cmp_eq_u16_e64 s2, 0x8000, v23
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(SKIP_2) | instid1(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_dual_cndmask_b32 v22, v22, v21 :: v_dual_lshlrev_b32 v25, 16, v24
; GFX11-FAKE16-NEXT: v_cmp_eq_f32_e32 vcc_lo, 0, v19
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v19, 16, v23
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v26, 16, v22
; GFX11-FAKE16-NEXT: s_and_b32 vcc_lo, vcc_lo, s0
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX11-FAKE16-NEXT: v_cmp_lt_f32_e64 s1, v19, v25
; GFX11-FAKE16-NEXT: v_cndmask_b32_e32 v17, v17, v20, vcc_lo
; GFX11-FAKE16-NEXT: v_cmp_eq_u16_e64 s0, 0x8000, v21
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_3) | instid1(VALU_DEP_4)
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v19, v24, v23, s1
; GFX11-FAKE16-NEXT: v_and_b32_e32 v24, 0xffff0000, v10
; GFX11-FAKE16-NEXT: v_cmp_u_f32_e64 s1, v27, v27
; GFX11-FAKE16-NEXT: v_and_b32_e32 v27, 0xffff0000, v1
-; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v20, 16, v19
; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
+; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v20, 16, v19
; GFX11-FAKE16-NEXT: v_cmp_u_f32_e32 vcc_lo, v24, v24
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v25, v29, v28, s1
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v29, 16, v1
@@ -7002,13 +6986,12 @@ define <16 x bfloat> @v_minimumnum_v16bf16(<16 x bfloat> %x, <16 x bfloat> %y) #
; GFX11-FAKE16-NEXT: v_cmp_eq_u16_e64 s0, 0x8000, v25
; GFX11-FAKE16-NEXT: v_cndmask_b32_e32 v21, v22, v21, vcc_lo
; GFX11-FAKE16-NEXT: v_cmp_lt_f32_e64 s3, v20, v26
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_3) | instid1(VALU_DEP_4)
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v20, v24, v25, s3
; GFX11-FAKE16-NEXT: v_and_b32_e32 v24, 0xffff0000, v9
; GFX11-FAKE16-NEXT: v_cmp_u_f32_e64 s3, v27, v27
; GFX11-FAKE16-NEXT: v_and_b32_e32 v27, 0xffff0000, v0
-; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v22, 16, v20
; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
+; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v22, 16, v20
; GFX11-FAKE16-NEXT: v_cmp_u_f32_e32 vcc_lo, v24, v24
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v26, v29, v28, s3
; GFX11-FAKE16-NEXT: v_lshrrev_b32_e32 v29, 16, v0
@@ -7021,13 +7004,12 @@ define <16 x bfloat> @v_minimumnum_v16bf16(<16 x bfloat> %x, <16 x bfloat> %y) #
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v22, 16, v26
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v23, 16, v24
; GFX11-FAKE16-NEXT: s_and_b32 vcc_lo, vcc_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_4) | instid1(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_cmp_lt_f32_e64 s1, v22, v23
; GFX11-FAKE16-NEXT: v_dual_cndmask_b32 v20, v20, v25 :: v_dual_lshlrev_b32 v25, 16, v7
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v22, v24, v26, s1
; GFX11-FAKE16-NEXT: v_cmp_u_f32_e64 s1, v27, v27
; GFX11-FAKE16-NEXT: v_and_b32_e32 v24, 0xffff0000, v8
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v27, 16, v22
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v23, v29, v28, s1
; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_2)
@@ -7054,14 +7036,13 @@ define <16 x bfloat> @v_minimumnum_v16bf16(<16 x bfloat> %x, <16 x bfloat> %y) #
; GFX11-FAKE16-NEXT: v_cmp_u_f32_e64 s1, v28, v28
; GFX11-FAKE16-NEXT: v_cndmask_b32_e32 v22, v22, v26, vcc_lo
; GFX11-FAKE16-NEXT: v_cmp_eq_u16_e64 s0, 0x8000, v23
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(SKIP_2) | instid1(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v28, 16, v24
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v6, v6, v14, s1
; GFX11-FAKE16-NEXT: v_cmp_lt_f32_e64 s1, v27, v25
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_cmp_eq_f32_e32 vcc_lo, 0, v28
-; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v27, 16, v6
; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3)
+; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v27, 16, v6
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v15, v15, v7, s1
; GFX11-FAKE16-NEXT: v_cmp_u_f32_e64 s1, v29, v29
; GFX11-FAKE16-NEXT: s_and_b32 vcc_lo, vcc_lo, s0
@@ -7072,18 +7053,17 @@ define <16 x bfloat> @v_minimumnum_v16bf16(<16 x bfloat> %x, <16 x bfloat> %y) #
; GFX11-FAKE16-NEXT: v_cmp_eq_f32_e64 s1, 0, v25
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v26, 16, v14
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v25, 16, v5
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_2)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_2) | instid1(VALU_DEP_4)
; GFX11-FAKE16-NEXT: v_cmp_lt_f32_e64 s3, v27, v26
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v26, 16, v13
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v14, v14, v6, s3
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_cmp_u_f32_e64 s3, v25, v25
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_1) | instid1(VALU_DEP_4)
; GFX11-FAKE16-NEXT: v_cmp_u_f32_e32 vcc_lo, v26, v26
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v25, 16, v3
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v24, 16, v14
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v5, v5, v13, s3
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_cndmask_b32_e32 v13, v13, v5, vcc_lo
; GFX11-FAKE16-NEXT: s_and_b32 vcc_lo, s1, s2
; GFX11-FAKE16-NEXT: v_dual_cndmask_b32 v7, v15, v7 :: v_dual_lshlrev_b32 v26, 16, v5
@@ -7094,17 +7074,15 @@ define <16 x bfloat> @v_minimumnum_v16bf16(<16 x bfloat> %x, <16 x bfloat> %y) #
; GFX11-FAKE16-NEXT: v_perm_b32 v7, v16, v7, 0x5040100
; GFX11-FAKE16-NEXT: v_cmp_u_f32_e64 s0, v15, v15
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v15, 16, v12
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_2) | instid1(VALU_DEP_2)
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v4, v4, v12, s0
; GFX11-FAKE16-NEXT: v_cmp_u_f32_e64 s0, v25, v25
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v25, 16, v11
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v3, v3, v11, s0
; GFX11-FAKE16-NEXT: v_cmp_lt_f32_e64 s0, v26, v24
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_2) | instid1(VALU_DEP_2)
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v27, 16, v3
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v13, v13, v5, s0
; GFX11-FAKE16-NEXT: v_cmp_u_f32_e64 s0, v15, v15
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v15, 16, v13
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v12, v12, v4, s0
; GFX11-FAKE16-NEXT: v_cmp_u_f32_e64 s0, v25, v25
@@ -7112,12 +7090,11 @@ define <16 x bfloat> @v_minimumnum_v16bf16(<16 x bfloat> %x, <16 x bfloat> %y) #
; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
; GFX11-FAKE16-NEXT: v_cmp_eq_f32_e64 s1, 0, v15
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v24, 16, v12
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(SKIP_2) | instid1(VALU_DEP_4)
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v11, v11, v3, s0
; GFX11-FAKE16-NEXT: v_cmp_eq_u16_e64 s0, 0x8000, v6
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v15, 16, v2
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
; GFX11-FAKE16-NEXT: v_cmp_lt_f32_e64 s3, v25, v24
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_4)
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v26, 16, v11
; GFX11-FAKE16-NEXT: s_and_b32 vcc_lo, vcc_lo, s0
; GFX11-FAKE16-NEXT: v_cmp_eq_u16_e64 s0, 0x8000, v4
@@ -7129,8 +7106,8 @@ define <16 x bfloat> @v_minimumnum_v16bf16(<16 x bfloat> %x, <16 x bfloat> %y) #
; GFX11-FAKE16-NEXT: v_dual_cndmask_b32 v5, v13, v5 :: v_dual_lshlrev_b32 v14, 16, v12
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v11, v11, v3, s3
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v15, 16, v8
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v2, v2, v10, s2
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_4)
; GFX11-FAKE16-NEXT: v_perm_b32 v5, v17, v5, 0x5040100
; GFX11-FAKE16-NEXT: v_cmp_eq_f32_e32 vcc_lo, 0, v14
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v13, 16, v11
@@ -7144,7 +7121,6 @@ define <16 x bfloat> @v_minimumnum_v16bf16(<16 x bfloat> %x, <16 x bfloat> %y) #
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v14, 16, v9
; GFX11-FAKE16-NEXT: v_cndmask_b32_e32 v4, v12, v4, vcc_lo
; GFX11-FAKE16-NEXT: v_perm_b32 v6, v18, v6, 0x5040100
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(SKIP_3) | instid1(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v1, v1, v9, s2
; GFX11-FAKE16-NEXT: v_cmp_u_f32_e64 s2, v13, v13
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v13, 16, v10
@@ -7152,31 +7128,28 @@ define <16 x bfloat> @v_minimumnum_v16bf16(<16 x bfloat> %x, <16 x bfloat> %y) #
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v0, v0, v8, s2
; GFX11-FAKE16-NEXT: v_cmp_u_f32_e64 s2, v14, v14
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v14, 16, v1
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v24, 16, v0
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v9, v9, v1, s2
; GFX11-FAKE16-NEXT: v_cmp_u_f32_e64 s2, v15, v15
; GFX11-FAKE16-NEXT: v_cmp_eq_u16_e64 s4, 0x8000, v0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_2) | instid1(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v8, v8, v0, s2
; GFX11-FAKE16-NEXT: v_cmp_u_f32_e64 s2, v13, v13
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v13, 16, v9
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_1) | instid1(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v15, 16, v8
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v10, v10, v2, s2
; GFX11-FAKE16-NEXT: v_cmp_lt_f32_e64 s2, v14, v13
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_2) | instid1(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v25, 16, v10
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v9, v9, v1, s2
; GFX11-FAKE16-NEXT: v_cmp_lt_f32_e64 s2, v24, v15
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_cmp_lt_f32_e64 s3, v26, v25
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_3) | instid1(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v13, 16, v9
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_1) | instid1(VALU_DEP_4)
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v8, v8, v0, s2
; GFX11-FAKE16-NEXT: v_cmp_eq_u16_e64 s2, 0x8000, v3
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v10, v10, v2, s3
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v14, 16, v8
; GFX11-FAKE16-NEXT: s_and_b32 s1, s1, s2
; GFX11-FAKE16-NEXT: v_cmp_eq_u16_e64 s2, 0x8000, v1
@@ -7622,7 +7595,7 @@ define <16 x bfloat> @v_minimumnum_v16bf16(<16 x bfloat> %x, <16 x bfloat> %y) #
; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX12-FAKE16-NEXT: v_lshlrev_b32_e32 v22, 16, v19
; GFX12-FAKE16-NEXT: v_cmp_u_f32_e64 s1, v20, v20
-; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_3) | instid1(VALU_DEP_3)
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX12-FAKE16-NEXT: v_cmp_lt_f32_e64 s0, v21, v22
; GFX12-FAKE16-NEXT: v_lshrrev_b32_e32 v21, 16, v13
; GFX12-FAKE16-NEXT: v_lshrrev_b32_e32 v22, 16, v5
@@ -7646,10 +7619,10 @@ define <16 x bfloat> @v_minimumnum_v16bf16(<16 x bfloat> %x, <16 x bfloat> %y) #
; GFX12-FAKE16-NEXT: v_lshlrev_b32_e32 v25, 16, v17
; GFX12-FAKE16-NEXT: v_cmp_u_f32_e64 s0, v21, v21
; GFX12-FAKE16-NEXT: s_wait_alu depctr_va_sdst(0)
-; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_2) | instid1(VALU_DEP_2)
; GFX12-FAKE16-NEXT: v_cndmask_b32_e64 v21, v23, v22, s0
; GFX12-FAKE16-NEXT: v_lshlrev_b32_e32 v24, 16, v20
; GFX12-FAKE16-NEXT: v_cmp_eq_u16_e64 s0, 0x8000, v18
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX12-FAKE16-NEXT: v_cmp_lt_f32_e64 s1, v24, v25
; GFX12-FAKE16-NEXT: v_lshrrev_b32_e32 v24, 16, v11
; GFX12-FAKE16-NEXT: v_lshrrev_b32_e32 v25, 16, v3
@@ -7670,9 +7643,9 @@ define <16 x bfloat> @v_minimumnum_v16bf16(<16 x bfloat> %x, <16 x bfloat> %y) #
; GFX12-FAKE16-NEXT: v_lshlrev_b32_e32 v27, 16, v22
; GFX12-FAKE16-NEXT: v_cmp_u_f32_e64 s1, v23, v23
; GFX12-FAKE16-NEXT: s_wait_alu depctr_va_sdst(0)
-; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX12-FAKE16-NEXT: v_cndmask_b32_e64 v23, v25, v24, s1
; GFX12-FAKE16-NEXT: s_wait_alu depctr_va_vcc(0)
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-FAKE16-NEXT: v_cndmask_b32_e32 v24, v24, v23, vcc_lo
; GFX12-FAKE16-NEXT: v_cmp_lt_f32_e32 vcc_lo, v26, v27
; GFX12-FAKE16-NEXT: v_and_b32_e32 v27, 0xffff0000, v2
@@ -7713,18 +7686,18 @@ define <16 x bfloat> @v_minimumnum_v16bf16(<16 x bfloat> %x, <16 x bfloat> %y) #
; GFX12-FAKE16-NEXT: v_cndmask_b32_e32 v21, v22, v21, vcc_lo
; GFX12-FAKE16-NEXT: v_cmp_lt_f32_e64 s3, v20, v26
; GFX12-FAKE16-NEXT: s_wait_alu depctr_va_sdst(0)
-; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_3) | instid1(VALU_DEP_4)
; GFX12-FAKE16-NEXT: v_cndmask_b32_e64 v20, v24, v25, s3
; GFX12-FAKE16-NEXT: v_and_b32_e32 v24, 0xffff0000, v9
; GFX12-FAKE16-NEXT: v_cmp_u_f32_e64 s3, v27, v27
; GFX12-FAKE16-NEXT: v_and_b32_e32 v27, 0xffff0000, v0
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
; GFX12-FAKE16-NEXT: v_lshlrev_b32_e32 v22, 16, v20
-; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(SKIP_4) | instid1(VALU_DEP_2)
; GFX12-FAKE16-NEXT: v_cmp_u_f32_e32 vcc_lo, v24, v24
; GFX12-FAKE16-NEXT: s_wait_alu depctr_va_sdst(0)
; GFX12-FAKE16-NEXT: v_cndmask_b32_e64 v26, v29, v28, s3
; GFX12-FAKE16-NEXT: v_lshrrev_b32_e32 v29, 16, v0
; GFX12-FAKE16-NEXT: s_wait_alu depctr_va_vcc(0)
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX12-FAKE16-NEXT: v_cndmask_b32_e32 v24, v28, v26, vcc_lo
; GFX12-FAKE16-NEXT: s_and_b32 vcc_lo, s1, s2
; GFX12-FAKE16-NEXT: v_lshrrev_b32_e32 v28, 16, v8
@@ -7734,7 +7707,7 @@ define <16 x bfloat> @v_minimumnum_v16bf16(<16 x bfloat> %x, <16 x bfloat> %y) #
; GFX12-FAKE16-NEXT: v_lshlrev_b32_e32 v22, 16, v26
; GFX12-FAKE16-NEXT: v_lshlrev_b32_e32 v23, 16, v24
; GFX12-FAKE16-NEXT: s_and_b32 vcc_lo, vcc_lo, s0
-; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_3) | instid1(VALU_DEP_2)
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-FAKE16-NEXT: v_cmp_lt_f32_e64 s1, v22, v23
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-FAKE16-NEXT: v_dual_cndmask_b32 v20, v20, v25 :: v_dual_lshlrev_b32 v25, 16, v7
@@ -7742,18 +7715,17 @@ define <16 x bfloat> @v_minimumnum_v16bf16(<16 x bfloat> %x, <16 x bfloat> %y) #
; GFX12-FAKE16-NEXT: v_cndmask_b32_e64 v22, v24, v26, s1
; GFX12-FAKE16-NEXT: v_cmp_u_f32_e64 s1, v27, v27
; GFX12-FAKE16-NEXT: v_and_b32_e32 v24, 0xffff0000, v8
-; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_1) | instid1(VALU_DEP_3)
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_2) | instid1(VALU_DEP_3)
; GFX12-FAKE16-NEXT: v_lshlrev_b32_e32 v27, 16, v22
; GFX12-FAKE16-NEXT: s_wait_alu depctr_va_sdst(0)
; GFX12-FAKE16-NEXT: v_cndmask_b32_e64 v23, v29, v28, s1
-; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX12-FAKE16-NEXT: v_cmp_u_f32_e32 vcc_lo, v24, v24
; GFX12-FAKE16-NEXT: s_wait_alu depctr_va_vcc(0)
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_3) | instid1(VALU_DEP_3)
; GFX12-FAKE16-NEXT: v_dual_cndmask_b32 v24, v28, v23 :: v_dual_lshlrev_b32 v29, 16, v14
; GFX12-FAKE16-NEXT: v_lshlrev_b32_e32 v28, 16, v15
; GFX12-FAKE16-NEXT: v_cmp_u_f32_e32 vcc_lo, v25, v25
; GFX12-FAKE16-NEXT: v_lshlrev_b32_e32 v25, 16, v23
-; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX12-FAKE16-NEXT: v_cmp_u_f32_e64 s0, v28, v28
; GFX12-FAKE16-NEXT: s_wait_alu depctr_va_vcc(0)
; GFX12-FAKE16-NEXT: v_cndmask_b32_e32 v7, v7, v15, vcc_lo
@@ -7781,7 +7753,6 @@ define <16 x bfloat> @v_minimumnum_v16bf16(<16 x bfloat> %x, <16 x bfloat> %y) #
; GFX12-FAKE16-NEXT: v_cmp_eq_f32_e32 vcc_lo, 0, v28
; GFX12-FAKE16-NEXT: v_lshlrev_b32_e32 v27, 16, v6
; GFX12-FAKE16-NEXT: s_wait_alu depctr_va_sdst(0)
-; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX12-FAKE16-NEXT: v_cndmask_b32_e64 v15, v15, v7, s1
; GFX12-FAKE16-NEXT: v_cmp_u_f32_e64 s1, v29, v29
; GFX12-FAKE16-NEXT: s_and_b32 vcc_lo, vcc_lo, s0
@@ -7793,20 +7764,19 @@ define <16 x bfloat> @v_minimumnum_v16bf16(<16 x bfloat> %x, <16 x bfloat> %y) #
; GFX12-FAKE16-NEXT: v_cmp_eq_f32_e64 s1, 0, v25
; GFX12-FAKE16-NEXT: v_lshlrev_b32_e32 v26, 16, v14
; GFX12-FAKE16-NEXT: v_lshlrev_b32_e32 v25, 16, v5
-; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_2) | instid1(VALU_DEP_2)
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_4) | instid1(VALU_DEP_3)
; GFX12-FAKE16-NEXT: v_cmp_lt_f32_e64 s3, v27, v26
; GFX12-FAKE16-NEXT: v_lshlrev_b32_e32 v26, 16, v13
; GFX12-FAKE16-NEXT: s_wait_alu depctr_va_sdst(0)
; GFX12-FAKE16-NEXT: v_cndmask_b32_e64 v14, v14, v6, s3
; GFX12-FAKE16-NEXT: v_cmp_u_f32_e64 s3, v25, v25
-; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_1) | instid1(VALU_DEP_4)
; GFX12-FAKE16-NEXT: v_cmp_u_f32_e32 vcc_lo, v26, v26
; GFX12-FAKE16-NEXT: v_lshlrev_b32_e32 v25, 16, v3
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(SKIP_3) | instid1(VALU_DEP_1)
; GFX12-FAKE16-NEXT: v_lshlrev_b32_e32 v24, 16, v14
; GFX12-FAKE16-NEXT: s_wait_alu depctr_va_sdst(0)
; GFX12-FAKE16-NEXT: v_cndmask_b32_e64 v5, v5, v13, s3
; GFX12-FAKE16-NEXT: s_wait_alu depctr_va_vcc(0)
-; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-FAKE16-NEXT: v_cndmask_b32_e32 v13, v13, v5, vcc_lo
; GFX12-FAKE16-NEXT: s_and_b32 vcc_lo, s1, s2
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
@@ -7819,19 +7789,17 @@ define <16 x bfloat> @v_minimumnum_v16bf16(<16 x bfloat> %x, <16 x bfloat> %y) #
; GFX12-FAKE16-NEXT: v_cmp_u_f32_e64 s0, v15, v15
; GFX12-FAKE16-NEXT: v_lshlrev_b32_e32 v15, 16, v12
; GFX12-FAKE16-NEXT: s_wait_alu depctr_va_sdst(0)
-; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_3) | instid1(VALU_DEP_2)
; GFX12-FAKE16-NEXT: v_cndmask_b32_e64 v4, v4, v12, s0
; GFX12-FAKE16-NEXT: v_cmp_u_f32_e64 s0, v25, v25
; GFX12-FAKE16-NEXT: v_lshlrev_b32_e32 v25, 16, v11
; GFX12-FAKE16-NEXT: s_wait_alu depctr_va_sdst(0)
; GFX12-FAKE16-NEXT: v_cndmask_b32_e64 v3, v3, v11, s0
; GFX12-FAKE16-NEXT: v_cmp_lt_f32_e64 s0, v26, v24
-; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_2)
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_3) | instid1(VALU_DEP_2)
; GFX12-FAKE16-NEXT: v_lshlrev_b32_e32 v27, 16, v3
; GFX12-FAKE16-NEXT: s_wait_alu depctr_va_sdst(0)
; GFX12-FAKE16-NEXT: v_cndmask_b32_e64 v13, v13, v5, s0
; GFX12-FAKE16-NEXT: v_cmp_u_f32_e64 s0, v15, v15
-; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX12-FAKE16-NEXT: v_lshlrev_b32_e32 v15, 16, v13
; GFX12-FAKE16-NEXT: s_wait_alu depctr_va_sdst(0)
; GFX12-FAKE16-NEXT: v_cndmask_b32_e64 v12, v12, v4, s0
@@ -7881,13 +7849,12 @@ define <16 x bfloat> @v_minimumnum_v16bf16(<16 x bfloat> %x, <16 x bfloat> %y) #
; GFX12-FAKE16-NEXT: v_lshlrev_b32_e32 v13, 16, v10
; GFX12-FAKE16-NEXT: v_perm_b32 v4, v21, v4, 0x5040100
; GFX12-FAKE16-NEXT: s_wait_alu depctr_va_sdst(0)
-; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_2) | instid1(VALU_DEP_3)
; GFX12-FAKE16-NEXT: v_cndmask_b32_e64 v0, v0, v8, s2
; GFX12-FAKE16-NEXT: v_cmp_u_f32_e64 s2, v14, v14
; GFX12-FAKE16-NEXT: v_lshlrev_b32_e32 v14, 16, v1
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX12-FAKE16-NEXT: v_lshlrev_b32_e32 v24, 16, v0
; GFX12-FAKE16-NEXT: s_wait_alu depctr_va_sdst(0)
-; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_3) | instid1(VALU_DEP_2)
; GFX12-FAKE16-NEXT: v_cndmask_b32_e64 v9, v9, v1, s2
; GFX12-FAKE16-NEXT: v_cmp_u_f32_e64 s2, v15, v15
; GFX12-FAKE16-NEXT: v_cmp_eq_u16_e64 s4, 0x8000, v0
@@ -7895,25 +7862,23 @@ define <16 x bfloat> @v_minimumnum_v16bf16(<16 x bfloat> %x, <16 x bfloat> %y) #
; GFX12-FAKE16-NEXT: v_cndmask_b32_e64 v8, v8, v0, s2
; GFX12-FAKE16-NEXT: v_cmp_u_f32_e64 s2, v13, v13
; GFX12-FAKE16-NEXT: v_lshlrev_b32_e32 v13, 16, v9
-; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_1) | instid1(VALU_DEP_3)
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_2) | instid1(VALU_DEP_3)
; GFX12-FAKE16-NEXT: v_lshlrev_b32_e32 v15, 16, v8
; GFX12-FAKE16-NEXT: s_wait_alu depctr_va_sdst(0)
; GFX12-FAKE16-NEXT: v_cndmask_b32_e64 v10, v10, v2, s2
-; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX12-FAKE16-NEXT: v_cmp_lt_f32_e64 s2, v14, v13
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_3) | instid1(VALU_DEP_3)
; GFX12-FAKE16-NEXT: v_lshlrev_b32_e32 v25, 16, v10
; GFX12-FAKE16-NEXT: s_wait_alu depctr_va_sdst(0)
-; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_3)
; GFX12-FAKE16-NEXT: v_cndmask_b32_e64 v9, v9, v1, s2
; GFX12-FAKE16-NEXT: v_cmp_lt_f32_e64 s2, v24, v15
; GFX12-FAKE16-NEXT: v_cmp_lt_f32_e64 s3, v26, v25
-; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_1) | instid1(VALU_DEP_3)
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_4) | instid1(VALU_DEP_3)
; GFX12-FAKE16-NEXT: v_lshlrev_b32_e32 v13, 16, v9
; GFX12-FAKE16-NEXT: s_wait_alu depctr_va_sdst(0)
; GFX12-FAKE16-NEXT: v_cndmask_b32_e64 v8, v8, v0, s2
; GFX12-FAKE16-NEXT: v_cmp_eq_u16_e64 s2, 0x8000, v3
; GFX12-FAKE16-NEXT: v_cndmask_b32_e64 v10, v10, v2, s3
-; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX12-FAKE16-NEXT: v_lshlrev_b32_e32 v14, 16, v8
; GFX12-FAKE16-NEXT: s_and_b32 s1, s1, s2
; GFX12-FAKE16-NEXT: v_cmp_eq_u16_e64 s2, 0x8000, v1
@@ -10734,11 +10699,10 @@ define <32 x bfloat> @v_minimumnum_v32bf16(<32 x bfloat> %x, <32 x bfloat> %y) #
; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v35.l, v31.l
; GFX11-TRUE16-NEXT: v_cmp_lt_f32_e64 s0, v96, v97
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_3)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_2) | instid1(VALU_DEP_3)
; GFX11-TRUE16-NEXT: v_lshlrev_b32_e32 v39, 16, v35
; GFX11-TRUE16-NEXT: v_cndmask_b16 v35.h, v80.l, v50.l, s8
; GFX11-TRUE16-NEXT: v_cndmask_b16 v33.l, v87.l, v32.l, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX11-TRUE16-NEXT: v_cmp_lt_f32_e32 vcc_lo, v36, v39
; GFX11-TRUE16-NEXT: v_cmp_eq_u16_e64 s0, 0x8000, v32.l
; GFX11-TRUE16-NEXT: v_cndmask_b16 v36.h, v81.l, v51.l, s9
@@ -11329,13 +11293,14 @@ define <32 x bfloat> @v_minimumnum_v32bf16(<32 x bfloat> %x, <32 x bfloat> %y) #
; GFX11-FAKE16-NEXT: v_and_b32_e32 v66, 0xffff0000, v31
; GFX11-FAKE16-NEXT: v_cndmask_b32_e32 v32, v33, v65, vcc_lo
; GFX11-FAKE16-NEXT: s_and_b32 vcc_lo, s24, s8
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_cndmask_b32_e32 v33, v54, v112, vcc_lo
; GFX11-FAKE16-NEXT: v_cmp_u_f32_e32 vcc_lo, v50, v50
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_3) | instid1(VALU_DEP_2)
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v67, 16, v32
; GFX11-FAKE16-NEXT: v_cmp_eq_u16_e64 s1, 0x8000, v32
; GFX11-FAKE16-NEXT: v_cndmask_b32_e32 v31, v31, v15, vcc_lo
; GFX11-FAKE16-NEXT: v_cmp_u_f32_e32 vcc_lo, v66, v66
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v66, 16, v31
; GFX11-FAKE16-NEXT: v_dual_cndmask_b32 v50, v65, v32 :: v_dual_lshlrev_b32 v65, 16, v15
; GFX11-FAKE16-NEXT: s_and_b32 vcc_lo, s25, s9
@@ -11347,12 +11312,13 @@ define <32 x bfloat> @v_minimumnum_v32bf16(<32 x bfloat> %x, <32 x bfloat> %y) #
; GFX11-FAKE16-NEXT: v_cmp_lt_f32_e32 vcc_lo, v65, v66
; GFX11-FAKE16-NEXT: v_cndmask_b32_e32 v31, v31, v15, vcc_lo
; GFX11-FAKE16-NEXT: v_cmp_lt_f32_e32 vcc_lo, v67, v68
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_4) | instid1(VALU_DEP_2)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v66, 16, v31
; GFX11-FAKE16-NEXT: v_cndmask_b32_e32 v50, v50, v32, vcc_lo
; GFX11-FAKE16-NEXT: s_and_b32 vcc_lo, s27, s11
; GFX11-FAKE16-NEXT: v_cndmask_b32_e32 v65, v83, v128, vcc_lo
; GFX11-FAKE16-NEXT: s_and_b32 vcc_lo, s28, s12
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v67, 16, v50
; GFX11-FAKE16-NEXT: v_cndmask_b32_e32 v49, v49, v130, vcc_lo
; GFX11-FAKE16-NEXT: s_and_b32 vcc_lo, s29, s13
@@ -11394,7 +11360,7 @@ define <32 x bfloat> @v_minimumnum_v32bf16(<32 x bfloat> %x, <32 x bfloat> %y) #
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v29, 16, v9
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v32, 16, v26
; GFX11-FAKE16-NEXT: s_and_b32 vcc_lo, vcc_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_3) | instid1(VALU_DEP_4)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX11-FAKE16-NEXT: v_cmp_u_f32_e64 s1, v29, v29
; GFX11-FAKE16-NEXT: v_cndmask_b32_e32 v12, v28, v12, vcc_lo
; GFX11-FAKE16-NEXT: v_cmp_eq_u16_e64 s0, 0x8000, v10
@@ -11402,19 +11368,18 @@ define <32 x bfloat> @v_minimumnum_v32bf16(<32 x bfloat> %x, <32 x bfloat> %y) #
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v9, v9, v25, s1
; GFX11-FAKE16-NEXT: v_cmp_eq_f32_e64 s1, 0, v31
; GFX11-FAKE16-NEXT: v_perm_b32 v12, v39, v12, 0x5040100
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(SKIP_2) | instid1(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v26, v26, v10, s2
; GFX11-FAKE16-NEXT: v_cmp_u_f32_e64 s2, v50, v50
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v31, 16, v9
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_2) | instid1(VALU_DEP_2)
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v28, 16, v26
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v25, v25, v9, s2
; GFX11-FAKE16-NEXT: v_cmp_eq_u16_e64 s2, 0x8000, v11
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v29, 16, v25
; GFX11-FAKE16-NEXT: s_and_b32 vcc_lo, s1, s2
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_cndmask_b32_e32 v11, v27, v11, vcc_lo
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v27, 16, v8
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_3) | instid1(VALU_DEP_4)
; GFX11-FAKE16-NEXT: v_cmp_lt_f32_e64 s1, v31, v29
; GFX11-FAKE16-NEXT: v_cmp_eq_f32_e32 vcc_lo, 0, v28
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v29, 16, v22
@@ -11423,6 +11388,7 @@ define <32 x bfloat> @v_minimumnum_v32bf16(<32 x bfloat> %x, <32 x bfloat> %y) #
; GFX11-FAKE16-NEXT: v_cmp_u_f32_e64 s1, v27, v27
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v27, 16, v24
; GFX11-FAKE16-NEXT: s_and_b32 vcc_lo, vcc_lo, s0
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_cndmask_b32_e32 v10, v26, v10, vcc_lo
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v8, v8, v24, s1
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v26, 16, v7
@@ -11450,15 +11416,14 @@ define <32 x bfloat> @v_minimumnum_v32bf16(<32 x bfloat> %x, <32 x bfloat> %y) #
; GFX11-FAKE16-NEXT: v_cmp_u_f32_e64 s1, v27, v27
; GFX11-FAKE16-NEXT: v_cndmask_b32_e32 v9, v25, v9, vcc_lo
; GFX11-FAKE16-NEXT: v_cmp_eq_u16_e64 s0, 0x8000, v8
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(SKIP_3) | instid1(VALU_DEP_4)
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v27, 16, v24
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v6, v6, v22, s1
; GFX11-FAKE16-NEXT: v_cmp_lt_f32_e64 s1, v28, v26
; GFX11-FAKE16-NEXT: v_perm_b32 v9, v52, v9, 0x5040100
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
; GFX11-FAKE16-NEXT: v_cmp_eq_f32_e32 vcc_lo, 0, v27
-; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v27, 16, v6
; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_4)
+; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v27, 16, v6
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v23, v23, v7, s1
; GFX11-FAKE16-NEXT: v_cmp_u_f32_e64 s1, v29, v29
; GFX11-FAKE16-NEXT: s_and_b32 vcc_lo, vcc_lo, s0
@@ -11473,15 +11438,14 @@ define <32 x bfloat> @v_minimumnum_v32bf16(<32 x bfloat> %x, <32 x bfloat> %y) #
; GFX11-FAKE16-NEXT: v_perm_b32 v8, v53, v8, 0x5040100
; GFX11-FAKE16-NEXT: v_cmp_lt_f32_e64 s3, v27, v26
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v26, 16, v21
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v22, v22, v6, s3
; GFX11-FAKE16-NEXT: v_cmp_u_f32_e64 s3, v25, v25
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_1) | instid1(VALU_DEP_4)
; GFX11-FAKE16-NEXT: v_cmp_u_f32_e32 vcc_lo, v26, v26
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v25, 16, v3
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v24, 16, v22
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v5, v5, v21, s3
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_cndmask_b32_e32 v21, v21, v5, vcc_lo
; GFX11-FAKE16-NEXT: s_and_b32 vcc_lo, s1, s2
; GFX11-FAKE16-NEXT: v_dual_cndmask_b32 v7, v23, v7 :: v_dual_lshlrev_b32 v26, 16, v5
@@ -11492,17 +11456,15 @@ define <32 x bfloat> @v_minimumnum_v32bf16(<32 x bfloat> %x, <32 x bfloat> %y) #
; GFX11-FAKE16-NEXT: v_perm_b32 v7, v55, v7, 0x5040100
; GFX11-FAKE16-NEXT: v_cmp_u_f32_e64 s0, v23, v23
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v23, 16, v20
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_2) | instid1(VALU_DEP_2)
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v4, v4, v20, s0
; GFX11-FAKE16-NEXT: v_cmp_u_f32_e64 s0, v25, v25
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v25, 16, v19
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v3, v3, v19, s0
; GFX11-FAKE16-NEXT: v_cmp_lt_f32_e64 s0, v26, v24
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_2) | instid1(VALU_DEP_2)
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v27, 16, v3
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v21, v21, v5, s0
; GFX11-FAKE16-NEXT: v_cmp_u_f32_e64 s0, v23, v23
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v23, 16, v21
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v20, v20, v4, s0
; GFX11-FAKE16-NEXT: v_cmp_u_f32_e64 s0, v25, v25
@@ -11510,12 +11472,11 @@ define <32 x bfloat> @v_minimumnum_v32bf16(<32 x bfloat> %x, <32 x bfloat> %y) #
; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
; GFX11-FAKE16-NEXT: v_cmp_eq_f32_e64 s1, 0, v23
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v24, 16, v20
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(SKIP_2) | instid1(VALU_DEP_4)
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v19, v19, v3, s0
; GFX11-FAKE16-NEXT: v_cmp_eq_u16_e64 s0, 0x8000, v6
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v23, 16, v2
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
; GFX11-FAKE16-NEXT: v_cmp_lt_f32_e64 s3, v25, v24
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_4)
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v26, 16, v19
; GFX11-FAKE16-NEXT: s_and_b32 vcc_lo, vcc_lo, s0
; GFX11-FAKE16-NEXT: v_cmp_eq_u16_e64 s0, 0x8000, v4
@@ -11527,8 +11488,8 @@ define <32 x bfloat> @v_minimumnum_v32bf16(<32 x bfloat> %x, <32 x bfloat> %y) #
; GFX11-FAKE16-NEXT: v_dual_cndmask_b32 v5, v21, v5 :: v_dual_lshlrev_b32 v22, 16, v20
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v19, v19, v3, s3
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v23, 16, v16
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v2, v2, v18, s2
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_4)
; GFX11-FAKE16-NEXT: v_perm_b32 v5, v33, v5, 0x5040100
; GFX11-FAKE16-NEXT: v_cmp_eq_f32_e32 vcc_lo, 0, v22
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v21, 16, v19
@@ -11542,7 +11503,6 @@ define <32 x bfloat> @v_minimumnum_v32bf16(<32 x bfloat> %x, <32 x bfloat> %y) #
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v22, 16, v17
; GFX11-FAKE16-NEXT: v_cndmask_b32_e32 v4, v20, v4, vcc_lo
; GFX11-FAKE16-NEXT: v_perm_b32 v6, v64, v6, 0x5040100
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(SKIP_3) | instid1(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v1, v1, v17, s2
; GFX11-FAKE16-NEXT: v_cmp_u_f32_e64 s2, v21, v21
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v21, 16, v18
@@ -11550,31 +11510,28 @@ define <32 x bfloat> @v_minimumnum_v32bf16(<32 x bfloat> %x, <32 x bfloat> %y) #
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v0, v0, v16, s2
; GFX11-FAKE16-NEXT: v_cmp_u_f32_e64 s2, v22, v22
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v22, 16, v1
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v24, 16, v0
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v17, v17, v1, s2
; GFX11-FAKE16-NEXT: v_cmp_u_f32_e64 s2, v23, v23
; GFX11-FAKE16-NEXT: v_cmp_eq_u16_e64 s4, 0x8000, v0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_2) | instid1(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v16, v16, v0, s2
; GFX11-FAKE16-NEXT: v_cmp_u_f32_e64 s2, v21, v21
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v21, 16, v17
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_1) | instid1(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v23, 16, v16
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v18, v18, v2, s2
; GFX11-FAKE16-NEXT: v_cmp_lt_f32_e64 s2, v22, v21
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_2) | instid1(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v25, 16, v18
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v17, v17, v1, s2
; GFX11-FAKE16-NEXT: v_cmp_lt_f32_e64 s2, v24, v23
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_cmp_lt_f32_e64 s3, v26, v25
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_3) | instid1(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v21, 16, v17
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_1) | instid1(VALU_DEP_4)
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v16, v16, v0, s2
; GFX11-FAKE16-NEXT: v_cmp_eq_u16_e64 s2, 0x8000, v3
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v18, v18, v2, s3
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v22, 16, v16
; GFX11-FAKE16-NEXT: s_and_b32 s1, s1, s2
; GFX11-FAKE16-NEXT: v_cmp_eq_u16_e64 s2, 0x8000, v1
@@ -11963,8 +11920,8 @@ define <32 x bfloat> @v_minimumnum_v32bf16(<32 x bfloat> %x, <32 x bfloat> %y) #
; GFX12-TRUE16-NEXT: v_lshlrev_b32_e32 v39, 16, v35
; GFX12-TRUE16-NEXT: v_cndmask_b16 v35.h, v80.l, v50.l, s8
; GFX12-TRUE16-NEXT: s_wait_alu depctr_va_sdst(0)
-; GFX12-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_3)
; GFX12-TRUE16-NEXT: v_cndmask_b16 v33.l, v87.l, v32.l, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX12-TRUE16-NEXT: v_cmp_lt_f32_e32 vcc_lo, v36, v39
; GFX12-TRUE16-NEXT: v_cmp_eq_u16_e64 s0, 0x8000, v32.l
; GFX12-TRUE16-NEXT: v_cndmask_b16 v36.h, v81.l, v51.l, s9
@@ -12738,12 +12695,11 @@ define <32 x bfloat> @v_minimumnum_v32bf16(<32 x bfloat> %x, <32 x bfloat> %y) #
; GFX12-FAKE16-NEXT: v_cndmask_b32_e64 v26, v26, v10, s2
; GFX12-FAKE16-NEXT: v_cmp_u_f32_e64 s2, v50, v50
; GFX12-FAKE16-NEXT: v_lshlrev_b32_e32 v31, 16, v9
-; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_1) | instid1(VALU_DEP_3)
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_3) | instid1(VALU_DEP_2)
; GFX12-FAKE16-NEXT: v_lshlrev_b32_e32 v28, 16, v26
; GFX12-FAKE16-NEXT: s_wait_alu depctr_va_sdst(0)
; GFX12-FAKE16-NEXT: v_cndmask_b32_e64 v25, v25, v9, s2
; GFX12-FAKE16-NEXT: v_cmp_eq_u16_e64 s2, 0x8000, v11
-; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX12-FAKE16-NEXT: v_lshlrev_b32_e32 v29, 16, v25
; GFX12-FAKE16-NEXT: s_and_b32 vcc_lo, s1, s2
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
@@ -12814,16 +12770,16 @@ define <32 x bfloat> @v_minimumnum_v32bf16(<32 x bfloat> %x, <32 x bfloat> %y) #
; GFX12-FAKE16-NEXT: v_cmp_lt_f32_e64 s3, v27, v26
; GFX12-FAKE16-NEXT: v_lshlrev_b32_e32 v26, 16, v21
; GFX12-FAKE16-NEXT: s_wait_alu depctr_va_sdst(0)
-; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_3)
; GFX12-FAKE16-NEXT: v_cndmask_b32_e64 v22, v22, v6, s3
; GFX12-FAKE16-NEXT: v_cmp_u_f32_e64 s3, v25, v25
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_1) | instid1(VALU_DEP_4)
; GFX12-FAKE16-NEXT: v_cmp_u_f32_e32 vcc_lo, v26, v26
; GFX12-FAKE16-NEXT: v_lshlrev_b32_e32 v25, 16, v3
-; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(SKIP_3) | instid1(VALU_DEP_1)
; GFX12-FAKE16-NEXT: v_lshlrev_b32_e32 v24, 16, v22
; GFX12-FAKE16-NEXT: s_wait_alu depctr_va_sdst(0)
; GFX12-FAKE16-NEXT: v_cndmask_b32_e64 v5, v5, v21, s3
; GFX12-FAKE16-NEXT: s_wait_alu depctr_va_vcc(0)
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-FAKE16-NEXT: v_cndmask_b32_e32 v21, v21, v5, vcc_lo
; GFX12-FAKE16-NEXT: s_and_b32 vcc_lo, s1, s2
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
@@ -12836,19 +12792,17 @@ define <32 x bfloat> @v_minimumnum_v32bf16(<32 x bfloat> %x, <32 x bfloat> %y) #
; GFX12-FAKE16-NEXT: v_cmp_u_f32_e64 s0, v23, v23
; GFX12-FAKE16-NEXT: v_lshlrev_b32_e32 v23, 16, v20
; GFX12-FAKE16-NEXT: s_wait_alu depctr_va_sdst(0)
-; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_3) | instid1(VALU_DEP_2)
; GFX12-FAKE16-NEXT: v_cndmask_b32_e64 v4, v4, v20, s0
; GFX12-FAKE16-NEXT: v_cmp_u_f32_e64 s0, v25, v25
; GFX12-FAKE16-NEXT: v_lshlrev_b32_e32 v25, 16, v19
; GFX12-FAKE16-NEXT: s_wait_alu depctr_va_sdst(0)
; GFX12-FAKE16-NEXT: v_cndmask_b32_e64 v3, v3, v19, s0
; GFX12-FAKE16-NEXT: v_cmp_lt_f32_e64 s0, v26, v24
-; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_2)
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_3) | instid1(VALU_DEP_2)
; GFX12-FAKE16-NEXT: v_lshlrev_b32_e32 v27, 16, v3
; GFX12-FAKE16-NEXT: s_wait_alu depctr_va_sdst(0)
; GFX12-FAKE16-NEXT: v_cndmask_b32_e64 v21, v21, v5, s0
; GFX12-FAKE16-NEXT: v_cmp_u_f32_e64 s0, v23, v23
-; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX12-FAKE16-NEXT: v_lshlrev_b32_e32 v23, 16, v21
; GFX12-FAKE16-NEXT: s_wait_alu depctr_va_sdst(0)
; GFX12-FAKE16-NEXT: v_cndmask_b32_e64 v20, v20, v4, s0
@@ -12898,13 +12852,12 @@ define <32 x bfloat> @v_minimumnum_v32bf16(<32 x bfloat> %x, <32 x bfloat> %y) #
; GFX12-FAKE16-NEXT: v_lshlrev_b32_e32 v21, 16, v18
; GFX12-FAKE16-NEXT: v_perm_b32 v4, v54, v4, 0x5040100
; GFX12-FAKE16-NEXT: s_wait_alu depctr_va_sdst(0)
-; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_2) | instid1(VALU_DEP_3)
; GFX12-FAKE16-NEXT: v_cndmask_b32_e64 v0, v0, v16, s2
; GFX12-FAKE16-NEXT: v_cmp_u_f32_e64 s2, v22, v22
; GFX12-FAKE16-NEXT: v_lshlrev_b32_e32 v22, 16, v1
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX12-FAKE16-NEXT: v_lshlrev_b32_e32 v24, 16, v0
; GFX12-FAKE16-NEXT: s_wait_alu depctr_va_sdst(0)
-; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_3) | instid1(VALU_DEP_2)
; GFX12-FAKE16-NEXT: v_cndmask_b32_e64 v17, v17, v1, s2
; GFX12-FAKE16-NEXT: v_cmp_u_f32_e64 s2, v23, v23
; GFX12-FAKE16-NEXT: v_cmp_eq_u16_e64 s4, 0x8000, v0
@@ -12912,25 +12865,23 @@ define <32 x bfloat> @v_minimumnum_v32bf16(<32 x bfloat> %x, <32 x bfloat> %y) #
; GFX12-FAKE16-NEXT: v_cndmask_b32_e64 v16, v16, v0, s2
; GFX12-FAKE16-NEXT: v_cmp_u_f32_e64 s2, v21, v21
; GFX12-FAKE16-NEXT: v_lshlrev_b32_e32 v21, 16, v17
-; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_1) | instid1(VALU_DEP_3)
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_2) | instid1(VALU_DEP_3)
; GFX12-FAKE16-NEXT: v_lshlrev_b32_e32 v23, 16, v16
; GFX12-FAKE16-NEXT: s_wait_alu depctr_va_sdst(0)
; GFX12-FAKE16-NEXT: v_cndmask_b32_e64 v18, v18, v2, s2
-; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX12-FAKE16-NEXT: v_cmp_lt_f32_e64 s2, v22, v21
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_3) | instid1(VALU_DEP_3)
; GFX12-FAKE16-NEXT: v_lshlrev_b32_e32 v25, 16, v18
; GFX12-FAKE16-NEXT: s_wait_alu depctr_va_sdst(0)
-; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_3)
; GFX12-FAKE16-NEXT: v_cndmask_b32_e64 v17, v17, v1, s2
; GFX12-FAKE16-NEXT: v_cmp_lt_f32_e64 s2, v24, v23
; GFX12-FAKE16-NEXT: v_cmp_lt_f32_e64 s3, v26, v25
-; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_1) | instid1(VALU_DEP_3)
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_4) | instid1(VALU_DEP_3)
; GFX12-FAKE16-NEXT: v_lshlrev_b32_e32 v21, 16, v17
; GFX12-FAKE16-NEXT: s_wait_alu depctr_va_sdst(0)
; GFX12-FAKE16-NEXT: v_cndmask_b32_e64 v16, v16, v0, s2
; GFX12-FAKE16-NEXT: v_cmp_eq_u16_e64 s2, 0x8000, v3
; GFX12-FAKE16-NEXT: v_cndmask_b32_e64 v18, v18, v2, s3
-; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX12-FAKE16-NEXT: v_lshlrev_b32_e32 v22, 16, v16
; GFX12-FAKE16-NEXT: s_and_b32 s1, s1, s2
; GFX12-FAKE16-NEXT: v_cmp_eq_u16_e64 s2, 0x8000, v1
@@ -13103,7 +13054,7 @@ define bfloat @v_minimumnum_bf16_no_ieee(bfloat %x, bfloat %y) #0 {
; GFX11-FAKE16-NEXT: v_cndmask_b32_e32 v1, v1, v0, vcc_lo
; GFX11-FAKE16-NEXT: v_cmp_eq_u16_e32 vcc_lo, 0x8000, v0
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v2, 16, v1
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_cmp_eq_f32_e64 s0, 0, v2
; GFX11-FAKE16-NEXT: s_and_b32 vcc_lo, s0, vcc_lo
; GFX11-FAKE16-NEXT: v_cndmask_b32_e32 v0, v1, v0, vcc_lo
@@ -13366,24 +13317,23 @@ define <2 x bfloat> @v_minimumnum_v2bf16_no_ieee(<2 x bfloat> %x, <2 x bfloat> %
; GFX11-TRUE16-NEXT: v_cmp_u_f32_e64 s1, v6, v6
; GFX11-TRUE16-NEXT: v_cmp_u_f32_e64 s2, v7, v7
; GFX11-TRUE16-NEXT: v_cndmask_b16 v2.l, v5.l, v3.l, vcc_lo
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-TRUE16-NEXT: v_cndmask_b16 v0.l, v0.l, v1.l, s0
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-TRUE16-NEXT: v_cndmask_b16 v3.l, v3.l, v2.l, s1
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_2) | instid1(VALU_DEP_4)
; GFX11-TRUE16-NEXT: v_cndmask_b16 v1.l, v1.l, v0.l, s2
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v4.l, v2.l
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v5.l, v0.l
-; GFX11-TRUE16-NEXT: v_mov_b16_e32 v6.l, v3.l
; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
+; GFX11-TRUE16-NEXT: v_mov_b16_e32 v6.l, v3.l
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v7.l, v1.l
-; GFX11-TRUE16-NEXT: v_lshlrev_b32_e32 v4, 16, v4
; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
+; GFX11-TRUE16-NEXT: v_lshlrev_b32_e32 v4, 16, v4
; GFX11-TRUE16-NEXT: v_lshlrev_b32_e32 v5, 16, v5
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
; GFX11-TRUE16-NEXT: v_lshlrev_b32_e32 v6, 16, v6
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-TRUE16-NEXT: v_lshlrev_b32_e32 v7, 16, v7
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-TRUE16-NEXT: v_cmp_lt_f32_e32 vcc_lo, v4, v6
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX11-TRUE16-NEXT: v_cmp_lt_f32_e64 s0, v5, v7
; GFX11-TRUE16-NEXT: v_cndmask_b16 v3.l, v3.l, v2.l, vcc_lo
; GFX11-TRUE16-NEXT: v_cndmask_b16 v1.l, v1.l, v0.l, s0
@@ -13442,8 +13392,8 @@ define <2 x bfloat> @v_minimumnum_v2bf16_no_ieee(<2 x bfloat> %x, <2 x bfloat> %
; GFX11-FAKE16-NEXT: v_cmp_eq_f32_e64 s2, 0, v5
; GFX11-FAKE16-NEXT: v_cndmask_b32_e32 v2, v3, v2, vcc_lo
; GFX11-FAKE16-NEXT: s_and_b32 vcc_lo, s2, s1
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_cndmask_b32_e32 v0, v1, v0, vcc_lo
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_perm_b32 v0, v2, v0, 0x5040100
; GFX11-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -13489,17 +13439,17 @@ define <2 x bfloat> @v_minimumnum_v2bf16_no_ieee(<2 x bfloat> %x, <2 x bfloat> %
; GFX12-TRUE16-NEXT: s_wait_alu depctr_va_vcc(0)
; GFX12-TRUE16-NEXT: v_cndmask_b16 v3.l, v3.l, v2.l, vcc_lo
; GFX12-TRUE16-NEXT: s_wait_alu depctr_va_sdst(0)
-; GFX12-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_3) | instid1(VALU_DEP_4)
; GFX12-TRUE16-NEXT: v_cndmask_b16 v1.l, v1.l, v0.l, s0
; GFX12-TRUE16-NEXT: v_cmp_eq_u16_e32 vcc_lo, 0x8000, v2.l
; GFX12-TRUE16-NEXT: v_cmp_eq_u16_e64 s0, 0x8000, v0.l
; GFX12-TRUE16-NEXT: v_mov_b16_e32 v4.l, v3.l
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX12-TRUE16-NEXT: v_mov_b16_e32 v5.l, v1.l
-; GFX12-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX12-TRUE16-NEXT: v_lshlrev_b32_e32 v4, 16, v4
-; GFX12-TRUE16-NEXT: v_lshlrev_b32_e32 v5, 16, v5
; GFX12-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX12-TRUE16-NEXT: v_lshlrev_b32_e32 v5, 16, v5
; GFX12-TRUE16-NEXT: v_cmp_eq_f32_e64 s1, 0, v4
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX12-TRUE16-NEXT: v_cmp_eq_f32_e64 s2, 0, v5
; GFX12-TRUE16-NEXT: s_and_b32 s1, s1, vcc_lo
; GFX12-TRUE16-NEXT: s_and_b32 s0, s2, s0
@@ -13845,7 +13795,6 @@ define <3 x bfloat> @v_minimumnum_v3bf16_no_ieee(<3 x bfloat> %x, <3 x bfloat> %
; GFX11-TRUE16-NEXT: v_cmp_lt_f32_e64 s1, v11, v10
; GFX11-TRUE16-NEXT: v_cndmask_b16 v5.l, v5.l, v4.l, vcc_lo
; GFX11-TRUE16-NEXT: v_cmp_eq_u16_e32 vcc_lo, 0x8000, v4.l
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_4)
; GFX11-TRUE16-NEXT: v_cndmask_b16 v3.l, v3.l, v1.l, s0
; GFX11-TRUE16-NEXT: v_cmp_eq_u16_e64 s0, 0x8000, v1.l
; GFX11-TRUE16-NEXT: v_cndmask_b16 v2.l, v2.l, v0.l, s1
@@ -13913,19 +13862,20 @@ define <3 x bfloat> @v_minimumnum_v3bf16_no_ieee(<3 x bfloat> %x, <3 x bfloat> %
; GFX11-FAKE16-NEXT: v_cmp_lt_f32_e32 vcc_lo, v8, v7
; GFX11-FAKE16-NEXT: v_cndmask_b32_e32 v3, v3, v1, vcc_lo
; GFX11-FAKE16-NEXT: v_cmp_eq_f32_e32 vcc_lo, 0, v6
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_2) | instid1(VALU_DEP_2)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v6, 16, v3
; GFX11-FAKE16-NEXT: s_and_b32 vcc_lo, vcc_lo, s0
; GFX11-FAKE16-NEXT: v_dual_cndmask_b32 v4, v5, v4 :: v_dual_lshlrev_b32 v7, 16, v2
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-FAKE16-NEXT: v_cmp_eq_f32_e32 vcc_lo, 0, v6
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_cmp_eq_f32_e64 s1, 0, v7
; GFX11-FAKE16-NEXT: s_and_b32 s0, s1, s2
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v0, v2, v0, s0
; GFX11-FAKE16-NEXT: v_cmp_eq_u16_e64 s0, 0x8000, v1
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX11-FAKE16-NEXT: v_perm_b32 v0, v4, v0, 0x5040100
; GFX11-FAKE16-NEXT: s_and_b32 vcc_lo, vcc_lo, s0
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_cndmask_b32_e32 v1, v3, v1, vcc_lo
; GFX11-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -14531,28 +14481,27 @@ define <4 x bfloat> @v_minimumnum_v4bf16_no_ieee(<4 x bfloat> %x, <4 x bfloat> %
; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
; GFX11-FAKE16-NEXT: v_cmp_lt_f32_e64 s1, v13, v12
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v10, 16, v2
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_2) | instid1(VALU_DEP_4)
; GFX11-FAKE16-NEXT: v_cmp_lt_f32_e64 s0, v8, v9
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v3, v3, v1, s1
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_4)
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v7, v7, v6, s0
; GFX11-FAKE16-NEXT: v_cmp_lt_f32_e64 s0, v11, v10
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_2) | instid1(VALU_DEP_2)
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v8, 16, v7
; GFX11-FAKE16-NEXT: v_cndmask_b32_e64 v2, v2, v0, s0
; GFX11-FAKE16-NEXT: v_cmp_eq_u16_e64 s0, 0x8000, v4
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX11-FAKE16-NEXT: v_lshlrev_b32_e32 v9, 16, v2
; GFX11-FAKE16-NEXT: s_and_b32 vcc_lo, vcc_lo, s0
; GFX11-FAKE16-NEXT: v_cmp_eq_u16_e64 s0, 0x8000, v6
; GFX11-FAKE16-NEXT: v_dual_cndmask_b32 v4, v5, v4 :: v_dual_lshlrev_b32 v5, 16, v3
; GFX11-FAKE16-NEXT: v_cmp_eq_f32_e32 vcc_lo, 0, v8
; GFX11-FAKE16-NEXT: v_cmp_eq_f32_e64 s1, 0, v9
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_cmp_eq_f32_e64 s3, 0, v5
; GFX11-FAKE16-NEXT: s_and_b32 vcc_lo, vcc_lo, s0
; GFX11-FAKE16-NEXT: v_cndmask_b32_e32 v5, v7, v6, vcc_lo
; GFX11-FAKE16-NEXT: s_and_b32 vcc_lo, s1, s2
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_cndmask_b32_e32 v0, v2, v0, vcc_lo
; GFX11-FAKE16-NEXT: s_and_b32 vcc_lo, s3, s4
; GFX11-FAKE16-NEXT: v_cndmask_b32_e32 v1, v3, v1, vcc_lo
@@ -14717,18 +14666,17 @@ define <4 x bfloat> @v_minimumnum_v4bf16_no_ieee(<4 x bfloat> %x, <4 x bfloat> %
; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_3)
; GFX12-FAKE16-NEXT: v_lshlrev_b32_e32 v10, 16, v2
; GFX12-FAKE16-NEXT: v_cmp_lt_f32_e64 s0, v8, v9
-; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_1) | instid1(VALU_DEP_2)
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_4) | instid1(VALU_DEP_3)
; GFX12-FAKE16-NEXT: v_cmp_lt_f32_e64 s1, v13, v12
; GFX12-FAKE16-NEXT: s_wait_alu depctr_va_sdst(0)
; GFX12-FAKE16-NEXT: v_cndmask_b32_e64 v7, v7, v6, s0
; GFX12-FAKE16-NEXT: v_cmp_lt_f32_e64 s0, v11, v10
-; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_3)
; GFX12-FAKE16-NEXT: v_cndmask_b32_e64 v3, v3, v1, s1
; GFX12-FAKE16-NEXT: v_lshlrev_b32_e32 v8, 16, v7
; GFX12-FAKE16-NEXT: s_wait_alu depctr_va_sdst(0)
-; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX12-FAKE16-NEXT: v_cndmask_b32_e64 v2, v2, v0, s0
; GFX12-FAKE16-NEXT: v_cmp_eq_u16_e64 s0, 0x8000, v4
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX12-FAKE16-NEXT: v_lshlrev_b32_e32 v9, 16, v2
; GFX12-FAKE16-NEXT: s_and_b32 vcc_lo, vcc_lo, s0
; GFX12-FAKE16-NEXT: v_cmp_eq_u16_e64 s0, 0x8000, v6
diff --git a/llvm/test/CodeGen/AMDGPU/mixed-vmem-types.ll b/llvm/test/CodeGen/AMDGPU/mixed-vmem-types.ll
index 7fc67b832a9d82..7c1b330efb95ae 100644
--- a/llvm/test/CodeGen/AMDGPU/mixed-vmem-types.ll
+++ b/llvm/test/CodeGen/AMDGPU/mixed-vmem-types.ll
@@ -110,23 +110,26 @@ define amdgpu_cs void @mixed_vmem_types_sampler_reads(i32 inreg %globalTable, i3
; GFX12-GISEL-NEXT: s_wait_loadcnt 0x0
; GFX12-GISEL-NEXT: v_readfirstlane_b32 s4, v4
; GFX12-GISEL-NEXT: s_cmp_eq_f32 s0, 0
+; GFX12-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_3) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX12-GISEL-NEXT: s_cselect_b32 s0, 1, 0
; GFX12-GISEL-NEXT: s_cmp_eq_u32 s1, 0xac0
; GFX12-GISEL-NEXT: s_cselect_b32 s1, 1, 0
; GFX12-GISEL-NEXT: s_cmp_eq_u32 s2, 0xac0
+; GFX12-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_3)
; GFX12-GISEL-NEXT: s_cselect_b32 s2, 1, 0
; GFX12-GISEL-NEXT: s_cmp_eq_f32 s3, 1.0
; GFX12-GISEL-NEXT: s_cselect_b32 s3, 1, 0
; GFX12-GISEL-NEXT: s_cmp_eq_u32 s4, 0xac0
-; GFX12-GISEL-NEXT: s_cselect_b32 s4, 1, 0
; GFX12-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
+; GFX12-GISEL-NEXT: s_cselect_b32 s4, 1, 0
; GFX12-GISEL-NEXT: s_and_b32 s3, s4, s3
-; GFX12-GISEL-NEXT: s_and_b32 s2, s3, s2
; GFX12-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
+; GFX12-GISEL-NEXT: s_and_b32 s2, s3, s2
; GFX12-GISEL-NEXT: s_and_b32 s1, s2, s1
+; GFX12-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-GISEL-NEXT: s_and_b32 s0, s1, s0
-; GFX12-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX12-GISEL-NEXT: s_cmp_lg_u32 s0, 0
+; GFX12-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-GISEL-NEXT: s_cselect_b32 s0, 1, 0
; GFX12-GISEL-NEXT: v_mov_b32_e32 v0, s0
; GFX12-GISEL-NEXT: buffer_store_b32 v0, off, s[40:43], null
@@ -254,7 +257,7 @@ define amdgpu_cs void @mixed_vmem_types_bvh_reads(i32 inreg %descTable0, i32 inr
; GFX12-NEXT: v_readfirstlane_b32 s19, v15
; GFX12-NEXT: s_wait_alu depctr_va_sdst(0)
; GFX12-NEXT: v_cmpx_eq_u64_e32 s[16:17], v[12:13]
-; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_2)
+; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX12-NEXT: v_cmpx_eq_u64_e32 s[18:19], v[14:15]
; GFX12-NEXT: image_bvh64_intersect_ray v[15:18], [v[0:1], v2, v[3:5], v[6:8], v[9:11]], s[16:19]
; GFX12-NEXT: s_and_not1_wrexec_b32 s21, s21
@@ -313,7 +316,7 @@ define amdgpu_cs void @mixed_vmem_types_bvh_reads(i32 inreg %descTable0, i32 inr
; GFX12-GISEL-NEXT: v_cmp_eq_u64_e32 vcc_lo, s[20:21], v[12:13]
; GFX12-GISEL-NEXT: v_cmp_eq_u64_e64 s0, s[22:23], v[14:15]
; GFX12-GISEL-NEXT: s_and_b32 s0, vcc_lo, s0
-; GFX12-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX12-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
; GFX12-GISEL-NEXT: s_and_saveexec_b32 s0, s0
; GFX12-GISEL-NEXT: image_bvh64_intersect_ray v[12:15], [v[0:1], v2, v[3:5], v[6:8], v[9:11]], s[20:23]
; GFX12-GISEL-NEXT: ; implicit-def: $vgpr12
@@ -342,6 +345,7 @@ define amdgpu_cs void @mixed_vmem_types_bvh_reads(i32 inreg %descTable0, i32 inr
; GFX12-GISEL-NEXT: v_readfirstlane_b32 s0, v2
; GFX12-GISEL-NEXT: s_cselect_b32 s2, 1, 0
; GFX12-GISEL-NEXT: s_cmp_eq_u32 s1, 0xac0
+; GFX12-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-GISEL-NEXT: s_cselect_b32 s1, 1, 0
; GFX12-GISEL-NEXT: s_cmp_eq_u32 s0, 0xac0
; GFX12-GISEL-NEXT: v_cmp_eq_f32_e64 s0, 0, v13
@@ -349,6 +353,7 @@ define amdgpu_cs void @mixed_vmem_types_bvh_reads(i32 inreg %descTable0, i32 inr
; GFX12-GISEL-NEXT: s_and_b32 s1, s1, 1
; GFX12-GISEL-NEXT: s_and_b32 s3, s3, vcc_lo
; GFX12-GISEL-NEXT: s_cmp_lg_u32 s1, 0
+; GFX12-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_3) | instid1(SALU_CYCLE_1)
; GFX12-GISEL-NEXT: s_cselect_b32 s1, exec_lo, 0
; GFX12-GISEL-NEXT: s_and_b32 s2, s2, 1
; GFX12-GISEL-NEXT: s_and_b32 s1, s3, s1
diff --git a/llvm/test/CodeGen/AMDGPU/mubuf-legalize-operands-non-ptr-intrinsics.ll b/llvm/test/CodeGen/AMDGPU/mubuf-legalize-operands-non-ptr-intrinsics.ll
index 610d80e9bba092..cf60c0521e8e8c 100644
--- a/llvm/test/CodeGen/AMDGPU/mubuf-legalize-operands-non-ptr-intrinsics.ll
+++ b/llvm/test/CodeGen/AMDGPU/mubuf-legalize-operands-non-ptr-intrinsics.ll
@@ -890,10 +890,11 @@ define void @mubuf_vgpr_outside_entry(<4 x i32> %i, <4 x i32> %j, i32 %c, ptr ad
; GFX1100_W32-NEXT: s_cbranch_execnz .LBB2_1
; GFX1100_W32-NEXT: ; %bb.2:
; GFX1100_W32-NEXT: s_mov_b32 exec_lo, s5
+; GFX1100_W32-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX1100_W32-NEXT: v_and_b32_e32 v0, 0x3ff, v31
; GFX1100_W32-NEXT: s_mov_b32 s5, exec_lo
-; GFX1100_W32-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1100_W32-NEXT: v_cmpx_eq_u32_e32 0, v0
+; GFX1100_W32-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1100_W32-NEXT: s_cbranch_execz .LBB2_6
; GFX1100_W32-NEXT: ; %bb.3: ; %bb1
; GFX1100_W32-NEXT: v_mov_b32_e32 v0, s4
@@ -948,10 +949,11 @@ define void @mubuf_vgpr_outside_entry(<4 x i32> %i, <4 x i32> %j, i32 %c, ptr ad
; GFX1100_W64-NEXT: s_cbranch_execnz .LBB2_1
; GFX1100_W64-NEXT: ; %bb.2:
; GFX1100_W64-NEXT: s_mov_b64 exec, s[6:7]
+; GFX1100_W64-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX1100_W64-NEXT: v_and_b32_e32 v0, 0x3ff, v31
; GFX1100_W64-NEXT: s_mov_b64 s[6:7], exec
-; GFX1100_W64-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1100_W64-NEXT: v_cmpx_eq_u32_e32 0, v0
+; GFX1100_W64-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1100_W64-NEXT: s_cbranch_execz .LBB2_6
; GFX1100_W64-NEXT: ; %bb.3: ; %bb1
; GFX1100_W64-NEXT: v_mov_b32_e32 v0, s4
diff --git a/llvm/test/CodeGen/AMDGPU/mubuf-legalize-operands.ll b/llvm/test/CodeGen/AMDGPU/mubuf-legalize-operands.ll
index 69114085448886..820e712235fd98 100644
--- a/llvm/test/CodeGen/AMDGPU/mubuf-legalize-operands.ll
+++ b/llvm/test/CodeGen/AMDGPU/mubuf-legalize-operands.ll
@@ -910,10 +910,11 @@ define void @mubuf_vgpr_outside_entry(ptr addrspace(8) %i, ptr addrspace(8) %j,
; GFX1100_W32-NEXT: s_cbranch_execnz .LBB2_1
; GFX1100_W32-NEXT: ; %bb.2:
; GFX1100_W32-NEXT: s_mov_b32 exec_lo, s5
+; GFX1100_W32-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX1100_W32-NEXT: v_and_b32_e32 v0, 0x3ff, v31
; GFX1100_W32-NEXT: s_mov_b32 s5, exec_lo
-; GFX1100_W32-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1100_W32-NEXT: v_cmpx_eq_u32_e32 0, v0
+; GFX1100_W32-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1100_W32-NEXT: s_cbranch_execz .LBB2_6
; GFX1100_W32-NEXT: ; %bb.3: ; %bb1
; GFX1100_W32-NEXT: v_mov_b32_e32 v0, s4
@@ -968,10 +969,11 @@ define void @mubuf_vgpr_outside_entry(ptr addrspace(8) %i, ptr addrspace(8) %j,
; GFX1100_W64-NEXT: s_cbranch_execnz .LBB2_1
; GFX1100_W64-NEXT: ; %bb.2:
; GFX1100_W64-NEXT: s_mov_b64 exec, s[6:7]
+; GFX1100_W64-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX1100_W64-NEXT: v_and_b32_e32 v0, 0x3ff, v31
; GFX1100_W64-NEXT: s_mov_b64 s[6:7], exec
-; GFX1100_W64-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1100_W64-NEXT: v_cmpx_eq_u32_e32 0, v0
+; GFX1100_W64-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1100_W64-NEXT: s_cbranch_execz .LBB2_6
; GFX1100_W64-NEXT: ; %bb.3: ; %bb1
; GFX1100_W64-NEXT: v_mov_b32_e32 v0, s4
diff --git a/llvm/test/CodeGen/AMDGPU/mul.ll b/llvm/test/CodeGen/AMDGPU/mul.ll
index 1fad5619a628c0..f0b3a69a4387b2 100644
--- a/llvm/test/CodeGen/AMDGPU/mul.ll
+++ b/llvm/test/CodeGen/AMDGPU/mul.ll
@@ -2835,6 +2835,7 @@ define amdgpu_kernel void @mul32_in_branch(ptr addrspace(1) %out, ptr addrspace(
; GFX11-NEXT: s_mov_b32 s7, 0
; GFX11-NEXT: s_waitcnt lgkmcnt(0)
; GFX11-NEXT: s_cmp_lg_u32 s0, 0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc0 .LBB15_2
; GFX11-NEXT: ; %bb.1: ; %else
; GFX11-NEXT: s_mul_i32 s6, s0, s1
@@ -2846,7 +2847,7 @@ define amdgpu_kernel void @mul32_in_branch(ptr addrspace(1) %out, ptr addrspace(
; GFX11-NEXT: s_load_b128 s[0:3], s[4:5], 0x24
; GFX11-NEXT: s_and_b32 s4, s7, exec_lo
; GFX11-NEXT: s_cselect_b32 s4, 1, 0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_cmp_lg_u32 s4, 1
; GFX11-NEXT: s_cbranch_scc1 .LBB15_5
; GFX11-NEXT: ; %bb.4: ; %if
@@ -2873,6 +2874,7 @@ define amdgpu_kernel void @mul32_in_branch(ptr addrspace(1) %out, ptr addrspace(
; GFX12-NEXT: s_mov_b32 s7, 0
; GFX12-NEXT: s_wait_kmcnt 0x0
; GFX12-NEXT: s_cmp_lg_u32 s0, 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_scc0 .LBB15_2
; GFX12-NEXT: ; %bb.1: ; %else
; GFX12-NEXT: s_mul_i32 s6, s0, s1
@@ -2884,7 +2886,7 @@ define amdgpu_kernel void @mul32_in_branch(ptr addrspace(1) %out, ptr addrspace(
; GFX12-NEXT: s_load_b128 s[0:3], s[4:5], 0x24
; GFX12-NEXT: s_and_b32 s4, s7, exec_lo
; GFX12-NEXT: s_cselect_b32 s4, 1, 0
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_lg_u32 s4, 1
; GFX12-NEXT: s_cbranch_scc1 .LBB15_5
; GFX12-NEXT: ; %bb.4: ; %if
@@ -2915,6 +2917,7 @@ define amdgpu_kernel void @mul32_in_branch(ptr addrspace(1) %out, ptr addrspace(
; GFX1250-NEXT: s_mov_b32 s7, 0
; GFX1250-NEXT: s_wait_kmcnt 0x0
; GFX1250-NEXT: s_cmp_lg_u32 s0, 0
+; GFX1250-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1250-NEXT: s_cbranch_scc0 .LBB15_2
; GFX1250-NEXT: ; %bb.1: ; %else
; GFX1250-NEXT: s_mul_i32 s6, s0, s1
@@ -2927,7 +2930,7 @@ define amdgpu_kernel void @mul32_in_branch(ptr addrspace(1) %out, ptr addrspace(
; GFX1250-NEXT: s_wait_xcnt 0x0
; GFX1250-NEXT: s_and_b32 s4, s7, exec_lo
; GFX1250-NEXT: s_cselect_b32 s4, 1, 0
-; GFX1250-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1250-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1250-NEXT: s_cmp_lg_u32 s4, 1
; GFX1250-NEXT: s_cbranch_scc1 .LBB15_5
; GFX1250-NEXT: ; %bb.4: ; %if
@@ -2954,6 +2957,7 @@ define amdgpu_kernel void @mul32_in_branch(ptr addrspace(1) %out, ptr addrspace(
; GFX13-NEXT: s_mov_b32 s7, 0
; GFX13-NEXT: s_wait_kmcnt 0x0
; GFX13-NEXT: s_cmp_lg_u32 s0, 0
+; GFX13-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX13-NEXT: s_cbranch_scc0 .LBB15_2
; GFX13-NEXT: ; %bb.1: ; %else
; GFX13-NEXT: s_mul_i32 s6, s0, s1
@@ -2965,7 +2969,7 @@ define amdgpu_kernel void @mul32_in_branch(ptr addrspace(1) %out, ptr addrspace(
; GFX13-NEXT: s_load_b128 s[0:3], s[4:5], 0x24 nv
; GFX13-NEXT: s_and_b32 s4, s7, exec_lo
; GFX13-NEXT: s_cselect_b32 s4, 1, 0
-; GFX13-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX13-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX13-NEXT: s_cmp_lg_u32 s4, 1
; GFX13-NEXT: s_cbranch_scc1 .LBB15_5
; GFX13-NEXT: ; %bb.4: ; %if
@@ -3201,6 +3205,7 @@ define amdgpu_kernel void @mul64_in_branch(ptr addrspace(1) %out, ptr addrspace(
; GFX11-NEXT: s_load_b256 s[0:7], s[4:5], 0x24
; GFX11-NEXT: s_waitcnt lgkmcnt(0)
; GFX11-NEXT: s_cmp_lg_u64 s[4:5], 0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc0 .LBB16_2
; GFX11-NEXT: ; %bb.1: ; %else
; GFX11-NEXT: s_mul_i32 s7, s4, s7
@@ -3219,6 +3224,7 @@ define amdgpu_kernel void @mul64_in_branch(ptr addrspace(1) %out, ptr addrspace(
; GFX11-NEXT: s_and_b32 s6, s6, exec_lo
; GFX11-NEXT: s_cselect_b32 s6, 1, 0
; GFX11-NEXT: s_cmp_lg_u32 s6, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB16_5
; GFX11-NEXT: ; %bb.4: ; %if
; GFX11-NEXT: s_mov_b32 s7, 0x31016000
@@ -3241,6 +3247,7 @@ define amdgpu_kernel void @mul64_in_branch(ptr addrspace(1) %out, ptr addrspace(
; GFX12-NEXT: s_load_b256 s[0:7], s[4:5], 0x24
; GFX12-NEXT: s_wait_kmcnt 0x0
; GFX12-NEXT: s_cmp_lg_u64 s[4:5], 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_scc0 .LBB16_2
; GFX12-NEXT: ; %bb.1: ; %else
; GFX12-NEXT: s_mul_u64 s[4:5], s[4:5], s[6:7]
@@ -3254,6 +3261,7 @@ define amdgpu_kernel void @mul64_in_branch(ptr addrspace(1) %out, ptr addrspace(
; GFX12-NEXT: s_and_b32 s6, s6, exec_lo
; GFX12-NEXT: s_cselect_b32 s6, 1, 0
; GFX12-NEXT: s_cmp_lg_u32 s6, 1
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cbranch_scc1 .LBB16_5
; GFX12-NEXT: ; %bb.4: ; %if
; GFX12-NEXT: s_mov_b32 s7, 0x31016000
@@ -3280,6 +3288,7 @@ define amdgpu_kernel void @mul64_in_branch(ptr addrspace(1) %out, ptr addrspace(
; GFX1250-NEXT: s_load_b256 s[8:15], s[4:5], 0x24 nv
; GFX1250-NEXT: s_wait_kmcnt 0x0
; GFX1250-NEXT: s_cmp_lg_u64 s[12:13], 0
+; GFX1250-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1250-NEXT: s_cbranch_scc0 .LBB16_2
; GFX1250-NEXT: ; %bb.1: ; %else
; GFX1250-NEXT: s_mul_u64 s[0:1], s[12:13], s[14:15]
@@ -3293,6 +3302,7 @@ define amdgpu_kernel void @mul64_in_branch(ptr addrspace(1) %out, ptr addrspace(
; GFX1250-NEXT: s_and_b32 s2, s2, exec_lo
; GFX1250-NEXT: s_cselect_b32 s2, 1, 0
; GFX1250-NEXT: s_cmp_lg_u32 s2, 1
+; GFX1250-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1250-NEXT: s_cbranch_scc1 .LBB16_5
; GFX1250-NEXT: ; %bb.4: ; %if
; GFX1250-NEXT: s_mov_b32 s3, 0x31016000
@@ -3315,6 +3325,7 @@ define amdgpu_kernel void @mul64_in_branch(ptr addrspace(1) %out, ptr addrspace(
; GFX13-NEXT: s_load_b256 s[0:7], s[4:5], 0x24 nv
; GFX13-NEXT: s_wait_kmcnt 0x0
; GFX13-NEXT: s_cmp_lg_u64 s[4:5], 0
+; GFX13-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX13-NEXT: s_cbranch_scc0 .LBB16_2
; GFX13-NEXT: ; %bb.1: ; %else
; GFX13-NEXT: s_mul_u64 s[4:5], s[4:5], s[6:7]
@@ -3328,6 +3339,7 @@ define amdgpu_kernel void @mul64_in_branch(ptr addrspace(1) %out, ptr addrspace(
; GFX13-NEXT: s_and_b32 s6, s6, exec_lo
; GFX13-NEXT: s_cselect_b32 s6, 1, 0
; GFX13-NEXT: s_cmp_lg_u32 s6, 1
+; GFX13-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX13-NEXT: s_cbranch_scc1 .LBB16_5
; GFX13-NEXT: ; %bb.4: ; %if
; GFX13-NEXT: s_mov_b32 s7, 0x31016000
@@ -4005,16 +4017,15 @@ define amdgpu_kernel void @v_mul_i128(ptr addrspace(1) %out, ptr addrspace(1) %a
; GFX11-NEXT: v_mad_u64_u32 v[2:3], null, v0, v5, v[9:10]
; GFX11-NEXT: v_add3_u32 v14, v14, v16, v11
; GFX11-NEXT: v_mul_lo_u32 v11, v7, v0
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_1) | instid1(VALU_DEP_4)
; GFX11-NEXT: v_add_co_u32 v3, s0, v12, v3
; GFX11-NEXT: v_add_co_ci_u32_e64 v4, null, 0, 0, s0
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-NEXT: v_mad_u64_u32 v[9:10], null, v6, v0, v[13:14]
-; GFX11-NEXT: v_mad_u64_u32 v[6:7], null, v1, v5, v[3:4]
; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX11-NEXT: v_mad_u64_u32 v[6:7], null, v1, v5, v[3:4]
; GFX11-NEXT: v_add3_u32 v0, v11, v10, v17
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-NEXT: v_add_co_u32 v10, vcc_lo, v6, v9
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v11, null, v7, v0, vcc_lo
; GFX11-NEXT: v_mov_b32_e32 v9, v2
; GFX11-NEXT: global_store_b128 v15, v[8:11], s[2:3]
@@ -4043,16 +4054,15 @@ define amdgpu_kernel void @v_mul_i128(ptr addrspace(1) %out, ptr addrspace(1) %a
; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_3)
; GFX12-NEXT: v_mad_co_u64_u32 v[9:10], null, v0, v5, v[9:10]
; GFX12-NEXT: v_add3_u32 v3, v3, v14, v11
-; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_3)
; GFX12-NEXT: v_add_co_u32 v10, s0, v12, v10
; GFX12-NEXT: v_add_co_ci_u32_e64 v11, null, 0, 0, s0
-; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX12-NEXT: v_mad_co_u64_u32 v[2:3], null, v6, v0, v[2:3]
-; GFX12-NEXT: v_mad_co_u64_u32 v[0:1], null, v1, v5, v[10:11]
; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX12-NEXT: v_mad_co_u64_u32 v[0:1], null, v1, v5, v[10:11]
; GFX12-NEXT: v_add3_u32 v3, v7, v3, v4
+; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX12-NEXT: v_add_co_u32 v10, vcc_lo, v0, v2
-; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-NEXT: v_add_co_ci_u32_e64 v11, null, v1, v3, vcc_lo
; GFX12-NEXT: global_store_b128 v13, v[8:11], s[2:3]
; GFX12-NEXT: s_endpgm
@@ -4119,16 +4129,15 @@ define amdgpu_kernel void @v_mul_i128(ptr addrspace(1) %out, ptr addrspace(1) %a
; GFX13-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_3)
; GFX13-NEXT: v_mad_co_u64_u32 v[9:10], null, v0, v5, v[9:10]
; GFX13-NEXT: v_add3_u32 v3, v3, v14, v11
-; GFX13-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX13-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_3)
; GFX13-NEXT: v_add_co_u32 v10, s0, v12, v10
; GFX13-NEXT: v_add_co_ci_u32_e64 v11, null, 0, 0, s0
-; GFX13-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX13-NEXT: v_mad_co_u64_u32 v[2:3], null, v6, v0, v[2:3]
-; GFX13-NEXT: v_mad_co_u64_u32 v[0:1], null, v1, v5, v[10:11]
; GFX13-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX13-NEXT: v_mad_co_u64_u32 v[0:1], null, v1, v5, v[10:11]
; GFX13-NEXT: v_add3_u32 v3, v7, v3, v4
+; GFX13-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX13-NEXT: v_add_co_u32 v10, vcc_lo, v0, v2
-; GFX13-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX13-NEXT: v_add_co_ci_u32_e64 v11, null, v1, v3, vcc_lo
; GFX13-NEXT: global_store_b128 v13, v[8:11], s[2:3] scale_offset
; GFX13-NEXT: s_endpgm
diff --git a/llvm/test/CodeGen/AMDGPU/narrow_math_for_and.ll b/llvm/test/CodeGen/AMDGPU/narrow_math_for_and.ll
index ac3883659f84ad..e809681bbecf12 100644
--- a/llvm/test/CodeGen/AMDGPU/narrow_math_for_and.ll
+++ b/llvm/test/CodeGen/AMDGPU/narrow_math_for_and.ll
@@ -142,7 +142,7 @@ define i64 @no_narrow_add(i64 %a, i64 %b) {
; CHECK-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; CHECK-NEXT: v_and_b32_e32 v0, 0x80000000, v0
; CHECK-NEXT: v_and_b32_e32 v1, 0x80000000, v2
-; CHECK-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; CHECK-NEXT: s_delay_alu instid0(VALU_DEP_1)
; CHECK-NEXT: v_add_co_u32 v0, s0, v0, v1
; CHECK-NEXT: v_add_co_ci_u32_e64 v1, null, 0, 0, s0
; CHECK-NEXT: s_setpc_b64 s[30:31]
@@ -157,7 +157,7 @@ define i64 @no_narrow_add_1(i64 %a, i64 %b) {
; CHECK: ; %bb.0:
; CHECK-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; CHECK-NEXT: v_and_b32_e32 v1, 1, v2
-; CHECK-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; CHECK-NEXT: s_delay_alu instid0(VALU_DEP_1)
; CHECK-NEXT: v_add_co_u32 v0, s0, v0, v1
; CHECK-NEXT: v_add_co_ci_u32_e64 v1, null, 0, 0, s0
; CHECK-NEXT: s_setpc_b64 s[30:31]
@@ -175,10 +175,9 @@ define <2 x i64> @no_narrow_add_vec(<2 x i64> %a, <2 x i64> %b) #0 {
; CHECK-NEXT: v_and_b32_e32 v1, 0x80000000, v4
; CHECK-NEXT: v_and_b32_e32 v2, 30, v2
; CHECK-NEXT: v_and_b32_e32 v3, 0x7ffffffe, v6
-; CHECK-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_1)
+; CHECK-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_1) | instid1(VALU_DEP_3)
; CHECK-NEXT: v_add_co_u32 v0, s0, v0, v1
; CHECK-NEXT: v_add_co_ci_u32_e64 v1, null, 0, 0, s0
-; CHECK-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_1)
; CHECK-NEXT: v_add_co_u32 v2, s0, v2, v3
; CHECK-NEXT: v_add_co_ci_u32_e64 v3, null, 0, 0, s0
; CHECK-NEXT: s_setpc_b64 s[30:31]
diff --git a/llvm/test/CodeGen/AMDGPU/no-dup-inst-prefetch.ll b/llvm/test/CodeGen/AMDGPU/no-dup-inst-prefetch.ll
index 22f17fa4a6dc9f..5734fe872442af 100644
--- a/llvm/test/CodeGen/AMDGPU/no-dup-inst-prefetch.ll
+++ b/llvm/test/CodeGen/AMDGPU/no-dup-inst-prefetch.ll
@@ -64,10 +64,11 @@ define amdgpu_cs void @_amdgpu_cs_main(float %0, i32 %1) {
; GFX12-NEXT: .LBB0_1: ; %Flow
; GFX12-NEXT: ; in Loop: Header=BB0_2 Depth=1
; GFX12-NEXT: s_or_b32 exec_lo, exec_lo, s3
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: v_mov_b32_e32 v1, v0
; GFX12-NEXT: s_and_b32 s0, exec_lo, s2
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_or_b32 s1, s0, s1
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_and_not1_b32 exec_lo, exec_lo, s1
; GFX12-NEXT: s_cbranch_execz .LBB0_4
; GFX12-NEXT: .LBB0_2: ; %bb
diff --git a/llvm/test/CodeGen/AMDGPU/offset-split-flat.ll b/llvm/test/CodeGen/AMDGPU/offset-split-flat.ll
index 592cc5b7fac915..b9bf91e79280fa 100644
--- a/llvm/test/CodeGen/AMDGPU/offset-split-flat.ll
+++ b/llvm/test/CodeGen/AMDGPU/offset-split-flat.ll
@@ -263,7 +263,6 @@ define i8 @flat_inst_valu_offset_13bit_max(ptr %p) {
; GFX11-SDAG-TRUE16: ; %bb.0:
; GFX11-SDAG-TRUE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-SDAG-TRUE16-NEXT: v_add_co_u32 v0, vcc_lo, 0x1000, v0
-; GFX11-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-SDAG-TRUE16-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-SDAG-TRUE16-NEXT: flat_load_d16_u8 v0, v[0:1] offset:4095
; GFX11-SDAG-TRUE16-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
@@ -273,7 +272,6 @@ define i8 @flat_inst_valu_offset_13bit_max(ptr %p) {
; GFX11-SDAG-FAKE16: ; %bb.0:
; GFX11-SDAG-FAKE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-SDAG-FAKE16-NEXT: v_add_co_u32 v0, vcc_lo, 0x1000, v0
-; GFX11-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-SDAG-FAKE16-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-SDAG-FAKE16-NEXT: flat_load_u8 v0, v[0:1] offset:4095
; GFX11-SDAG-FAKE16-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
@@ -314,7 +312,6 @@ define i8 @flat_inst_valu_offset_13bit_max(ptr %p) {
; GFX11-GISEL: ; %bb.0:
; GFX11-GISEL-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, 0x1fff, v0
-; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-GISEL-NEXT: flat_load_u8 v0, v[0:1]
; GFX11-GISEL-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
@@ -358,7 +355,6 @@ define i8 @flat_inst_valu_offset_24bit_max(ptr %p) {
; GFX11-SDAG-TRUE16: ; %bb.0:
; GFX11-SDAG-TRUE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-SDAG-TRUE16-NEXT: v_add_co_u32 v0, vcc_lo, 0x7ff000, v0
-; GFX11-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-SDAG-TRUE16-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-SDAG-TRUE16-NEXT: flat_load_d16_u8 v0, v[0:1] offset:4095
; GFX11-SDAG-TRUE16-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
@@ -368,7 +364,6 @@ define i8 @flat_inst_valu_offset_24bit_max(ptr %p) {
; GFX11-SDAG-FAKE16: ; %bb.0:
; GFX11-SDAG-FAKE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-SDAG-FAKE16-NEXT: v_add_co_u32 v0, vcc_lo, 0x7ff000, v0
-; GFX11-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-SDAG-FAKE16-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-SDAG-FAKE16-NEXT: flat_load_u8 v0, v[0:1] offset:4095
; GFX11-SDAG-FAKE16-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
@@ -409,7 +404,6 @@ define i8 @flat_inst_valu_offset_24bit_max(ptr %p) {
; GFX11-GISEL: ; %bb.0:
; GFX11-GISEL-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, 0x7fffff, v0
-; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-GISEL-NEXT: flat_load_u8 v0, v[0:1]
; GFX11-GISEL-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
@@ -453,7 +447,6 @@ define i8 @flat_inst_valu_offset_neg_11bit_max(ptr %p) {
; GFX11-SDAG-TRUE16: ; %bb.0:
; GFX11-SDAG-TRUE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-SDAG-TRUE16-NEXT: v_add_co_u32 v0, vcc_lo, 0xfffff800, v0
-; GFX11-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-SDAG-TRUE16-NEXT: v_add_co_ci_u32_e64 v1, null, -1, v1, vcc_lo
; GFX11-SDAG-TRUE16-NEXT: flat_load_d16_u8 v0, v[0:1]
; GFX11-SDAG-TRUE16-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
@@ -463,7 +456,6 @@ define i8 @flat_inst_valu_offset_neg_11bit_max(ptr %p) {
; GFX11-SDAG-FAKE16: ; %bb.0:
; GFX11-SDAG-FAKE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-SDAG-FAKE16-NEXT: v_add_co_u32 v0, vcc_lo, 0xfffff800, v0
-; GFX11-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-SDAG-FAKE16-NEXT: v_add_co_ci_u32_e64 v1, null, -1, v1, vcc_lo
; GFX11-SDAG-FAKE16-NEXT: flat_load_u8 v0, v[0:1]
; GFX11-SDAG-FAKE16-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
@@ -495,7 +487,6 @@ define i8 @flat_inst_valu_offset_neg_11bit_max(ptr %p) {
; GFX11-GISEL: ; %bb.0:
; GFX11-GISEL-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, 0xfffff800, v0
-; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, -1, v1, vcc_lo
; GFX11-GISEL-NEXT: flat_load_u8 v0, v[0:1]
; GFX11-GISEL-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
@@ -539,7 +530,6 @@ define i8 @flat_inst_valu_offset_neg_12bit_max(ptr %p) {
; GFX11-SDAG-TRUE16: ; %bb.0:
; GFX11-SDAG-TRUE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-SDAG-TRUE16-NEXT: v_add_co_u32 v0, vcc_lo, 0xfffff000, v0
-; GFX11-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-SDAG-TRUE16-NEXT: v_add_co_ci_u32_e64 v1, null, -1, v1, vcc_lo
; GFX11-SDAG-TRUE16-NEXT: flat_load_d16_u8 v0, v[0:1]
; GFX11-SDAG-TRUE16-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
@@ -549,7 +539,6 @@ define i8 @flat_inst_valu_offset_neg_12bit_max(ptr %p) {
; GFX11-SDAG-FAKE16: ; %bb.0:
; GFX11-SDAG-FAKE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-SDAG-FAKE16-NEXT: v_add_co_u32 v0, vcc_lo, 0xfffff000, v0
-; GFX11-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-SDAG-FAKE16-NEXT: v_add_co_ci_u32_e64 v1, null, -1, v1, vcc_lo
; GFX11-SDAG-FAKE16-NEXT: flat_load_u8 v0, v[0:1]
; GFX11-SDAG-FAKE16-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
@@ -581,7 +570,6 @@ define i8 @flat_inst_valu_offset_neg_12bit_max(ptr %p) {
; GFX11-GISEL: ; %bb.0:
; GFX11-GISEL-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, 0xfffff000, v0
-; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, -1, v1, vcc_lo
; GFX11-GISEL-NEXT: flat_load_u8 v0, v[0:1]
; GFX11-GISEL-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
@@ -625,7 +613,6 @@ define i8 @flat_inst_valu_offset_neg_13bit_max(ptr %p) {
; GFX11-SDAG-TRUE16: ; %bb.0:
; GFX11-SDAG-TRUE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-SDAG-TRUE16-NEXT: v_add_co_u32 v0, vcc_lo, 0xffffe000, v0
-; GFX11-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-SDAG-TRUE16-NEXT: v_add_co_ci_u32_e64 v1, null, -1, v1, vcc_lo
; GFX11-SDAG-TRUE16-NEXT: flat_load_d16_u8 v0, v[0:1]
; GFX11-SDAG-TRUE16-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
@@ -635,7 +622,6 @@ define i8 @flat_inst_valu_offset_neg_13bit_max(ptr %p) {
; GFX11-SDAG-FAKE16: ; %bb.0:
; GFX11-SDAG-FAKE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-SDAG-FAKE16-NEXT: v_add_co_u32 v0, vcc_lo, 0xffffe000, v0
-; GFX11-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-SDAG-FAKE16-NEXT: v_add_co_ci_u32_e64 v1, null, -1, v1, vcc_lo
; GFX11-SDAG-FAKE16-NEXT: flat_load_u8 v0, v[0:1]
; GFX11-SDAG-FAKE16-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
@@ -667,7 +653,6 @@ define i8 @flat_inst_valu_offset_neg_13bit_max(ptr %p) {
; GFX11-GISEL: ; %bb.0:
; GFX11-GISEL-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, 0xffffe000, v0
-; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, -1, v1, vcc_lo
; GFX11-GISEL-NEXT: flat_load_u8 v0, v[0:1]
; GFX11-GISEL-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
@@ -711,7 +696,6 @@ define i8 @flat_inst_valu_offset_neg_24bit_max(ptr %p) {
; GFX11-SDAG-TRUE16: ; %bb.0:
; GFX11-SDAG-TRUE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-SDAG-TRUE16-NEXT: v_add_co_u32 v0, vcc_lo, 0xff800000, v0
-; GFX11-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-SDAG-TRUE16-NEXT: v_add_co_ci_u32_e64 v1, null, -1, v1, vcc_lo
; GFX11-SDAG-TRUE16-NEXT: flat_load_d16_u8 v0, v[0:1]
; GFX11-SDAG-TRUE16-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
@@ -721,7 +705,6 @@ define i8 @flat_inst_valu_offset_neg_24bit_max(ptr %p) {
; GFX11-SDAG-FAKE16: ; %bb.0:
; GFX11-SDAG-FAKE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-SDAG-FAKE16-NEXT: v_add_co_u32 v0, vcc_lo, 0xff800000, v0
-; GFX11-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-SDAG-FAKE16-NEXT: v_add_co_ci_u32_e64 v1, null, -1, v1, vcc_lo
; GFX11-SDAG-FAKE16-NEXT: flat_load_u8 v0, v[0:1]
; GFX11-SDAG-FAKE16-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
@@ -753,7 +736,6 @@ define i8 @flat_inst_valu_offset_neg_24bit_max(ptr %p) {
; GFX11-GISEL: ; %bb.0:
; GFX11-GISEL-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, 0xff800000, v0
-; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, -1, v1, vcc_lo
; GFX11-GISEL-NEXT: flat_load_u8 v0, v[0:1]
; GFX11-GISEL-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
@@ -873,7 +855,6 @@ define i8 @flat_inst_valu_offset_2x_12bit_max(ptr %p) {
; GFX11-SDAG-TRUE16: ; %bb.0:
; GFX11-SDAG-TRUE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-SDAG-TRUE16-NEXT: v_add_co_u32 v0, vcc_lo, 0x1000, v0
-; GFX11-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-SDAG-TRUE16-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-SDAG-TRUE16-NEXT: flat_load_d16_u8 v0, v[0:1] offset:4095
; GFX11-SDAG-TRUE16-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
@@ -883,7 +864,6 @@ define i8 @flat_inst_valu_offset_2x_12bit_max(ptr %p) {
; GFX11-SDAG-FAKE16: ; %bb.0:
; GFX11-SDAG-FAKE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-SDAG-FAKE16-NEXT: v_add_co_u32 v0, vcc_lo, 0x1000, v0
-; GFX11-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-SDAG-FAKE16-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-SDAG-FAKE16-NEXT: flat_load_u8 v0, v[0:1] offset:4095
; GFX11-SDAG-FAKE16-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
@@ -924,7 +904,6 @@ define i8 @flat_inst_valu_offset_2x_12bit_max(ptr %p) {
; GFX11-GISEL: ; %bb.0:
; GFX11-GISEL-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, 0x1fff, v0
-; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-GISEL-NEXT: flat_load_u8 v0, v[0:1]
; GFX11-GISEL-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
@@ -968,7 +947,6 @@ define i8 @flat_inst_valu_offset_2x_13bit_max(ptr %p) {
; GFX11-SDAG-TRUE16: ; %bb.0:
; GFX11-SDAG-TRUE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-SDAG-TRUE16-NEXT: v_add_co_u32 v0, vcc_lo, 0x3000, v0
-; GFX11-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-SDAG-TRUE16-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-SDAG-TRUE16-NEXT: flat_load_d16_u8 v0, v[0:1] offset:4095
; GFX11-SDAG-TRUE16-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
@@ -978,7 +956,6 @@ define i8 @flat_inst_valu_offset_2x_13bit_max(ptr %p) {
; GFX11-SDAG-FAKE16: ; %bb.0:
; GFX11-SDAG-FAKE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-SDAG-FAKE16-NEXT: v_add_co_u32 v0, vcc_lo, 0x3000, v0
-; GFX11-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-SDAG-FAKE16-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-SDAG-FAKE16-NEXT: flat_load_u8 v0, v[0:1] offset:4095
; GFX11-SDAG-FAKE16-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
@@ -1019,7 +996,6 @@ define i8 @flat_inst_valu_offset_2x_13bit_max(ptr %p) {
; GFX11-GISEL: ; %bb.0:
; GFX11-GISEL-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, 0x3fff, v0
-; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-GISEL-NEXT: flat_load_u8 v0, v[0:1]
; GFX11-GISEL-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
@@ -1063,7 +1039,6 @@ define i8 @flat_inst_valu_offset_2x_24bit_max(ptr %p) {
; GFX11-SDAG-TRUE16: ; %bb.0:
; GFX11-SDAG-TRUE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-SDAG-TRUE16-NEXT: v_add_co_u32 v0, vcc_lo, 0xfff000, v0
-; GFX11-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-SDAG-TRUE16-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-SDAG-TRUE16-NEXT: flat_load_d16_u8 v0, v[0:1] offset:4094
; GFX11-SDAG-TRUE16-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
@@ -1073,7 +1048,6 @@ define i8 @flat_inst_valu_offset_2x_24bit_max(ptr %p) {
; GFX11-SDAG-FAKE16: ; %bb.0:
; GFX11-SDAG-FAKE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-SDAG-FAKE16-NEXT: v_add_co_u32 v0, vcc_lo, 0xfff000, v0
-; GFX11-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-SDAG-FAKE16-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-SDAG-FAKE16-NEXT: flat_load_u8 v0, v[0:1] offset:4094
; GFX11-SDAG-FAKE16-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
@@ -1120,7 +1094,6 @@ define i8 @flat_inst_valu_offset_2x_24bit_max(ptr %p) {
; GFX11-GISEL: ; %bb.0:
; GFX11-GISEL-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, 0xfffffe, v0
-; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-GISEL-NEXT: flat_load_u8 v0, v[0:1]
; GFX11-GISEL-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
@@ -1167,7 +1140,6 @@ define i8 @flat_inst_valu_offset_2x_neg_11bit_max(ptr %p) {
; GFX11-SDAG-TRUE16: ; %bb.0:
; GFX11-SDAG-TRUE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-SDAG-TRUE16-NEXT: v_add_co_u32 v0, vcc_lo, 0xfffff000, v0
-; GFX11-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-SDAG-TRUE16-NEXT: v_add_co_ci_u32_e64 v1, null, -1, v1, vcc_lo
; GFX11-SDAG-TRUE16-NEXT: flat_load_d16_u8 v0, v[0:1]
; GFX11-SDAG-TRUE16-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
@@ -1177,7 +1149,6 @@ define i8 @flat_inst_valu_offset_2x_neg_11bit_max(ptr %p) {
; GFX11-SDAG-FAKE16: ; %bb.0:
; GFX11-SDAG-FAKE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-SDAG-FAKE16-NEXT: v_add_co_u32 v0, vcc_lo, 0xfffff000, v0
-; GFX11-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-SDAG-FAKE16-NEXT: v_add_co_ci_u32_e64 v1, null, -1, v1, vcc_lo
; GFX11-SDAG-FAKE16-NEXT: flat_load_u8 v0, v[0:1]
; GFX11-SDAG-FAKE16-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
@@ -1209,7 +1180,6 @@ define i8 @flat_inst_valu_offset_2x_neg_11bit_max(ptr %p) {
; GFX11-GISEL: ; %bb.0:
; GFX11-GISEL-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, 0xfffff000, v0
-; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, -1, v1, vcc_lo
; GFX11-GISEL-NEXT: flat_load_u8 v0, v[0:1]
; GFX11-GISEL-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
@@ -1253,7 +1223,6 @@ define i8 @flat_inst_valu_offset_2x_neg_12bit_max(ptr %p) {
; GFX11-SDAG-TRUE16: ; %bb.0:
; GFX11-SDAG-TRUE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-SDAG-TRUE16-NEXT: v_add_co_u32 v0, vcc_lo, 0xffffe000, v0
-; GFX11-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-SDAG-TRUE16-NEXT: v_add_co_ci_u32_e64 v1, null, -1, v1, vcc_lo
; GFX11-SDAG-TRUE16-NEXT: flat_load_d16_u8 v0, v[0:1]
; GFX11-SDAG-TRUE16-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
@@ -1263,7 +1232,6 @@ define i8 @flat_inst_valu_offset_2x_neg_12bit_max(ptr %p) {
; GFX11-SDAG-FAKE16: ; %bb.0:
; GFX11-SDAG-FAKE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-SDAG-FAKE16-NEXT: v_add_co_u32 v0, vcc_lo, 0xffffe000, v0
-; GFX11-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-SDAG-FAKE16-NEXT: v_add_co_ci_u32_e64 v1, null, -1, v1, vcc_lo
; GFX11-SDAG-FAKE16-NEXT: flat_load_u8 v0, v[0:1]
; GFX11-SDAG-FAKE16-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
@@ -1295,7 +1263,6 @@ define i8 @flat_inst_valu_offset_2x_neg_12bit_max(ptr %p) {
; GFX11-GISEL: ; %bb.0:
; GFX11-GISEL-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, 0xffffe000, v0
-; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, -1, v1, vcc_lo
; GFX11-GISEL-NEXT: flat_load_u8 v0, v[0:1]
; GFX11-GISEL-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
@@ -1339,7 +1306,6 @@ define i8 @flat_inst_valu_offset_2x_neg_13bit_max(ptr %p) {
; GFX11-SDAG-TRUE16: ; %bb.0:
; GFX11-SDAG-TRUE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-SDAG-TRUE16-NEXT: v_add_co_u32 v0, vcc_lo, 0xffffc000, v0
-; GFX11-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-SDAG-TRUE16-NEXT: v_add_co_ci_u32_e64 v1, null, -1, v1, vcc_lo
; GFX11-SDAG-TRUE16-NEXT: flat_load_d16_u8 v0, v[0:1]
; GFX11-SDAG-TRUE16-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
@@ -1349,7 +1315,6 @@ define i8 @flat_inst_valu_offset_2x_neg_13bit_max(ptr %p) {
; GFX11-SDAG-FAKE16: ; %bb.0:
; GFX11-SDAG-FAKE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-SDAG-FAKE16-NEXT: v_add_co_u32 v0, vcc_lo, 0xffffc000, v0
-; GFX11-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-SDAG-FAKE16-NEXT: v_add_co_ci_u32_e64 v1, null, -1, v1, vcc_lo
; GFX11-SDAG-FAKE16-NEXT: flat_load_u8 v0, v[0:1]
; GFX11-SDAG-FAKE16-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
@@ -1381,7 +1346,6 @@ define i8 @flat_inst_valu_offset_2x_neg_13bit_max(ptr %p) {
; GFX11-GISEL: ; %bb.0:
; GFX11-GISEL-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, 0xffffc000, v0
-; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, -1, v1, vcc_lo
; GFX11-GISEL-NEXT: flat_load_u8 v0, v[0:1]
; GFX11-GISEL-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
@@ -1425,7 +1389,6 @@ define i8 @flat_inst_valu_offset_2x_neg_24bit_max(ptr %p) {
; GFX11-SDAG-TRUE16: ; %bb.0:
; GFX11-SDAG-TRUE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-SDAG-TRUE16-NEXT: v_add_co_u32 v0, vcc_lo, 0xff000001, v0
-; GFX11-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-SDAG-TRUE16-NEXT: v_add_co_ci_u32_e64 v1, null, -1, v1, vcc_lo
; GFX11-SDAG-TRUE16-NEXT: flat_load_d16_u8 v0, v[0:1]
; GFX11-SDAG-TRUE16-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
@@ -1435,7 +1398,6 @@ define i8 @flat_inst_valu_offset_2x_neg_24bit_max(ptr %p) {
; GFX11-SDAG-FAKE16: ; %bb.0:
; GFX11-SDAG-FAKE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-SDAG-FAKE16-NEXT: v_add_co_u32 v0, vcc_lo, 0xff000001, v0
-; GFX11-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-SDAG-FAKE16-NEXT: v_add_co_ci_u32_e64 v1, null, -1, v1, vcc_lo
; GFX11-SDAG-FAKE16-NEXT: flat_load_u8 v0, v[0:1]
; GFX11-SDAG-FAKE16-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
@@ -1473,7 +1435,6 @@ define i8 @flat_inst_valu_offset_2x_neg_24bit_max(ptr %p) {
; GFX11-GISEL: ; %bb.0:
; GFX11-GISEL-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, 0xff000001, v0
-; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, -1, v1, vcc_lo
; GFX11-GISEL-NEXT: flat_load_u8 v0, v[0:1]
; GFX11-GISEL-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
@@ -1521,7 +1482,6 @@ define i8 @flat_inst_valu_offset_64bit_11bit_split0(ptr %p) {
; GFX11-SDAG-TRUE16: ; %bb.0:
; GFX11-SDAG-TRUE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-SDAG-TRUE16-NEXT: v_add_co_u32 v0, vcc_lo, 0, v0
-; GFX11-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-SDAG-TRUE16-NEXT: v_add_co_ci_u32_e64 v1, null, 2, v1, vcc_lo
; GFX11-SDAG-TRUE16-NEXT: flat_load_d16_u8 v0, v[0:1] offset:2047
; GFX11-SDAG-TRUE16-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
@@ -1531,7 +1491,6 @@ define i8 @flat_inst_valu_offset_64bit_11bit_split0(ptr %p) {
; GFX11-SDAG-FAKE16: ; %bb.0:
; GFX11-SDAG-FAKE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-SDAG-FAKE16-NEXT: v_add_co_u32 v0, vcc_lo, 0, v0
-; GFX11-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-SDAG-FAKE16-NEXT: v_add_co_ci_u32_e64 v1, null, 2, v1, vcc_lo
; GFX11-SDAG-FAKE16-NEXT: flat_load_u8 v0, v[0:1] offset:2047
; GFX11-SDAG-FAKE16-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
@@ -1578,7 +1537,6 @@ define i8 @flat_inst_valu_offset_64bit_11bit_split0(ptr %p) {
; GFX11-GISEL: ; %bb.0:
; GFX11-GISEL-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, 0x7ff, v0
-; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 2, v1, vcc_lo
; GFX11-GISEL-NEXT: flat_load_u8 v0, v[0:1]
; GFX11-GISEL-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
@@ -1626,7 +1584,6 @@ define i8 @flat_inst_valu_offset_64bit_11bit_split1(ptr %p) {
; GFX11-SDAG-TRUE16: ; %bb.0:
; GFX11-SDAG-TRUE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-SDAG-TRUE16-NEXT: v_add_co_u32 v0, vcc_lo, 0, v0
-; GFX11-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-SDAG-TRUE16-NEXT: v_add_co_ci_u32_e64 v1, null, 2, v1, vcc_lo
; GFX11-SDAG-TRUE16-NEXT: flat_load_d16_u8 v0, v[0:1] offset:2048
; GFX11-SDAG-TRUE16-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
@@ -1636,7 +1593,6 @@ define i8 @flat_inst_valu_offset_64bit_11bit_split1(ptr %p) {
; GFX11-SDAG-FAKE16: ; %bb.0:
; GFX11-SDAG-FAKE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-SDAG-FAKE16-NEXT: v_add_co_u32 v0, vcc_lo, 0, v0
-; GFX11-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-SDAG-FAKE16-NEXT: v_add_co_ci_u32_e64 v1, null, 2, v1, vcc_lo
; GFX11-SDAG-FAKE16-NEXT: flat_load_u8 v0, v[0:1] offset:2048
; GFX11-SDAG-FAKE16-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
@@ -1683,7 +1639,6 @@ define i8 @flat_inst_valu_offset_64bit_11bit_split1(ptr %p) {
; GFX11-GISEL: ; %bb.0:
; GFX11-GISEL-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, 0x800, v0
-; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 2, v1, vcc_lo
; GFX11-GISEL-NEXT: flat_load_u8 v0, v[0:1]
; GFX11-GISEL-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
@@ -1731,7 +1686,6 @@ define i8 @flat_inst_valu_offset_64bit_12bit_split0(ptr %p) {
; GFX11-SDAG-TRUE16: ; %bb.0:
; GFX11-SDAG-TRUE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-SDAG-TRUE16-NEXT: v_add_co_u32 v0, vcc_lo, 0, v0
-; GFX11-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-SDAG-TRUE16-NEXT: v_add_co_ci_u32_e64 v1, null, 2, v1, vcc_lo
; GFX11-SDAG-TRUE16-NEXT: flat_load_d16_u8 v0, v[0:1] offset:4095
; GFX11-SDAG-TRUE16-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
@@ -1741,7 +1695,6 @@ define i8 @flat_inst_valu_offset_64bit_12bit_split0(ptr %p) {
; GFX11-SDAG-FAKE16: ; %bb.0:
; GFX11-SDAG-FAKE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-SDAG-FAKE16-NEXT: v_add_co_u32 v0, vcc_lo, 0, v0
-; GFX11-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-SDAG-FAKE16-NEXT: v_add_co_ci_u32_e64 v1, null, 2, v1, vcc_lo
; GFX11-SDAG-FAKE16-NEXT: flat_load_u8 v0, v[0:1] offset:4095
; GFX11-SDAG-FAKE16-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
@@ -1788,7 +1741,6 @@ define i8 @flat_inst_valu_offset_64bit_12bit_split0(ptr %p) {
; GFX11-GISEL: ; %bb.0:
; GFX11-GISEL-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, 0xfff, v0
-; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 2, v1, vcc_lo
; GFX11-GISEL-NEXT: flat_load_u8 v0, v[0:1]
; GFX11-GISEL-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
@@ -1836,7 +1788,6 @@ define i8 @flat_inst_valu_offset_64bit_12bit_split1(ptr %p) {
; GFX11-SDAG-TRUE16: ; %bb.0:
; GFX11-SDAG-TRUE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-SDAG-TRUE16-NEXT: v_add_co_u32 v0, vcc_lo, 0x1000, v0
-; GFX11-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-SDAG-TRUE16-NEXT: v_add_co_ci_u32_e64 v1, null, 2, v1, vcc_lo
; GFX11-SDAG-TRUE16-NEXT: flat_load_d16_u8 v0, v[0:1]
; GFX11-SDAG-TRUE16-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
@@ -1846,7 +1797,6 @@ define i8 @flat_inst_valu_offset_64bit_12bit_split1(ptr %p) {
; GFX11-SDAG-FAKE16: ; %bb.0:
; GFX11-SDAG-FAKE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-SDAG-FAKE16-NEXT: v_add_co_u32 v0, vcc_lo, 0x1000, v0
-; GFX11-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-SDAG-FAKE16-NEXT: v_add_co_ci_u32_e64 v1, null, 2, v1, vcc_lo
; GFX11-SDAG-FAKE16-NEXT: flat_load_u8 v0, v[0:1]
; GFX11-SDAG-FAKE16-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
@@ -1884,7 +1834,6 @@ define i8 @flat_inst_valu_offset_64bit_12bit_split1(ptr %p) {
; GFX11-GISEL: ; %bb.0:
; GFX11-GISEL-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, 0x1000, v0
-; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 2, v1, vcc_lo
; GFX11-GISEL-NEXT: flat_load_u8 v0, v[0:1]
; GFX11-GISEL-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
@@ -1932,7 +1881,6 @@ define i8 @flat_inst_valu_offset_64bit_13bit_split0(ptr %p) {
; GFX11-SDAG-TRUE16: ; %bb.0:
; GFX11-SDAG-TRUE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-SDAG-TRUE16-NEXT: v_add_co_u32 v0, vcc_lo, 0x1000, v0
-; GFX11-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-SDAG-TRUE16-NEXT: v_add_co_ci_u32_e64 v1, null, 2, v1, vcc_lo
; GFX11-SDAG-TRUE16-NEXT: flat_load_d16_u8 v0, v[0:1] offset:4095
; GFX11-SDAG-TRUE16-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
@@ -1942,7 +1890,6 @@ define i8 @flat_inst_valu_offset_64bit_13bit_split0(ptr %p) {
; GFX11-SDAG-FAKE16: ; %bb.0:
; GFX11-SDAG-FAKE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-SDAG-FAKE16-NEXT: v_add_co_u32 v0, vcc_lo, 0x1000, v0
-; GFX11-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-SDAG-FAKE16-NEXT: v_add_co_ci_u32_e64 v1, null, 2, v1, vcc_lo
; GFX11-SDAG-FAKE16-NEXT: flat_load_u8 v0, v[0:1] offset:4095
; GFX11-SDAG-FAKE16-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
@@ -1989,7 +1936,6 @@ define i8 @flat_inst_valu_offset_64bit_13bit_split0(ptr %p) {
; GFX11-GISEL: ; %bb.0:
; GFX11-GISEL-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, 0x1fff, v0
-; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 2, v1, vcc_lo
; GFX11-GISEL-NEXT: flat_load_u8 v0, v[0:1]
; GFX11-GISEL-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
@@ -2037,7 +1983,6 @@ define i8 @flat_inst_valu_offset_64bit_13bit_split1(ptr %p) {
; GFX11-SDAG-TRUE16: ; %bb.0:
; GFX11-SDAG-TRUE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-SDAG-TRUE16-NEXT: v_add_co_u32 v0, vcc_lo, 0x2000, v0
-; GFX11-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-SDAG-TRUE16-NEXT: v_add_co_ci_u32_e64 v1, null, 2, v1, vcc_lo
; GFX11-SDAG-TRUE16-NEXT: flat_load_d16_u8 v0, v[0:1]
; GFX11-SDAG-TRUE16-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
@@ -2047,7 +1992,6 @@ define i8 @flat_inst_valu_offset_64bit_13bit_split1(ptr %p) {
; GFX11-SDAG-FAKE16: ; %bb.0:
; GFX11-SDAG-FAKE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-SDAG-FAKE16-NEXT: v_add_co_u32 v0, vcc_lo, 0x2000, v0
-; GFX11-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-SDAG-FAKE16-NEXT: v_add_co_ci_u32_e64 v1, null, 2, v1, vcc_lo
; GFX11-SDAG-FAKE16-NEXT: flat_load_u8 v0, v[0:1]
; GFX11-SDAG-FAKE16-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
@@ -2085,7 +2029,6 @@ define i8 @flat_inst_valu_offset_64bit_13bit_split1(ptr %p) {
; GFX11-GISEL: ; %bb.0:
; GFX11-GISEL-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, 0x2000, v0
-; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 2, v1, vcc_lo
; GFX11-GISEL-NEXT: flat_load_u8 v0, v[0:1]
; GFX11-GISEL-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
@@ -2134,7 +2077,6 @@ define i8 @flat_inst_valu_offset_64bit_11bit_neg_high_split0(ptr %p) {
; GFX11-SDAG-TRUE16: ; %bb.0:
; GFX11-SDAG-TRUE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-SDAG-TRUE16-NEXT: v_add_co_u32 v0, vcc_lo, 0x7ff, v0
-; GFX11-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-SDAG-TRUE16-NEXT: v_add_co_ci_u32_e64 v1, null, 0x80000000, v1, vcc_lo
; GFX11-SDAG-TRUE16-NEXT: flat_load_d16_u8 v0, v[0:1]
; GFX11-SDAG-TRUE16-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
@@ -2144,7 +2086,6 @@ define i8 @flat_inst_valu_offset_64bit_11bit_neg_high_split0(ptr %p) {
; GFX11-SDAG-FAKE16: ; %bb.0:
; GFX11-SDAG-FAKE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-SDAG-FAKE16-NEXT: v_add_co_u32 v0, vcc_lo, 0x7ff, v0
-; GFX11-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-SDAG-FAKE16-NEXT: v_add_co_ci_u32_e64 v1, null, 0x80000000, v1, vcc_lo
; GFX11-SDAG-FAKE16-NEXT: flat_load_u8 v0, v[0:1]
; GFX11-SDAG-FAKE16-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
@@ -2192,7 +2133,6 @@ define i8 @flat_inst_valu_offset_64bit_11bit_neg_high_split0(ptr %p) {
; GFX11-GISEL: ; %bb.0:
; GFX11-GISEL-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, 0x7ff, v0
-; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0x80000000, v1, vcc_lo
; GFX11-GISEL-NEXT: flat_load_u8 v0, v[0:1]
; GFX11-GISEL-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
@@ -2241,7 +2181,6 @@ define i8 @flat_inst_valu_offset_64bit_11bit_neg_high_split1(ptr %p) {
; GFX11-SDAG-TRUE16: ; %bb.0:
; GFX11-SDAG-TRUE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-SDAG-TRUE16-NEXT: v_add_co_u32 v0, vcc_lo, 0x800, v0
-; GFX11-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-SDAG-TRUE16-NEXT: v_add_co_ci_u32_e64 v1, null, 0x80000000, v1, vcc_lo
; GFX11-SDAG-TRUE16-NEXT: flat_load_d16_u8 v0, v[0:1]
; GFX11-SDAG-TRUE16-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
@@ -2251,7 +2190,6 @@ define i8 @flat_inst_valu_offset_64bit_11bit_neg_high_split1(ptr %p) {
; GFX11-SDAG-FAKE16: ; %bb.0:
; GFX11-SDAG-FAKE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-SDAG-FAKE16-NEXT: v_add_co_u32 v0, vcc_lo, 0x800, v0
-; GFX11-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-SDAG-FAKE16-NEXT: v_add_co_ci_u32_e64 v1, null, 0x80000000, v1, vcc_lo
; GFX11-SDAG-FAKE16-NEXT: flat_load_u8 v0, v[0:1]
; GFX11-SDAG-FAKE16-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
@@ -2299,7 +2237,6 @@ define i8 @flat_inst_valu_offset_64bit_11bit_neg_high_split1(ptr %p) {
; GFX11-GISEL: ; %bb.0:
; GFX11-GISEL-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, 0x800, v0
-; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0x80000000, v1, vcc_lo
; GFX11-GISEL-NEXT: flat_load_u8 v0, v[0:1]
; GFX11-GISEL-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
@@ -2348,7 +2285,6 @@ define i8 @flat_inst_valu_offset_64bit_12bit_neg_high_split0(ptr %p) {
; GFX11-SDAG-TRUE16: ; %bb.0:
; GFX11-SDAG-TRUE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-SDAG-TRUE16-NEXT: v_add_co_u32 v0, vcc_lo, 0xfff, v0
-; GFX11-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-SDAG-TRUE16-NEXT: v_add_co_ci_u32_e64 v1, null, 0x80000000, v1, vcc_lo
; GFX11-SDAG-TRUE16-NEXT: flat_load_d16_u8 v0, v[0:1]
; GFX11-SDAG-TRUE16-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
@@ -2358,7 +2294,6 @@ define i8 @flat_inst_valu_offset_64bit_12bit_neg_high_split0(ptr %p) {
; GFX11-SDAG-FAKE16: ; %bb.0:
; GFX11-SDAG-FAKE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-SDAG-FAKE16-NEXT: v_add_co_u32 v0, vcc_lo, 0xfff, v0
-; GFX11-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-SDAG-FAKE16-NEXT: v_add_co_ci_u32_e64 v1, null, 0x80000000, v1, vcc_lo
; GFX11-SDAG-FAKE16-NEXT: flat_load_u8 v0, v[0:1]
; GFX11-SDAG-FAKE16-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
@@ -2406,7 +2341,6 @@ define i8 @flat_inst_valu_offset_64bit_12bit_neg_high_split0(ptr %p) {
; GFX11-GISEL: ; %bb.0:
; GFX11-GISEL-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, 0xfff, v0
-; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0x80000000, v1, vcc_lo
; GFX11-GISEL-NEXT: flat_load_u8 v0, v[0:1]
; GFX11-GISEL-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
@@ -2455,7 +2389,6 @@ define i8 @flat_inst_valu_offset_64bit_12bit_neg_high_split1(ptr %p) {
; GFX11-SDAG-TRUE16: ; %bb.0:
; GFX11-SDAG-TRUE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-SDAG-TRUE16-NEXT: v_add_co_u32 v0, vcc_lo, 0x1000, v0
-; GFX11-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-SDAG-TRUE16-NEXT: v_add_co_ci_u32_e64 v1, null, 0x80000000, v1, vcc_lo
; GFX11-SDAG-TRUE16-NEXT: flat_load_d16_u8 v0, v[0:1]
; GFX11-SDAG-TRUE16-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
@@ -2465,7 +2398,6 @@ define i8 @flat_inst_valu_offset_64bit_12bit_neg_high_split1(ptr %p) {
; GFX11-SDAG-FAKE16: ; %bb.0:
; GFX11-SDAG-FAKE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-SDAG-FAKE16-NEXT: v_add_co_u32 v0, vcc_lo, 0x1000, v0
-; GFX11-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-SDAG-FAKE16-NEXT: v_add_co_ci_u32_e64 v1, null, 0x80000000, v1, vcc_lo
; GFX11-SDAG-FAKE16-NEXT: flat_load_u8 v0, v[0:1]
; GFX11-SDAG-FAKE16-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
@@ -2513,7 +2445,6 @@ define i8 @flat_inst_valu_offset_64bit_12bit_neg_high_split1(ptr %p) {
; GFX11-GISEL: ; %bb.0:
; GFX11-GISEL-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, 0x1000, v0
-; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0x80000000, v1, vcc_lo
; GFX11-GISEL-NEXT: flat_load_u8 v0, v[0:1]
; GFX11-GISEL-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
@@ -2562,7 +2493,6 @@ define i8 @flat_inst_valu_offset_64bit_13bit_neg_high_split0(ptr %p) {
; GFX11-SDAG-TRUE16: ; %bb.0:
; GFX11-SDAG-TRUE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-SDAG-TRUE16-NEXT: v_add_co_u32 v0, vcc_lo, 0x1fff, v0
-; GFX11-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-SDAG-TRUE16-NEXT: v_add_co_ci_u32_e64 v1, null, 0x80000000, v1, vcc_lo
; GFX11-SDAG-TRUE16-NEXT: flat_load_d16_u8 v0, v[0:1]
; GFX11-SDAG-TRUE16-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
@@ -2572,7 +2502,6 @@ define i8 @flat_inst_valu_offset_64bit_13bit_neg_high_split0(ptr %p) {
; GFX11-SDAG-FAKE16: ; %bb.0:
; GFX11-SDAG-FAKE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-SDAG-FAKE16-NEXT: v_add_co_u32 v0, vcc_lo, 0x1fff, v0
-; GFX11-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-SDAG-FAKE16-NEXT: v_add_co_ci_u32_e64 v1, null, 0x80000000, v1, vcc_lo
; GFX11-SDAG-FAKE16-NEXT: flat_load_u8 v0, v[0:1]
; GFX11-SDAG-FAKE16-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
@@ -2620,7 +2549,6 @@ define i8 @flat_inst_valu_offset_64bit_13bit_neg_high_split0(ptr %p) {
; GFX11-GISEL: ; %bb.0:
; GFX11-GISEL-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, 0x1fff, v0
-; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0x80000000, v1, vcc_lo
; GFX11-GISEL-NEXT: flat_load_u8 v0, v[0:1]
; GFX11-GISEL-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
@@ -2669,7 +2597,6 @@ define i8 @flat_inst_valu_offset_64bit_13bit_neg_high_split1(ptr %p) {
; GFX11-SDAG-TRUE16: ; %bb.0:
; GFX11-SDAG-TRUE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-SDAG-TRUE16-NEXT: v_add_co_u32 v0, vcc_lo, 0x2000, v0
-; GFX11-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-SDAG-TRUE16-NEXT: v_add_co_ci_u32_e64 v1, null, 0x80000000, v1, vcc_lo
; GFX11-SDAG-TRUE16-NEXT: flat_load_d16_u8 v0, v[0:1]
; GFX11-SDAG-TRUE16-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
@@ -2679,7 +2606,6 @@ define i8 @flat_inst_valu_offset_64bit_13bit_neg_high_split1(ptr %p) {
; GFX11-SDAG-FAKE16: ; %bb.0:
; GFX11-SDAG-FAKE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-SDAG-FAKE16-NEXT: v_add_co_u32 v0, vcc_lo, 0x2000, v0
-; GFX11-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-SDAG-FAKE16-NEXT: v_add_co_ci_u32_e64 v1, null, 0x80000000, v1, vcc_lo
; GFX11-SDAG-FAKE16-NEXT: flat_load_u8 v0, v[0:1]
; GFX11-SDAG-FAKE16-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
@@ -2727,7 +2653,6 @@ define i8 @flat_inst_valu_offset_64bit_13bit_neg_high_split1(ptr %p) {
; GFX11-GISEL: ; %bb.0:
; GFX11-GISEL-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, 0x2000, v0
-; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0x80000000, v1, vcc_lo
; GFX11-GISEL-NEXT: flat_load_u8 v0, v[0:1]
; GFX11-GISEL-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
@@ -3053,7 +2978,6 @@ define amdgpu_kernel void @flat_inst_salu_offset_13bit_max(ptr %p) {
; GFX11-SDAG-TRUE16-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
; GFX11-SDAG-TRUE16-NEXT: s_waitcnt lgkmcnt(0)
; GFX11-SDAG-TRUE16-NEXT: v_add_co_u32 v0, s0, 0x1000, s0
-; GFX11-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-SDAG-TRUE16-NEXT: v_add_co_ci_u32_e64 v1, null, 0, s1, s0
; GFX11-SDAG-TRUE16-NEXT: flat_load_d16_u8 v0, v[0:1] offset:4095 glc dlc
; GFX11-SDAG-TRUE16-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
@@ -3065,7 +2989,6 @@ define amdgpu_kernel void @flat_inst_salu_offset_13bit_max(ptr %p) {
; GFX11-SDAG-FAKE16-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
; GFX11-SDAG-FAKE16-NEXT: s_waitcnt lgkmcnt(0)
; GFX11-SDAG-FAKE16-NEXT: v_add_co_u32 v0, s0, 0x1000, s0
-; GFX11-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-SDAG-FAKE16-NEXT: v_add_co_ci_u32_e64 v1, null, 0, s1, s0
; GFX11-SDAG-FAKE16-NEXT: flat_load_u8 v0, v[0:1] offset:4095 glc dlc
; GFX11-SDAG-FAKE16-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
@@ -3165,7 +3088,6 @@ define amdgpu_kernel void @flat_inst_salu_offset_neg_11bit_max(ptr %p) {
; GFX11-SDAG-TRUE16-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
; GFX11-SDAG-TRUE16-NEXT: s_waitcnt lgkmcnt(0)
; GFX11-SDAG-TRUE16-NEXT: v_add_co_u32 v0, s0, 0xfffff800, s0
-; GFX11-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-SDAG-TRUE16-NEXT: v_add_co_ci_u32_e64 v1, null, -1, s1, s0
; GFX11-SDAG-TRUE16-NEXT: flat_load_d16_u8 v0, v[0:1] glc dlc
; GFX11-SDAG-TRUE16-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
@@ -3177,7 +3099,6 @@ define amdgpu_kernel void @flat_inst_salu_offset_neg_11bit_max(ptr %p) {
; GFX11-SDAG-FAKE16-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
; GFX11-SDAG-FAKE16-NEXT: s_waitcnt lgkmcnt(0)
; GFX11-SDAG-FAKE16-NEXT: v_add_co_u32 v0, s0, 0xfffff800, s0
-; GFX11-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-SDAG-FAKE16-NEXT: v_add_co_ci_u32_e64 v1, null, -1, s1, s0
; GFX11-SDAG-FAKE16-NEXT: flat_load_u8 v0, v[0:1] glc dlc
; GFX11-SDAG-FAKE16-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
@@ -3277,7 +3198,6 @@ define amdgpu_kernel void @flat_inst_salu_offset_neg_12bit_max(ptr %p) {
; GFX11-SDAG-TRUE16-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
; GFX11-SDAG-TRUE16-NEXT: s_waitcnt lgkmcnt(0)
; GFX11-SDAG-TRUE16-NEXT: v_add_co_u32 v0, s0, 0xfffff000, s0
-; GFX11-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-SDAG-TRUE16-NEXT: v_add_co_ci_u32_e64 v1, null, -1, s1, s0
; GFX11-SDAG-TRUE16-NEXT: flat_load_d16_u8 v0, v[0:1] glc dlc
; GFX11-SDAG-TRUE16-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
@@ -3289,7 +3209,6 @@ define amdgpu_kernel void @flat_inst_salu_offset_neg_12bit_max(ptr %p) {
; GFX11-SDAG-FAKE16-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
; GFX11-SDAG-FAKE16-NEXT: s_waitcnt lgkmcnt(0)
; GFX11-SDAG-FAKE16-NEXT: v_add_co_u32 v0, s0, 0xfffff000, s0
-; GFX11-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-SDAG-FAKE16-NEXT: v_add_co_ci_u32_e64 v1, null, -1, s1, s0
; GFX11-SDAG-FAKE16-NEXT: flat_load_u8 v0, v[0:1] glc dlc
; GFX11-SDAG-FAKE16-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
@@ -3389,7 +3308,6 @@ define amdgpu_kernel void @flat_inst_salu_offset_neg_13bit_max(ptr %p) {
; GFX11-SDAG-TRUE16-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
; GFX11-SDAG-TRUE16-NEXT: s_waitcnt lgkmcnt(0)
; GFX11-SDAG-TRUE16-NEXT: v_add_co_u32 v0, s0, 0xffffe000, s0
-; GFX11-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-SDAG-TRUE16-NEXT: v_add_co_ci_u32_e64 v1, null, -1, s1, s0
; GFX11-SDAG-TRUE16-NEXT: flat_load_d16_u8 v0, v[0:1] glc dlc
; GFX11-SDAG-TRUE16-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
@@ -3401,7 +3319,6 @@ define amdgpu_kernel void @flat_inst_salu_offset_neg_13bit_max(ptr %p) {
; GFX11-SDAG-FAKE16-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
; GFX11-SDAG-FAKE16-NEXT: s_waitcnt lgkmcnt(0)
; GFX11-SDAG-FAKE16-NEXT: v_add_co_u32 v0, s0, 0xffffe000, s0
-; GFX11-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-SDAG-FAKE16-NEXT: v_add_co_ci_u32_e64 v1, null, -1, s1, s0
; GFX11-SDAG-FAKE16-NEXT: flat_load_u8 v0, v[0:1] glc dlc
; GFX11-SDAG-FAKE16-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
@@ -3591,7 +3508,6 @@ define amdgpu_kernel void @flat_inst_salu_offset_2x_12bit_max(ptr %p) {
; GFX11-SDAG-TRUE16-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
; GFX11-SDAG-TRUE16-NEXT: s_waitcnt lgkmcnt(0)
; GFX11-SDAG-TRUE16-NEXT: v_add_co_u32 v0, s0, 0x1000, s0
-; GFX11-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-SDAG-TRUE16-NEXT: v_add_co_ci_u32_e64 v1, null, 0, s1, s0
; GFX11-SDAG-TRUE16-NEXT: flat_load_d16_u8 v0, v[0:1] offset:4095 glc dlc
; GFX11-SDAG-TRUE16-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
@@ -3603,7 +3519,6 @@ define amdgpu_kernel void @flat_inst_salu_offset_2x_12bit_max(ptr %p) {
; GFX11-SDAG-FAKE16-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
; GFX11-SDAG-FAKE16-NEXT: s_waitcnt lgkmcnt(0)
; GFX11-SDAG-FAKE16-NEXT: v_add_co_u32 v0, s0, 0x1000, s0
-; GFX11-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-SDAG-FAKE16-NEXT: v_add_co_ci_u32_e64 v1, null, 0, s1, s0
; GFX11-SDAG-FAKE16-NEXT: flat_load_u8 v0, v[0:1] offset:4095 glc dlc
; GFX11-SDAG-FAKE16-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
@@ -3703,7 +3618,6 @@ define amdgpu_kernel void @flat_inst_salu_offset_2x_13bit_max(ptr %p) {
; GFX11-SDAG-TRUE16-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
; GFX11-SDAG-TRUE16-NEXT: s_waitcnt lgkmcnt(0)
; GFX11-SDAG-TRUE16-NEXT: v_add_co_u32 v0, s0, 0x3000, s0
-; GFX11-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-SDAG-TRUE16-NEXT: v_add_co_ci_u32_e64 v1, null, 0, s1, s0
; GFX11-SDAG-TRUE16-NEXT: flat_load_d16_u8 v0, v[0:1] offset:4095 glc dlc
; GFX11-SDAG-TRUE16-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
@@ -3715,7 +3629,6 @@ define amdgpu_kernel void @flat_inst_salu_offset_2x_13bit_max(ptr %p) {
; GFX11-SDAG-FAKE16-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
; GFX11-SDAG-FAKE16-NEXT: s_waitcnt lgkmcnt(0)
; GFX11-SDAG-FAKE16-NEXT: v_add_co_u32 v0, s0, 0x3000, s0
-; GFX11-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-SDAG-FAKE16-NEXT: v_add_co_ci_u32_e64 v1, null, 0, s1, s0
; GFX11-SDAG-FAKE16-NEXT: flat_load_u8 v0, v[0:1] offset:4095 glc dlc
; GFX11-SDAG-FAKE16-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
@@ -3815,7 +3728,6 @@ define amdgpu_kernel void @flat_inst_salu_offset_2x_neg_11bit_max(ptr %p) {
; GFX11-SDAG-TRUE16-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
; GFX11-SDAG-TRUE16-NEXT: s_waitcnt lgkmcnt(0)
; GFX11-SDAG-TRUE16-NEXT: v_add_co_u32 v0, s0, 0xfffff000, s0
-; GFX11-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-SDAG-TRUE16-NEXT: v_add_co_ci_u32_e64 v1, null, -1, s1, s0
; GFX11-SDAG-TRUE16-NEXT: flat_load_d16_u8 v0, v[0:1] glc dlc
; GFX11-SDAG-TRUE16-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
@@ -3827,7 +3739,6 @@ define amdgpu_kernel void @flat_inst_salu_offset_2x_neg_11bit_max(ptr %p) {
; GFX11-SDAG-FAKE16-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
; GFX11-SDAG-FAKE16-NEXT: s_waitcnt lgkmcnt(0)
; GFX11-SDAG-FAKE16-NEXT: v_add_co_u32 v0, s0, 0xfffff000, s0
-; GFX11-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-SDAG-FAKE16-NEXT: v_add_co_ci_u32_e64 v1, null, -1, s1, s0
; GFX11-SDAG-FAKE16-NEXT: flat_load_u8 v0, v[0:1] glc dlc
; GFX11-SDAG-FAKE16-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
@@ -3927,7 +3838,6 @@ define amdgpu_kernel void @flat_inst_salu_offset_2x_neg_12bit_max(ptr %p) {
; GFX11-SDAG-TRUE16-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
; GFX11-SDAG-TRUE16-NEXT: s_waitcnt lgkmcnt(0)
; GFX11-SDAG-TRUE16-NEXT: v_add_co_u32 v0, s0, 0xffffe000, s0
-; GFX11-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-SDAG-TRUE16-NEXT: v_add_co_ci_u32_e64 v1, null, -1, s1, s0
; GFX11-SDAG-TRUE16-NEXT: flat_load_d16_u8 v0, v[0:1] glc dlc
; GFX11-SDAG-TRUE16-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
@@ -3939,7 +3849,6 @@ define amdgpu_kernel void @flat_inst_salu_offset_2x_neg_12bit_max(ptr %p) {
; GFX11-SDAG-FAKE16-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
; GFX11-SDAG-FAKE16-NEXT: s_waitcnt lgkmcnt(0)
; GFX11-SDAG-FAKE16-NEXT: v_add_co_u32 v0, s0, 0xffffe000, s0
-; GFX11-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-SDAG-FAKE16-NEXT: v_add_co_ci_u32_e64 v1, null, -1, s1, s0
; GFX11-SDAG-FAKE16-NEXT: flat_load_u8 v0, v[0:1] glc dlc
; GFX11-SDAG-FAKE16-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
@@ -4039,7 +3948,6 @@ define amdgpu_kernel void @flat_inst_salu_offset_2x_neg_13bit_max(ptr %p) {
; GFX11-SDAG-TRUE16-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
; GFX11-SDAG-TRUE16-NEXT: s_waitcnt lgkmcnt(0)
; GFX11-SDAG-TRUE16-NEXT: v_add_co_u32 v0, s0, 0xffffc000, s0
-; GFX11-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-SDAG-TRUE16-NEXT: v_add_co_ci_u32_e64 v1, null, -1, s1, s0
; GFX11-SDAG-TRUE16-NEXT: flat_load_d16_u8 v0, v[0:1] glc dlc
; GFX11-SDAG-TRUE16-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
@@ -4051,7 +3959,6 @@ define amdgpu_kernel void @flat_inst_salu_offset_2x_neg_13bit_max(ptr %p) {
; GFX11-SDAG-FAKE16-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
; GFX11-SDAG-FAKE16-NEXT: s_waitcnt lgkmcnt(0)
; GFX11-SDAG-FAKE16-NEXT: v_add_co_u32 v0, s0, 0xffffc000, s0
-; GFX11-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-SDAG-FAKE16-NEXT: v_add_co_ci_u32_e64 v1, null, -1, s1, s0
; GFX11-SDAG-FAKE16-NEXT: flat_load_u8 v0, v[0:1] glc dlc
; GFX11-SDAG-FAKE16-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
@@ -4151,7 +4058,6 @@ define amdgpu_kernel void @flat_inst_salu_offset_64bit_11bit_split0(ptr %p) {
; GFX11-SDAG-TRUE16-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
; GFX11-SDAG-TRUE16-NEXT: s_waitcnt lgkmcnt(0)
; GFX11-SDAG-TRUE16-NEXT: v_add_co_u32 v0, s0, 0, s0
-; GFX11-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-SDAG-TRUE16-NEXT: v_add_co_ci_u32_e64 v1, null, 2, s1, s0
; GFX11-SDAG-TRUE16-NEXT: flat_load_d16_u8 v0, v[0:1] offset:2047 glc dlc
; GFX11-SDAG-TRUE16-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
@@ -4163,7 +4069,6 @@ define amdgpu_kernel void @flat_inst_salu_offset_64bit_11bit_split0(ptr %p) {
; GFX11-SDAG-FAKE16-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
; GFX11-SDAG-FAKE16-NEXT: s_waitcnt lgkmcnt(0)
; GFX11-SDAG-FAKE16-NEXT: v_add_co_u32 v0, s0, 0, s0
-; GFX11-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-SDAG-FAKE16-NEXT: v_add_co_ci_u32_e64 v1, null, 2, s1, s0
; GFX11-SDAG-FAKE16-NEXT: flat_load_u8 v0, v[0:1] offset:2047 glc dlc
; GFX11-SDAG-FAKE16-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
@@ -4270,7 +4175,6 @@ define amdgpu_kernel void @flat_inst_salu_offset_64bit_11bit_split1(ptr %p) {
; GFX11-SDAG-TRUE16-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
; GFX11-SDAG-TRUE16-NEXT: s_waitcnt lgkmcnt(0)
; GFX11-SDAG-TRUE16-NEXT: v_add_co_u32 v0, s0, 0, s0
-; GFX11-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-SDAG-TRUE16-NEXT: v_add_co_ci_u32_e64 v1, null, 2, s1, s0
; GFX11-SDAG-TRUE16-NEXT: flat_load_d16_u8 v0, v[0:1] offset:2048 glc dlc
; GFX11-SDAG-TRUE16-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
@@ -4282,7 +4186,6 @@ define amdgpu_kernel void @flat_inst_salu_offset_64bit_11bit_split1(ptr %p) {
; GFX11-SDAG-FAKE16-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
; GFX11-SDAG-FAKE16-NEXT: s_waitcnt lgkmcnt(0)
; GFX11-SDAG-FAKE16-NEXT: v_add_co_u32 v0, s0, 0, s0
-; GFX11-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-SDAG-FAKE16-NEXT: v_add_co_ci_u32_e64 v1, null, 2, s1, s0
; GFX11-SDAG-FAKE16-NEXT: flat_load_u8 v0, v[0:1] offset:2048 glc dlc
; GFX11-SDAG-FAKE16-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
@@ -4389,7 +4292,6 @@ define amdgpu_kernel void @flat_inst_salu_offset_64bit_12bit_split0(ptr %p) {
; GFX11-SDAG-TRUE16-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
; GFX11-SDAG-TRUE16-NEXT: s_waitcnt lgkmcnt(0)
; GFX11-SDAG-TRUE16-NEXT: v_add_co_u32 v0, s0, 0, s0
-; GFX11-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-SDAG-TRUE16-NEXT: v_add_co_ci_u32_e64 v1, null, 2, s1, s0
; GFX11-SDAG-TRUE16-NEXT: flat_load_d16_u8 v0, v[0:1] offset:4095 glc dlc
; GFX11-SDAG-TRUE16-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
@@ -4401,7 +4303,6 @@ define amdgpu_kernel void @flat_inst_salu_offset_64bit_12bit_split0(ptr %p) {
; GFX11-SDAG-FAKE16-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
; GFX11-SDAG-FAKE16-NEXT: s_waitcnt lgkmcnt(0)
; GFX11-SDAG-FAKE16-NEXT: v_add_co_u32 v0, s0, 0, s0
-; GFX11-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-SDAG-FAKE16-NEXT: v_add_co_ci_u32_e64 v1, null, 2, s1, s0
; GFX11-SDAG-FAKE16-NEXT: flat_load_u8 v0, v[0:1] offset:4095 glc dlc
; GFX11-SDAG-FAKE16-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
@@ -4509,7 +4410,6 @@ define amdgpu_kernel void @flat_inst_salu_offset_64bit_12bit_split1(ptr %p) {
; GFX11-SDAG-TRUE16-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
; GFX11-SDAG-TRUE16-NEXT: s_waitcnt lgkmcnt(0)
; GFX11-SDAG-TRUE16-NEXT: v_add_co_u32 v0, s0, 0x1000, s0
-; GFX11-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-SDAG-TRUE16-NEXT: v_add_co_ci_u32_e64 v1, null, 2, s1, s0
; GFX11-SDAG-TRUE16-NEXT: flat_load_d16_u8 v0, v[0:1] glc dlc
; GFX11-SDAG-TRUE16-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
@@ -4521,7 +4421,6 @@ define amdgpu_kernel void @flat_inst_salu_offset_64bit_12bit_split1(ptr %p) {
; GFX11-SDAG-FAKE16-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
; GFX11-SDAG-FAKE16-NEXT: s_waitcnt lgkmcnt(0)
; GFX11-SDAG-FAKE16-NEXT: v_add_co_u32 v0, s0, 0x1000, s0
-; GFX11-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-SDAG-FAKE16-NEXT: v_add_co_ci_u32_e64 v1, null, 2, s1, s0
; GFX11-SDAG-FAKE16-NEXT: flat_load_u8 v0, v[0:1] glc dlc
; GFX11-SDAG-FAKE16-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
@@ -4629,7 +4528,6 @@ define amdgpu_kernel void @flat_inst_salu_offset_64bit_13bit_split0(ptr %p) {
; GFX11-SDAG-TRUE16-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
; GFX11-SDAG-TRUE16-NEXT: s_waitcnt lgkmcnt(0)
; GFX11-SDAG-TRUE16-NEXT: v_add_co_u32 v0, s0, 0x1000, s0
-; GFX11-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-SDAG-TRUE16-NEXT: v_add_co_ci_u32_e64 v1, null, 2, s1, s0
; GFX11-SDAG-TRUE16-NEXT: flat_load_d16_u8 v0, v[0:1] offset:4095 glc dlc
; GFX11-SDAG-TRUE16-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
@@ -4641,7 +4539,6 @@ define amdgpu_kernel void @flat_inst_salu_offset_64bit_13bit_split0(ptr %p) {
; GFX11-SDAG-FAKE16-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
; GFX11-SDAG-FAKE16-NEXT: s_waitcnt lgkmcnt(0)
; GFX11-SDAG-FAKE16-NEXT: v_add_co_u32 v0, s0, 0x1000, s0
-; GFX11-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-SDAG-FAKE16-NEXT: v_add_co_ci_u32_e64 v1, null, 2, s1, s0
; GFX11-SDAG-FAKE16-NEXT: flat_load_u8 v0, v[0:1] offset:4095 glc dlc
; GFX11-SDAG-FAKE16-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
@@ -4749,7 +4646,6 @@ define amdgpu_kernel void @flat_inst_salu_offset_64bit_13bit_split1(ptr %p) {
; GFX11-SDAG-TRUE16-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
; GFX11-SDAG-TRUE16-NEXT: s_waitcnt lgkmcnt(0)
; GFX11-SDAG-TRUE16-NEXT: v_add_co_u32 v0, s0, 0x2000, s0
-; GFX11-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-SDAG-TRUE16-NEXT: v_add_co_ci_u32_e64 v1, null, 2, s1, s0
; GFX11-SDAG-TRUE16-NEXT: flat_load_d16_u8 v0, v[0:1] glc dlc
; GFX11-SDAG-TRUE16-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
@@ -4761,7 +4657,6 @@ define amdgpu_kernel void @flat_inst_salu_offset_64bit_13bit_split1(ptr %p) {
; GFX11-SDAG-FAKE16-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
; GFX11-SDAG-FAKE16-NEXT: s_waitcnt lgkmcnt(0)
; GFX11-SDAG-FAKE16-NEXT: v_add_co_u32 v0, s0, 0x2000, s0
-; GFX11-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-SDAG-FAKE16-NEXT: v_add_co_ci_u32_e64 v1, null, 2, s1, s0
; GFX11-SDAG-FAKE16-NEXT: flat_load_u8 v0, v[0:1] glc dlc
; GFX11-SDAG-FAKE16-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
@@ -4871,7 +4766,7 @@ define amdgpu_kernel void @flat_inst_salu_offset_64bit_11bit_neg_high_split0(ptr
; GFX11-SDAG-TRUE16-NEXT: s_waitcnt lgkmcnt(0)
; GFX11-SDAG-TRUE16-NEXT: v_mov_b32_e32 v1, s1
; GFX11-SDAG-TRUE16-NEXT: v_add_co_u32 v0, vcc_lo, 0x7ff, s0
-; GFX11-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX11-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX11-SDAG-TRUE16-NEXT: v_add_co_ci_u32_e64 v1, null, 0x80000000, v1, vcc_lo
; GFX11-SDAG-TRUE16-NEXT: flat_load_d16_u8 v0, v[0:1] glc dlc
; GFX11-SDAG-TRUE16-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
@@ -4884,7 +4779,7 @@ define amdgpu_kernel void @flat_inst_salu_offset_64bit_11bit_neg_high_split0(ptr
; GFX11-SDAG-FAKE16-NEXT: s_waitcnt lgkmcnt(0)
; GFX11-SDAG-FAKE16-NEXT: v_mov_b32_e32 v1, s1
; GFX11-SDAG-FAKE16-NEXT: v_add_co_u32 v0, vcc_lo, 0x7ff, s0
-; GFX11-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX11-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX11-SDAG-FAKE16-NEXT: v_add_co_ci_u32_e64 v1, null, 0x80000000, v1, vcc_lo
; GFX11-SDAG-FAKE16-NEXT: flat_load_u8 v0, v[0:1] glc dlc
; GFX11-SDAG-FAKE16-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
@@ -4897,7 +4792,7 @@ define amdgpu_kernel void @flat_inst_salu_offset_64bit_11bit_neg_high_split0(ptr
; GFX12-SDAG-TRUE16-NEXT: s_wait_kmcnt 0x0
; GFX12-SDAG-TRUE16-NEXT: v_mov_b32_e32 v1, s1
; GFX12-SDAG-TRUE16-NEXT: v_add_co_u32 v0, vcc_lo, 0x800000, s0
-; GFX12-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX12-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX12-SDAG-TRUE16-NEXT: v_add_co_ci_u32_e64 v1, null, 0x80000000, v1, vcc_lo
; GFX12-SDAG-TRUE16-NEXT: flat_load_d16_u8 v0, v[0:1] offset:-8386561 scope:SCOPE_SYS
; GFX12-SDAG-TRUE16-NEXT: s_wait_loadcnt_dscnt 0x0
@@ -4910,7 +4805,7 @@ define amdgpu_kernel void @flat_inst_salu_offset_64bit_11bit_neg_high_split0(ptr
; GFX12-SDAG-FAKE16-NEXT: s_wait_kmcnt 0x0
; GFX12-SDAG-FAKE16-NEXT: v_mov_b32_e32 v1, s1
; GFX12-SDAG-FAKE16-NEXT: v_add_co_u32 v0, vcc_lo, 0x800000, s0
-; GFX12-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX12-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX12-SDAG-FAKE16-NEXT: v_add_co_ci_u32_e64 v1, null, 0x80000000, v1, vcc_lo
; GFX12-SDAG-FAKE16-NEXT: flat_load_u8 v0, v[0:1] offset:-8386561 scope:SCOPE_SYS
; GFX12-SDAG-FAKE16-NEXT: s_wait_loadcnt_dscnt 0x0
@@ -4996,7 +4891,7 @@ define amdgpu_kernel void @flat_inst_salu_offset_64bit_11bit_neg_high_split1(ptr
; GFX11-SDAG-TRUE16-NEXT: s_waitcnt lgkmcnt(0)
; GFX11-SDAG-TRUE16-NEXT: v_mov_b32_e32 v1, s1
; GFX11-SDAG-TRUE16-NEXT: v_add_co_u32 v0, vcc_lo, 0x800, s0
-; GFX11-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX11-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX11-SDAG-TRUE16-NEXT: v_add_co_ci_u32_e64 v1, null, 0x80000000, v1, vcc_lo
; GFX11-SDAG-TRUE16-NEXT: flat_load_d16_u8 v0, v[0:1] glc dlc
; GFX11-SDAG-TRUE16-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
@@ -5009,7 +4904,7 @@ define amdgpu_kernel void @flat_inst_salu_offset_64bit_11bit_neg_high_split1(ptr
; GFX11-SDAG-FAKE16-NEXT: s_waitcnt lgkmcnt(0)
; GFX11-SDAG-FAKE16-NEXT: v_mov_b32_e32 v1, s1
; GFX11-SDAG-FAKE16-NEXT: v_add_co_u32 v0, vcc_lo, 0x800, s0
-; GFX11-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX11-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX11-SDAG-FAKE16-NEXT: v_add_co_ci_u32_e64 v1, null, 0x80000000, v1, vcc_lo
; GFX11-SDAG-FAKE16-NEXT: flat_load_u8 v0, v[0:1] glc dlc
; GFX11-SDAG-FAKE16-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
@@ -5022,7 +4917,7 @@ define amdgpu_kernel void @flat_inst_salu_offset_64bit_11bit_neg_high_split1(ptr
; GFX12-SDAG-TRUE16-NEXT: s_wait_kmcnt 0x0
; GFX12-SDAG-TRUE16-NEXT: v_mov_b32_e32 v1, s1
; GFX12-SDAG-TRUE16-NEXT: v_add_co_u32 v0, vcc_lo, 0x800000, s0
-; GFX12-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX12-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX12-SDAG-TRUE16-NEXT: v_add_co_ci_u32_e64 v1, null, 0x80000000, v1, vcc_lo
; GFX12-SDAG-TRUE16-NEXT: flat_load_d16_u8 v0, v[0:1] offset:-8386560 scope:SCOPE_SYS
; GFX12-SDAG-TRUE16-NEXT: s_wait_loadcnt_dscnt 0x0
@@ -5035,7 +4930,7 @@ define amdgpu_kernel void @flat_inst_salu_offset_64bit_11bit_neg_high_split1(ptr
; GFX12-SDAG-FAKE16-NEXT: s_wait_kmcnt 0x0
; GFX12-SDAG-FAKE16-NEXT: v_mov_b32_e32 v1, s1
; GFX12-SDAG-FAKE16-NEXT: v_add_co_u32 v0, vcc_lo, 0x800000, s0
-; GFX12-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX12-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX12-SDAG-FAKE16-NEXT: v_add_co_ci_u32_e64 v1, null, 0x80000000, v1, vcc_lo
; GFX12-SDAG-FAKE16-NEXT: flat_load_u8 v0, v[0:1] offset:-8386560 scope:SCOPE_SYS
; GFX12-SDAG-FAKE16-NEXT: s_wait_loadcnt_dscnt 0x0
@@ -5121,7 +5016,7 @@ define amdgpu_kernel void @flat_inst_salu_offset_64bit_12bit_neg_high_split0(ptr
; GFX11-SDAG-TRUE16-NEXT: s_waitcnt lgkmcnt(0)
; GFX11-SDAG-TRUE16-NEXT: v_mov_b32_e32 v1, s1
; GFX11-SDAG-TRUE16-NEXT: v_add_co_u32 v0, vcc_lo, 0xfff, s0
-; GFX11-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX11-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX11-SDAG-TRUE16-NEXT: v_add_co_ci_u32_e64 v1, null, 0x80000000, v1, vcc_lo
; GFX11-SDAG-TRUE16-NEXT: flat_load_d16_u8 v0, v[0:1] glc dlc
; GFX11-SDAG-TRUE16-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
@@ -5134,7 +5029,7 @@ define amdgpu_kernel void @flat_inst_salu_offset_64bit_12bit_neg_high_split0(ptr
; GFX11-SDAG-FAKE16-NEXT: s_waitcnt lgkmcnt(0)
; GFX11-SDAG-FAKE16-NEXT: v_mov_b32_e32 v1, s1
; GFX11-SDAG-FAKE16-NEXT: v_add_co_u32 v0, vcc_lo, 0xfff, s0
-; GFX11-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX11-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX11-SDAG-FAKE16-NEXT: v_add_co_ci_u32_e64 v1, null, 0x80000000, v1, vcc_lo
; GFX11-SDAG-FAKE16-NEXT: flat_load_u8 v0, v[0:1] glc dlc
; GFX11-SDAG-FAKE16-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
@@ -5147,7 +5042,7 @@ define amdgpu_kernel void @flat_inst_salu_offset_64bit_12bit_neg_high_split0(ptr
; GFX12-SDAG-TRUE16-NEXT: s_wait_kmcnt 0x0
; GFX12-SDAG-TRUE16-NEXT: v_mov_b32_e32 v1, s1
; GFX12-SDAG-TRUE16-NEXT: v_add_co_u32 v0, vcc_lo, 0x800000, s0
-; GFX12-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX12-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX12-SDAG-TRUE16-NEXT: v_add_co_ci_u32_e64 v1, null, 0x80000000, v1, vcc_lo
; GFX12-SDAG-TRUE16-NEXT: flat_load_d16_u8 v0, v[0:1] offset:-8384513 scope:SCOPE_SYS
; GFX12-SDAG-TRUE16-NEXT: s_wait_loadcnt_dscnt 0x0
@@ -5160,7 +5055,7 @@ define amdgpu_kernel void @flat_inst_salu_offset_64bit_12bit_neg_high_split0(ptr
; GFX12-SDAG-FAKE16-NEXT: s_wait_kmcnt 0x0
; GFX12-SDAG-FAKE16-NEXT: v_mov_b32_e32 v1, s1
; GFX12-SDAG-FAKE16-NEXT: v_add_co_u32 v0, vcc_lo, 0x800000, s0
-; GFX12-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX12-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX12-SDAG-FAKE16-NEXT: v_add_co_ci_u32_e64 v1, null, 0x80000000, v1, vcc_lo
; GFX12-SDAG-FAKE16-NEXT: flat_load_u8 v0, v[0:1] offset:-8384513 scope:SCOPE_SYS
; GFX12-SDAG-FAKE16-NEXT: s_wait_loadcnt_dscnt 0x0
@@ -5246,7 +5141,7 @@ define amdgpu_kernel void @flat_inst_salu_offset_64bit_12bit_neg_high_split1(ptr
; GFX11-SDAG-TRUE16-NEXT: s_waitcnt lgkmcnt(0)
; GFX11-SDAG-TRUE16-NEXT: v_mov_b32_e32 v1, s1
; GFX11-SDAG-TRUE16-NEXT: v_add_co_u32 v0, vcc_lo, 0x1000, s0
-; GFX11-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX11-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX11-SDAG-TRUE16-NEXT: v_add_co_ci_u32_e64 v1, null, 0x80000000, v1, vcc_lo
; GFX11-SDAG-TRUE16-NEXT: flat_load_d16_u8 v0, v[0:1] glc dlc
; GFX11-SDAG-TRUE16-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
@@ -5259,7 +5154,7 @@ define amdgpu_kernel void @flat_inst_salu_offset_64bit_12bit_neg_high_split1(ptr
; GFX11-SDAG-FAKE16-NEXT: s_waitcnt lgkmcnt(0)
; GFX11-SDAG-FAKE16-NEXT: v_mov_b32_e32 v1, s1
; GFX11-SDAG-FAKE16-NEXT: v_add_co_u32 v0, vcc_lo, 0x1000, s0
-; GFX11-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX11-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX11-SDAG-FAKE16-NEXT: v_add_co_ci_u32_e64 v1, null, 0x80000000, v1, vcc_lo
; GFX11-SDAG-FAKE16-NEXT: flat_load_u8 v0, v[0:1] glc dlc
; GFX11-SDAG-FAKE16-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
@@ -5272,7 +5167,7 @@ define amdgpu_kernel void @flat_inst_salu_offset_64bit_12bit_neg_high_split1(ptr
; GFX12-SDAG-TRUE16-NEXT: s_wait_kmcnt 0x0
; GFX12-SDAG-TRUE16-NEXT: v_mov_b32_e32 v1, s1
; GFX12-SDAG-TRUE16-NEXT: v_add_co_u32 v0, vcc_lo, 0x800000, s0
-; GFX12-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX12-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX12-SDAG-TRUE16-NEXT: v_add_co_ci_u32_e64 v1, null, 0x80000000, v1, vcc_lo
; GFX12-SDAG-TRUE16-NEXT: flat_load_d16_u8 v0, v[0:1] offset:-8384512 scope:SCOPE_SYS
; GFX12-SDAG-TRUE16-NEXT: s_wait_loadcnt_dscnt 0x0
@@ -5285,7 +5180,7 @@ define amdgpu_kernel void @flat_inst_salu_offset_64bit_12bit_neg_high_split1(ptr
; GFX12-SDAG-FAKE16-NEXT: s_wait_kmcnt 0x0
; GFX12-SDAG-FAKE16-NEXT: v_mov_b32_e32 v1, s1
; GFX12-SDAG-FAKE16-NEXT: v_add_co_u32 v0, vcc_lo, 0x800000, s0
-; GFX12-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX12-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX12-SDAG-FAKE16-NEXT: v_add_co_ci_u32_e64 v1, null, 0x80000000, v1, vcc_lo
; GFX12-SDAG-FAKE16-NEXT: flat_load_u8 v0, v[0:1] offset:-8384512 scope:SCOPE_SYS
; GFX12-SDAG-FAKE16-NEXT: s_wait_loadcnt_dscnt 0x0
@@ -5371,7 +5266,7 @@ define amdgpu_kernel void @flat_inst_salu_offset_64bit_13bit_neg_high_split0(ptr
; GFX11-SDAG-TRUE16-NEXT: s_waitcnt lgkmcnt(0)
; GFX11-SDAG-TRUE16-NEXT: v_mov_b32_e32 v1, s1
; GFX11-SDAG-TRUE16-NEXT: v_add_co_u32 v0, vcc_lo, 0x1fff, s0
-; GFX11-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX11-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX11-SDAG-TRUE16-NEXT: v_add_co_ci_u32_e64 v1, null, 0x80000000, v1, vcc_lo
; GFX11-SDAG-TRUE16-NEXT: flat_load_d16_u8 v0, v[0:1] glc dlc
; GFX11-SDAG-TRUE16-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
@@ -5384,7 +5279,7 @@ define amdgpu_kernel void @flat_inst_salu_offset_64bit_13bit_neg_high_split0(ptr
; GFX11-SDAG-FAKE16-NEXT: s_waitcnt lgkmcnt(0)
; GFX11-SDAG-FAKE16-NEXT: v_mov_b32_e32 v1, s1
; GFX11-SDAG-FAKE16-NEXT: v_add_co_u32 v0, vcc_lo, 0x1fff, s0
-; GFX11-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX11-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX11-SDAG-FAKE16-NEXT: v_add_co_ci_u32_e64 v1, null, 0x80000000, v1, vcc_lo
; GFX11-SDAG-FAKE16-NEXT: flat_load_u8 v0, v[0:1] glc dlc
; GFX11-SDAG-FAKE16-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
@@ -5397,7 +5292,7 @@ define amdgpu_kernel void @flat_inst_salu_offset_64bit_13bit_neg_high_split0(ptr
; GFX12-SDAG-TRUE16-NEXT: s_wait_kmcnt 0x0
; GFX12-SDAG-TRUE16-NEXT: v_mov_b32_e32 v1, s1
; GFX12-SDAG-TRUE16-NEXT: v_add_co_u32 v0, vcc_lo, 0x800000, s0
-; GFX12-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX12-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX12-SDAG-TRUE16-NEXT: v_add_co_ci_u32_e64 v1, null, 0x80000000, v1, vcc_lo
; GFX12-SDAG-TRUE16-NEXT: flat_load_d16_u8 v0, v[0:1] offset:-8380417 scope:SCOPE_SYS
; GFX12-SDAG-TRUE16-NEXT: s_wait_loadcnt_dscnt 0x0
@@ -5410,7 +5305,7 @@ define amdgpu_kernel void @flat_inst_salu_offset_64bit_13bit_neg_high_split0(ptr
; GFX12-SDAG-FAKE16-NEXT: s_wait_kmcnt 0x0
; GFX12-SDAG-FAKE16-NEXT: v_mov_b32_e32 v1, s1
; GFX12-SDAG-FAKE16-NEXT: v_add_co_u32 v0, vcc_lo, 0x800000, s0
-; GFX12-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX12-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX12-SDAG-FAKE16-NEXT: v_add_co_ci_u32_e64 v1, null, 0x80000000, v1, vcc_lo
; GFX12-SDAG-FAKE16-NEXT: flat_load_u8 v0, v[0:1] offset:-8380417 scope:SCOPE_SYS
; GFX12-SDAG-FAKE16-NEXT: s_wait_loadcnt_dscnt 0x0
@@ -5496,7 +5391,7 @@ define amdgpu_kernel void @flat_inst_salu_offset_64bit_13bit_neg_high_split1(ptr
; GFX11-SDAG-TRUE16-NEXT: s_waitcnt lgkmcnt(0)
; GFX11-SDAG-TRUE16-NEXT: v_mov_b32_e32 v1, s1
; GFX11-SDAG-TRUE16-NEXT: v_add_co_u32 v0, vcc_lo, 0x2000, s0
-; GFX11-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX11-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX11-SDAG-TRUE16-NEXT: v_add_co_ci_u32_e64 v1, null, 0x80000000, v1, vcc_lo
; GFX11-SDAG-TRUE16-NEXT: flat_load_d16_u8 v0, v[0:1] glc dlc
; GFX11-SDAG-TRUE16-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
@@ -5509,7 +5404,7 @@ define amdgpu_kernel void @flat_inst_salu_offset_64bit_13bit_neg_high_split1(ptr
; GFX11-SDAG-FAKE16-NEXT: s_waitcnt lgkmcnt(0)
; GFX11-SDAG-FAKE16-NEXT: v_mov_b32_e32 v1, s1
; GFX11-SDAG-FAKE16-NEXT: v_add_co_u32 v0, vcc_lo, 0x2000, s0
-; GFX11-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX11-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX11-SDAG-FAKE16-NEXT: v_add_co_ci_u32_e64 v1, null, 0x80000000, v1, vcc_lo
; GFX11-SDAG-FAKE16-NEXT: flat_load_u8 v0, v[0:1] glc dlc
; GFX11-SDAG-FAKE16-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
@@ -5522,7 +5417,7 @@ define amdgpu_kernel void @flat_inst_salu_offset_64bit_13bit_neg_high_split1(ptr
; GFX12-SDAG-TRUE16-NEXT: s_wait_kmcnt 0x0
; GFX12-SDAG-TRUE16-NEXT: v_mov_b32_e32 v1, s1
; GFX12-SDAG-TRUE16-NEXT: v_add_co_u32 v0, vcc_lo, 0x800000, s0
-; GFX12-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX12-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX12-SDAG-TRUE16-NEXT: v_add_co_ci_u32_e64 v1, null, 0x80000000, v1, vcc_lo
; GFX12-SDAG-TRUE16-NEXT: flat_load_d16_u8 v0, v[0:1] offset:-8380416 scope:SCOPE_SYS
; GFX12-SDAG-TRUE16-NEXT: s_wait_loadcnt_dscnt 0x0
@@ -5535,7 +5430,7 @@ define amdgpu_kernel void @flat_inst_salu_offset_64bit_13bit_neg_high_split1(ptr
; GFX12-SDAG-FAKE16-NEXT: s_wait_kmcnt 0x0
; GFX12-SDAG-FAKE16-NEXT: v_mov_b32_e32 v1, s1
; GFX12-SDAG-FAKE16-NEXT: v_add_co_u32 v0, vcc_lo, 0x800000, s0
-; GFX12-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX12-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX12-SDAG-FAKE16-NEXT: v_add_co_ci_u32_e64 v1, null, 0x80000000, v1, vcc_lo
; GFX12-SDAG-FAKE16-NEXT: flat_load_u8 v0, v[0:1] offset:-8380416 scope:SCOPE_SYS
; GFX12-SDAG-FAKE16-NEXT: s_wait_loadcnt_dscnt 0x0
diff --git a/llvm/test/CodeGen/AMDGPU/offset-split-global.ll b/llvm/test/CodeGen/AMDGPU/offset-split-global.ll
index 23243f86ac6a18..1b385e5d7e2151 100644
--- a/llvm/test/CodeGen/AMDGPU/offset-split-global.ll
+++ b/llvm/test/CodeGen/AMDGPU/offset-split-global.ll
@@ -268,7 +268,6 @@ define i8 @global_inst_valu_offset_13bit_max(ptr addrspace(1) %p) {
; GFX11-GISEL: ; %bb.0:
; GFX11-GISEL-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, 0x1fff, v0
-; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-GISEL-NEXT: global_load_u8 v0, v[0:1], off
; GFX11-GISEL-NEXT: s_waitcnt vmcnt(0)
@@ -307,7 +306,6 @@ define i8 @global_inst_valu_offset_13bit_max(ptr addrspace(1) %p) {
; GFX11-SDAG-TRUE16: ; %bb.0:
; GFX11-SDAG-TRUE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-SDAG-TRUE16-NEXT: v_add_co_u32 v0, vcc_lo, 0x1000, v0
-; GFX11-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-SDAG-TRUE16-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-SDAG-TRUE16-NEXT: global_load_d16_u8 v0, v[0:1], off offset:4095
; GFX11-SDAG-TRUE16-NEXT: s_waitcnt vmcnt(0)
@@ -317,7 +315,6 @@ define i8 @global_inst_valu_offset_13bit_max(ptr addrspace(1) %p) {
; GFX11-SDAG-FAKE16: ; %bb.0:
; GFX11-SDAG-FAKE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-SDAG-FAKE16-NEXT: v_add_co_u32 v0, vcc_lo, 0x1000, v0
-; GFX11-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-SDAG-FAKE16-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-SDAG-FAKE16-NEXT: global_load_u8 v0, v[0:1], off offset:4095
; GFX11-SDAG-FAKE16-NEXT: s_waitcnt vmcnt(0)
@@ -372,7 +369,6 @@ define i8 @global_inst_valu_offset_24bit_max(ptr addrspace(1) %p) {
; GFX11-GISEL: ; %bb.0:
; GFX11-GISEL-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, 0x7fffff, v0
-; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-GISEL-NEXT: global_load_u8 v0, v[0:1], off
; GFX11-GISEL-NEXT: s_waitcnt vmcnt(0)
@@ -411,7 +407,6 @@ define i8 @global_inst_valu_offset_24bit_max(ptr addrspace(1) %p) {
; GFX11-SDAG-TRUE16: ; %bb.0:
; GFX11-SDAG-TRUE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-SDAG-TRUE16-NEXT: v_add_co_u32 v0, vcc_lo, 0x7ff000, v0
-; GFX11-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-SDAG-TRUE16-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-SDAG-TRUE16-NEXT: global_load_d16_u8 v0, v[0:1], off offset:4095
; GFX11-SDAG-TRUE16-NEXT: s_waitcnt vmcnt(0)
@@ -421,7 +416,6 @@ define i8 @global_inst_valu_offset_24bit_max(ptr addrspace(1) %p) {
; GFX11-SDAG-FAKE16: ; %bb.0:
; GFX11-SDAG-FAKE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-SDAG-FAKE16-NEXT: v_add_co_u32 v0, vcc_lo, 0x7ff000, v0
-; GFX11-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-SDAG-FAKE16-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-SDAG-FAKE16-NEXT: global_load_u8 v0, v[0:1], off offset:4095
; GFX11-SDAG-FAKE16-NEXT: s_waitcnt vmcnt(0)
@@ -624,7 +618,6 @@ define i8 @global_inst_valu_offset_neg_13bit_max(ptr addrspace(1) %p) {
; GFX11-GISEL: ; %bb.0:
; GFX11-GISEL-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, 0xffffe000, v0
-; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, -1, v1, vcc_lo
; GFX11-GISEL-NEXT: global_load_u8 v0, v[0:1], off
; GFX11-GISEL-NEXT: s_waitcnt vmcnt(0)
@@ -645,7 +638,6 @@ define i8 @global_inst_valu_offset_neg_13bit_max(ptr addrspace(1) %p) {
; GFX11-SDAG-TRUE16: ; %bb.0:
; GFX11-SDAG-TRUE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-SDAG-TRUE16-NEXT: v_add_co_u32 v0, vcc_lo, 0xffffe000, v0
-; GFX11-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-SDAG-TRUE16-NEXT: v_add_co_ci_u32_e64 v1, null, -1, v1, vcc_lo
; GFX11-SDAG-TRUE16-NEXT: global_load_d16_u8 v0, v[0:1], off
; GFX11-SDAG-TRUE16-NEXT: s_waitcnt vmcnt(0)
@@ -655,7 +647,6 @@ define i8 @global_inst_valu_offset_neg_13bit_max(ptr addrspace(1) %p) {
; GFX11-SDAG-FAKE16: ; %bb.0:
; GFX11-SDAG-FAKE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-SDAG-FAKE16-NEXT: v_add_co_u32 v0, vcc_lo, 0xffffe000, v0
-; GFX11-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-SDAG-FAKE16-NEXT: v_add_co_ci_u32_e64 v1, null, -1, v1, vcc_lo
; GFX11-SDAG-FAKE16-NEXT: global_load_u8 v0, v[0:1], off
; GFX11-SDAG-FAKE16-NEXT: s_waitcnt vmcnt(0)
@@ -710,7 +701,6 @@ define i8 @global_inst_valu_offset_neg_24bit_max(ptr addrspace(1) %p) {
; GFX11-GISEL: ; %bb.0:
; GFX11-GISEL-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, 0xff800000, v0
-; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, -1, v1, vcc_lo
; GFX11-GISEL-NEXT: global_load_u8 v0, v[0:1], off
; GFX11-GISEL-NEXT: s_waitcnt vmcnt(0)
@@ -731,7 +721,6 @@ define i8 @global_inst_valu_offset_neg_24bit_max(ptr addrspace(1) %p) {
; GFX11-SDAG-TRUE16: ; %bb.0:
; GFX11-SDAG-TRUE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-SDAG-TRUE16-NEXT: v_add_co_u32 v0, vcc_lo, 0xff800000, v0
-; GFX11-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-SDAG-TRUE16-NEXT: v_add_co_ci_u32_e64 v1, null, -1, v1, vcc_lo
; GFX11-SDAG-TRUE16-NEXT: global_load_d16_u8 v0, v[0:1], off
; GFX11-SDAG-TRUE16-NEXT: s_waitcnt vmcnt(0)
@@ -741,7 +730,6 @@ define i8 @global_inst_valu_offset_neg_24bit_max(ptr addrspace(1) %p) {
; GFX11-SDAG-FAKE16: ; %bb.0:
; GFX11-SDAG-FAKE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-SDAG-FAKE16-NEXT: v_add_co_u32 v0, vcc_lo, 0xff800000, v0
-; GFX11-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-SDAG-FAKE16-NEXT: v_add_co_ci_u32_e64 v1, null, -1, v1, vcc_lo
; GFX11-SDAG-FAKE16-NEXT: global_load_u8 v0, v[0:1], off
; GFX11-SDAG-FAKE16-NEXT: s_waitcnt vmcnt(0)
@@ -880,7 +868,6 @@ define i8 @global_inst_valu_offset_2x_12bit_max(ptr addrspace(1) %p) {
; GFX11-GISEL: ; %bb.0:
; GFX11-GISEL-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, 0x1fff, v0
-; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-GISEL-NEXT: global_load_u8 v0, v[0:1], off
; GFX11-GISEL-NEXT: s_waitcnt vmcnt(0)
@@ -919,7 +906,6 @@ define i8 @global_inst_valu_offset_2x_12bit_max(ptr addrspace(1) %p) {
; GFX11-SDAG-TRUE16: ; %bb.0:
; GFX11-SDAG-TRUE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-SDAG-TRUE16-NEXT: v_add_co_u32 v0, vcc_lo, 0x1000, v0
-; GFX11-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-SDAG-TRUE16-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-SDAG-TRUE16-NEXT: global_load_d16_u8 v0, v[0:1], off offset:4095
; GFX11-SDAG-TRUE16-NEXT: s_waitcnt vmcnt(0)
@@ -929,7 +915,6 @@ define i8 @global_inst_valu_offset_2x_12bit_max(ptr addrspace(1) %p) {
; GFX11-SDAG-FAKE16: ; %bb.0:
; GFX11-SDAG-FAKE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-SDAG-FAKE16-NEXT: v_add_co_u32 v0, vcc_lo, 0x1000, v0
-; GFX11-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-SDAG-FAKE16-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-SDAG-FAKE16-NEXT: global_load_u8 v0, v[0:1], off offset:4095
; GFX11-SDAG-FAKE16-NEXT: s_waitcnt vmcnt(0)
@@ -984,7 +969,6 @@ define i8 @global_inst_valu_offset_2x_13bit_max(ptr addrspace(1) %p) {
; GFX11-GISEL: ; %bb.0:
; GFX11-GISEL-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, 0x3fff, v0
-; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-GISEL-NEXT: global_load_u8 v0, v[0:1], off
; GFX11-GISEL-NEXT: s_waitcnt vmcnt(0)
@@ -1023,7 +1007,6 @@ define i8 @global_inst_valu_offset_2x_13bit_max(ptr addrspace(1) %p) {
; GFX11-SDAG-TRUE16: ; %bb.0:
; GFX11-SDAG-TRUE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-SDAG-TRUE16-NEXT: v_add_co_u32 v0, vcc_lo, 0x3000, v0
-; GFX11-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-SDAG-TRUE16-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-SDAG-TRUE16-NEXT: global_load_d16_u8 v0, v[0:1], off offset:4095
; GFX11-SDAG-TRUE16-NEXT: s_waitcnt vmcnt(0)
@@ -1033,7 +1016,6 @@ define i8 @global_inst_valu_offset_2x_13bit_max(ptr addrspace(1) %p) {
; GFX11-SDAG-FAKE16: ; %bb.0:
; GFX11-SDAG-FAKE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-SDAG-FAKE16-NEXT: v_add_co_u32 v0, vcc_lo, 0x3000, v0
-; GFX11-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-SDAG-FAKE16-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-SDAG-FAKE16-NEXT: global_load_u8 v0, v[0:1], off offset:4095
; GFX11-SDAG-FAKE16-NEXT: s_waitcnt vmcnt(0)
@@ -1088,7 +1070,6 @@ define i8 @global_inst_valu_offset_2x_24bit_max(ptr addrspace(1) %p) {
; GFX11-GISEL: ; %bb.0:
; GFX11-GISEL-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, 0xfffffe, v0
-; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-GISEL-NEXT: global_load_u8 v0, v[0:1], off
; GFX11-GISEL-NEXT: s_waitcnt vmcnt(0)
@@ -1130,7 +1111,6 @@ define i8 @global_inst_valu_offset_2x_24bit_max(ptr addrspace(1) %p) {
; GFX11-SDAG-TRUE16: ; %bb.0:
; GFX11-SDAG-TRUE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-SDAG-TRUE16-NEXT: v_add_co_u32 v0, vcc_lo, 0xfff000, v0
-; GFX11-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-SDAG-TRUE16-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-SDAG-TRUE16-NEXT: global_load_d16_u8 v0, v[0:1], off offset:4094
; GFX11-SDAG-TRUE16-NEXT: s_waitcnt vmcnt(0)
@@ -1140,7 +1120,6 @@ define i8 @global_inst_valu_offset_2x_24bit_max(ptr addrspace(1) %p) {
; GFX11-SDAG-FAKE16: ; %bb.0:
; GFX11-SDAG-FAKE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-SDAG-FAKE16-NEXT: v_add_co_u32 v0, vcc_lo, 0xfff000, v0
-; GFX11-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-SDAG-FAKE16-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-SDAG-FAKE16-NEXT: global_load_u8 v0, v[0:1], off offset:4094
; GFX11-SDAG-FAKE16-NEXT: s_waitcnt vmcnt(0)
@@ -1276,7 +1255,6 @@ define i8 @global_inst_valu_offset_2x_neg_12bit_max(ptr addrspace(1) %p) {
; GFX11-GISEL: ; %bb.0:
; GFX11-GISEL-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, 0xffffe000, v0
-; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, -1, v1, vcc_lo
; GFX11-GISEL-NEXT: global_load_u8 v0, v[0:1], off
; GFX11-GISEL-NEXT: s_waitcnt vmcnt(0)
@@ -1297,7 +1275,6 @@ define i8 @global_inst_valu_offset_2x_neg_12bit_max(ptr addrspace(1) %p) {
; GFX11-SDAG-TRUE16: ; %bb.0:
; GFX11-SDAG-TRUE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-SDAG-TRUE16-NEXT: v_add_co_u32 v0, vcc_lo, 0xffffe000, v0
-; GFX11-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-SDAG-TRUE16-NEXT: v_add_co_ci_u32_e64 v1, null, -1, v1, vcc_lo
; GFX11-SDAG-TRUE16-NEXT: global_load_d16_u8 v0, v[0:1], off
; GFX11-SDAG-TRUE16-NEXT: s_waitcnt vmcnt(0)
@@ -1307,7 +1284,6 @@ define i8 @global_inst_valu_offset_2x_neg_12bit_max(ptr addrspace(1) %p) {
; GFX11-SDAG-FAKE16: ; %bb.0:
; GFX11-SDAG-FAKE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-SDAG-FAKE16-NEXT: v_add_co_u32 v0, vcc_lo, 0xffffe000, v0
-; GFX11-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-SDAG-FAKE16-NEXT: v_add_co_ci_u32_e64 v1, null, -1, v1, vcc_lo
; GFX11-SDAG-FAKE16-NEXT: global_load_u8 v0, v[0:1], off
; GFX11-SDAG-FAKE16-NEXT: s_waitcnt vmcnt(0)
@@ -1362,7 +1338,6 @@ define i8 @global_inst_valu_offset_2x_neg_13bit_max(ptr addrspace(1) %p) {
; GFX11-GISEL: ; %bb.0:
; GFX11-GISEL-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, 0xffffc000, v0
-; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, -1, v1, vcc_lo
; GFX11-GISEL-NEXT: global_load_u8 v0, v[0:1], off
; GFX11-GISEL-NEXT: s_waitcnt vmcnt(0)
@@ -1383,7 +1358,6 @@ define i8 @global_inst_valu_offset_2x_neg_13bit_max(ptr addrspace(1) %p) {
; GFX11-SDAG-TRUE16: ; %bb.0:
; GFX11-SDAG-TRUE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-SDAG-TRUE16-NEXT: v_add_co_u32 v0, vcc_lo, 0xffffc000, v0
-; GFX11-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-SDAG-TRUE16-NEXT: v_add_co_ci_u32_e64 v1, null, -1, v1, vcc_lo
; GFX11-SDAG-TRUE16-NEXT: global_load_d16_u8 v0, v[0:1], off
; GFX11-SDAG-TRUE16-NEXT: s_waitcnt vmcnt(0)
@@ -1393,7 +1367,6 @@ define i8 @global_inst_valu_offset_2x_neg_13bit_max(ptr addrspace(1) %p) {
; GFX11-SDAG-FAKE16: ; %bb.0:
; GFX11-SDAG-FAKE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-SDAG-FAKE16-NEXT: v_add_co_u32 v0, vcc_lo, 0xffffc000, v0
-; GFX11-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-SDAG-FAKE16-NEXT: v_add_co_ci_u32_e64 v1, null, -1, v1, vcc_lo
; GFX11-SDAG-FAKE16-NEXT: global_load_u8 v0, v[0:1], off
; GFX11-SDAG-FAKE16-NEXT: s_waitcnt vmcnt(0)
@@ -1448,7 +1421,6 @@ define i8 @global_inst_valu_offset_2x_neg_24bit_max(ptr addrspace(1) %p) {
; GFX11-GISEL: ; %bb.0:
; GFX11-GISEL-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, 0xff000001, v0
-; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, -1, v1, vcc_lo
; GFX11-GISEL-NEXT: global_load_u8 v0, v[0:1], off
; GFX11-GISEL-NEXT: s_waitcnt vmcnt(0)
@@ -1490,7 +1462,6 @@ define i8 @global_inst_valu_offset_2x_neg_24bit_max(ptr addrspace(1) %p) {
; GFX11-SDAG-TRUE16: ; %bb.0:
; GFX11-SDAG-TRUE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-SDAG-TRUE16-NEXT: v_add_co_u32 v0, vcc_lo, 0xff001000, v0
-; GFX11-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-SDAG-TRUE16-NEXT: v_add_co_ci_u32_e64 v1, null, -1, v1, vcc_lo
; GFX11-SDAG-TRUE16-NEXT: global_load_d16_u8 v0, v[0:1], off offset:-4095
; GFX11-SDAG-TRUE16-NEXT: s_waitcnt vmcnt(0)
@@ -1500,7 +1471,6 @@ define i8 @global_inst_valu_offset_2x_neg_24bit_max(ptr addrspace(1) %p) {
; GFX11-SDAG-FAKE16: ; %bb.0:
; GFX11-SDAG-FAKE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-SDAG-FAKE16-NEXT: v_add_co_u32 v0, vcc_lo, 0xff001000, v0
-; GFX11-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-SDAG-FAKE16-NEXT: v_add_co_ci_u32_e64 v1, null, -1, v1, vcc_lo
; GFX11-SDAG-FAKE16-NEXT: global_load_u8 v0, v[0:1], off offset:-4095
; GFX11-SDAG-FAKE16-NEXT: s_waitcnt vmcnt(0)
@@ -1563,7 +1533,6 @@ define i8 @global_inst_valu_offset_64bit_11bit_split0(ptr addrspace(1) %p) {
; GFX11-GISEL: ; %bb.0:
; GFX11-GISEL-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, 0x7ff, v0
-; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 2, v1, vcc_lo
; GFX11-GISEL-NEXT: global_load_u8 v0, v[0:1], off
; GFX11-GISEL-NEXT: s_waitcnt vmcnt(0)
@@ -1605,7 +1574,6 @@ define i8 @global_inst_valu_offset_64bit_11bit_split0(ptr addrspace(1) %p) {
; GFX11-SDAG-TRUE16: ; %bb.0:
; GFX11-SDAG-TRUE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-SDAG-TRUE16-NEXT: v_add_co_u32 v0, vcc_lo, 0, v0
-; GFX11-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-SDAG-TRUE16-NEXT: v_add_co_ci_u32_e64 v1, null, 2, v1, vcc_lo
; GFX11-SDAG-TRUE16-NEXT: global_load_d16_u8 v0, v[0:1], off offset:2047
; GFX11-SDAG-TRUE16-NEXT: s_waitcnt vmcnt(0)
@@ -1615,7 +1583,6 @@ define i8 @global_inst_valu_offset_64bit_11bit_split0(ptr addrspace(1) %p) {
; GFX11-SDAG-FAKE16: ; %bb.0:
; GFX11-SDAG-FAKE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-SDAG-FAKE16-NEXT: v_add_co_u32 v0, vcc_lo, 0, v0
-; GFX11-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-SDAG-FAKE16-NEXT: v_add_co_ci_u32_e64 v1, null, 2, v1, vcc_lo
; GFX11-SDAG-FAKE16-NEXT: global_load_u8 v0, v[0:1], off offset:2047
; GFX11-SDAG-FAKE16-NEXT: s_waitcnt vmcnt(0)
@@ -1677,7 +1644,6 @@ define i8 @global_inst_valu_offset_64bit_11bit_split1(ptr addrspace(1) %p) {
; GFX11-GISEL: ; %bb.0:
; GFX11-GISEL-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, 0x800, v0
-; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 2, v1, vcc_lo
; GFX11-GISEL-NEXT: global_load_u8 v0, v[0:1], off
; GFX11-GISEL-NEXT: s_waitcnt vmcnt(0)
@@ -1710,7 +1676,6 @@ define i8 @global_inst_valu_offset_64bit_11bit_split1(ptr addrspace(1) %p) {
; GFX11-SDAG-TRUE16: ; %bb.0:
; GFX11-SDAG-TRUE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-SDAG-TRUE16-NEXT: v_add_co_u32 v0, vcc_lo, 0, v0
-; GFX11-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-SDAG-TRUE16-NEXT: v_add_co_ci_u32_e64 v1, null, 2, v1, vcc_lo
; GFX11-SDAG-TRUE16-NEXT: global_load_d16_u8 v0, v[0:1], off offset:2048
; GFX11-SDAG-TRUE16-NEXT: s_waitcnt vmcnt(0)
@@ -1720,7 +1685,6 @@ define i8 @global_inst_valu_offset_64bit_11bit_split1(ptr addrspace(1) %p) {
; GFX11-SDAG-FAKE16: ; %bb.0:
; GFX11-SDAG-FAKE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-SDAG-FAKE16-NEXT: v_add_co_u32 v0, vcc_lo, 0, v0
-; GFX11-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-SDAG-FAKE16-NEXT: v_add_co_ci_u32_e64 v1, null, 2, v1, vcc_lo
; GFX11-SDAG-FAKE16-NEXT: global_load_u8 v0, v[0:1], off offset:2048
; GFX11-SDAG-FAKE16-NEXT: s_waitcnt vmcnt(0)
@@ -1782,7 +1746,6 @@ define i8 @global_inst_valu_offset_64bit_12bit_split0(ptr addrspace(1) %p) {
; GFX11-GISEL: ; %bb.0:
; GFX11-GISEL-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, 0xfff, v0
-; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 2, v1, vcc_lo
; GFX11-GISEL-NEXT: global_load_u8 v0, v[0:1], off
; GFX11-GISEL-NEXT: s_waitcnt vmcnt(0)
@@ -1824,7 +1787,6 @@ define i8 @global_inst_valu_offset_64bit_12bit_split0(ptr addrspace(1) %p) {
; GFX11-SDAG-TRUE16: ; %bb.0:
; GFX11-SDAG-TRUE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-SDAG-TRUE16-NEXT: v_add_co_u32 v0, vcc_lo, 0, v0
-; GFX11-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-SDAG-TRUE16-NEXT: v_add_co_ci_u32_e64 v1, null, 2, v1, vcc_lo
; GFX11-SDAG-TRUE16-NEXT: global_load_d16_u8 v0, v[0:1], off offset:4095
; GFX11-SDAG-TRUE16-NEXT: s_waitcnt vmcnt(0)
@@ -1834,7 +1796,6 @@ define i8 @global_inst_valu_offset_64bit_12bit_split0(ptr addrspace(1) %p) {
; GFX11-SDAG-FAKE16: ; %bb.0:
; GFX11-SDAG-FAKE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-SDAG-FAKE16-NEXT: v_add_co_u32 v0, vcc_lo, 0, v0
-; GFX11-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-SDAG-FAKE16-NEXT: v_add_co_ci_u32_e64 v1, null, 2, v1, vcc_lo
; GFX11-SDAG-FAKE16-NEXT: global_load_u8 v0, v[0:1], off offset:4095
; GFX11-SDAG-FAKE16-NEXT: s_waitcnt vmcnt(0)
@@ -1896,7 +1857,6 @@ define i8 @global_inst_valu_offset_64bit_12bit_split1(ptr addrspace(1) %p) {
; GFX11-GISEL: ; %bb.0:
; GFX11-GISEL-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, 0x1000, v0
-; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 2, v1, vcc_lo
; GFX11-GISEL-NEXT: global_load_u8 v0, v[0:1], off
; GFX11-GISEL-NEXT: s_waitcnt vmcnt(0)
@@ -1920,7 +1880,6 @@ define i8 @global_inst_valu_offset_64bit_12bit_split1(ptr addrspace(1) %p) {
; GFX11-SDAG-TRUE16: ; %bb.0:
; GFX11-SDAG-TRUE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-SDAG-TRUE16-NEXT: v_add_co_u32 v0, vcc_lo, 0x1000, v0
-; GFX11-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-SDAG-TRUE16-NEXT: v_add_co_ci_u32_e64 v1, null, 2, v1, vcc_lo
; GFX11-SDAG-TRUE16-NEXT: global_load_d16_u8 v0, v[0:1], off
; GFX11-SDAG-TRUE16-NEXT: s_waitcnt vmcnt(0)
@@ -1930,7 +1889,6 @@ define i8 @global_inst_valu_offset_64bit_12bit_split1(ptr addrspace(1) %p) {
; GFX11-SDAG-FAKE16: ; %bb.0:
; GFX11-SDAG-FAKE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-SDAG-FAKE16-NEXT: v_add_co_u32 v0, vcc_lo, 0x1000, v0
-; GFX11-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-SDAG-FAKE16-NEXT: v_add_co_ci_u32_e64 v1, null, 2, v1, vcc_lo
; GFX11-SDAG-FAKE16-NEXT: global_load_u8 v0, v[0:1], off
; GFX11-SDAG-FAKE16-NEXT: s_waitcnt vmcnt(0)
@@ -1992,7 +1950,6 @@ define i8 @global_inst_valu_offset_64bit_13bit_split0(ptr addrspace(1) %p) {
; GFX11-GISEL: ; %bb.0:
; GFX11-GISEL-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, 0x1fff, v0
-; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 2, v1, vcc_lo
; GFX11-GISEL-NEXT: global_load_u8 v0, v[0:1], off
; GFX11-GISEL-NEXT: s_waitcnt vmcnt(0)
@@ -2034,7 +1991,6 @@ define i8 @global_inst_valu_offset_64bit_13bit_split0(ptr addrspace(1) %p) {
; GFX11-SDAG-TRUE16: ; %bb.0:
; GFX11-SDAG-TRUE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-SDAG-TRUE16-NEXT: v_add_co_u32 v0, vcc_lo, 0x1000, v0
-; GFX11-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-SDAG-TRUE16-NEXT: v_add_co_ci_u32_e64 v1, null, 2, v1, vcc_lo
; GFX11-SDAG-TRUE16-NEXT: global_load_d16_u8 v0, v[0:1], off offset:4095
; GFX11-SDAG-TRUE16-NEXT: s_waitcnt vmcnt(0)
@@ -2044,7 +2000,6 @@ define i8 @global_inst_valu_offset_64bit_13bit_split0(ptr addrspace(1) %p) {
; GFX11-SDAG-FAKE16: ; %bb.0:
; GFX11-SDAG-FAKE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-SDAG-FAKE16-NEXT: v_add_co_u32 v0, vcc_lo, 0x1000, v0
-; GFX11-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-SDAG-FAKE16-NEXT: v_add_co_ci_u32_e64 v1, null, 2, v1, vcc_lo
; GFX11-SDAG-FAKE16-NEXT: global_load_u8 v0, v[0:1], off offset:4095
; GFX11-SDAG-FAKE16-NEXT: s_waitcnt vmcnt(0)
@@ -2106,7 +2061,6 @@ define i8 @global_inst_valu_offset_64bit_13bit_split1(ptr addrspace(1) %p) {
; GFX11-GISEL: ; %bb.0:
; GFX11-GISEL-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, 0x2000, v0
-; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 2, v1, vcc_lo
; GFX11-GISEL-NEXT: global_load_u8 v0, v[0:1], off
; GFX11-GISEL-NEXT: s_waitcnt vmcnt(0)
@@ -2130,7 +2084,6 @@ define i8 @global_inst_valu_offset_64bit_13bit_split1(ptr addrspace(1) %p) {
; GFX11-SDAG-TRUE16: ; %bb.0:
; GFX11-SDAG-TRUE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-SDAG-TRUE16-NEXT: v_add_co_u32 v0, vcc_lo, 0x2000, v0
-; GFX11-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-SDAG-TRUE16-NEXT: v_add_co_ci_u32_e64 v1, null, 2, v1, vcc_lo
; GFX11-SDAG-TRUE16-NEXT: global_load_d16_u8 v0, v[0:1], off
; GFX11-SDAG-TRUE16-NEXT: s_waitcnt vmcnt(0)
@@ -2140,7 +2093,6 @@ define i8 @global_inst_valu_offset_64bit_13bit_split1(ptr addrspace(1) %p) {
; GFX11-SDAG-FAKE16: ; %bb.0:
; GFX11-SDAG-FAKE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-SDAG-FAKE16-NEXT: v_add_co_u32 v0, vcc_lo, 0x2000, v0
-; GFX11-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-SDAG-FAKE16-NEXT: v_add_co_ci_u32_e64 v1, null, 2, v1, vcc_lo
; GFX11-SDAG-FAKE16-NEXT: global_load_u8 v0, v[0:1], off
; GFX11-SDAG-FAKE16-NEXT: s_waitcnt vmcnt(0)
@@ -2203,7 +2155,6 @@ define i8 @global_inst_valu_offset_64bit_11bit_neg_high_split0(ptr addrspace(1)
; GFX11-GISEL: ; %bb.0:
; GFX11-GISEL-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, 0x7ff, v0
-; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0x80000000, v1, vcc_lo
; GFX11-GISEL-NEXT: global_load_u8 v0, v[0:1], off
; GFX11-GISEL-NEXT: s_waitcnt vmcnt(0)
@@ -2246,7 +2197,6 @@ define i8 @global_inst_valu_offset_64bit_11bit_neg_high_split0(ptr addrspace(1)
; GFX11-SDAG-TRUE16: ; %bb.0:
; GFX11-SDAG-TRUE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-SDAG-TRUE16-NEXT: v_add_co_u32 v0, vcc_lo, 0x1000, v0
-; GFX11-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-SDAG-TRUE16-NEXT: v_add_co_ci_u32_e64 v1, null, 0x80000000, v1, vcc_lo
; GFX11-SDAG-TRUE16-NEXT: global_load_d16_u8 v0, v[0:1], off offset:-2049
; GFX11-SDAG-TRUE16-NEXT: s_waitcnt vmcnt(0)
@@ -2256,7 +2206,6 @@ define i8 @global_inst_valu_offset_64bit_11bit_neg_high_split0(ptr addrspace(1)
; GFX11-SDAG-FAKE16: ; %bb.0:
; GFX11-SDAG-FAKE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-SDAG-FAKE16-NEXT: v_add_co_u32 v0, vcc_lo, 0x1000, v0
-; GFX11-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-SDAG-FAKE16-NEXT: v_add_co_ci_u32_e64 v1, null, 0x80000000, v1, vcc_lo
; GFX11-SDAG-FAKE16-NEXT: global_load_u8 v0, v[0:1], off offset:-2049
; GFX11-SDAG-FAKE16-NEXT: s_waitcnt vmcnt(0)
@@ -2319,7 +2268,6 @@ define i8 @global_inst_valu_offset_64bit_11bit_neg_high_split1(ptr addrspace(1)
; GFX11-GISEL: ; %bb.0:
; GFX11-GISEL-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, 0x800, v0
-; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0x80000000, v1, vcc_lo
; GFX11-GISEL-NEXT: global_load_u8 v0, v[0:1], off
; GFX11-GISEL-NEXT: s_waitcnt vmcnt(0)
@@ -2353,7 +2301,6 @@ define i8 @global_inst_valu_offset_64bit_11bit_neg_high_split1(ptr addrspace(1)
; GFX11-SDAG-TRUE16: ; %bb.0:
; GFX11-SDAG-TRUE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-SDAG-TRUE16-NEXT: v_add_co_u32 v0, vcc_lo, 0x1000, v0
-; GFX11-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-SDAG-TRUE16-NEXT: v_add_co_ci_u32_e64 v1, null, 0x80000000, v1, vcc_lo
; GFX11-SDAG-TRUE16-NEXT: global_load_d16_u8 v0, v[0:1], off offset:-2048
; GFX11-SDAG-TRUE16-NEXT: s_waitcnt vmcnt(0)
@@ -2363,7 +2310,6 @@ define i8 @global_inst_valu_offset_64bit_11bit_neg_high_split1(ptr addrspace(1)
; GFX11-SDAG-FAKE16: ; %bb.0:
; GFX11-SDAG-FAKE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-SDAG-FAKE16-NEXT: v_add_co_u32 v0, vcc_lo, 0x1000, v0
-; GFX11-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-SDAG-FAKE16-NEXT: v_add_co_ci_u32_e64 v1, null, 0x80000000, v1, vcc_lo
; GFX11-SDAG-FAKE16-NEXT: global_load_u8 v0, v[0:1], off offset:-2048
; GFX11-SDAG-FAKE16-NEXT: s_waitcnt vmcnt(0)
@@ -2426,7 +2372,6 @@ define i8 @global_inst_valu_offset_64bit_12bit_neg_high_split0(ptr addrspace(1)
; GFX11-GISEL: ; %bb.0:
; GFX11-GISEL-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, 0xfff, v0
-; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0x80000000, v1, vcc_lo
; GFX11-GISEL-NEXT: global_load_u8 v0, v[0:1], off
; GFX11-GISEL-NEXT: s_waitcnt vmcnt(0)
@@ -2469,7 +2414,6 @@ define i8 @global_inst_valu_offset_64bit_12bit_neg_high_split0(ptr addrspace(1)
; GFX11-SDAG-TRUE16: ; %bb.0:
; GFX11-SDAG-TRUE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-SDAG-TRUE16-NEXT: v_add_co_u32 v0, vcc_lo, 0x1000, v0
-; GFX11-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-SDAG-TRUE16-NEXT: v_add_co_ci_u32_e64 v1, null, 0x80000000, v1, vcc_lo
; GFX11-SDAG-TRUE16-NEXT: global_load_d16_u8 v0, v[0:1], off offset:-1
; GFX11-SDAG-TRUE16-NEXT: s_waitcnt vmcnt(0)
@@ -2479,7 +2423,6 @@ define i8 @global_inst_valu_offset_64bit_12bit_neg_high_split0(ptr addrspace(1)
; GFX11-SDAG-FAKE16: ; %bb.0:
; GFX11-SDAG-FAKE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-SDAG-FAKE16-NEXT: v_add_co_u32 v0, vcc_lo, 0x1000, v0
-; GFX11-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-SDAG-FAKE16-NEXT: v_add_co_ci_u32_e64 v1, null, 0x80000000, v1, vcc_lo
; GFX11-SDAG-FAKE16-NEXT: global_load_u8 v0, v[0:1], off offset:-1
; GFX11-SDAG-FAKE16-NEXT: s_waitcnt vmcnt(0)
@@ -2542,7 +2485,6 @@ define i8 @global_inst_valu_offset_64bit_12bit_neg_high_split1(ptr addrspace(1)
; GFX11-GISEL: ; %bb.0:
; GFX11-GISEL-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, 0x1000, v0
-; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0x80000000, v1, vcc_lo
; GFX11-GISEL-NEXT: global_load_u8 v0, v[0:1], off
; GFX11-GISEL-NEXT: s_waitcnt vmcnt(0)
@@ -2576,7 +2518,6 @@ define i8 @global_inst_valu_offset_64bit_12bit_neg_high_split1(ptr addrspace(1)
; GFX11-SDAG-TRUE16: ; %bb.0:
; GFX11-SDAG-TRUE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-SDAG-TRUE16-NEXT: v_add_co_u32 v0, vcc_lo, 0x1000, v0
-; GFX11-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-SDAG-TRUE16-NEXT: v_add_co_ci_u32_e64 v1, null, 0x80000000, v1, vcc_lo
; GFX11-SDAG-TRUE16-NEXT: global_load_d16_u8 v0, v[0:1], off
; GFX11-SDAG-TRUE16-NEXT: s_waitcnt vmcnt(0)
@@ -2586,7 +2527,6 @@ define i8 @global_inst_valu_offset_64bit_12bit_neg_high_split1(ptr addrspace(1)
; GFX11-SDAG-FAKE16: ; %bb.0:
; GFX11-SDAG-FAKE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-SDAG-FAKE16-NEXT: v_add_co_u32 v0, vcc_lo, 0x1000, v0
-; GFX11-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-SDAG-FAKE16-NEXT: v_add_co_ci_u32_e64 v1, null, 0x80000000, v1, vcc_lo
; GFX11-SDAG-FAKE16-NEXT: global_load_u8 v0, v[0:1], off
; GFX11-SDAG-FAKE16-NEXT: s_waitcnt vmcnt(0)
@@ -2649,7 +2589,6 @@ define i8 @global_inst_valu_offset_64bit_13bit_neg_high_split0(ptr addrspace(1)
; GFX11-GISEL: ; %bb.0:
; GFX11-GISEL-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, 0x1fff, v0
-; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0x80000000, v1, vcc_lo
; GFX11-GISEL-NEXT: global_load_u8 v0, v[0:1], off
; GFX11-GISEL-NEXT: s_waitcnt vmcnt(0)
@@ -2692,7 +2631,6 @@ define i8 @global_inst_valu_offset_64bit_13bit_neg_high_split0(ptr addrspace(1)
; GFX11-SDAG-TRUE16: ; %bb.0:
; GFX11-SDAG-TRUE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-SDAG-TRUE16-NEXT: v_add_co_u32 v0, vcc_lo, 0x2000, v0
-; GFX11-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-SDAG-TRUE16-NEXT: v_add_co_ci_u32_e64 v1, null, 0x80000000, v1, vcc_lo
; GFX11-SDAG-TRUE16-NEXT: global_load_d16_u8 v0, v[0:1], off offset:-1
; GFX11-SDAG-TRUE16-NEXT: s_waitcnt vmcnt(0)
@@ -2702,7 +2640,6 @@ define i8 @global_inst_valu_offset_64bit_13bit_neg_high_split0(ptr addrspace(1)
; GFX11-SDAG-FAKE16: ; %bb.0:
; GFX11-SDAG-FAKE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-SDAG-FAKE16-NEXT: v_add_co_u32 v0, vcc_lo, 0x2000, v0
-; GFX11-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-SDAG-FAKE16-NEXT: v_add_co_ci_u32_e64 v1, null, 0x80000000, v1, vcc_lo
; GFX11-SDAG-FAKE16-NEXT: global_load_u8 v0, v[0:1], off offset:-1
; GFX11-SDAG-FAKE16-NEXT: s_waitcnt vmcnt(0)
@@ -2765,7 +2702,6 @@ define i8 @global_inst_valu_offset_64bit_13bit_neg_high_split1(ptr addrspace(1)
; GFX11-GISEL: ; %bb.0:
; GFX11-GISEL-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, 0x2000, v0
-; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, 0x80000000, v1, vcc_lo
; GFX11-GISEL-NEXT: global_load_u8 v0, v[0:1], off
; GFX11-GISEL-NEXT: s_waitcnt vmcnt(0)
@@ -2799,7 +2735,6 @@ define i8 @global_inst_valu_offset_64bit_13bit_neg_high_split1(ptr addrspace(1)
; GFX11-SDAG-TRUE16: ; %bb.0:
; GFX11-SDAG-TRUE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-SDAG-TRUE16-NEXT: v_add_co_u32 v0, vcc_lo, 0x2000, v0
-; GFX11-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-SDAG-TRUE16-NEXT: v_add_co_ci_u32_e64 v1, null, 0x80000000, v1, vcc_lo
; GFX11-SDAG-TRUE16-NEXT: global_load_d16_u8 v0, v[0:1], off
; GFX11-SDAG-TRUE16-NEXT: s_waitcnt vmcnt(0)
@@ -2809,7 +2744,6 @@ define i8 @global_inst_valu_offset_64bit_13bit_neg_high_split1(ptr addrspace(1)
; GFX11-SDAG-FAKE16: ; %bb.0:
; GFX11-SDAG-FAKE16-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-SDAG-FAKE16-NEXT: v_add_co_u32 v0, vcc_lo, 0x2000, v0
-; GFX11-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-SDAG-FAKE16-NEXT: v_add_co_ci_u32_e64 v1, null, 0x80000000, v1, vcc_lo
; GFX11-SDAG-FAKE16-NEXT: global_load_u8 v0, v[0:1], off
; GFX11-SDAG-FAKE16-NEXT: s_waitcnt vmcnt(0)
@@ -3442,7 +3376,6 @@ define amdgpu_kernel void @global_inst_salu_offset_neg_13bit_max(ptr addrspace(1
; GFX11-SDAG-TRUE16-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
; GFX11-SDAG-TRUE16-NEXT: s_waitcnt lgkmcnt(0)
; GFX11-SDAG-TRUE16-NEXT: v_add_co_u32 v0, s0, 0xffffe000, s0
-; GFX11-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-SDAG-TRUE16-NEXT: v_add_co_ci_u32_e64 v1, null, -1, s1, s0
; GFX11-SDAG-TRUE16-NEXT: global_load_d16_u8 v0, v[0:1], off glc dlc
; GFX11-SDAG-TRUE16-NEXT: s_waitcnt vmcnt(0)
@@ -3454,7 +3387,6 @@ define amdgpu_kernel void @global_inst_salu_offset_neg_13bit_max(ptr addrspace(1
; GFX11-SDAG-FAKE16-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
; GFX11-SDAG-FAKE16-NEXT: s_waitcnt lgkmcnt(0)
; GFX11-SDAG-FAKE16-NEXT: v_add_co_u32 v0, s0, 0xffffe000, s0
-; GFX11-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-SDAG-FAKE16-NEXT: v_add_co_ci_u32_e64 v1, null, -1, s1, s0
; GFX11-SDAG-FAKE16-NEXT: global_load_u8 v0, v[0:1], off glc dlc
; GFX11-SDAG-FAKE16-NEXT: s_waitcnt vmcnt(0)
@@ -3909,7 +3841,6 @@ define amdgpu_kernel void @global_inst_salu_offset_2x_neg_12bit_max(ptr addrspac
; GFX11-SDAG-TRUE16-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
; GFX11-SDAG-TRUE16-NEXT: s_waitcnt lgkmcnt(0)
; GFX11-SDAG-TRUE16-NEXT: v_add_co_u32 v0, s0, 0xffffe000, s0
-; GFX11-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-SDAG-TRUE16-NEXT: v_add_co_ci_u32_e64 v1, null, -1, s1, s0
; GFX11-SDAG-TRUE16-NEXT: global_load_d16_u8 v0, v[0:1], off glc dlc
; GFX11-SDAG-TRUE16-NEXT: s_waitcnt vmcnt(0)
@@ -3921,7 +3852,6 @@ define amdgpu_kernel void @global_inst_salu_offset_2x_neg_12bit_max(ptr addrspac
; GFX11-SDAG-FAKE16-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
; GFX11-SDAG-FAKE16-NEXT: s_waitcnt lgkmcnt(0)
; GFX11-SDAG-FAKE16-NEXT: v_add_co_u32 v0, s0, 0xffffe000, s0
-; GFX11-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-SDAG-FAKE16-NEXT: v_add_co_ci_u32_e64 v1, null, -1, s1, s0
; GFX11-SDAG-FAKE16-NEXT: global_load_u8 v0, v[0:1], off glc dlc
; GFX11-SDAG-FAKE16-NEXT: s_waitcnt vmcnt(0)
@@ -4018,7 +3948,6 @@ define amdgpu_kernel void @global_inst_salu_offset_2x_neg_13bit_max(ptr addrspac
; GFX11-SDAG-TRUE16-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
; GFX11-SDAG-TRUE16-NEXT: s_waitcnt lgkmcnt(0)
; GFX11-SDAG-TRUE16-NEXT: v_add_co_u32 v0, s0, 0xffffc000, s0
-; GFX11-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-SDAG-TRUE16-NEXT: v_add_co_ci_u32_e64 v1, null, -1, s1, s0
; GFX11-SDAG-TRUE16-NEXT: global_load_d16_u8 v0, v[0:1], off glc dlc
; GFX11-SDAG-TRUE16-NEXT: s_waitcnt vmcnt(0)
@@ -4030,7 +3959,6 @@ define amdgpu_kernel void @global_inst_salu_offset_2x_neg_13bit_max(ptr addrspac
; GFX11-SDAG-FAKE16-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
; GFX11-SDAG-FAKE16-NEXT: s_waitcnt lgkmcnt(0)
; GFX11-SDAG-FAKE16-NEXT: v_add_co_u32 v0, s0, 0xffffc000, s0
-; GFX11-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-SDAG-FAKE16-NEXT: v_add_co_ci_u32_e64 v1, null, -1, s1, s0
; GFX11-SDAG-FAKE16-NEXT: global_load_u8 v0, v[0:1], off glc dlc
; GFX11-SDAG-FAKE16-NEXT: s_waitcnt vmcnt(0)
@@ -4131,7 +4059,6 @@ define amdgpu_kernel void @global_inst_salu_offset_64bit_11bit_split0(ptr addrsp
; GFX11-SDAG-TRUE16-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
; GFX11-SDAG-TRUE16-NEXT: s_waitcnt lgkmcnt(0)
; GFX11-SDAG-TRUE16-NEXT: v_add_co_u32 v0, s0, 0, s0
-; GFX11-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-SDAG-TRUE16-NEXT: v_add_co_ci_u32_e64 v1, null, 2, s1, s0
; GFX11-SDAG-TRUE16-NEXT: global_load_d16_u8 v0, v[0:1], off offset:2047 glc dlc
; GFX11-SDAG-TRUE16-NEXT: s_waitcnt vmcnt(0)
@@ -4143,7 +4070,6 @@ define amdgpu_kernel void @global_inst_salu_offset_64bit_11bit_split0(ptr addrsp
; GFX11-SDAG-FAKE16-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
; GFX11-SDAG-FAKE16-NEXT: s_waitcnt lgkmcnt(0)
; GFX11-SDAG-FAKE16-NEXT: v_add_co_u32 v0, s0, 0, s0
-; GFX11-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-SDAG-FAKE16-NEXT: v_add_co_ci_u32_e64 v1, null, 2, s1, s0
; GFX11-SDAG-FAKE16-NEXT: global_load_u8 v0, v[0:1], off offset:2047 glc dlc
; GFX11-SDAG-FAKE16-NEXT: s_waitcnt vmcnt(0)
@@ -4248,7 +4174,6 @@ define amdgpu_kernel void @global_inst_salu_offset_64bit_11bit_split1(ptr addrsp
; GFX11-SDAG-TRUE16-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
; GFX11-SDAG-TRUE16-NEXT: s_waitcnt lgkmcnt(0)
; GFX11-SDAG-TRUE16-NEXT: v_add_co_u32 v0, s0, 0, s0
-; GFX11-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-SDAG-TRUE16-NEXT: v_add_co_ci_u32_e64 v1, null, 2, s1, s0
; GFX11-SDAG-TRUE16-NEXT: global_load_d16_u8 v0, v[0:1], off offset:2048 glc dlc
; GFX11-SDAG-TRUE16-NEXT: s_waitcnt vmcnt(0)
@@ -4260,7 +4185,6 @@ define amdgpu_kernel void @global_inst_salu_offset_64bit_11bit_split1(ptr addrsp
; GFX11-SDAG-FAKE16-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
; GFX11-SDAG-FAKE16-NEXT: s_waitcnt lgkmcnt(0)
; GFX11-SDAG-FAKE16-NEXT: v_add_co_u32 v0, s0, 0, s0
-; GFX11-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-SDAG-FAKE16-NEXT: v_add_co_ci_u32_e64 v1, null, 2, s1, s0
; GFX11-SDAG-FAKE16-NEXT: global_load_u8 v0, v[0:1], off offset:2048 glc dlc
; GFX11-SDAG-FAKE16-NEXT: s_waitcnt vmcnt(0)
@@ -4365,7 +4289,6 @@ define amdgpu_kernel void @global_inst_salu_offset_64bit_12bit_split0(ptr addrsp
; GFX11-SDAG-TRUE16-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
; GFX11-SDAG-TRUE16-NEXT: s_waitcnt lgkmcnt(0)
; GFX11-SDAG-TRUE16-NEXT: v_add_co_u32 v0, s0, 0, s0
-; GFX11-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-SDAG-TRUE16-NEXT: v_add_co_ci_u32_e64 v1, null, 2, s1, s0
; GFX11-SDAG-TRUE16-NEXT: global_load_d16_u8 v0, v[0:1], off offset:4095 glc dlc
; GFX11-SDAG-TRUE16-NEXT: s_waitcnt vmcnt(0)
@@ -4377,7 +4300,6 @@ define amdgpu_kernel void @global_inst_salu_offset_64bit_12bit_split0(ptr addrsp
; GFX11-SDAG-FAKE16-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
; GFX11-SDAG-FAKE16-NEXT: s_waitcnt lgkmcnt(0)
; GFX11-SDAG-FAKE16-NEXT: v_add_co_u32 v0, s0, 0, s0
-; GFX11-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-SDAG-FAKE16-NEXT: v_add_co_ci_u32_e64 v1, null, 2, s1, s0
; GFX11-SDAG-FAKE16-NEXT: global_load_u8 v0, v[0:1], off offset:4095 glc dlc
; GFX11-SDAG-FAKE16-NEXT: s_waitcnt vmcnt(0)
@@ -4482,7 +4404,6 @@ define amdgpu_kernel void @global_inst_salu_offset_64bit_12bit_split1(ptr addrsp
; GFX11-SDAG-TRUE16-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
; GFX11-SDAG-TRUE16-NEXT: s_waitcnt lgkmcnt(0)
; GFX11-SDAG-TRUE16-NEXT: v_add_co_u32 v0, s0, 0x1000, s0
-; GFX11-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-SDAG-TRUE16-NEXT: v_add_co_ci_u32_e64 v1, null, 2, s1, s0
; GFX11-SDAG-TRUE16-NEXT: global_load_d16_u8 v0, v[0:1], off glc dlc
; GFX11-SDAG-TRUE16-NEXT: s_waitcnt vmcnt(0)
@@ -4494,7 +4415,6 @@ define amdgpu_kernel void @global_inst_salu_offset_64bit_12bit_split1(ptr addrsp
; GFX11-SDAG-FAKE16-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
; GFX11-SDAG-FAKE16-NEXT: s_waitcnt lgkmcnt(0)
; GFX11-SDAG-FAKE16-NEXT: v_add_co_u32 v0, s0, 0x1000, s0
-; GFX11-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-SDAG-FAKE16-NEXT: v_add_co_ci_u32_e64 v1, null, 2, s1, s0
; GFX11-SDAG-FAKE16-NEXT: global_load_u8 v0, v[0:1], off glc dlc
; GFX11-SDAG-FAKE16-NEXT: s_waitcnt vmcnt(0)
@@ -4599,7 +4519,6 @@ define amdgpu_kernel void @global_inst_salu_offset_64bit_13bit_split0(ptr addrsp
; GFX11-SDAG-TRUE16-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
; GFX11-SDAG-TRUE16-NEXT: s_waitcnt lgkmcnt(0)
; GFX11-SDAG-TRUE16-NEXT: v_add_co_u32 v0, s0, 0x1000, s0
-; GFX11-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-SDAG-TRUE16-NEXT: v_add_co_ci_u32_e64 v1, null, 2, s1, s0
; GFX11-SDAG-TRUE16-NEXT: global_load_d16_u8 v0, v[0:1], off offset:4095 glc dlc
; GFX11-SDAG-TRUE16-NEXT: s_waitcnt vmcnt(0)
@@ -4611,7 +4530,6 @@ define amdgpu_kernel void @global_inst_salu_offset_64bit_13bit_split0(ptr addrsp
; GFX11-SDAG-FAKE16-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
; GFX11-SDAG-FAKE16-NEXT: s_waitcnt lgkmcnt(0)
; GFX11-SDAG-FAKE16-NEXT: v_add_co_u32 v0, s0, 0x1000, s0
-; GFX11-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-SDAG-FAKE16-NEXT: v_add_co_ci_u32_e64 v1, null, 2, s1, s0
; GFX11-SDAG-FAKE16-NEXT: global_load_u8 v0, v[0:1], off offset:4095 glc dlc
; GFX11-SDAG-FAKE16-NEXT: s_waitcnt vmcnt(0)
@@ -4716,7 +4634,6 @@ define amdgpu_kernel void @global_inst_salu_offset_64bit_13bit_split1(ptr addrsp
; GFX11-SDAG-TRUE16-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
; GFX11-SDAG-TRUE16-NEXT: s_waitcnt lgkmcnt(0)
; GFX11-SDAG-TRUE16-NEXT: v_add_co_u32 v0, s0, 0x2000, s0
-; GFX11-SDAG-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-SDAG-TRUE16-NEXT: v_add_co_ci_u32_e64 v1, null, 2, s1, s0
; GFX11-SDAG-TRUE16-NEXT: global_load_d16_u8 v0, v[0:1], off glc dlc
; GFX11-SDAG-TRUE16-NEXT: s_waitcnt vmcnt(0)
@@ -4728,7 +4645,6 @@ define amdgpu_kernel void @global_inst_salu_offset_64bit_13bit_split1(ptr addrsp
; GFX11-SDAG-FAKE16-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
; GFX11-SDAG-FAKE16-NEXT: s_waitcnt lgkmcnt(0)
; GFX11-SDAG-FAKE16-NEXT: v_add_co_u32 v0, s0, 0x2000, s0
-; GFX11-SDAG-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-SDAG-FAKE16-NEXT: v_add_co_ci_u32_e64 v1, null, 2, s1, s0
; GFX11-SDAG-FAKE16-NEXT: global_load_u8 v0, v[0:1], off glc dlc
; GFX11-SDAG-FAKE16-NEXT: s_waitcnt vmcnt(0)
diff --git a/llvm/test/CodeGen/AMDGPU/optimize-ds-bvh-stack-pre-ra.ll b/llvm/test/CodeGen/AMDGPU/optimize-ds-bvh-stack-pre-ra.ll
index f764ca92a07623..4812752869cf5a 100644
--- a/llvm/test/CodeGen/AMDGPU/optimize-ds-bvh-stack-pre-ra.ll
+++ b/llvm/test/CodeGen/AMDGPU/optimize-ds-bvh-stack-pre-ra.ll
@@ -12,6 +12,7 @@ define amdgpu_gs void @test_ds_bvh_stack_push4_pop1(i32 %addr, i32 %data.0, i64
; CHECK-NEXT: v_mov_b32_e32 v31, v20
; CHECK-NEXT: image_bvh8_intersect_ray v[21:30], [v[2:3], v[4:5], v[31:33], v[8:10], v11], s[0:3]
; CHECK-NEXT: v_cmpx_eq_f32_e32 0, v20
+; CHECK-NEXT: s_delay_alu instid0(VALU_DEP_1)
; CHECK-NEXT: s_cbranch_execz .LBB0_2
; CHECK-NEXT: ; %bb.1: ; %if
; CHECK-NEXT: global_load_b64 v[6:7], v[12:13], off
@@ -106,6 +107,7 @@ define amdgpu_gs void @test_ds_bvh_stack_push8_pop1(i32 %addr, i32 %data.0, i64
; CHECK-NEXT: v_mov_b32_e32 v31, v20
; CHECK-NEXT: image_bvh8_intersect_ray v[21:30], [v[2:3], v[4:5], v[31:33], v[8:10], v11], s[0:3]
; CHECK-NEXT: v_cmpx_eq_f32_e32 0, v20
+; CHECK-NEXT: s_delay_alu instid0(VALU_DEP_1)
; CHECK-NEXT: s_cbranch_execz .LBB1_2
; CHECK-NEXT: ; %bb.1: ; %if
; CHECK-NEXT: global_load_b64 v[6:7], v[12:13], off
@@ -208,6 +210,7 @@ define amdgpu_gs void @test_ds_bvh_stack_push8_pop2(i32 %addr, i32 %data.0, i64
; CHECK-NEXT: v_mov_b32_e32 v31, v20
; CHECK-NEXT: image_bvh8_intersect_ray v[21:30], [v[2:3], v[4:5], v[31:33], v[8:10], v11], s[0:3]
; CHECK-NEXT: v_cmpx_eq_f32_e32 0, v20
+; CHECK-NEXT: s_delay_alu instid0(VALU_DEP_1)
; CHECK-NEXT: s_cbranch_execz .LBB2_2
; CHECK-NEXT: ; %bb.1: ; %if
; CHECK-NEXT: global_load_b64 v[6:7], v[12:13], off
diff --git a/llvm/test/CodeGen/AMDGPU/preemit-peephole-scc-liveness-issue215745.ll b/llvm/test/CodeGen/AMDGPU/preemit-peephole-scc-liveness-issue215745.ll
index d92c079843e30b..7238b8fc6c98bb 100644
--- a/llvm/test/CodeGen/AMDGPU/preemit-peephole-scc-liveness-issue215745.ll
+++ b/llvm/test/CodeGen/AMDGPU/preemit-peephole-scc-liveness-issue215745.ll
@@ -57,18 +57,21 @@ define amdgpu_kernel void @triangular_loop_scc_liveness(i64 %n) {
; GFX1100-NEXT: ; in Loop: Header=BB0_1 Depth=1
; GFX1100-NEXT: s_cselect_b32 s3, 0, s1
; GFX1100-NEXT: s_cselect_b32 s2, 0, s0
-; GFX1100-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
+; GFX1100-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1100-NEXT: s_cmp_eq_u64 s[2:3], 0
; GFX1100-NEXT: s_cselect_b32 s2, -1, 0
+; GFX1100-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1100-NEXT: s_and_b32 vcc_lo, exec_lo, s2
; GFX1100-NEXT: .LBB0_3: ; %TransitionBlock
; GFX1100-NEXT: ; Parent Loop BB0_1 Depth=1
; GFX1100-NEXT: ; => This Inner Loop Header: Depth=2
+; GFX1100-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1100-NEXT: s_cbranch_vccz .LBB0_3
; GFX1100-NEXT: ; %bb.4: ; %loop.exit.guard
; GFX1100-NEXT: ; in Loop: Header=BB0_1 Depth=1
; GFX1100-NEXT: s_mov_b64 s[2:3], 0
; GFX1100-NEXT: s_mov_b32 vcc_lo, 0
+; GFX1100-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1100-NEXT: s_cbranch_vccz .LBB0_1
; GFX1100-NEXT: ; %bb.5: ; %DummyReturnBlock
; GFX1100-NEXT: s_endpgm
diff --git a/llvm/test/CodeGen/AMDGPU/promote-alloca-vector-dynamic-idx-bitcasts-llc.ll b/llvm/test/CodeGen/AMDGPU/promote-alloca-vector-dynamic-idx-bitcasts-llc.ll
index 10287c131acdbb..5e8d43895358a8 100644
--- a/llvm/test/CodeGen/AMDGPU/promote-alloca-vector-dynamic-idx-bitcasts-llc.ll
+++ b/llvm/test/CodeGen/AMDGPU/promote-alloca-vector-dynamic-idx-bitcasts-llc.ll
@@ -175,6 +175,7 @@ define amdgpu_kernel void @test_bitcast_llc_v64i16_v8i16(ptr addrspace(1) %out,
; GFX11-NEXT: v_mov_b32_e32 v4, 0
; GFX11-NEXT: s_waitcnt lgkmcnt(0)
; GFX11-NEXT: s_lshl_b32 m0, s2, 2
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: v_movrels_b32_e32 v3, v3
; GFX11-NEXT: v_movrels_b32_e32 v2, v2
; GFX11-NEXT: v_movrels_b32_e32 v1, v1
@@ -188,6 +189,7 @@ define amdgpu_kernel void @test_bitcast_llc_v64i16_v8i16(ptr addrspace(1) %out,
; GFX12-NEXT: v_mov_b32_e32 v4, 0
; GFX12-NEXT: s_wait_kmcnt 0x0
; GFX12-NEXT: s_lshl_b32 m0, s2, 2
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: v_movrels_b32_e32 v3, v3
; GFX12-NEXT: v_movrels_b32_e32 v2, v2
; GFX12-NEXT: v_movrels_b32_e32 v1, v1
@@ -228,6 +230,7 @@ define amdgpu_kernel void @test_bitcast_llc_v32i32_v4i32(ptr addrspace(1) %out,
; GFX11-NEXT: v_mov_b32_e32 v4, 0
; GFX11-NEXT: s_waitcnt lgkmcnt(0)
; GFX11-NEXT: s_lshl_b32 m0, s2, 2
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: v_movrels_b32_e32 v3, v3
; GFX11-NEXT: v_movrels_b32_e32 v2, v2
; GFX11-NEXT: v_movrels_b32_e32 v1, v1
@@ -241,6 +244,7 @@ define amdgpu_kernel void @test_bitcast_llc_v32i32_v4i32(ptr addrspace(1) %out,
; GFX12-NEXT: v_mov_b32_e32 v4, 0
; GFX12-NEXT: s_wait_kmcnt 0x0
; GFX12-NEXT: s_lshl_b32 m0, s2, 2
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: v_movrels_b32_e32 v3, v3
; GFX12-NEXT: v_movrels_b32_e32 v2, v2
; GFX12-NEXT: v_movrels_b32_e32 v1, v1
@@ -286,6 +290,7 @@ define amdgpu_kernel void @test_bitcast_llc_v16i64_v4i256(ptr addrspace(1) %out,
; GFX11-NEXT: v_mov_b32_e32 v8, 0
; GFX11-NEXT: s_waitcnt lgkmcnt(0)
; GFX11-NEXT: s_lshl_b32 m0, s2, 3
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: v_movrels_b32_e32 v3, v3
; GFX11-NEXT: v_movrels_b32_e32 v2, v2
; GFX11-NEXT: v_movrels_b32_e32 v1, v1
@@ -305,6 +310,7 @@ define amdgpu_kernel void @test_bitcast_llc_v16i64_v4i256(ptr addrspace(1) %out,
; GFX12-NEXT: v_mov_b32_e32 v8, 0
; GFX12-NEXT: s_wait_kmcnt 0x0
; GFX12-NEXT: s_lshl_b32 m0, s2, 3
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: v_movrels_b32_e32 v3, v3
; GFX12-NEXT: v_movrels_b32_e32 v2, v2
; GFX12-NEXT: v_movrels_b32_e32 v1, v1
diff --git a/llvm/test/CodeGen/AMDGPU/promote-constOffset-to-imm.ll b/llvm/test/CodeGen/AMDGPU/promote-constOffset-to-imm.ll
index b113380874cd38..c571bca74bd0ba 100644
--- a/llvm/test/CodeGen/AMDGPU/promote-constOffset-to-imm.ll
+++ b/llvm/test/CodeGen/AMDGPU/promote-constOffset-to-imm.ll
@@ -252,17 +252,16 @@ define amdgpu_kernel void @clmem_read_simplified(ptr addrspace(1) %buffer) {
; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-NEXT: v_and_b32_e32 v16, 0xffff8000, v1
; GFX11-NEXT: v_lshlrev_b32_e32 v0, 3, v0
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX11-NEXT: v_add_co_u32 v1, s0, s34, v16
; GFX11-NEXT: v_add_co_ci_u32_e64 v2, null, s35, 0, s0
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_u32 v0, vcc_lo, v1, v0
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v2, vcc_lo
; GFX11-NEXT: s_clause 0x1
; GFX11-NEXT: global_load_b64 v[2:3], v[0:1], off
; GFX11-NEXT: global_load_b64 v[4:5], v[0:1], off offset:2048
; GFX11-NEXT: v_add_co_u32 v6, vcc_lo, v0, 0x2000
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v7, null, 0, v1, vcc_lo
; GFX11-NEXT: v_add_co_u32 v8, vcc_lo, 0x1000, v0
; GFX11-NEXT: v_add_co_ci_u32_e64 v9, null, 0, v1, vcc_lo
@@ -273,7 +272,6 @@ define amdgpu_kernel void @clmem_read_simplified(ptr addrspace(1) %buffer) {
; GFX11-NEXT: global_load_b64 v[6:7], v[6:7], off
; GFX11-NEXT: v_add_co_ci_u32_e64 v13, null, 0, v1, vcc_lo
; GFX11-NEXT: v_add_co_u32 v0, vcc_lo, 0x3000, v0
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-NEXT: s_clause 0x2
; GFX11-NEXT: global_load_b64 v[12:13], v[12:13], off offset:2048
@@ -281,31 +279,30 @@ define amdgpu_kernel void @clmem_read_simplified(ptr addrspace(1) %buffer) {
; GFX11-NEXT: global_load_b64 v[0:1], v[0:1], off offset:2048
; GFX11-NEXT: s_waitcnt vmcnt(6)
; GFX11-NEXT: v_add_co_u32 v2, vcc_lo, v4, v2
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX11-NEXT: v_add_co_ci_u32_e64 v3, null, v5, v3, vcc_lo
; GFX11-NEXT: s_waitcnt vmcnt(5)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-NEXT: v_add_co_u32 v2, vcc_lo, v10, v2
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX11-NEXT: v_add_co_ci_u32_e64 v3, null, v11, v3, vcc_lo
; GFX11-NEXT: s_waitcnt vmcnt(4)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-NEXT: v_add_co_u32 v2, vcc_lo, v8, v2
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX11-NEXT: v_add_co_ci_u32_e64 v3, null, v9, v3, vcc_lo
; GFX11-NEXT: s_waitcnt vmcnt(3)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-NEXT: v_add_co_u32 v2, vcc_lo, v6, v2
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX11-NEXT: v_add_co_ci_u32_e64 v3, null, v7, v3, vcc_lo
; GFX11-NEXT: s_waitcnt vmcnt(2)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-NEXT: v_add_co_u32 v2, vcc_lo, v12, v2
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX11-NEXT: v_add_co_ci_u32_e64 v3, null, v13, v3, vcc_lo
; GFX11-NEXT: s_waitcnt vmcnt(1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-NEXT: v_add_co_u32 v2, vcc_lo, v14, v2
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX11-NEXT: v_add_co_ci_u32_e64 v3, null, v15, v3, vcc_lo
; GFX11-NEXT: s_waitcnt vmcnt(0)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, v1, v3, vcc_lo
; GFX11-NEXT: global_store_b64 v16, v[0:1], s[34:35]
; GFX11-NEXT: s_endpgm
@@ -829,11 +826,11 @@ define hidden amdgpu_kernel void @clmem_read(ptr addrspace(1) %buffer) {
; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_and_b32_e32 v6, 0xfe000000, v1
; GFX11-NEXT: v_lshl_or_b32 v0, v0, 3, v6
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX11-NEXT: v_add_co_u32 v0, s0, s34, v0
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, s35, 0, s0
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_u32 v0, vcc_lo, 0x2800, v0
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-NEXT: .LBB1_1: ; %for.cond.preheader
; GFX11-NEXT: ; =>This Loop Header: Depth=1
@@ -844,17 +841,15 @@ define hidden amdgpu_kernel void @clmem_read(ptr addrspace(1) %buffer) {
; GFX11-NEXT: .LBB1_2: ; %for.body
; GFX11-NEXT: ; Parent Loop BB1_1 Depth=1
; GFX11-NEXT: ; => This Inner Loop Header: Depth=2
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-NEXT: v_add_co_u32 v8, vcc_lo, 0xffffe000, v4
; GFX11-NEXT: v_add_co_ci_u32_e64 v9, null, -1, v5, vcc_lo
; GFX11-NEXT: v_add_co_u32 v10, vcc_lo, 0xfffff000, v4
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_2) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v11, null, -1, v5, vcc_lo
; GFX11-NEXT: global_load_b64 v[12:13], v[8:9], off offset:-2048
; GFX11-NEXT: v_add_co_u32 v22, vcc_lo, v4, 0x2000
; GFX11-NEXT: v_add_co_ci_u32_e64 v23, null, 0, v5, vcc_lo
; GFX11-NEXT: v_add_co_u32 v24, vcc_lo, 0x1000, v4
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v25, null, 0, v5, vcc_lo
; GFX11-NEXT: global_load_b64 v[26:27], v[22:23], off offset:-4096
; GFX11-NEXT: v_add_co_u32 v28, vcc_lo, 0x2000, v4
@@ -871,61 +866,60 @@ define hidden amdgpu_kernel void @clmem_read(ptr addrspace(1) %buffer) {
; GFX11-NEXT: global_load_b64 v[22:23], v[22:23], off
; GFX11-NEXT: global_load_b64 v[28:29], v[28:29], off offset:2048
; GFX11-NEXT: v_add_co_u32 v4, vcc_lo, 0x10000, v4
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v5, null, 0, v5, vcc_lo
; GFX11-NEXT: s_addk_i32 s1, 0x2000
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_4) | instid1(VALU_DEP_2)
; GFX11-NEXT: s_cmp_gt_u32 s1, 0x3fffff
; GFX11-NEXT: s_waitcnt vmcnt(10)
; GFX11-NEXT: v_add_co_u32 v2, s0, v12, v2
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX11-NEXT: v_add_co_ci_u32_e64 v3, null, v13, v3, s0
; GFX11-NEXT: s_waitcnt vmcnt(7)
; GFX11-NEXT: v_add_co_u32 v2, s0, v8, v2
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX11-NEXT: v_add_co_ci_u32_e64 v3, null, v9, v3, s0
; GFX11-NEXT: s_waitcnt vmcnt(6)
; GFX11-NEXT: v_add_co_u32 v2, s0, v10, v2
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX11-NEXT: v_add_co_ci_u32_e64 v3, null, v11, v3, s0
; GFX11-NEXT: s_waitcnt vmcnt(5)
; GFX11-NEXT: v_add_co_u32 v2, s0, v14, v2
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX11-NEXT: v_add_co_ci_u32_e64 v3, null, v15, v3, s0
; GFX11-NEXT: s_waitcnt vmcnt(4)
; GFX11-NEXT: v_add_co_u32 v2, s0, v16, v2
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX11-NEXT: v_add_co_ci_u32_e64 v3, null, v17, v3, s0
; GFX11-NEXT: s_waitcnt vmcnt(3)
; GFX11-NEXT: v_add_co_u32 v2, s0, v18, v2
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX11-NEXT: v_add_co_ci_u32_e64 v3, null, v19, v3, s0
; GFX11-NEXT: s_waitcnt vmcnt(2)
; GFX11-NEXT: v_add_co_u32 v2, s0, v20, v2
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-NEXT: v_add_co_ci_u32_e64 v3, null, v21, v3, s0
; GFX11-NEXT: v_add_co_u32 v2, s0, v26, v2
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-NEXT: v_add_co_ci_u32_e64 v3, null, v27, v3, s0
; GFX11-NEXT: v_add_co_u32 v2, s0, v24, v2
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX11-NEXT: v_add_co_ci_u32_e64 v3, null, v25, v3, s0
; GFX11-NEXT: s_waitcnt vmcnt(1)
; GFX11-NEXT: v_add_co_u32 v2, s0, v22, v2
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX11-NEXT: v_add_co_ci_u32_e64 v3, null, v23, v3, s0
; GFX11-NEXT: s_waitcnt vmcnt(0)
; GFX11-NEXT: v_add_co_u32 v2, vcc_lo, v28, v2
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX11-NEXT: v_add_co_ci_u32_e64 v3, null, v29, v3, vcc_lo
; GFX11-NEXT: s_cbranch_scc0 .LBB1_2
; GFX11-NEXT: ; %bb.3: ; %while.cond.loopexit
; GFX11-NEXT: ; in Loop: Header=BB1_1 Depth=1
; GFX11-NEXT: v_sub_co_u32 v7, s0, v7, 1
; GFX11-NEXT: s_and_b32 vcc_lo, exec_lo, s0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_vccz .LBB1_1
; GFX11-NEXT: ; %bb.4: ; %while.end
; GFX11-NEXT: v_add_co_u32 v0, s0, s34, v6
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, s35, 0, s0
; GFX11-NEXT: global_store_b64 v[0:1], v[2:3], off
; GFX11-NEXT: s_endpgm
@@ -1256,17 +1250,16 @@ define amdgpu_kernel void @Address32(ptr addrspace(1) %buffer) {
; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-NEXT: v_and_b32_e32 v6, 0xffff8000, v1
; GFX11-NEXT: v_lshlrev_b32_e32 v0, 2, v0
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX11-NEXT: v_add_co_u32 v1, s0, s34, v6
; GFX11-NEXT: v_add_co_ci_u32_e64 v2, null, s35, 0, s0
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_u32 v0, vcc_lo, v1, v0
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v2, vcc_lo
; GFX11-NEXT: s_clause 0x1
; GFX11-NEXT: global_load_b32 v7, v[0:1], off
; GFX11-NEXT: global_load_b32 v8, v[0:1], off offset:1024
; GFX11-NEXT: v_add_co_u32 v2, vcc_lo, 0x1000, v0
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v1, vcc_lo
; GFX11-NEXT: v_add_co_u32 v4, vcc_lo, v0, 0x2000
; GFX11-NEXT: v_add_co_ci_u32_e64 v5, null, 0, v1, vcc_lo
@@ -1278,7 +1271,6 @@ define amdgpu_kernel void @Address32(ptr addrspace(1) %buffer) {
; GFX11-NEXT: global_load_b32 v13, v[2:3], off offset:2048
; GFX11-NEXT: global_load_b32 v2, v[2:3], off offset:3072
; GFX11-NEXT: v_add_co_u32 v0, vcc_lo, 0x2000, v0
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-NEXT: s_clause 0x1
; GFX11-NEXT: global_load_b32 v3, v[4:5], off
@@ -1515,14 +1507,14 @@ define amdgpu_kernel void @Offset64(ptr addrspace(1) %buffer) {
; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-NEXT: v_and_b32_e32 v8, 0xffff8000, v1
; GFX11-NEXT: v_lshlrev_b32_e32 v0, 3, v0
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX11-NEXT: v_add_co_u32 v1, s0, s34, v8
; GFX11-NEXT: v_add_co_ci_u32_e64 v2, null, s35, 0, s0
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_u32 v0, vcc_lo, v1, v0
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v2, vcc_lo
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_u32 v2, vcc_lo, 0xfffff000, v0
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX11-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v1, vcc_lo
; GFX11-NEXT: s_clause 0x2
; GFX11-NEXT: global_load_b64 v[4:5], v[0:1], off
@@ -1532,15 +1524,14 @@ define amdgpu_kernel void @Offset64(ptr addrspace(1) %buffer) {
; GFX11-NEXT: global_load_b64 v[0:1], v[0:1], off
; GFX11-NEXT: s_waitcnt vmcnt(2)
; GFX11-NEXT: v_add_co_u32 v4, vcc_lo, v6, v4
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX11-NEXT: v_add_co_ci_u32_e64 v5, null, v7, v5, vcc_lo
; GFX11-NEXT: s_waitcnt vmcnt(1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-NEXT: v_add_co_u32 v2, vcc_lo, v2, v4
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX11-NEXT: v_add_co_ci_u32_e64 v3, null, v3, v5, vcc_lo
; GFX11-NEXT: s_waitcnt vmcnt(0)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, v1, v3, vcc_lo
; GFX11-NEXT: global_store_b64 v8, v[0:1], s[34:35]
; GFX11-NEXT: s_endpgm
@@ -1730,17 +1721,16 @@ define amdgpu_kernel void @p32Offset64(ptr addrspace(1) %buffer) {
; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-NEXT: v_and_b32_e32 v6, 0xffff8000, v1
; GFX11-NEXT: v_lshlrev_b32_e32 v0, 2, v0
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX11-NEXT: v_add_co_u32 v1, s0, s34, v6
; GFX11-NEXT: v_add_co_ci_u32_e64 v2, null, s35, 0, s0
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_u32 v0, vcc_lo, v1, v0
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v2, vcc_lo
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_u32 v2, vcc_lo, 0x7ffff000, v0
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX11-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v1, vcc_lo
; GFX11-NEXT: v_add_co_u32 v4, vcc_lo, 0x80000000, v0
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v5, null, 0, v1, vcc_lo
; GFX11-NEXT: s_clause 0x3
; GFX11-NEXT: global_load_b32 v0, v[0:1], off
@@ -1983,16 +1973,16 @@ define amdgpu_kernel void @DiffBase(ptr addrspace(1) %buffer1,
; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_and_b32_e32 v12, 0xffff8000, v0
; GFX11-NEXT: v_add_co_u32 v2, s0, s36, v12
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_3)
; GFX11-NEXT: v_add_co_ci_u32_e64 v3, null, s37, 0, s0
; GFX11-NEXT: v_add_co_u32 v8, s0, s38, v12
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_3)
; GFX11-NEXT: v_add_co_u32 v0, vcc_lo, 0x1000, v2
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_4) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v3, vcc_lo
; GFX11-NEXT: v_add_co_u32 v2, vcc_lo, v2, 0x2000
; GFX11-NEXT: v_add_co_ci_u32_e64 v9, null, s39, 0, s0
; GFX11-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
; GFX11-NEXT: v_add_co_u32 v4, vcc_lo, 0x2000, v8
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX11-NEXT: v_add_co_ci_u32_e64 v5, null, 0, v9, vcc_lo
; GFX11-NEXT: v_add_co_u32 v8, vcc_lo, 0x3000, v8
; GFX11-NEXT: global_load_b64 v[6:7], v[2:3], off offset:-4096
@@ -2005,19 +1995,17 @@ define amdgpu_kernel void @DiffBase(ptr addrspace(1) %buffer1,
; GFX11-NEXT: global_load_b64 v[8:9], v[8:9], off offset:2048
; GFX11-NEXT: s_waitcnt vmcnt(4)
; GFX11-NEXT: v_add_co_u32 v0, vcc_lo, v0, v6
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_2) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, v1, v7, vcc_lo
; GFX11-NEXT: s_waitcnt vmcnt(2)
; GFX11-NEXT: v_add_co_u32 v4, vcc_lo, v10, v4
; GFX11-NEXT: v_add_co_ci_u32_e64 v5, null, v11, v5, vcc_lo
; GFX11-NEXT: s_waitcnt vmcnt(1)
; GFX11-NEXT: v_add_co_u32 v0, vcc_lo, v2, v0
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_2) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, v3, v1, vcc_lo
; GFX11-NEXT: s_waitcnt vmcnt(0)
; GFX11-NEXT: v_add_co_u32 v2, vcc_lo, v8, v4
; GFX11-NEXT: v_add_co_ci_u32_e64 v3, null, v9, v5, vcc_lo
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, v1, v3, vcc_lo
; GFX11-NEXT: global_store_b64 v12, v[0:1], s[36:37]
@@ -2305,14 +2293,14 @@ define amdgpu_kernel void @ReverseOrder(ptr addrspace(1) %buffer) {
; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-NEXT: v_and_b32_e32 v16, 0xffff8000, v1
; GFX11-NEXT: v_lshlrev_b32_e32 v0, 3, v0
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX11-NEXT: v_add_co_u32 v1, s0, s34, v16
; GFX11-NEXT: v_add_co_ci_u32_e64 v2, null, s35, 0, s0
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_u32 v0, vcc_lo, v1, v0
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v2, vcc_lo
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_u32 v2, vcc_lo, 0x3000, v0
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX11-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v1, vcc_lo
; GFX11-NEXT: v_add_co_u32 v8, vcc_lo, 0x2000, v0
; GFX11-NEXT: s_clause 0x2
@@ -2321,7 +2309,6 @@ define amdgpu_kernel void @ReverseOrder(ptr addrspace(1) %buffer) {
; GFX11-NEXT: global_load_b64 v[2:3], v[2:3], off
; GFX11-NEXT: v_add_co_ci_u32_e64 v9, null, 0, v1, vcc_lo
; GFX11-NEXT: v_add_co_u32 v10, vcc_lo, 0x1000, v0
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v11, null, 0, v1, vcc_lo
; GFX11-NEXT: s_clause 0x4
; GFX11-NEXT: global_load_b64 v[12:13], v[8:9], off offset:2048
@@ -2331,30 +2318,29 @@ define amdgpu_kernel void @ReverseOrder(ptr addrspace(1) %buffer) {
; GFX11-NEXT: global_load_b64 v[0:1], v[0:1], off offset:2048
; GFX11-NEXT: s_waitcnt vmcnt(6)
; GFX11-NEXT: v_add_co_u32 v4, vcc_lo, v6, v4
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX11-NEXT: v_add_co_ci_u32_e64 v5, null, v7, v5, vcc_lo
; GFX11-NEXT: s_waitcnt vmcnt(5)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-NEXT: v_add_co_u32 v2, vcc_lo, v2, v4
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX11-NEXT: v_add_co_ci_u32_e64 v3, null, v3, v5, vcc_lo
; GFX11-NEXT: s_waitcnt vmcnt(4)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-NEXT: v_add_co_u32 v2, vcc_lo, v12, v2
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX11-NEXT: v_add_co_ci_u32_e64 v3, null, v13, v3, vcc_lo
; GFX11-NEXT: s_waitcnt vmcnt(2)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-NEXT: v_add_co_u32 v2, vcc_lo, v8, v2
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX11-NEXT: v_add_co_ci_u32_e64 v3, null, v9, v3, vcc_lo
; GFX11-NEXT: s_waitcnt vmcnt(1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-NEXT: v_add_co_u32 v2, vcc_lo, v10, v2
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-NEXT: v_add_co_ci_u32_e64 v3, null, v11, v3, vcc_lo
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-NEXT: v_add_co_u32 v2, vcc_lo, v14, v2
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX11-NEXT: v_add_co_ci_u32_e64 v3, null, v15, v3, vcc_lo
; GFX11-NEXT: s_waitcnt vmcnt(0)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, v1, v3, vcc_lo
; GFX11-NEXT: global_store_b64 v16, v[0:1], s[34:35]
; GFX11-NEXT: s_endpgm
@@ -2538,14 +2524,14 @@ define hidden amdgpu_kernel void @negativeoffset(ptr addrspace(1) nocapture %buf
; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-NEXT: v_and_b32_e32 v4, 0xffff8000, v1
; GFX11-NEXT: v_lshlrev_b32_e32 v0, 3, v0
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX11-NEXT: v_add_co_u32 v1, s0, s34, v4
; GFX11-NEXT: v_add_co_ci_u32_e64 v2, null, s35, 0, s0
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_u32 v0, vcc_lo, v1, v0
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v2, vcc_lo
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_add_co_u32 v2, vcc_lo, 0x1000, v0
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX11-NEXT: v_add_co_ci_u32_e64 v3, null, -1, v1, vcc_lo
; GFX11-NEXT: v_add_nc_u32_e32 v1, -1, v1
; GFX11-NEXT: s_clause 0x1
@@ -2553,7 +2539,6 @@ define hidden amdgpu_kernel void @negativeoffset(ptr addrspace(1) nocapture %buf
; GFX11-NEXT: global_load_b64 v[0:1], v[0:1], off
; GFX11-NEXT: s_waitcnt vmcnt(0)
; GFX11-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, v1, v3, vcc_lo
; GFX11-NEXT: global_store_b64 v4, v[0:1], s[34:35]
; GFX11-NEXT: s_endpgm
@@ -2644,7 +2629,6 @@ define amdgpu_kernel void @negativeoffsetnullptr(ptr %buffer) {
; GFX11-TRUE16: ; %bb.0: ; %entry
; GFX11-TRUE16-NEXT: s_mov_b64 s[0:1], src_private_base
; GFX11-TRUE16-NEXT: v_add_co_u32 v0, s0, -1, 0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_add_co_ci_u32_e64 v1, null, -1, s1, s0
; GFX11-TRUE16-NEXT: s_mov_b32 s0, 0
; GFX11-TRUE16-NEXT: flat_load_d16_u8 v0, v[0:1]
@@ -2656,6 +2640,7 @@ define amdgpu_kernel void @negativeoffsetnullptr(ptr %buffer) {
; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_or_b32 s0, s1, s0
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB8_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %end
; GFX11-TRUE16-NEXT: s_endpgm
@@ -2664,7 +2649,6 @@ define amdgpu_kernel void @negativeoffsetnullptr(ptr %buffer) {
; GFX11-FAKE16: ; %bb.0: ; %entry
; GFX11-FAKE16-NEXT: s_mov_b64 s[0:1], src_private_base
; GFX11-FAKE16-NEXT: v_add_co_u32 v0, s0, -1, 0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_add_co_ci_u32_e64 v1, null, -1, s1, s0
; GFX11-FAKE16-NEXT: s_mov_b32 s0, 0
; GFX11-FAKE16-NEXT: flat_load_u8 v0, v[0:1]
@@ -2676,6 +2660,7 @@ define amdgpu_kernel void @negativeoffsetnullptr(ptr %buffer) {
; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_or_b32 s0, s1, s0
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB8_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %end
; GFX11-FAKE16-NEXT: s_endpgm
diff --git a/llvm/test/CodeGen/AMDGPU/pseudo-scalar-transcendental.ll b/llvm/test/CodeGen/AMDGPU/pseudo-scalar-transcendental.ll
index 8de7a0de592c7e..eb98fb488aab02 100644
--- a/llvm/test/CodeGen/AMDGPU/pseudo-scalar-transcendental.ll
+++ b/llvm/test/CodeGen/AMDGPU/pseudo-scalar-transcendental.ll
@@ -10,27 +10,28 @@ define amdgpu_cs float @v_s_exp_f32(float inreg %src) {
; GFX12-SDAG-LABEL: v_s_exp_f32:
; GFX12-SDAG: ; %bb.0:
; GFX12-SDAG-NEXT: s_cmp_lt_f32 s0, 0xc2fc0000
+; GFX12-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_3) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-SDAG-NEXT: s_cselect_b32 s1, 0x42800000, 0
-; GFX12-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_2)
; GFX12-SDAG-NEXT: s_add_f32 s0, s0, s1
; GFX12-SDAG-NEXT: s_cselect_b32 s1, 0x1f800000, 1.0
+; GFX12-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_2) | instskip(NEXT) | instid1(TRANS32_DEP_1)
; GFX12-SDAG-NEXT: v_s_exp_f32 s0, s0
-; GFX12-SDAG-NEXT: s_delay_alu instid0(TRANS32_DEP_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_2)
; GFX12-SDAG-NEXT: s_mul_f32 s0, s0, s1
; GFX12-SDAG-NEXT: s_wait_alu depctr_sa_sdst(0)
+; GFX12-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_2)
; GFX12-SDAG-NEXT: v_mov_b32_e32 v0, s0
; GFX12-SDAG-NEXT: ; return to shader part epilog
;
; GFX12-GISEL-LABEL: v_s_exp_f32:
; GFX12-GISEL: ; %bb.0:
; GFX12-GISEL-NEXT: s_cmp_lt_f32 s0, 0xc2fc0000
+; GFX12-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_3) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-GISEL-NEXT: s_cselect_b32 s1, 0x42800000, 0
-; GFX12-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_2)
; GFX12-GISEL-NEXT: s_add_f32 s0, s0, s1
; GFX12-GISEL-NEXT: s_cselect_b32 s1, 0xffffffc0, 0
+; GFX12-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_2) | instskip(SKIP_1) | instid1(TRANS32_DEP_1)
; GFX12-GISEL-NEXT: v_s_exp_f32 s0, s0
; GFX12-GISEL-NEXT: s_wait_alu depctr_va_sdst(0)
-; GFX12-GISEL-NEXT: s_delay_alu instid0(TRANS32_DEP_1)
; GFX12-GISEL-NEXT: v_ldexp_f32 v0, s0, s1
; GFX12-GISEL-NEXT: ; return to shader part epilog
;
@@ -115,20 +116,22 @@ define amdgpu_cs float @v_s_log_f32(float inreg %src) {
; GFX12-SDAG-LABEL: v_s_log_f32:
; GFX12-SDAG: ; %bb.0:
; GFX12-SDAG-NEXT: s_cmp_lt_f32 s0, 0x800000
+; GFX12-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_3) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-SDAG-NEXT: s_cselect_b32 s1, 0x4f800000, 1.0
-; GFX12-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_2)
; GFX12-SDAG-NEXT: s_mul_f32 s0, s0, s1
; GFX12-SDAG-NEXT: s_cselect_b32 s1, 0x42000000, 0
+; GFX12-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_2) | instskip(NEXT) | instid1(TRANS32_DEP_1)
; GFX12-SDAG-NEXT: v_s_log_f32 s0, s0
-; GFX12-SDAG-NEXT: s_delay_alu instid0(TRANS32_DEP_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_2)
; GFX12-SDAG-NEXT: s_sub_f32 s0, s0, s1
; GFX12-SDAG-NEXT: s_wait_alu depctr_sa_sdst(0)
+; GFX12-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_2)
; GFX12-SDAG-NEXT: v_mov_b32_e32 v0, s0
; GFX12-SDAG-NEXT: ; return to shader part epilog
;
; GFX12-GISEL-LABEL: v_s_log_f32:
; GFX12-GISEL: ; %bb.0:
; GFX12-GISEL-NEXT: s_cmp_lt_f32 s0, 0x800000
+; GFX12-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_3)
; GFX12-GISEL-NEXT: s_cselect_b32 s2, 1, 0
; GFX12-GISEL-NEXT: s_cselect_b32 s1, 1, 0
; GFX12-GISEL-NEXT: s_lshl_b32 s2, s2, 5
@@ -508,7 +511,7 @@ define amdgpu_cs float @v_s_sqrt_f32(float inreg %src) {
; GFX12-SDAG: ; %bb.0:
; GFX12-SDAG-NEXT: s_mul_f32 s1, s0, 0x4f800000
; GFX12-SDAG-NEXT: s_cmp_lt_f32 s0, 0xf800000
-; GFX12-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_2) | instskip(NEXT) | instid1(SALU_CYCLE_1)
+; GFX12-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_3) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-SDAG-NEXT: s_cselect_b32 s1, s1, s0
; GFX12-SDAG-NEXT: v_s_sqrt_f32 s2, s1
; GFX12-SDAG-NEXT: s_mov_b32 s4, s1
@@ -519,19 +522,19 @@ define amdgpu_cs float @v_s_sqrt_f32(float inreg %src) {
; GFX12-SDAG-NEXT: s_fmac_f32 s4, s5, s2
; GFX12-SDAG-NEXT: s_mov_b32 s5, s1
; GFX12-SDAG-NEXT: s_cmp_le_f32 s4, 0
+; GFX12-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_3) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX12-SDAG-NEXT: s_cselect_b32 s3, s3, s2
; GFX12-SDAG-NEXT: s_add_co_i32 s4, s2, 1
-; GFX12-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-SDAG-NEXT: s_xor_b32 s6, s4, 0x80000000
+; GFX12-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_3)
; GFX12-SDAG-NEXT: s_fmac_f32 s5, s6, s2
-; GFX12-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_3)
; GFX12-SDAG-NEXT: s_cmp_gt_f32 s5, 0
+; GFX12-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_3) | instskip(SKIP_4) | instid1(SALU_CYCLE_1)
; GFX12-SDAG-NEXT: s_cselect_b32 s2, s4, s3
; GFX12-SDAG-NEXT: s_cmp_lt_f32 s0, 0xf800000
; GFX12-SDAG-NEXT: s_mul_f32 s0, s2, 0x37800000
; GFX12-SDAG-NEXT: v_cmp_class_f32_e64 s3, s1, 0x260
; GFX12-SDAG-NEXT: s_wait_alu depctr_sa_sdst(0)
-; GFX12-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-SDAG-NEXT: s_cselect_b32 s0, s0, s2
; GFX12-SDAG-NEXT: s_and_b32 s2, s3, exec_lo
; GFX12-SDAG-NEXT: s_wait_alu depctr_sa_sdst(0)
@@ -544,29 +547,30 @@ define amdgpu_cs float @v_s_sqrt_f32(float inreg %src) {
; GFX12-GISEL: ; %bb.0:
; GFX12-GISEL-NEXT: s_cmp_lt_f32 s0, 0xf800000
; GFX12-GISEL-NEXT: s_mul_f32 s2, s0, 0x4f800000
+; GFX12-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_2) | instskip(NEXT) | instid1(SALU_CYCLE_2)
; GFX12-GISEL-NEXT: s_cselect_b32 s1, 1, 0
-; GFX12-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_2) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-GISEL-NEXT: s_cselect_b32 s0, s2, s0
+; GFX12-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(TRANS32_DEP_1)
; GFX12-GISEL-NEXT: v_s_sqrt_f32 s2, s0
; GFX12-GISEL-NEXT: s_mov_b32 s4, s0
; GFX12-GISEL-NEXT: s_mov_b32 s6, s0
-; GFX12-GISEL-NEXT: s_delay_alu instid0(TRANS32_DEP_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-GISEL-NEXT: s_add_co_i32 s3, s2, -1
+; GFX12-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-GISEL-NEXT: s_xor_b32 s5, s3, 0x80000000
-; GFX12-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX12-GISEL-NEXT: s_fmac_f32 s4, s5, s2
; GFX12-GISEL-NEXT: s_add_co_i32 s5, s2, 1
+; GFX12-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-GISEL-NEXT: s_xor_b32 s7, s5, 0x80000000
-; GFX12-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_2)
; GFX12-GISEL-NEXT: s_cmp_le_f32 s4, 0
; GFX12-GISEL-NEXT: s_fmac_f32 s6, s7, s2
+; GFX12-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_2) | instskip(NEXT) | instid1(SALU_CYCLE_2)
; GFX12-GISEL-NEXT: s_cselect_b32 s2, s3, s2
; GFX12-GISEL-NEXT: s_cmp_gt_f32 s6, 0
; GFX12-GISEL-NEXT: v_cmp_class_f32_e64 s3, s0, 0x260
+; GFX12-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_2) | instskip(SKIP_2) | instid1(SALU_CYCLE_3)
; GFX12-GISEL-NEXT: s_cselect_b32 s2, s5, s2
; GFX12-GISEL-NEXT: s_cmp_lg_u32 s1, 0
; GFX12-GISEL-NEXT: s_mul_f32 s4, s2, 0x37800000
-; GFX12-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_3)
; GFX12-GISEL-NEXT: s_cselect_b32 s1, s4, s2
; GFX12-GISEL-NEXT: s_cmp_lg_u32 s3, 0
; GFX12-GISEL-NEXT: s_wait_alu depctr_sa_sdst(0)
@@ -669,23 +673,23 @@ define amdgpu_cs float @srcmods_abs_f32(float inreg %src) {
; GFX12-SDAG-LABEL: srcmods_abs_f32:
; GFX12-SDAG: ; %bb.0:
; GFX12-SDAG-NEXT: s_bitset0_b32 s0, 31
-; GFX12-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
+; GFX12-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_3)
; GFX12-SDAG-NEXT: s_cmp_lt_f32 s0, 0x800000
; GFX12-SDAG-NEXT: s_cselect_b32 s1, 0x4f800000, 1.0
+; GFX12-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_2)
; GFX12-SDAG-NEXT: s_mul_f32 s0, s0, s1
; GFX12-SDAG-NEXT: s_cselect_b32 s1, 0x42000000, 0
-; GFX12-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_2) | instskip(NEXT) | instid1(TRANS32_DEP_1)
; GFX12-SDAG-NEXT: v_s_log_f32 s0, s0
+; GFX12-SDAG-NEXT: s_delay_alu instid0(TRANS32_DEP_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_2)
; GFX12-SDAG-NEXT: s_sub_f32 s0, s0, s1
; GFX12-SDAG-NEXT: s_wait_alu depctr_sa_sdst(0)
-; GFX12-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_2)
; GFX12-SDAG-NEXT: v_mov_b32_e32 v0, s0
; GFX12-SDAG-NEXT: ; return to shader part epilog
;
; GFX12-GISEL-LABEL: srcmods_abs_f32:
; GFX12-GISEL: ; %bb.0:
; GFX12-GISEL-NEXT: s_and_b32 s1, s0, 0x7fffffff
-; GFX12-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX12-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_3)
; GFX12-GISEL-NEXT: s_cmp_lt_f32 s1, 0x800000
; GFX12-GISEL-NEXT: s_cselect_b32 s2, 1, 0
; GFX12-GISEL-NEXT: s_cselect_b32 s1, 1, 0
@@ -729,21 +733,22 @@ define amdgpu_cs float @srcmods_neg_f32(float inreg %src) {
; GFX12-SDAG: ; %bb.0:
; GFX12-SDAG-NEXT: s_xor_b32 s1, s0, 0x80000000
; GFX12-SDAG-NEXT: s_cmp_gt_f32 s0, 0x80800000
+; GFX12-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_3) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-SDAG-NEXT: s_cselect_b32 s0, 0x4f800000, 1.0
-; GFX12-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_2)
; GFX12-SDAG-NEXT: s_mul_f32 s0, s1, s0
; GFX12-SDAG-NEXT: s_cselect_b32 s1, 0x42000000, 0
+; GFX12-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_2) | instskip(NEXT) | instid1(TRANS32_DEP_1)
; GFX12-SDAG-NEXT: v_s_log_f32 s0, s0
-; GFX12-SDAG-NEXT: s_delay_alu instid0(TRANS32_DEP_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_2)
; GFX12-SDAG-NEXT: s_sub_f32 s0, s0, s1
; GFX12-SDAG-NEXT: s_wait_alu depctr_sa_sdst(0)
+; GFX12-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_2)
; GFX12-SDAG-NEXT: v_mov_b32_e32 v0, s0
; GFX12-SDAG-NEXT: ; return to shader part epilog
;
; GFX12-GISEL-LABEL: srcmods_neg_f32:
; GFX12-GISEL: ; %bb.0:
; GFX12-GISEL-NEXT: s_xor_b32 s1, s0, 0x80000000
-; GFX12-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX12-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_3)
; GFX12-GISEL-NEXT: s_cmp_lt_f32 s1, 0x800000
; GFX12-GISEL-NEXT: s_cselect_b32 s2, 1, 0
; GFX12-GISEL-NEXT: s_cselect_b32 s1, 1, 0
diff --git a/llvm/test/CodeGen/AMDGPU/ptradd-sdag.ll b/llvm/test/CodeGen/AMDGPU/ptradd-sdag.ll
index 42b4282dc2c5c2..87cbcb321c4c59 100644
--- a/llvm/test/CodeGen/AMDGPU/ptradd-sdag.ll
+++ b/llvm/test/CodeGen/AMDGPU/ptradd-sdag.ll
@@ -41,10 +41,10 @@ define ptr @gep_as0(ptr %p, i64 %offset) {
; GFX11: ; %bb.0: ; %entry
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: v_lshlrev_b64 v[2:3], 2, v[2:3]
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, v1, v3, vcc_lo
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-NEXT: v_add_co_u32 v0, vcc_lo, v0, 5
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-NEXT: s_setpc_b64 s[30:31]
@@ -196,10 +196,10 @@ define ptr @multi_gep_as0(ptr %p, i64 %offset) {
; GFX11: ; %bb.0: ; %entry
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: v_lshlrev_b64 v[2:3], 2, v[2:3]
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, v1, v3, vcc_lo
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-NEXT: v_add_co_u32 v0, vcc_lo, v0, 5
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX11-NEXT: s_setpc_b64 s[30:31]
diff --git a/llvm/test/CodeGen/AMDGPU/reassoc-mul-add-1-to-mad.ll b/llvm/test/CodeGen/AMDGPU/reassoc-mul-add-1-to-mad.ll
index f9c85fa324fe19..b64d271a03f0a6 100644
--- a/llvm/test/CodeGen/AMDGPU/reassoc-mul-add-1-to-mad.ll
+++ b/llvm/test/CodeGen/AMDGPU/reassoc-mul-add-1-to-mad.ll
@@ -2047,12 +2047,12 @@ define i64 @v_mul_sub_1_i64(i64 %x, i64 %y) {
; GFX13-NEXT: s_wait_bvhcnt 0x0
; GFX13-NEXT: s_wait_kmcnt 0x0
; GFX13-NEXT: v_add_co_u32 v2, vcc_lo, v2, -1
-; GFX13-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX13-NEXT: v_add_co_ci_u32_e64 v3, null, -1, v3, vcc_lo
+; GFX13-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX13-NEXT: v_mul_lo_u32 v4, v1, v2
-; GFX13-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX13-NEXT: v_mul_lo_u32 v3, v0, v3
; GFX13-NEXT: v_mad_co_u64_u32 v[0:1], null, v0, v2, 0
+; GFX13-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX13-NEXT: v_add3_u32 v1, v1, v3, v4
; GFX13-NEXT: s_set_pc_i64 s[30:31]
%sub = sub i64 %y, 1
@@ -2139,12 +2139,12 @@ define i64 @v_mul_sub_1_i64_commute(i64 %x, i64 %y) {
; GFX13-NEXT: s_wait_bvhcnt 0x0
; GFX13-NEXT: s_wait_kmcnt 0x0
; GFX13-NEXT: v_add_co_u32 v2, vcc_lo, v2, -1
-; GFX13-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX13-NEXT: v_add_co_ci_u32_e64 v3, null, -1, v3, vcc_lo
+; GFX13-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX13-NEXT: v_mul_lo_u32 v4, v2, v1
-; GFX13-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX13-NEXT: v_mul_lo_u32 v3, v3, v0
; GFX13-NEXT: v_mad_co_u64_u32 v[0:1], null, v2, v0, 0
+; GFX13-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX13-NEXT: v_add3_u32 v1, v1, v4, v3
; GFX13-NEXT: s_set_pc_i64 s[30:31]
%sub = sub i64 %y, 1
@@ -2234,7 +2234,7 @@ define i64 @v_mul_sub_x_i64(i64 %x, i64 %y) {
; GFX13-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX13-NEXT: v_add3_u32 v3, v3, v5, v4
; GFX13-NEXT: v_sub_co_u32 v0, vcc_lo, v2, v0
-; GFX13-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX13-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX13-NEXT: v_sub_co_ci_u32_e64 v1, null, v3, v1, vcc_lo
; GFX13-NEXT: s_set_pc_i64 s[30:31]
%mul = mul i64 %x, %y
@@ -2321,12 +2321,12 @@ define i64 @v_mul_add_2_i64(i64 %x, i64 %y) {
; GFX13-NEXT: s_wait_bvhcnt 0x0
; GFX13-NEXT: s_wait_kmcnt 0x0
; GFX13-NEXT: v_add_co_u32 v2, vcc_lo, v2, 2
-; GFX13-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX13-NEXT: v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
+; GFX13-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX13-NEXT: v_mul_lo_u32 v4, v1, v2
-; GFX13-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX13-NEXT: v_mul_lo_u32 v3, v0, v3
; GFX13-NEXT: v_mad_co_u64_u32 v[0:1], null, v0, v2, 0
+; GFX13-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX13-NEXT: v_add3_u32 v1, v1, v3, v4
; GFX13-NEXT: s_set_pc_i64 s[30:31]
%add = add i64 %y, 2
@@ -2413,12 +2413,12 @@ define i64 @v_mul_sub_2_i64(i64 %x, i64 %y) {
; GFX13-NEXT: s_wait_bvhcnt 0x0
; GFX13-NEXT: s_wait_kmcnt 0x0
; GFX13-NEXT: v_add_co_u32 v2, vcc_lo, v2, -2
-; GFX13-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX13-NEXT: v_add_co_ci_u32_e64 v3, null, -1, v3, vcc_lo
+; GFX13-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX13-NEXT: v_mul_lo_u32 v4, v1, v2
-; GFX13-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX13-NEXT: v_mul_lo_u32 v3, v0, v3
; GFX13-NEXT: v_mad_co_u64_u32 v[0:1], null, v0, v2, 0
+; GFX13-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX13-NEXT: v_add3_u32 v1, v1, v3, v4
; GFX13-NEXT: s_set_pc_i64 s[30:31]
%sub = sub i64 %y, 2
@@ -5373,16 +5373,15 @@ define amdgpu_kernel void @compute_mad(ptr addrspace(4) %i18, ptr addrspace(4) %
; GFX13-NEXT: v_add_nc_u32_e32 v3, v4, v1
; GFX13-NEXT: v_mad_co_u64_u32 v[0:1], null, s5, s4, v[0:1]
; GFX13-NEXT: v_mul_lo_u32 v1, v3, v2
-; GFX13-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX13-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX13-NEXT: v_add_co_u32 v2, s2, s2, v0
; GFX13-NEXT: v_add_co_ci_u32_e64 v3, null, s3, 0, s2
-; GFX13-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX13-NEXT: v_mad_co_u64_u32 v[4:5], null, v1, v4, v[1:2]
-; GFX13-NEXT: v_lshlrev_b64_e32 v[2:3], 2, v[2:3]
; GFX13-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX13-NEXT: v_lshlrev_b64_e32 v[2:3], 2, v[2:3]
; GFX13-NEXT: v_mad_co_u64_u32 v[0:1], null, v4, v1, v[4:5]
+; GFX13-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_3)
; GFX13-NEXT: v_add_co_u32 v1, vcc_lo, s0, v2
-; GFX13-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX13-NEXT: v_add_co_ci_u32_e64 v2, null, s1, v3, vcc_lo
; GFX13-NEXT: global_store_b32 v[1:2], v0, off
; GFX13-NEXT: s_endpgm
diff --git a/llvm/test/CodeGen/AMDGPU/release-vgprs-spill.ll b/llvm/test/CodeGen/AMDGPU/release-vgprs-spill.ll
index b62cb923f30a32..bcb10e8a6b657a 100644
--- a/llvm/test/CodeGen/AMDGPU/release-vgprs-spill.ll
+++ b/llvm/test/CodeGen/AMDGPU/release-vgprs-spill.ll
@@ -27,7 +27,6 @@ define amdgpu_kernel void @f(ptr %ptr, i32 %arg) "amdgpu-flat-work-group-size"="
; CHECK-NEXT: scratch_load_b32 v2, off, off
; CHECK-NEXT: s_waitcnt vmcnt(1)
; CHECK-NEXT: v_add_co_u32 v0, s0, s2, v0
-; CHECK-NEXT: s_delay_alu instid0(VALU_DEP_1)
; CHECK-NEXT: v_add_co_ci_u32_e64 v1, null, s3, 0, s0
; CHECK-NEXT: s_waitcnt vmcnt(0)
; CHECK-NEXT: global_store_b32 v[0:1], v2, off
diff --git a/llvm/test/CodeGen/AMDGPU/rename-independent-subregs-unused-lanes.ll b/llvm/test/CodeGen/AMDGPU/rename-independent-subregs-unused-lanes.ll
index 9fb6be62a9d66a..65a9af531ba837 100644
--- a/llvm/test/CodeGen/AMDGPU/rename-independent-subregs-unused-lanes.ll
+++ b/llvm/test/CodeGen/AMDGPU/rename-independent-subregs-unused-lanes.ll
@@ -15,15 +15,18 @@ define <7 x i32> @multiple_predecessor_unused_lanes(<7 x i32> %ha, i32 %h.sel) {
; CHECK-NEXT: v_dual_mov_b32 v7, v0 :: v_dual_and_b32 v0, 3, v8
; CHECK-NEXT: s_mov_b32 s0, exec_lo
; CHECK-NEXT: v_cmpx_lt_i32_e32 0, v0
+; CHECK-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; CHECK-NEXT: s_xor_b32 s0, exec_lo, s0
; CHECK-NEXT: s_cbranch_execz .LBB0_6
; CHECK-NEXT: ; %bb.1: ; %LeafBlock
; CHECK-NEXT: s_mov_b32 s1, exec_lo
; CHECK-NEXT: v_cmpx_ne_u32_e32 1, v0
+; CHECK-NEXT: s_delay_alu instid0(VALU_DEP_1)
; CHECK-NEXT: s_xor_b32 s1, exec_lo, s1
; CHECK-NEXT: ; %bb.2: ; %h.default
; CHECK-NEXT: v_xor_b32_e32 v12, 1, v5
; CHECK-NEXT: ; %bb.3: ; %Flow
+; CHECK-NEXT: s_delay_alu instid0(VALU_DEP_2)
; CHECK-NEXT: s_and_not1_saveexec_b32 s1, s1
; CHECK-NEXT: s_cbranch_execz .LBB0_5
; CHECK-NEXT: ; %bb.4: ; %h.shuffle
@@ -41,6 +44,7 @@ define <7 x i32> @multiple_predecessor_unused_lanes(<7 x i32> %ha, i32 %h.sel) {
; CHECK-NEXT: ; implicit-def: $vgpr5
; CHECK-NEXT: ; implicit-def: $vgpr1
; CHECK-NEXT: .LBB0_6: ; %Flow2
+; CHECK-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; CHECK-NEXT: s_and_not1_saveexec_b32 s0, s0
; CHECK-NEXT: s_cbranch_execz .LBB0_8
; CHECK-NEXT: ; %bb.7: ; %h.add
@@ -53,7 +57,7 @@ define <7 x i32> @multiple_predecessor_unused_lanes(<7 x i32> %ha, i32 %h.sel) {
; CHECK-NEXT: v_or_b32_e32 v7, 1, v7
; CHECK-NEXT: .LBB0_8: ; %UnifiedReturnBlock
; CHECK-NEXT: s_or_b32 exec_lo, exec_lo, s0
-; CHECK-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; CHECK-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; CHECK-NEXT: v_dual_mov_b32 v0, v7 :: v_dual_mov_b32 v1, v8
; CHECK-NEXT: v_dual_mov_b32 v2, v9 :: v_dual_mov_b32 v3, v10
; CHECK-NEXT: v_dual_mov_b32 v4, v11 :: v_dual_mov_b32 v5, v12
diff --git a/llvm/test/CodeGen/AMDGPU/repeated-divisor.ll b/llvm/test/CodeGen/AMDGPU/repeated-divisor.ll
index 8438b8b089cc4e..ffa2421f933c20 100644
--- a/llvm/test/CodeGen/AMDGPU/repeated-divisor.ll
+++ b/llvm/test/CodeGen/AMDGPU/repeated-divisor.ll
@@ -89,10 +89,10 @@ define <2 x float> @v_repeat_divisor_f32_x2(float %x, float %y, float %D) #0 {
; GFX11-NEXT: v_fma_f32 v4, -v4, v10, v7
; GFX11-NEXT: v_div_fmas_f32 v3, v3, v5, v8
; GFX11-NEXT: s_mov_b32 vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: v_div_fmas_f32 v4, v4, v6, v10
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-NEXT: v_div_fixup_f32 v0, v3, v2, v0
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX11-NEXT: v_div_fixup_f32 v1, v4, v2, v1
; GFX11-NEXT: s_setpc_b64 s[30:31]
%div0 = fdiv float %x, %D
@@ -228,10 +228,10 @@ define <2 x float> @v_repeat_divisor_f32_x2_arcp_daz(float %x, float %y, float %
; GFX11-NEXT: v_fmac_f32_e32 v6, v7, v4
; GFX11-NEXT: v_fma_f32 v3, -v3, v6, v5
; GFX11-NEXT: s_denorm_mode 12
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: v_div_fmas_f32 v3, v3, v4, v6
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_div_fixup_f32 v2, v3, v2, 1.0
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_mul_f32_e32 v0, v0, v2
; GFX11-NEXT: v_mul_f32_e32 v1, v1, v2
; GFX11-NEXT: s_setpc_b64 s[30:31]
diff --git a/llvm/test/CodeGen/AMDGPU/required-export-priority.ll b/llvm/test/CodeGen/AMDGPU/required-export-priority.ll
index b6e2e75c9d28ca..d7cc4be0febad7 100644
--- a/llvm/test/CodeGen/AMDGPU/required-export-priority.ll
+++ b/llvm/test/CodeGen/AMDGPU/required-export-priority.ll
@@ -179,6 +179,7 @@ define amdgpu_ps void @test_if_export_f32(i32 %flag, float %x, float %y, float %
; GFX11: ; %bb.0:
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v0
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_cbranch_execz .LBB9_2
; GFX11-NEXT: ; %bb.1: ; %exp
; GFX11-NEXT: exp mrt0, v1, v2, v3, v4
@@ -190,6 +191,7 @@ define amdgpu_ps void @test_if_export_f32(i32 %flag, float %x, float %y, float %
; GFX1150-NEXT: s_setprio 2
; GFX1150-NEXT: s_mov_b32 s0, exec_lo
; GFX1150-NEXT: v_cmpx_ne_u32_e32 0, v0
+; GFX1150-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1150-NEXT: s_cbranch_execz .LBB9_2
; GFX1150-NEXT: ; %bb.1: ; %exp
; GFX1150-NEXT: exp mrt0, v1, v2, v3, v4
@@ -216,6 +218,7 @@ define amdgpu_ps void @test_if_export_vm_f32(i32 %flag, float %x, float %y, floa
; GFX11: ; %bb.0:
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v0
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_cbranch_execz .LBB10_2
; GFX11-NEXT: ; %bb.1: ; %exp
; GFX11-NEXT: exp mrt0, v1, v2, v3, v4
@@ -227,6 +230,7 @@ define amdgpu_ps void @test_if_export_vm_f32(i32 %flag, float %x, float %y, floa
; GFX1150-NEXT: s_setprio 2
; GFX1150-NEXT: s_mov_b32 s0, exec_lo
; GFX1150-NEXT: v_cmpx_ne_u32_e32 0, v0
+; GFX1150-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1150-NEXT: s_cbranch_execz .LBB10_2
; GFX1150-NEXT: ; %bb.1: ; %exp
; GFX1150-NEXT: exp mrt0, v1, v2, v3, v4
@@ -253,6 +257,7 @@ define amdgpu_ps void @test_if_export_done_f32(i32 %flag, float %x, float %y, fl
; GFX11: ; %bb.0:
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v0
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_cbranch_execz .LBB11_2
; GFX11-NEXT: ; %bb.1: ; %exp
; GFX11-NEXT: exp mrt0, v1, v2, v3, v4 done
@@ -264,6 +269,7 @@ define amdgpu_ps void @test_if_export_done_f32(i32 %flag, float %x, float %y, fl
; GFX1150-NEXT: s_setprio 2
; GFX1150-NEXT: s_mov_b32 s0, exec_lo
; GFX1150-NEXT: v_cmpx_ne_u32_e32 0, v0
+; GFX1150-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1150-NEXT: s_cbranch_execz .LBB11_2
; GFX1150-NEXT: ; %bb.1: ; %exp
; GFX1150-NEXT: exp mrt0, v1, v2, v3, v4 done
@@ -290,6 +296,7 @@ define amdgpu_ps void @test_if_export_vm_done_f32(i32 %flag, float %x, float %y,
; GFX11: ; %bb.0:
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: v_cmpx_ne_u32_e32 0, v0
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_cbranch_execz .LBB12_2
; GFX11-NEXT: ; %bb.1: ; %exp
; GFX11-NEXT: exp mrt0, v1, v2, v3, v4 done
@@ -301,6 +308,7 @@ define amdgpu_ps void @test_if_export_vm_done_f32(i32 %flag, float %x, float %y,
; GFX1150-NEXT: s_setprio 2
; GFX1150-NEXT: s_mov_b32 s0, exec_lo
; GFX1150-NEXT: v_cmpx_ne_u32_e32 0, v0
+; GFX1150-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1150-NEXT: s_cbranch_execz .LBB12_2
; GFX1150-NEXT: ; %bb.1: ; %exp
; GFX1150-NEXT: exp mrt0, v1, v2, v3, v4 done
diff --git a/llvm/test/CodeGen/AMDGPU/s-barrier-signal-var-gep.ll b/llvm/test/CodeGen/AMDGPU/s-barrier-signal-var-gep.ll
index d248a0a8007aca..552364e3f00f25 100644
--- a/llvm/test/CodeGen/AMDGPU/s-barrier-signal-var-gep.ll
+++ b/llvm/test/CodeGen/AMDGPU/s-barrier-signal-var-gep.ll
@@ -20,6 +20,7 @@ define amdgpu_kernel void @signal_var_bar0() {
; CHECK-NEXT: v_nop
; CHECK-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; CHECK-NEXT: s_mov_b32 m0, 0x100001
+; CHECK-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; CHECK-NEXT: s_barrier_init m0
; CHECK-NEXT: s_barrier_signal 1
; CHECK-NEXT: s_barrier_wait 0
@@ -35,6 +36,7 @@ define amdgpu_kernel void @signal_var_bar0() {
; CHECK-OBJ-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; CHECK-OBJ-SDAG-NEXT: s_bfe_u32 s0, s0, 0x60004
; CHECK-OBJ-SDAG-NEXT: s_or_b32 m0, s0, 0x100000
+; CHECK-OBJ-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; CHECK-OBJ-SDAG-NEXT: s_barrier_init m0
; CHECK-OBJ-SDAG-NEXT: s_mov_b32 m0, s0
; CHECK-OBJ-SDAG-NEXT: s_barrier_signal m0
@@ -51,6 +53,7 @@ define amdgpu_kernel void @signal_var_bar0() {
; CHECK-OBJ-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; CHECK-OBJ-GISEL-NEXT: s_and_b32 s0, s0, 63
; CHECK-OBJ-GISEL-NEXT: s_or_b32 m0, s0, 0x100000
+; CHECK-OBJ-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; CHECK-OBJ-GISEL-NEXT: s_barrier_init m0
; CHECK-OBJ-GISEL-NEXT: s_mov_b32 m0, s0
; CHECK-OBJ-GISEL-NEXT: s_barrier_signal m0
@@ -70,6 +73,7 @@ define amdgpu_kernel void @signal_var_bar1() {
; CHECK-NEXT: v_nop
; CHECK-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; CHECK-NEXT: s_mov_b32 m0, 0x100002
+; CHECK-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; CHECK-NEXT: s_barrier_init m0
; CHECK-NEXT: s_barrier_signal 2
; CHECK-NEXT: s_barrier_wait 1
@@ -85,6 +89,7 @@ define amdgpu_kernel void @signal_var_bar1() {
; CHECK-OBJ-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; CHECK-OBJ-SDAG-NEXT: s_bfe_u32 s0, s0, 0x60004
; CHECK-OBJ-SDAG-NEXT: s_or_b32 m0, s0, 0x100000
+; CHECK-OBJ-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; CHECK-OBJ-SDAG-NEXT: s_barrier_init m0
; CHECK-OBJ-SDAG-NEXT: s_mov_b32 m0, s0
; CHECK-OBJ-SDAG-NEXT: s_barrier_signal m0
@@ -101,10 +106,11 @@ define amdgpu_kernel void @signal_var_bar1() {
; CHECK-OBJ-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; CHECK-OBJ-GISEL-NEXT: s_lshr_b32 s0, s0, 4
; CHECK-OBJ-GISEL-NEXT: s_and_b32 s0, s0, 63
-; CHECK-OBJ-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; CHECK-OBJ-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; CHECK-OBJ-GISEL-NEXT: s_or_b32 m0, s0, 0x100000
; CHECK-OBJ-GISEL-NEXT: s_barrier_init m0
; CHECK-OBJ-GISEL-NEXT: s_mov_b32 m0, s0
+; CHECK-OBJ-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; CHECK-OBJ-GISEL-NEXT: s_barrier_signal m0
; CHECK-OBJ-GISEL-NEXT: s_barrier_wait 1
; CHECK-OBJ-GISEL-NEXT: s_endpgm
@@ -127,6 +133,7 @@ define amdgpu_kernel void @signal_var_misaligned() {
; CHECK-NEXT: v_nop
; CHECK-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; CHECK-NEXT: s_mov_b32 m0, 0x100001
+; CHECK-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; CHECK-NEXT: s_barrier_init m0
; CHECK-NEXT: s_barrier_signal 1
; CHECK-NEXT: s_barrier_wait 1
@@ -142,6 +149,7 @@ define amdgpu_kernel void @signal_var_misaligned() {
; CHECK-OBJ-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; CHECK-OBJ-SDAG-NEXT: s_bfe_u32 s0, s0, 0x60004
; CHECK-OBJ-SDAG-NEXT: s_or_b32 m0, s0, 0x100000
+; CHECK-OBJ-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; CHECK-OBJ-SDAG-NEXT: s_barrier_init m0
; CHECK-OBJ-SDAG-NEXT: s_mov_b32 m0, s0
; CHECK-OBJ-SDAG-NEXT: s_barrier_signal m0
@@ -158,10 +166,11 @@ define amdgpu_kernel void @signal_var_misaligned() {
; CHECK-OBJ-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; CHECK-OBJ-GISEL-NEXT: s_lshr_b32 s0, s0, 4
; CHECK-OBJ-GISEL-NEXT: s_and_b32 s0, s0, 63
-; CHECK-OBJ-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; CHECK-OBJ-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; CHECK-OBJ-GISEL-NEXT: s_or_b32 m0, s0, 0x100000
; CHECK-OBJ-GISEL-NEXT: s_barrier_init m0
; CHECK-OBJ-GISEL-NEXT: s_mov_b32 m0, s0
+; CHECK-OBJ-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; CHECK-OBJ-GISEL-NEXT: s_barrier_signal m0
; CHECK-OBJ-GISEL-NEXT: s_barrier_wait 1
; CHECK-OBJ-GISEL-NEXT: s_endpgm
@@ -189,6 +198,7 @@ define amdgpu_kernel void @signal_var_dynamic(i32 %idx) {
; CHECK-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; CHECK-SDAG-NEXT: s_bfe_u32 s0, s0, 0x60004
; CHECK-SDAG-NEXT: s_or_b32 m0, s0, 0x100000
+; CHECK-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; CHECK-SDAG-NEXT: s_barrier_init m0
; CHECK-SDAG-NEXT: s_mov_b32 m0, s0
; CHECK-SDAG-NEXT: s_barrier_signal m0
@@ -210,6 +220,7 @@ define amdgpu_kernel void @signal_var_dynamic(i32 %idx) {
; CHECK-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; CHECK-GISEL-NEXT: s_and_b32 s0, s0, 63
; CHECK-GISEL-NEXT: s_or_b32 m0, s0, 0x100000
+; CHECK-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; CHECK-GISEL-NEXT: s_barrier_init m0
; CHECK-GISEL-NEXT: s_mov_b32 m0, s0
; CHECK-GISEL-NEXT: s_barrier_signal m0
@@ -228,6 +239,7 @@ define amdgpu_kernel void @signal_var_dynamic(i32 %idx) {
; CHECK-OBJ-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; CHECK-OBJ-SDAG-NEXT: s_bfe_u32 s0, s0, 0x60004
; CHECK-OBJ-SDAG-NEXT: s_or_b32 m0, s0, 0x100000
+; CHECK-OBJ-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; CHECK-OBJ-SDAG-NEXT: s_barrier_init m0
; CHECK-OBJ-SDAG-NEXT: s_mov_b32 m0, s0
; CHECK-OBJ-SDAG-NEXT: s_barrier_signal m0
@@ -249,6 +261,7 @@ define amdgpu_kernel void @signal_var_dynamic(i32 %idx) {
; CHECK-OBJ-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; CHECK-OBJ-GISEL-NEXT: s_and_b32 s0, s0, 63
; CHECK-OBJ-GISEL-NEXT: s_or_b32 m0, s0, 0x100000
+; CHECK-OBJ-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; CHECK-OBJ-GISEL-NEXT: s_barrier_init m0
; CHECK-OBJ-GISEL-NEXT: s_mov_b32 m0, s0
; CHECK-OBJ-GISEL-NEXT: s_barrier_signal m0
diff --git a/llvm/test/CodeGen/AMDGPU/s-barrier.ll b/llvm/test/CodeGen/AMDGPU/s-barrier.ll
index ed07b9af504877..0e9257ffc1d04b 100644
--- a/llvm/test/CodeGen/AMDGPU/s-barrier.ll
+++ b/llvm/test/CodeGen/AMDGPU/s-barrier.ll
@@ -15,6 +15,7 @@ define void @func1() {
; GFX12-SDAG-NEXT: s_wait_bvhcnt 0x0
; GFX12-SDAG-NEXT: s_wait_kmcnt 0x0
; GFX12-SDAG-NEXT: s_mov_b32 m0, 3
+; GFX12-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX12-SDAG-NEXT: s_barrier_join m0
; GFX12-SDAG-NEXT: s_mov_b32 m0, 0x70003
; GFX12-SDAG-NEXT: s_barrier_signal m0
@@ -48,6 +49,7 @@ define void @func2() {
; GFX12-SDAG-NEXT: s_wait_bvhcnt 0x0
; GFX12-SDAG-NEXT: s_wait_kmcnt 0x0
; GFX12-SDAG-NEXT: s_mov_b32 m0, 2
+; GFX12-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX12-SDAG-NEXT: s_barrier_join m0
; GFX12-SDAG-NEXT: s_mov_b32 m0, 0x70002
; GFX12-SDAG-NEXT: s_barrier_signal m0
@@ -86,12 +88,13 @@ define amdgpu_kernel void @kernel1(ptr addrspace(1) %out, ptr addrspace(3) %in)
; GFX12-SDAG-NEXT: s_mov_b32 s32, 0
; GFX12-SDAG-NEXT: s_wait_kmcnt 0x0
; GFX12-SDAG-NEXT: s_bfe_u32 s2, s2, 0x60004
-; GFX12-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX12-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
; GFX12-SDAG-NEXT: s_or_b32 s3, s2, 0x90000
; GFX12-SDAG-NEXT: s_cmp_eq_u32 0, 0
; GFX12-SDAG-NEXT: s_mov_b32 m0, s3
; GFX12-SDAG-NEXT: s_barrier_init m0
; GFX12-SDAG-NEXT: s_mov_b32 m0, 0xc0001
+; GFX12-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX12-SDAG-NEXT: s_barrier_signal m0
; GFX12-SDAG-NEXT: s_mov_b32 m0, s3
; GFX12-SDAG-NEXT: s_barrier_signal m0
@@ -105,6 +108,7 @@ define amdgpu_kernel void @kernel1(ptr addrspace(1) %out, ptr addrspace(3) %in)
; GFX12-SDAG-NEXT: s_barrier_leave
; GFX12-SDAG-NEXT: s_get_barrier_state s3, m0
; GFX12-SDAG-NEXT: s_mov_b32 m0, s2
+; GFX12-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-SDAG-NEXT: s_get_barrier_state s2, m0
; GFX12-SDAG-NEXT: s_wait_kmcnt 0x0
; GFX12-SDAG-NEXT: s_getpc_b64 s[2:3]
@@ -146,10 +150,12 @@ define amdgpu_kernel void @kernel1(ptr addrspace(1) %out, ptr addrspace(3) %in)
; GFX12-GISEL-NEXT: s_or_b32 s1, s0, 0x90000
; GFX12-GISEL-NEXT: s_cmp_eq_u32 0, 0
; GFX12-GISEL-NEXT: s_mov_b32 m0, s1
+; GFX12-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX12-GISEL-NEXT: s_barrier_init m0
; GFX12-GISEL-NEXT: s_mov_b32 m0, 0xc0001
; GFX12-GISEL-NEXT: s_barrier_signal m0
; GFX12-GISEL-NEXT: s_mov_b32 m0, s1
+; GFX12-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-GISEL-NEXT: s_barrier_signal m0
; GFX12-GISEL-NEXT: s_mov_b32 m0, s0
; GFX12-GISEL-NEXT: s_barrier_signal -1
@@ -280,6 +286,7 @@ define void @signal_var_cnt0_dynamic_bar(ptr addrspace(3) inreg %bar) {
; GFX12-SDAG-NEXT: s_wait_bvhcnt 0x0
; GFX12-SDAG-NEXT: s_wait_kmcnt 0x0
; GFX12-SDAG-NEXT: s_bfe_u32 m0, s0, 0x60004
+; GFX12-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-SDAG-NEXT: s_barrier_signal m0
; GFX12-SDAG-NEXT: s_setpc_b64 s[30:31]
;
@@ -293,6 +300,7 @@ define void @signal_var_cnt0_dynamic_bar(ptr addrspace(3) inreg %bar) {
; GFX12-GISEL-NEXT: s_lshr_b32 s0, s0, 4
; GFX12-GISEL-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-GISEL-NEXT: s_and_b32 m0, s0, 63
+; GFX12-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-GISEL-NEXT: s_barrier_signal m0
; GFX12-GISEL-NEXT: s_setpc_b64 s[30:31]
call void @llvm.amdgcn.s.barrier.signal.var(ptr addrspace(3) %bar, i32 0)
@@ -313,6 +321,7 @@ define void @barrier_init_dynamic_cnt(ptr addrspace(3) inreg %bar, i32 inreg %cn
; GFX12-SDAG-NEXT: s_lshl_b32 s1, s1, 16
; GFX12-SDAG-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-SDAG-NEXT: s_or_b32 m0, s1, s0
+; GFX12-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-SDAG-NEXT: s_barrier_init m0
; GFX12-SDAG-NEXT: s_setpc_b64 s[30:31]
;
@@ -330,6 +339,7 @@ define void @barrier_init_dynamic_cnt(ptr addrspace(3) inreg %bar, i32 inreg %cn
; GFX12-GISEL-NEXT: s_lshl_b32 s1, s1, 16
; GFX12-GISEL-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-GISEL-NEXT: s_or_b32 m0, s0, s1
+; GFX12-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-GISEL-NEXT: s_barrier_init m0
; GFX12-GISEL-NEXT: s_setpc_b64 s[30:31]
call void @llvm.amdgcn.s.barrier.init(ptr addrspace(3) %bar, i32 %cnt)
@@ -350,6 +360,7 @@ define void @signal_var_dynamic_cnt(ptr addrspace(3) inreg %bar, i32 inreg %cnt)
; GFX12-SDAG-NEXT: s_lshl_b32 s1, s1, 16
; GFX12-SDAG-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-SDAG-NEXT: s_or_b32 m0, s1, s0
+; GFX12-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-SDAG-NEXT: s_barrier_signal m0
; GFX12-SDAG-NEXT: s_setpc_b64 s[30:31]
;
@@ -367,6 +378,7 @@ define void @signal_var_dynamic_cnt(ptr addrspace(3) inreg %bar, i32 inreg %cnt)
; GFX12-GISEL-NEXT: s_lshl_b32 s1, s1, 16
; GFX12-GISEL-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-GISEL-NEXT: s_or_b32 m0, s0, s1
+; GFX12-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-GISEL-NEXT: s_barrier_signal m0
; GFX12-GISEL-NEXT: s_setpc_b64 s[30:31]
call void @llvm.amdgcn.s.barrier.signal.var(ptr addrspace(3) %bar, i32 %cnt)
diff --git a/llvm/test/CodeGen/AMDGPU/s-cluster-barrier.ll b/llvm/test/CodeGen/AMDGPU/s-cluster-barrier.ll
index 27b613606a9243..c3ee8b1eca1514 100644
--- a/llvm/test/CodeGen/AMDGPU/s-cluster-barrier.ll
+++ b/llvm/test/CodeGen/AMDGPU/s-cluster-barrier.ll
@@ -10,10 +10,11 @@ define amdgpu_kernel void @kernel1() #0 {
; GFX12-NEXT: v_nop
; GFX12-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX12-NEXT: s_cmp_eq_u32 0, 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_2)
; GFX12-NEXT: s_barrier_signal_isfirst -1
; GFX12-NEXT: s_barrier_wait -1
; GFX12-NEXT: s_cselect_b32 s0, 1, 0
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cmp_lg_u32 s0, 1
; GFX12-NEXT: s_cbranch_scc1 .LBB0_2
; GFX12-NEXT: ; %bb.1:
diff --git a/llvm/test/CodeGen/AMDGPU/s-wakeup-barrier.ll b/llvm/test/CodeGen/AMDGPU/s-wakeup-barrier.ll
index d9e8a954ddffdc..bfd559d3b6445f 100644
--- a/llvm/test/CodeGen/AMDGPU/s-wakeup-barrier.ll
+++ b/llvm/test/CodeGen/AMDGPU/s-wakeup-barrier.ll
@@ -18,6 +18,7 @@ define amdgpu_kernel void @kernel1(ptr addrspace(1) %out, ptr addrspace(3) %in)
; GFX1250-SDAG-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-SDAG-NEXT: s_load_b32 s0, s[4:5], 0x2c nv
; GFX1250-SDAG-NEXT: s_mov_b32 m0, 1
+; GFX1250-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
; GFX1250-SDAG-NEXT: s_wakeup_barrier m0
; GFX1250-SDAG-NEXT: s_wait_kmcnt 0x0
; GFX1250-SDAG-NEXT: s_bfe_u32 m0, s0, 0x60004
@@ -34,7 +35,7 @@ define amdgpu_kernel void @kernel1(ptr addrspace(1) %out, ptr addrspace(3) %in)
; GFX1250-GISEL-NEXT: s_wakeup_barrier 1
; GFX1250-GISEL-NEXT: s_wait_kmcnt 0x0
; GFX1250-GISEL-NEXT: s_lshr_b32 s0, s0, 4
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1250-GISEL-NEXT: s_and_b32 m0, s0, 63
; GFX1250-GISEL-NEXT: s_wakeup_barrier m0
; GFX1250-GISEL-NEXT: s_endpgm
diff --git a/llvm/test/CodeGen/AMDGPU/sad.ll b/llvm/test/CodeGen/AMDGPU/sad.ll
index c73e60022b6231..91cda87f220ed0 100644
--- a/llvm/test/CodeGen/AMDGPU/sad.ll
+++ b/llvm/test/CodeGen/AMDGPU/sad.ll
@@ -229,9 +229,10 @@ define amdgpu_kernel void @v_sad_u32_pat2(ptr addrspace(1) %out, i32 %a, i32 %b,
; GFX12-5-GISEL-NEXT: s_cselect_b32 s4, 1, 0
; GFX12-5-GISEL-NEXT: s_sub_co_i32 s0, s0, s1
; GFX12-5-GISEL-NEXT: s_cmp_lg_u32 s4, 0
-; GFX12-5-GISEL-NEXT: s_cselect_b32 s0, s0, s3
; GFX12-5-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
+; GFX12-5-GISEL-NEXT: s_cselect_b32 s0, s0, s3
; GFX12-5-GISEL-NEXT: s_add_co_i32 s0, s0, s2
+; GFX12-5-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-5-GISEL-NEXT: v_mov_b32_e32 v0, s0
; GFX12-5-GISEL-NEXT: global_store_b32 v1, v0, s[6:7]
; GFX12-5-GISEL-NEXT: s_endpgm
@@ -758,9 +759,10 @@ define amdgpu_kernel void @v_sad_u32_multi_use_sub_pat2(ptr addrspace(1) %out, i
; GFX12-5-GISEL-NEXT: s_cselect_b32 s4, 1, 0
; GFX12-5-GISEL-NEXT: s_sub_co_i32 s0, s0, s1
; GFX12-5-GISEL-NEXT: s_cmp_lg_u32 s4, 0
-; GFX12-5-GISEL-NEXT: s_cselect_b32 s1, s0, s3
; GFX12-5-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
+; GFX12-5-GISEL-NEXT: s_cselect_b32 s1, s0, s3
; GFX12-5-GISEL-NEXT: s_add_co_i32 s1, s1, s2
+; GFX12-5-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-5-GISEL-NEXT: v_dual_mov_b32 v0, s0 :: v_dual_mov_b32 v1, s1
; GFX12-5-GISEL-NEXT: scratch_store_b32 off, v0, s0 scope:SCOPE_SYS
; GFX12-5-GISEL-NEXT: s_wait_storecnt 0x0
@@ -867,9 +869,10 @@ define amdgpu_kernel void @v_sad_u32_multi_use_select_pat2(ptr addrspace(1) %out
; GFX12-5-GISEL-NEXT: s_cselect_b32 s4, 1, 0
; GFX12-5-GISEL-NEXT: s_sub_co_i32 s0, s0, s1
; GFX12-5-GISEL-NEXT: s_cmp_lg_u32 s4, 0
-; GFX12-5-GISEL-NEXT: s_cselect_b32 s0, s0, s3
; GFX12-5-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
+; GFX12-5-GISEL-NEXT: s_cselect_b32 s0, s0, s3
; GFX12-5-GISEL-NEXT: s_add_co_i32 s1, s0, s2
+; GFX12-5-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-5-GISEL-NEXT: v_dual_mov_b32 v0, s0 :: v_dual_mov_b32 v1, s1
; GFX12-5-GISEL-NEXT: scratch_store_b32 off, v0, s0 scope:SCOPE_SYS
; GFX12-5-GISEL-NEXT: s_wait_storecnt 0x0
@@ -1133,10 +1136,12 @@ define amdgpu_kernel void @v_sad_u32_vector_pat2(ptr addrspace(1) %out, <4 x i32
; GFX12-5-GISEL-NEXT: v_mov_b32_e32 v4, 0
; GFX12-5-GISEL-NEXT: s_wait_kmcnt 0x0
; GFX12-5-GISEL-NEXT: s_cmp_gt_u32 s8, s12
+; GFX12-5-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX12-5-GISEL-NEXT: s_cselect_b32 s16, 1, 0
; GFX12-5-GISEL-NEXT: s_cmp_gt_u32 s9, s13
; GFX12-5-GISEL-NEXT: s_cselect_b32 s17, 1, 0
; GFX12-5-GISEL-NEXT: s_cmp_gt_u32 s10, s14
+; GFX12-5-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX12-5-GISEL-NEXT: s_cselect_b32 s18, 1, 0
; GFX12-5-GISEL-NEXT: s_cmp_gt_u32 s11, s15
; GFX12-5-GISEL-NEXT: s_cselect_b32 s19, 1, 0
@@ -1149,10 +1154,12 @@ define amdgpu_kernel void @v_sad_u32_vector_pat2(ptr addrspace(1) %out, <4 x i32
; GFX12-5-GISEL-NEXT: s_sub_co_i32 s10, s14, s10
; GFX12-5-GISEL-NEXT: s_sub_co_i32 s11, s15, s11
; GFX12-5-GISEL-NEXT: s_cmp_lg_u32 s16, 0
+; GFX12-5-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX12-5-GISEL-NEXT: s_cselect_b32 s4, s4, s8
; GFX12-5-GISEL-NEXT: s_cmp_lg_u32 s17, 0
; GFX12-5-GISEL-NEXT: s_cselect_b32 s5, s5, s9
; GFX12-5-GISEL-NEXT: s_cmp_lg_u32 s18, 0
+; GFX12-5-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX12-5-GISEL-NEXT: s_cselect_b32 s8, s20, s10
; GFX12-5-GISEL-NEXT: s_cmp_lg_u32 s19, 0
; GFX12-5-GISEL-NEXT: s_cselect_b32 s9, s21, s11
@@ -1354,16 +1361,16 @@ define amdgpu_kernel void @v_sad_u32_i16_pat2(ptr addrspace(1) %out) {
; GFX12-5-GISEL-NEXT: v_readfirstlane_b32 s3, v1
; GFX12-5-GISEL-NEXT: v_mov_b32_e32 v1, 0
; GFX12-5-GISEL-NEXT: s_and_b32 s5, 0xffff, s3
-; GFX12-5-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX12-5-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-5-GISEL-NEXT: s_cmp_gt_u32 s4, s5
; GFX12-5-GISEL-NEXT: s_cselect_b32 s4, 1, 0
; GFX12-5-GISEL-NEXT: s_sub_co_i32 s6, s2, s3
; GFX12-5-GISEL-NEXT: s_sub_co_i32 s2, s3, s2
; GFX12-5-GISEL-NEXT: s_cmp_lg_u32 s4, 0
+; GFX12-5-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
; GFX12-5-GISEL-NEXT: s_cselect_b32 s2, s6, s2
; GFX12-5-GISEL-NEXT: v_readfirstlane_b32 s5, v2
; GFX12-5-GISEL-NEXT: s_add_co_i32 s2, s2, s5
-; GFX12-5-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-5-GISEL-NEXT: v_mov_b16_e32 v0.l, s2
; GFX12-5-GISEL-NEXT: s_wait_kmcnt 0x0
; GFX12-5-GISEL-NEXT: global_store_b16 v1, v0, s[0:1]
@@ -1562,6 +1569,7 @@ define amdgpu_kernel void @v_sad_u32_i8_pat2(ptr addrspace(1) %out) {
; GFX12-5-GISEL-NEXT: v_readfirstlane_b32 s3, v1
; GFX12-5-GISEL-NEXT: v_mov_b32_e32 v1, 0
; GFX12-5-GISEL-NEXT: s_cmp_gt_u32 s2, s3
+; GFX12-5-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_3) | instid1(SALU_CYCLE_1)
; GFX12-5-GISEL-NEXT: s_cselect_b32 s4, 1, 0
; GFX12-5-GISEL-NEXT: s_sub_co_i32 s6, s2, s3
; GFX12-5-GISEL-NEXT: s_sub_co_i32 s2, s3, s2
@@ -1661,16 +1669,17 @@ define amdgpu_kernel void @s_sad_u32_i8_pat2(ptr addrspace(1) %out, i8 zeroext %
; GFX12-5-GISEL-NEXT: s_lshr_b32 s3, s2, 8
; GFX12-5-GISEL-NEXT: s_and_b32 s4, s2, 0xff
; GFX12-5-GISEL-NEXT: s_and_b32 s5, s3, 0xff
-; GFX12-5-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX12-5-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-5-GISEL-NEXT: s_cmp_gt_u32 s4, s5
; GFX12-5-GISEL-NEXT: s_cselect_b32 s4, 1, 0
; GFX12-5-GISEL-NEXT: s_sub_co_i32 s5, s2, s3
; GFX12-5-GISEL-NEXT: s_sub_co_i32 s3, s3, s2
; GFX12-5-GISEL-NEXT: s_cmp_lg_u32 s4, 0
+; GFX12-5-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX12-5-GISEL-NEXT: s_cselect_b32 s3, s5, s3
; GFX12-5-GISEL-NEXT: s_lshr_b32 s2, s2, 16
-; GFX12-5-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-5-GISEL-NEXT: s_add_co_i32 s2, s3, s2
+; GFX12-5-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-5-GISEL-NEXT: v_mov_b32_e32 v0, s2
; GFX12-5-GISEL-NEXT: global_store_b8 v1, v0, s[0:1]
; GFX12-5-GISEL-NEXT: s_endpgm
@@ -1730,11 +1739,11 @@ define amdgpu_kernel void @v_sad_u32_mismatched_operands_pat1(ptr addrspace(1) %
; GFX12-5-SDAG-NEXT: s_wait_kmcnt 0x0
; GFX12-5-SDAG-NEXT: s_max_u32 s4, s0, s1
; GFX12-5-SDAG-NEXT: s_cmp_le_u32 s0, s1
-; GFX12-5-SDAG-NEXT: s_cselect_b32 s0, s0, s3
; GFX12-5-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
+; GFX12-5-SDAG-NEXT: s_cselect_b32 s0, s0, s3
; GFX12-5-SDAG-NEXT: s_sub_co_i32 s0, s4, s0
+; GFX12-5-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-5-SDAG-NEXT: s_add_co_i32 s0, s0, s2
-; GFX12-5-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-5-SDAG-NEXT: v_dual_mov_b32 v0, 0 :: v_dual_mov_b32 v1, s0
; GFX12-5-SDAG-NEXT: global_store_b32 v0, v1, s[6:7]
; GFX12-5-SDAG-NEXT: s_endpgm
@@ -1752,11 +1761,11 @@ define amdgpu_kernel void @v_sad_u32_mismatched_operands_pat1(ptr addrspace(1) %
; GFX12-5-GISEL-NEXT: s_wait_kmcnt 0x0
; GFX12-5-GISEL-NEXT: s_max_u32 s4, s0, s1
; GFX12-5-GISEL-NEXT: s_cmp_le_u32 s0, s1
-; GFX12-5-GISEL-NEXT: s_cselect_b32 s0, s0, s3
; GFX12-5-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
+; GFX12-5-GISEL-NEXT: s_cselect_b32 s0, s0, s3
; GFX12-5-GISEL-NEXT: s_sub_co_i32 s0, s4, s0
+; GFX12-5-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-5-GISEL-NEXT: s_add_co_i32 s0, s0, s2
-; GFX12-5-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-5-GISEL-NEXT: v_mov_b32_e32 v0, s0
; GFX12-5-GISEL-NEXT: global_store_b32 v1, v0, s[6:7]
; GFX12-5-GISEL-NEXT: s_endpgm
@@ -1840,9 +1849,10 @@ define amdgpu_kernel void @v_sad_u32_mismatched_operands_pat2(ptr addrspace(1) %
; GFX12-5-GISEL-NEXT: s_cselect_b32 s4, 1, 0
; GFX12-5-GISEL-NEXT: s_sub_co_i32 s0, s0, s3
; GFX12-5-GISEL-NEXT: s_cmp_lg_u32 s4, 0
-; GFX12-5-GISEL-NEXT: s_cselect_b32 s0, s0, s1
; GFX12-5-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
+; GFX12-5-GISEL-NEXT: s_cselect_b32 s0, s0, s1
; GFX12-5-GISEL-NEXT: s_add_co_i32 s0, s0, s2
+; GFX12-5-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-5-GISEL-NEXT: v_mov_b32_e32 v0, s0
; GFX12-5-GISEL-NEXT: global_store_b32 v1, v0, s[6:7]
; GFX12-5-GISEL-NEXT: s_endpgm
diff --git a/llvm/test/CodeGen/AMDGPU/saddo.ll b/llvm/test/CodeGen/AMDGPU/saddo.ll
index 9d8869c765e295..6649cfa6b442d2 100644
--- a/llvm/test/CodeGen/AMDGPU/saddo.ll
+++ b/llvm/test/CodeGen/AMDGPU/saddo.ll
@@ -604,14 +604,14 @@ define amdgpu_kernel void @v_saddo_i64(ptr addrspace(1) %out, ptr addrspace(1) %
; GFX11-NEXT: global_load_b64 v[2:3], v6, s[10:11]
; GFX11-NEXT: s_waitcnt vmcnt(0)
; GFX11-NEXT: v_add_co_u32 v4, vcc_lo, v0, v2
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX11-NEXT: v_add_co_ci_u32_e64 v5, null, v1, v3, vcc_lo
; GFX11-NEXT: v_cmp_gt_i32_e64 s0, 0, v3
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: v_cmp_lt_i64_e32 vcc_lo, v[4:5], v[0:1]
; GFX11-NEXT: s_xor_b32 s0, s0, vcc_lo
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-NEXT: s_cselect_b32 s0, 1, 0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: v_mov_b32_e32 v0, s0
; GFX11-NEXT: s_clause 0x1
; GFX11-NEXT: global_store_b64 v6, v[4:5], s[4:5]
diff --git a/llvm/test/CodeGen/AMDGPU/scale-offset-global.ll b/llvm/test/CodeGen/AMDGPU/scale-offset-global.ll
index d0b52dc945d3d6..32ed7e869c18c1 100644
--- a/llvm/test/CodeGen/AMDGPU/scale-offset-global.ll
+++ b/llvm/test/CodeGen/AMDGPU/scale-offset-global.ll
@@ -74,7 +74,7 @@ define amdgpu_ps float @global_load_b32_idxprom_wrong_stride(ptr addrspace(1) al
; GFX13-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX13-SDAG-NEXT: v_lshlrev_b64_e32 v[0:1], 3, v[0:1]
; GFX13-SDAG-NEXT: v_add_co_u32 v0, vcc_lo, s0, v0
-; GFX13-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX13-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX13-SDAG-NEXT: v_add_co_ci_u32_e64 v1, null, s1, v1, vcc_lo
; GFX13-SDAG-NEXT: global_load_b32 v0, v[0:1], off
; GFX13-SDAG-NEXT: s_wait_loadcnt 0x0
@@ -87,7 +87,7 @@ define amdgpu_ps float @global_load_b32_idxprom_wrong_stride(ptr addrspace(1) al
; GFX13-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX13-GISEL-NEXT: v_lshlrev_b64_e32 v[0:1], 3, v[0:1]
; GFX13-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, v2, v0
-; GFX13-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX13-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX13-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, v3, v1, vcc_lo
; GFX13-GISEL-NEXT: global_load_b32 v0, v[0:1], off
; GFX13-GISEL-NEXT: s_wait_loadcnt 0x0
diff --git a/llvm/test/CodeGen/AMDGPU/scmp.ll b/llvm/test/CodeGen/AMDGPU/scmp.ll
index 6ad3fee5344ccc..6ca5283f7f70e5 100644
--- a/llvm/test/CodeGen/AMDGPU/scmp.ll
+++ b/llvm/test/CodeGen/AMDGPU/scmp.ll
@@ -93,19 +93,17 @@ define i32 @scmp_i128(i128 %a, i128 %b) {
; GFX12-GISEL-NEXT: v_cndmask_b32_e64 v8, 0, 1, vcc_lo
; GFX12-GISEL-NEXT: v_cmp_gt_i64_e32 vcc_lo, v[2:3], v[6:7]
; GFX12-GISEL-NEXT: s_wait_alu depctr_va_sdst(0)
-; GFX12-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX12-GISEL-NEXT: v_cndmask_b32_e64 v0, 0, 1, s0
; GFX12-GISEL-NEXT: v_cmp_lt_i64_e64 s0, v[2:3], v[6:7]
; GFX12-GISEL-NEXT: s_wait_alu depctr_va_vcc(0)
; GFX12-GISEL-NEXT: v_cndmask_b32_e64 v9, 0, 1, vcc_lo
; GFX12-GISEL-NEXT: v_cmp_eq_u64_e32 vcc_lo, v[2:3], v[6:7]
; GFX12-GISEL-NEXT: s_wait_alu depctr_va_sdst(0)
-; GFX12-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_2) | instid1(VALU_DEP_2)
; GFX12-GISEL-NEXT: v_cndmask_b32_e64 v1, 0, 1, s0
; GFX12-GISEL-NEXT: s_wait_alu depctr_va_vcc(0)
; GFX12-GISEL-NEXT: v_cndmask_b32_e32 v2, v9, v8, vcc_lo
+; GFX12-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX12-GISEL-NEXT: v_cndmask_b32_e32 v0, v1, v0, vcc_lo
-; GFX12-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-GISEL-NEXT: v_sub_nc_u32_e32 v0, v2, v0
; GFX12-GISEL-NEXT: s_setpc_b64 s[30:31]
%r = call i32 @llvm.scmp.i32.i128(i128 %a, i128 %b)
@@ -704,19 +702,23 @@ define i32 @scmp_i128_uniform(i128 inreg %a, i128 inreg %b) {
; GFX12-GISEL-NEXT: v_cmp_lt_u64_e64 s0, s[0:1], s[16:17]
; GFX12-GISEL-NEXT: v_cmp_lt_i64_e64 s6, s[2:3], s[18:19]
; GFX12-GISEL-NEXT: s_cmp_lg_u32 s4, 0
+; GFX12-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX12-GISEL-NEXT: s_cselect_b32 s4, 1, 0
; GFX12-GISEL-NEXT: s_cmp_lg_u32 s5, 0
; GFX12-GISEL-NEXT: s_cselect_b32 s5, 1, 0
; GFX12-GISEL-NEXT: s_cmp_eq_u64 s[2:3], s[18:19]
+; GFX12-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX12-GISEL-NEXT: s_cselect_b32 s1, 1, 0
; GFX12-GISEL-NEXT: s_cmp_lg_u32 s0, 0
; GFX12-GISEL-NEXT: s_cselect_b32 s0, 1, 0
; GFX12-GISEL-NEXT: s_cmp_lg_u32 s6, 0
+; GFX12-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX12-GISEL-NEXT: s_cselect_b32 s6, 1, 0
; GFX12-GISEL-NEXT: s_cmp_eq_u64 s[2:3], s[18:19]
; GFX12-GISEL-NEXT: s_cselect_b32 s2, 1, 0
; GFX12-GISEL-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-GISEL-NEXT: s_cmp_lg_u32 s1, 0
+; GFX12-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX12-GISEL-NEXT: s_cselect_b32 s1, s4, s5
; GFX12-GISEL-NEXT: s_cmp_lg_u32 s2, 0
; GFX12-GISEL-NEXT: s_cselect_b32 s0, s0, s6
@@ -791,11 +793,13 @@ define i32 @scmp_i64_uniform(i64 inreg %a, i64 inreg %b) {
; GFX12-GISEL-NEXT: v_cmp_gt_i64_e64 s4, s[0:1], s[2:3]
; GFX12-GISEL-NEXT: v_cmp_lt_i64_e64 s0, s[0:1], s[2:3]
; GFX12-GISEL-NEXT: s_cmp_lg_u32 s4, 0
+; GFX12-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX12-GISEL-NEXT: s_cselect_b32 s1, 1, 0
; GFX12-GISEL-NEXT: s_cmp_lg_u32 s0, 0
; GFX12-GISEL-NEXT: s_cselect_b32 s0, 1, 0
; GFX12-GISEL-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-GISEL-NEXT: s_cmp_lg_u32 s1, 0
+; GFX12-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX12-GISEL-NEXT: s_cselect_b32 s1, 1, 0
; GFX12-GISEL-NEXT: s_cmp_lg_u32 s0, 0
; GFX12-GISEL-NEXT: s_cselect_b32 s0, 1, 0
@@ -842,6 +846,7 @@ define i32 @scmp_i32_uniform(i32 inreg %a, i32 inreg %b) {
; GFX12-SDAG-NEXT: s_wait_bvhcnt 0x0
; GFX12-SDAG-NEXT: s_wait_kmcnt 0x0
; GFX12-SDAG-NEXT: s_cmp_gt_i32 s0, s1
+; GFX12-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-SDAG-NEXT: s_cselect_b32 s2, 1, 0
; GFX12-SDAG-NEXT: s_cmp_ge_i32 s0, s1
; GFX12-SDAG-NEXT: s_wait_alu depctr_sa_sdst(0)
@@ -858,11 +863,13 @@ define i32 @scmp_i32_uniform(i32 inreg %a, i32 inreg %b) {
; GFX12-GISEL-NEXT: s_wait_bvhcnt 0x0
; GFX12-GISEL-NEXT: s_wait_kmcnt 0x0
; GFX12-GISEL-NEXT: s_cmp_gt_i32 s0, s1
+; GFX12-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX12-GISEL-NEXT: s_cselect_b32 s2, 1, 0
; GFX12-GISEL-NEXT: s_cmp_lt_i32 s0, s1
; GFX12-GISEL-NEXT: s_cselect_b32 s0, 1, 0
; GFX12-GISEL-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-GISEL-NEXT: s_cmp_lg_u32 s2, 0
+; GFX12-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX12-GISEL-NEXT: s_cselect_b32 s1, 1, 0
; GFX12-GISEL-NEXT: s_cmp_lg_u32 s0, 0
; GFX12-GISEL-NEXT: s_cselect_b32 s0, 1, 0
@@ -916,6 +923,7 @@ define i32 @scmp_i16_uniform(i16 inreg %a, i16 inreg %b) {
; GFX12-SDAG-NEXT: s_sext_i32_i16 s0, s0
; GFX12-SDAG-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-SDAG-NEXT: s_cmp_gt_i32 s0, s1
+; GFX12-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-SDAG-NEXT: s_cselect_b32 s2, 1, 0
; GFX12-SDAG-NEXT: s_cmp_ge_i32 s0, s1
; GFX12-SDAG-NEXT: s_wait_alu depctr_sa_sdst(0)
@@ -935,11 +943,13 @@ define i32 @scmp_i16_uniform(i16 inreg %a, i16 inreg %b) {
; GFX12-GISEL-NEXT: s_sext_i32_i16 s1, s1
; GFX12-GISEL-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-GISEL-NEXT: s_cmp_gt_i32 s0, s1
+; GFX12-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX12-GISEL-NEXT: s_cselect_b32 s2, 1, 0
; GFX12-GISEL-NEXT: s_cmp_lt_i32 s0, s1
; GFX12-GISEL-NEXT: s_cselect_b32 s0, 1, 0
; GFX12-GISEL-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-GISEL-NEXT: s_cmp_lg_u32 s2, 0
+; GFX12-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX12-GISEL-NEXT: s_cselect_b32 s1, 1, 0
; GFX12-GISEL-NEXT: s_cmp_lg_u32 s0, 0
; GFX12-GISEL-NEXT: s_cselect_b32 s0, 1, 0
@@ -996,6 +1006,7 @@ define i32 @scmp_i8_uniform(i8 inreg %a, i8 inreg %b) {
; GFX12-SDAG-NEXT: s_sext_i32_i16 s0, s0
; GFX12-SDAG-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-SDAG-NEXT: s_cmp_gt_i32 s0, s1
+; GFX12-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-SDAG-NEXT: s_cselect_b32 s2, 1, 0
; GFX12-SDAG-NEXT: s_cmp_ge_i32 s0, s1
; GFX12-SDAG-NEXT: s_wait_alu depctr_sa_sdst(0)
@@ -1015,11 +1026,13 @@ define i32 @scmp_i8_uniform(i8 inreg %a, i8 inreg %b) {
; GFX12-GISEL-NEXT: s_sext_i32_i8 s1, s1
; GFX12-GISEL-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-GISEL-NEXT: s_cmp_gt_i32 s0, s1
+; GFX12-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX12-GISEL-NEXT: s_cselect_b32 s2, 1, 0
; GFX12-GISEL-NEXT: s_cmp_lt_i32 s0, s1
; GFX12-GISEL-NEXT: s_cselect_b32 s0, 1, 0
; GFX12-GISEL-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-GISEL-NEXT: s_cmp_lg_u32 s2, 0
+; GFX12-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX12-GISEL-NEXT: s_cselect_b32 s1, 1, 0
; GFX12-GISEL-NEXT: s_cmp_lg_u32 s0, 0
; GFX12-GISEL-NEXT: s_cselect_b32 s0, 1, 0
@@ -1127,19 +1140,23 @@ define <2 x i32> @scmp_v2i16_uniform(<2 x i16> inreg %a, <2 x i16> inreg %b) {
; GFX12-GISEL-NEXT: s_cselect_b32 s4, 1, 0
; GFX12-GISEL-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-GISEL-NEXT: s_cmp_gt_i32 s2, s1
+; GFX12-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX12-GISEL-NEXT: s_cselect_b32 s5, 1, 0
; GFX12-GISEL-NEXT: s_cmp_lt_i32 s0, s3
; GFX12-GISEL-NEXT: s_cselect_b32 s0, 1, 0
; GFX12-GISEL-NEXT: s_cmp_lt_i32 s2, s1
+; GFX12-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX12-GISEL-NEXT: s_cselect_b32 s1, 1, 0
; GFX12-GISEL-NEXT: s_cmp_lg_u32 s4, 0
; GFX12-GISEL-NEXT: s_cselect_b32 s2, 1, 0
; GFX12-GISEL-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-GISEL-NEXT: s_cmp_lg_u32 s5, 0
+; GFX12-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX12-GISEL-NEXT: s_cselect_b32 s3, 1, 0
; GFX12-GISEL-NEXT: s_cmp_lg_u32 s0, 0
; GFX12-GISEL-NEXT: s_cselect_b32 s0, 1, 0
; GFX12-GISEL-NEXT: s_cmp_lg_u32 s1, 0
+; GFX12-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-GISEL-NEXT: s_cselect_b32 s1, 1, 0
; GFX12-GISEL-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-GISEL-NEXT: s_sub_co_i32 s0, s2, s0
@@ -1265,6 +1282,7 @@ define <4 x i32> @scmp_v4i8_uniform(<4 x i8> inreg %a, <4 x i8> inreg %b) {
; GFX12-SDAG-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-SDAG-NEXT: s_cselect_b32 s6, s8, -1
; GFX12-SDAG-NEXT: s_cmp_gt_i32 s1, s0
+; GFX12-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-SDAG-NEXT: s_cselect_b32 s7, 1, 0
; GFX12-SDAG-NEXT: s_cmp_ge_i32 s1, s0
; GFX12-SDAG-NEXT: s_sext_i32_i16 s0, s5
@@ -1281,6 +1299,7 @@ define <4 x i32> @scmp_v4i8_uniform(<4 x i8> inreg %a, <4 x i8> inreg %b) {
; GFX12-SDAG-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-SDAG-NEXT: s_cselect_b32 s3, s5, -1
; GFX12-SDAG-NEXT: s_cmp_gt_i32 s1, s0
+; GFX12-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-SDAG-NEXT: s_cselect_b32 s4, 1, 0
; GFX12-SDAG-NEXT: s_cmp_ge_i32 s1, s0
; GFX12-SDAG-NEXT: s_wait_alu depctr_sa_sdst(0)
@@ -1315,31 +1334,38 @@ define <4 x i32> @scmp_v4i8_uniform(<4 x i8> inreg %a, <4 x i8> inreg %b) {
; GFX12-GISEL-NEXT: s_cselect_b32 s9, 1, 0
; GFX12-GISEL-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-GISEL-NEXT: s_cmp_gt_i32 s3, s10
+; GFX12-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX12-GISEL-NEXT: s_cselect_b32 s11, 1, 0
; GFX12-GISEL-NEXT: s_cmp_lt_i32 s0, s4
; GFX12-GISEL-NEXT: s_cselect_b32 s0, 1, 0
; GFX12-GISEL-NEXT: s_cmp_lt_i32 s1, s6
+; GFX12-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX12-GISEL-NEXT: s_cselect_b32 s1, 1, 0
; GFX12-GISEL-NEXT: s_cmp_lt_i32 s2, s8
; GFX12-GISEL-NEXT: s_cselect_b32 s2, 1, 0
; GFX12-GISEL-NEXT: s_cmp_lt_i32 s3, s10
+; GFX12-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX12-GISEL-NEXT: s_cselect_b32 s3, 1, 0
; GFX12-GISEL-NEXT: s_cmp_lg_u32 s5, 0
; GFX12-GISEL-NEXT: s_cselect_b32 s4, 1, 0
; GFX12-GISEL-NEXT: s_cmp_lg_u32 s7, 0
+; GFX12-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX12-GISEL-NEXT: s_cselect_b32 s5, 1, 0
; GFX12-GISEL-NEXT: s_cmp_lg_u32 s9, 0
; GFX12-GISEL-NEXT: s_cselect_b32 s6, 1, 0
; GFX12-GISEL-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-GISEL-NEXT: s_cmp_lg_u32 s11, 0
+; GFX12-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX12-GISEL-NEXT: s_cselect_b32 s7, 1, 0
; GFX12-GISEL-NEXT: s_cmp_lg_u32 s0, 0
; GFX12-GISEL-NEXT: s_cselect_b32 s0, 1, 0
; GFX12-GISEL-NEXT: s_cmp_lg_u32 s1, 0
+; GFX12-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX12-GISEL-NEXT: s_cselect_b32 s1, 1, 0
; GFX12-GISEL-NEXT: s_cmp_lg_u32 s2, 0
; GFX12-GISEL-NEXT: s_cselect_b32 s2, 1, 0
; GFX12-GISEL-NEXT: s_cmp_lg_u32 s3, 0
+; GFX12-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-GISEL-NEXT: s_cselect_b32 s3, 1, 0
; GFX12-GISEL-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-GISEL-NEXT: s_sub_co_i32 s0, s4, s0
diff --git a/llvm/test/CodeGen/AMDGPU/scratch-pointer-sink.ll b/llvm/test/CodeGen/AMDGPU/scratch-pointer-sink.ll
index f842bf24613aac..da384dfa1bd750 100644
--- a/llvm/test/CodeGen/AMDGPU/scratch-pointer-sink.ll
+++ b/llvm/test/CodeGen/AMDGPU/scratch-pointer-sink.ll
@@ -7,6 +7,7 @@ define amdgpu_gfx i32 @sink_scratch_pointer(ptr addrspace(5) %stack, i32 inreg %
; GCN: ; %bb.0:
; GCN-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GCN-NEXT: s_cmp_lg_u32 s4, 0
+; GCN-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GCN-NEXT: s_cbranch_scc0 .LBB0_2
; GCN-NEXT: ; %bb.1: ; %bb2
; GCN-NEXT: scratch_load_b32 v0, v0, off offset:-4
@@ -22,6 +23,7 @@ define amdgpu_gfx i32 @sink_scratch_pointer(ptr addrspace(5) %stack, i32 inreg %
; GISEL: ; %bb.0:
; GISEL-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GISEL-NEXT: s_cmp_lg_u32 s4, 0
+; GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GISEL-NEXT: s_cbranch_scc0 .LBB0_2
; GISEL-NEXT: ; %bb.1: ; %bb2
; GISEL-NEXT: scratch_load_b32 v0, v0, off offset:-4
diff --git a/llvm/test/CodeGen/AMDGPU/select-fabs-fneg-extract.v2f16.ll b/llvm/test/CodeGen/AMDGPU/select-fabs-fneg-extract.v2f16.ll
index 2ccfb6219dc419..ca30a6a77a9edb 100644
--- a/llvm/test/CodeGen/AMDGPU/select-fabs-fneg-extract.v2f16.ll
+++ b/llvm/test/CodeGen/AMDGPU/select-fabs-fneg-extract.v2f16.ll
@@ -67,10 +67,9 @@ define <2 x half> @add_select_fabs_fabs_v2f16(<2 x i32> %c, <2 x half> %x, <2 x
; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v0, 16, v2
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v1, 16, v3
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-TRUE16-NEXT: v_cndmask_b16 v2.l, v3.l, v2.l, s0
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_cndmask_b16 v2.h, v1.l, v0.l, vcc_lo
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_pk_add_f16 v0, v2, v4
; GFX11-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -179,11 +178,10 @@ define { <2 x half>, <2 x half> } @add_select_multi_use_lhs_fabs_fabs_v2f16(<2 x
; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v0, 16, v2
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v1, 16, v3
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-TRUE16-NEXT: v_cndmask_b16 v3.l, v3.l, v2.l, s0
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX11-TRUE16-NEXT: v_cndmask_b16 v3.h, v1.l, v0.l, vcc_lo
; GFX11-TRUE16-NEXT: v_pk_add_f16 v1, v2, v4
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX11-TRUE16-NEXT: v_pk_add_f16 v0, v3, v5
; GFX11-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -279,11 +277,10 @@ define { <2 x half>, <2 x half> } @add_select_multi_store_use_lhs_fabs_fabs_v2f1
; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v0, 16, v2
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v1, 16, v3
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-TRUE16-NEXT: v_cndmask_b16 v3.l, v3.l, v2.l, s0
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX11-TRUE16-NEXT: v_cndmask_b16 v3.h, v1.l, v0.l, vcc_lo
; GFX11-TRUE16-NEXT: v_mov_b32_e32 v1, v2
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX11-TRUE16-NEXT: v_pk_add_f16 v0, v3, v4
; GFX11-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -395,11 +392,10 @@ define { <2 x half>, <2 x half> } @add_select_multi_use_rhs_fabs_fabs_v2f16(<2 x
; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v0, 16, v2
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v1, 16, v3
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-TRUE16-NEXT: v_cndmask_b16 v2.l, v3.l, v2.l, s0
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX11-TRUE16-NEXT: v_cndmask_b16 v2.h, v1.l, v0.l, vcc_lo
; GFX11-TRUE16-NEXT: v_pk_add_f16 v1, v3, v5
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX11-TRUE16-NEXT: v_pk_add_f16 v0, v2, v4
; GFX11-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -488,11 +484,11 @@ define <2 x half> @add_select_fabs_var_v2f16(<2 x i32> %c, <2 x half> %x, <2 x h
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, 0, v1
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v1, 16, v3
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e64 s0, 0, v0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v5, 16, v2
; GFX11-TRUE16-NEXT: v_cndmask_b16 v0.l, v3.l, v2.l, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_cndmask_b16 v0.h, v1.l, v5.l, vcc_lo
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_pk_add_f16 v0, v0, v4
; GFX11-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -576,11 +572,11 @@ define <2 x half> @add_select_fabs_negk_v2f16(<2 x i32> %c, <2 x half> %x, <2 x
; GFX11-TRUE16-NEXT: v_and_b32_e32 v2, 0x7fff7fff, v2
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, 0, v1
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e64 s0, 0, v0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v0, 16, v2
; GFX11-TRUE16-NEXT: v_cndmask_b16 v1.l, 0xbc00, v2.l, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_cndmask_b16 v1.h, 0xbc00, v0.l, vcc_lo
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_pk_add_f16 v0, v1, v3
; GFX11-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -662,10 +658,9 @@ define <2 x half> @add_select_fabs_negk_negk_v2f16(<2 x i32> %c, <2 x half> %x)
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v3.l, 0xbc00
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, 0, v1
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e64 s0, 0, v0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_cndmask_b16 v0.h, v3.l, 0xc000, vcc_lo
; GFX11-TRUE16-NEXT: v_cndmask_b16 v0.l, v3.l, 0xc000, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_pk_add_f16 v0, v2, v0 neg_lo:[0,1] neg_hi:[0,1]
; GFX11-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -745,10 +740,9 @@ define <2 x half> @add_select_posk_posk_v2f16(<2 x i32> %c, <2 x half> %x) {
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v3.l, 0x3c00
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, 0, v1
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e64 s0, 0, v0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_cndmask_b16 v0.h, v3.l, 0x4000, vcc_lo
; GFX11-TRUE16-NEXT: v_cndmask_b16 v0.l, v3.l, 0x4000, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_pk_add_f16 v0, v0, v2
; GFX11-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -1003,11 +997,11 @@ define <2 x half> @add_select_fabs_posk_v2f16(<2 x i32> %c, <2 x half> %x, <2 x
; GFX11-TRUE16-NEXT: v_and_b32_e32 v2, 0x7fff7fff, v2
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, 0, v1
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e64 s0, 0, v0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v0, 16, v2
; GFX11-TRUE16-NEXT: v_cndmask_b16 v1.l, 0x3c00, v2.l, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_cndmask_b16 v1.h, 0x3c00, v0.l, vcc_lo
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_pk_add_f16 v0, v1, v3
; GFX11-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -1172,10 +1166,9 @@ define <2 x half> @add_select_fneg_fneg_v2f16(<2 x i32> %c, <2 x half> %x, <2 x
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v1, 16, v2
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v5, 16, v3
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e64 s0, 0, v0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_cndmask_b16 v0.h, v5.l, v1.l, vcc_lo
; GFX11-TRUE16-NEXT: v_cndmask_b16 v0.l, v3.l, v2.l, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_pk_add_f16 v0, v4, v0 neg_lo:[0,1] neg_hi:[0,1]
; GFX11-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -1270,11 +1263,10 @@ define { <2 x half>, <2 x half> } @add_select_multi_use_lhs_fneg_fneg_v2f16(<2 x
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v1, 16, v2
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v6, 16, v3
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e64 s0, 0, v0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_2) | instid1(VALU_DEP_2)
; GFX11-TRUE16-NEXT: v_cndmask_b16 v0.h, v6.l, v1.l, vcc_lo
; GFX11-TRUE16-NEXT: v_cndmask_b16 v0.l, v3.l, v2.l, s0
; GFX11-TRUE16-NEXT: v_pk_add_f16 v1, v5, v2 neg_lo:[0,1] neg_hi:[0,1]
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX11-TRUE16-NEXT: v_pk_add_f16 v0, v4, v0 neg_lo:[0,1] neg_hi:[0,1]
; GFX11-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -1360,11 +1352,10 @@ define { <2 x half>, <2 x half> } @add_select_multi_store_use_lhs_fneg_fneg_v2f1
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v1, 16, v2
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v5, 16, v3
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e64 s0, 0, v0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_2) | instid1(VALU_DEP_2)
; GFX11-TRUE16-NEXT: v_cndmask_b16 v0.h, v5.l, v1.l, vcc_lo
; GFX11-TRUE16-NEXT: v_cndmask_b16 v0.l, v3.l, v2.l, s0
; GFX11-TRUE16-NEXT: v_xor_b32_e32 v1, 0x80008000, v2
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX11-TRUE16-NEXT: v_pk_add_f16 v0, v4, v0 neg_lo:[0,1] neg_hi:[0,1]
; GFX11-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -1462,11 +1453,10 @@ define { <2 x half>, <2 x half> } @add_select_multi_use_rhs_fneg_fneg_v2f16(<2 x
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v1, 16, v2
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v6, 16, v3
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e64 s0, 0, v0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_2) | instid1(VALU_DEP_2)
; GFX11-TRUE16-NEXT: v_cndmask_b16 v0.h, v6.l, v1.l, vcc_lo
; GFX11-TRUE16-NEXT: v_cndmask_b16 v0.l, v3.l, v2.l, s0
; GFX11-TRUE16-NEXT: v_pk_add_f16 v1, v5, v3 neg_lo:[0,1] neg_hi:[0,1]
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX11-TRUE16-NEXT: v_pk_add_f16 v0, v4, v0 neg_lo:[0,1] neg_hi:[0,1]
; GFX11-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -1552,11 +1542,11 @@ define <2 x half> @add_select_fneg_var_v2f16(<2 x i32> %c, <2 x half> %x, <2 x h
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, 0, v1
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v1, 16, v3
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e64 s0, 0, v0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v5, 16, v2
; GFX11-TRUE16-NEXT: v_cndmask_b16 v0.l, v3.l, v2.l, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_cndmask_b16 v0.h, v1.l, v5.l, vcc_lo
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_pk_add_f16 v0, v0, v4
; GFX11-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -1637,10 +1627,9 @@ define <2 x half> @add_select_fneg_negk_v2f16(<2 x i32> %c, <2 x half> %x, <2 x
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, 0, v1
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v1, 16, v2
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e64 s0, 0, v0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_cndmask_b16 v0.h, 0x3c00, v1.l, vcc_lo
; GFX11-TRUE16-NEXT: v_cndmask_b16 v0.l, 0x3c00, v2.l, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_pk_add_f16 v0, v3, v0 neg_lo:[0,1] neg_hi:[0,1]
; GFX11-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -1719,10 +1708,9 @@ define <2 x half> @add_select_fneg_inv2pi_v2f16(<2 x i32> %c, <2 x half> %x, <2
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, 0, v1
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v1, 16, v2
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e64 s0, 0, v0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_cndmask_b16 v0.h, 0xb118, v1.l, vcc_lo
; GFX11-TRUE16-NEXT: v_cndmask_b16 v0.l, 0xb118, v2.l, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_pk_add_f16 v0, v3, v0 neg_lo:[0,1] neg_hi:[0,1]
; GFX11-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -1801,10 +1789,9 @@ define <2 x half> @add_select_fneg_neginv2pi_v2f16(<2 x i32> %c, <2 x half> %x,
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, 0, v1
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v1, 16, v2
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e64 s0, 0, v0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_cndmask_b16 v0.h, 0x3118, v1.l, vcc_lo
; GFX11-TRUE16-NEXT: v_cndmask_b16 v0.l, 0x3118, v2.l, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_pk_add_f16 v0, v3, v0 neg_lo:[0,1] neg_hi:[0,1]
; GFX11-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -1884,10 +1871,9 @@ define <2 x half> @add_select_negk_negk_v2f16(<2 x i32> %c, <2 x half> %x) {
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v3.l, 0xbc00
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, 0, v1
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e64 s0, 0, v0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_cndmask_b16 v0.h, v3.l, 0xc000, vcc_lo
; GFX11-TRUE16-NEXT: v_cndmask_b16 v0.l, v3.l, 0xc000, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_pk_add_f16 v0, v0, v2
; GFX11-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -1966,10 +1952,9 @@ define <2 x half> @add_select_negliteralk_negliteralk_v2f16(<2 x i32> %c, <2 x h
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v3.l, 0xec00
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, 0, v1
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e64 s0, 0, v0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_cndmask_b16 v0.h, v3.l, 0xe800, vcc_lo
; GFX11-TRUE16-NEXT: v_cndmask_b16 v0.l, v3.l, 0xe800, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_pk_add_f16 v0, v0, v2
; GFX11-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -2048,10 +2033,9 @@ define <2 x half> @add_select_fneg_negk_negk_v2f16(<2 x i32> %c, <2 x half> %x)
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v3.l, 0xbc00
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, 0, v1
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e64 s0, 0, v0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_cndmask_b16 v0.h, v3.l, 0xc000, vcc_lo
; GFX11-TRUE16-NEXT: v_cndmask_b16 v0.l, v3.l, 0xc000, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_pk_add_f16 v0, v2, v0 neg_lo:[0,1] neg_hi:[0,1]
; GFX11-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -2130,10 +2114,9 @@ define <2 x half> @add_select_negk_fneg_v2f16(<2 x i32> %c, <2 x half> %x, <2 x
; GFX11-TRUE16-NEXT: v_cmp_ne_u32_e32 vcc_lo, 0, v1
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v1, 16, v2
; GFX11-TRUE16-NEXT: v_cmp_ne_u32_e64 s0, 0, v0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_cndmask_b16 v0.h, 0x3c00, v1.l, vcc_lo
; GFX11-TRUE16-NEXT: v_cndmask_b16 v0.l, 0x3c00, v2.l, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_pk_add_f16 v0, v3, v0 neg_lo:[0,1] neg_hi:[0,1]
; GFX11-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -2212,10 +2195,9 @@ define <2 x half> @add_select_fneg_posk_v2f16(<2 x i32> %c, <2 x half> %x, <2 x
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, 0, v1
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v1, 16, v2
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e64 s0, 0, v0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_cndmask_b16 v0.h, 0xbc00, v1.l, vcc_lo
; GFX11-TRUE16-NEXT: v_cndmask_b16 v0.l, 0xbc00, v2.l, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_pk_add_f16 v0, v3, v0 neg_lo:[0,1] neg_hi:[0,1]
; GFX11-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -2294,10 +2276,9 @@ define <2 x half> @add_select_posk_fneg_v2f16(<2 x i32> %c, <2 x half> %x, <2 x
; GFX11-TRUE16-NEXT: v_cmp_ne_u32_e32 vcc_lo, 0, v1
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v1, 16, v2
; GFX11-TRUE16-NEXT: v_cmp_ne_u32_e64 s0, 0, v0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_cndmask_b16 v0.h, 0xbc00, v1.l, vcc_lo
; GFX11-TRUE16-NEXT: v_cndmask_b16 v0.l, 0xbc00, v2.l, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_pk_add_f16 v0, v3, v0 neg_lo:[0,1] neg_hi:[0,1]
; GFX11-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -2383,10 +2364,9 @@ define <2 x half> @add_select_negfabs_fabs_v2f16(<2 x i32> %c, <2 x half> %x, <2
; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v0, 16, v2
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v1, 16, v3
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-TRUE16-NEXT: v_cndmask_b16 v2.l, v3.l, v2.l, s0
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_cndmask_b16 v2.h, v1.l, v0.l, vcc_lo
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_pk_add_f16 v0, v2, v4
; GFX11-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -2478,10 +2458,9 @@ define <2 x half> @add_select_fabs_negfabs_v2f16(<2 x i32> %c, <2 x half> %x, <2
; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v0, 16, v2
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v1, 16, v3
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-TRUE16-NEXT: v_cndmask_b16 v2.l, v3.l, v2.l, s0
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_cndmask_b16 v2.h, v1.l, v0.l, vcc_lo
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_pk_add_f16 v0, v2, v4
; GFX11-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -2573,10 +2552,9 @@ define <2 x half> @add_select_neg_fabs_v2f16(<2 x i32> %c, <2 x half> %x, <2 x h
; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v0, 16, v2
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v1, 16, v3
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-TRUE16-NEXT: v_cndmask_b16 v2.l, v3.l, v2.l, s0
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_cndmask_b16 v2.h, v1.l, v0.l, vcc_lo
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_pk_add_f16 v0, v2, v4
; GFX11-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -2667,10 +2645,9 @@ define <2 x half> @add_select_fabs_neg_v2f16(<2 x i32> %c, <2 x half> %x, <2 x h
; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v0, 16, v2
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v1, 16, v3
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-TRUE16-NEXT: v_cndmask_b16 v2.l, v3.l, v2.l, s0
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_cndmask_b16 v2.h, v1.l, v0.l, vcc_lo
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_pk_add_f16 v0, v2, v4
; GFX11-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -2755,11 +2732,11 @@ define <2 x half> @add_select_neg_negfabs_v2f16(<2 x i32> %c, <2 x half> %x, <2
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, 0, v1
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v1, 16, v2
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e64 s0, 0, v0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v5, 16, v3
; GFX11-TRUE16-NEXT: v_cndmask_b16 v0.l, v3.l, v2.l, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_cndmask_b16 v0.h, v5.l, v1.l, vcc_lo
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_pk_add_f16 v0, v4, v0 neg_lo:[0,1] neg_hi:[0,1]
; GFX11-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -2843,11 +2820,11 @@ define <2 x half> @add_select_negfabs_neg_v2f16(<2 x i32> %c, <2 x half> %x, <2
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, 0, v1
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v1, 16, v3
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e64 s0, 0, v0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v5, 16, v2
; GFX11-TRUE16-NEXT: v_cndmask_b16 v0.l, v2.l, v3.l, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_cndmask_b16 v0.h, v5.l, v1.l, vcc_lo
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_pk_add_f16 v0, v4, v0 neg_lo:[0,1] neg_hi:[0,1]
; GFX11-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -2933,11 +2910,11 @@ define <2 x half> @mul_select_negfabs_posk_v2f16(<2 x i32> %c, <2 x half> %x, <2
; GFX11-TRUE16-NEXT: v_or_b32_e32 v2, 0x80008000, v2
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, 0, v1
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e64 s0, 0, v0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v0, 16, v2
; GFX11-TRUE16-NEXT: v_cndmask_b16 v1.l, 0x4400, v2.l, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_cndmask_b16 v1.h, 0x4400, v0.l, vcc_lo
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_pk_mul_f16 v0, v1, v3
; GFX11-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -3109,11 +3086,11 @@ define <2 x half> @mul_select_negfabs_negk_v2f16(<2 x i32> %c, <2 x half> %x, <2
; GFX11-TRUE16-NEXT: v_or_b32_e32 v2, 0x80008000, v2
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, 0, v1
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e64 s0, 0, v0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v0, 16, v2
; GFX11-TRUE16-NEXT: v_cndmask_b16 v1.l, 0xc400, v2.l, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_cndmask_b16 v1.h, 0xc400, v0.l, vcc_lo
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_pk_mul_f16 v0, v1, v3
; GFX11-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -3295,8 +3272,8 @@ define <2 x half> @select_fneg_posk_src_add_v2f16(<2 x i32> %c, <2 x half> %x, <
; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_xor_b32_e32 v2, 0x80008000, v2
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v1, 16, v2
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-TRUE16-NEXT: v_cndmask_b16 v0.l, 0x4000, v2.l, s0
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX11-TRUE16-NEXT: v_cndmask_b16 v0.h, 0x4000, v1.l, vcc_lo
; GFX11-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -3374,10 +3351,9 @@ define <2 x half> @select_fneg_posk_src_add_v2f16_nsz(<2 x i32> %c, <2 x half> %
; GFX11-TRUE16-NEXT: v_pk_add_f16 v2, v2, -4.0 op_sel_hi:[1,0] neg_lo:[1,0] neg_hi:[1,0]
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, 0, v1
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e64 s0, 0, v0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v1, 16, v2
; GFX11-TRUE16-NEXT: v_cndmask_b16 v0.l, 0x4000, v2.l, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX11-TRUE16-NEXT: v_cndmask_b16 v0.h, 0x4000, v1.l, vcc_lo
; GFX11-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -3465,8 +3441,8 @@ define <2 x half> @select_fneg_posk_src_sub_v2f16(<2 x i32> %c, <2 x half> %x) {
; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_xor_b32_e32 v2, 0x80008000, v2
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v1, 16, v2
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-TRUE16-NEXT: v_cndmask_b16 v0.l, 0x4000, v2.l, s0
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX11-TRUE16-NEXT: v_cndmask_b16 v0.h, 0x4000, v1.l, vcc_lo
; GFX11-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -3544,10 +3520,9 @@ define <2 x half> @select_fneg_posk_src_sub_v2f16_nsz(<2 x i32> %c, <2 x half> %
; GFX11-TRUE16-NEXT: v_pk_add_f16 v2, v2, 4.0 op_sel_hi:[1,0] neg_lo:[1,0] neg_hi:[1,0]
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, 0, v1
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e64 s0, 0, v0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v1, 16, v2
; GFX11-TRUE16-NEXT: v_cndmask_b16 v0.l, 0x4000, v2.l, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX11-TRUE16-NEXT: v_cndmask_b16 v0.h, 0x4000, v1.l, vcc_lo
; GFX11-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -3624,10 +3599,9 @@ define <2 x half> @select_fneg_posk_src_mul_v2f16(<2 x i32> %c, <2 x half> %x) {
; GFX11-TRUE16-NEXT: v_pk_mul_f16 v2, v2, -4.0 op_sel_hi:[1,0]
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, 0, v1
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e64 s0, 0, v0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v1, 16, v2
; GFX11-TRUE16-NEXT: v_cndmask_b16 v0.l, 0x4000, v2.l, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX11-TRUE16-NEXT: v_cndmask_b16 v0.h, 0x4000, v1.l, vcc_lo
; GFX11-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -3804,8 +3778,8 @@ define <2 x half> @select_fneg_posk_src_fma_v2f16(<2 x i32> %c, <2 x half> %x, <
; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_xor_b32_e32 v2, 0x80008000, v2
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v1, 16, v2
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-TRUE16-NEXT: v_cndmask_b16 v0.l, 0x4000, v2.l, s0
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX11-TRUE16-NEXT: v_cndmask_b16 v0.h, 0x4000, v1.l, vcc_lo
; GFX11-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -3905,8 +3879,8 @@ define <2 x half> @select_fneg_posk_src_fmad_v2f16(<2 x i32> %c, <2 x half> %x,
; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_xor_b32_e32 v2, 0x80008000, v2
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v1, 16, v2
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-TRUE16-NEXT: v_cndmask_b16 v0.l, 0x4000, v2.l, s0
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX11-TRUE16-NEXT: v_cndmask_b16 v0.h, 0x4000, v1.l, vcc_lo
; GFX11-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -3994,10 +3968,9 @@ define <2 x half> @select_fneg_posk_src_fmad_v2f16_nsz(<2 x i32> %c, <2 x half>
; GFX11-TRUE16-NEXT: v_pk_fma_f16 v2, v2, -4.0, v3 op_sel_hi:[1,0,1] neg_lo:[0,0,1] neg_hi:[0,0,1]
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, 0, v1
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e64 s0, 0, v0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v1, 16, v2
; GFX11-TRUE16-NEXT: v_cndmask_b16 v0.l, 0x4000, v2.l, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX11-TRUE16-NEXT: v_cndmask_b16 v0.h, 0x4000, v1.l, vcc_lo
; GFX11-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
diff --git a/llvm/test/CodeGen/AMDGPU/select-flags-to-fmin-fmax.ll b/llvm/test/CodeGen/AMDGPU/select-flags-to-fmin-fmax.ll
index 5e1c454e5fc52c..c28593abf4dd66 100644
--- a/llvm/test/CodeGen/AMDGPU/select-flags-to-fmin-fmax.ll
+++ b/llvm/test/CodeGen/AMDGPU/select-flags-to-fmin-fmax.ll
@@ -1135,7 +1135,7 @@ define <2 x half> @v_test_fmin_legacy_ule_v2f16_safe(<2 x half> %a, <2 x half> %
; GFX1170-TRUE16-NEXT: v_lshrrev_b32_e32 v2, 16, v1
; GFX1170-TRUE16-NEXT: v_lshrrev_b32_e32 v3, 16, v0
; GFX1170-TRUE16-NEXT: v_cmp_ngt_f16_e64 s0, v0.l, v1.l
-; GFX1170-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX1170-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX1170-TRUE16-NEXT: v_cmp_ngt_f16_e32 vcc_lo, v3.l, v2.l
; GFX1170-TRUE16-NEXT: v_cndmask_b16 v0.l, v1.l, v0.l, s0
; GFX1170-TRUE16-NEXT: v_cndmask_b16 v0.h, v2.l, v3.l, vcc_lo
@@ -1164,7 +1164,7 @@ define <2 x half> @v_test_fmin_legacy_ule_v2f16_safe(<2 x half> %a, <2 x half> %
; GFX12-TRUE16-NEXT: v_lshrrev_b32_e32 v2, 16, v1
; GFX12-TRUE16-NEXT: v_lshrrev_b32_e32 v3, 16, v0
; GFX12-TRUE16-NEXT: v_cmp_ngt_f16_e64 s0, v0.l, v1.l
-; GFX12-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_2)
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX12-TRUE16-NEXT: v_cmp_ngt_f16_e32 vcc_lo, v3.l, v2.l
; GFX12-TRUE16-NEXT: s_wait_alu depctr_va_sdst(0)
; GFX12-TRUE16-NEXT: v_cndmask_b16 v0.l, v1.l, v0.l, s0
@@ -1234,7 +1234,7 @@ define <2 x half> @v_test_fmin_legacy_ule_v2f16_nnan_flag(<2 x half> %a, <2 x ha
; GFX1170-TRUE16-NEXT: v_lshrrev_b32_e32 v2, 16, v1
; GFX1170-TRUE16-NEXT: v_lshrrev_b32_e32 v3, 16, v0
; GFX1170-TRUE16-NEXT: v_cmp_ngt_f16_e64 s0, v0.l, v1.l
-; GFX1170-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX1170-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX1170-TRUE16-NEXT: v_cmp_ngt_f16_e32 vcc_lo, v3.l, v2.l
; GFX1170-TRUE16-NEXT: v_cndmask_b16 v0.l, v1.l, v0.l, s0
; GFX1170-TRUE16-NEXT: v_cndmask_b16 v0.h, v2.l, v3.l, vcc_lo
@@ -1263,7 +1263,7 @@ define <2 x half> @v_test_fmin_legacy_ule_v2f16_nnan_flag(<2 x half> %a, <2 x ha
; GFX12-TRUE16-NEXT: v_lshrrev_b32_e32 v2, 16, v1
; GFX12-TRUE16-NEXT: v_lshrrev_b32_e32 v3, 16, v0
; GFX12-TRUE16-NEXT: v_cmp_ngt_f16_e64 s0, v0.l, v1.l
-; GFX12-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_2)
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX12-TRUE16-NEXT: v_cmp_ngt_f16_e32 vcc_lo, v3.l, v2.l
; GFX12-TRUE16-NEXT: s_wait_alu depctr_va_sdst(0)
; GFX12-TRUE16-NEXT: v_cndmask_b16 v0.l, v1.l, v0.l, s0
@@ -1333,7 +1333,7 @@ define <2 x half> @v_test_fmin_legacy_ule_v2f16_nsz_flag(<2 x half> %a, <2 x hal
; GFX1170-TRUE16-NEXT: v_lshrrev_b32_e32 v2, 16, v1
; GFX1170-TRUE16-NEXT: v_lshrrev_b32_e32 v3, 16, v0
; GFX1170-TRUE16-NEXT: v_cmp_ngt_f16_e64 s0, v0.l, v1.l
-; GFX1170-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX1170-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX1170-TRUE16-NEXT: v_cmp_ngt_f16_e32 vcc_lo, v3.l, v2.l
; GFX1170-TRUE16-NEXT: v_cndmask_b16 v0.l, v1.l, v0.l, s0
; GFX1170-TRUE16-NEXT: v_cndmask_b16 v0.h, v2.l, v3.l, vcc_lo
@@ -1362,7 +1362,7 @@ define <2 x half> @v_test_fmin_legacy_ule_v2f16_nsz_flag(<2 x half> %a, <2 x hal
; GFX12-TRUE16-NEXT: v_lshrrev_b32_e32 v2, 16, v1
; GFX12-TRUE16-NEXT: v_lshrrev_b32_e32 v3, 16, v0
; GFX12-TRUE16-NEXT: v_cmp_ngt_f16_e64 s0, v0.l, v1.l
-; GFX12-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_2)
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX12-TRUE16-NEXT: v_cmp_ngt_f16_e32 vcc_lo, v3.l, v2.l
; GFX12-TRUE16-NEXT: s_wait_alu depctr_va_sdst(0)
; GFX12-TRUE16-NEXT: v_cndmask_b16 v0.l, v1.l, v0.l, s0
@@ -1476,7 +1476,7 @@ define <2 x half> @v_test_fmax_legacy_uge_v2f16_safe(<2 x half> %a, <2 x half> %
; GFX1170-TRUE16-NEXT: v_lshrrev_b32_e32 v2, 16, v1
; GFX1170-TRUE16-NEXT: v_lshrrev_b32_e32 v3, 16, v0
; GFX1170-TRUE16-NEXT: v_cmp_nlt_f16_e64 s0, v0.l, v1.l
-; GFX1170-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX1170-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX1170-TRUE16-NEXT: v_cmp_nlt_f16_e32 vcc_lo, v3.l, v2.l
; GFX1170-TRUE16-NEXT: v_cndmask_b16 v0.l, v1.l, v0.l, s0
; GFX1170-TRUE16-NEXT: v_cndmask_b16 v0.h, v2.l, v3.l, vcc_lo
@@ -1505,7 +1505,7 @@ define <2 x half> @v_test_fmax_legacy_uge_v2f16_safe(<2 x half> %a, <2 x half> %
; GFX12-TRUE16-NEXT: v_lshrrev_b32_e32 v2, 16, v1
; GFX12-TRUE16-NEXT: v_lshrrev_b32_e32 v3, 16, v0
; GFX12-TRUE16-NEXT: v_cmp_nlt_f16_e64 s0, v0.l, v1.l
-; GFX12-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_2)
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX12-TRUE16-NEXT: v_cmp_nlt_f16_e32 vcc_lo, v3.l, v2.l
; GFX12-TRUE16-NEXT: s_wait_alu depctr_va_sdst(0)
; GFX12-TRUE16-NEXT: v_cndmask_b16 v0.l, v1.l, v0.l, s0
@@ -1575,7 +1575,7 @@ define <2 x half> @v_test_fmax_legacy_uge_v2f16_nnan_flag(<2 x half> %a, <2 x ha
; GFX1170-TRUE16-NEXT: v_lshrrev_b32_e32 v2, 16, v1
; GFX1170-TRUE16-NEXT: v_lshrrev_b32_e32 v3, 16, v0
; GFX1170-TRUE16-NEXT: v_cmp_nlt_f16_e64 s0, v0.l, v1.l
-; GFX1170-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX1170-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX1170-TRUE16-NEXT: v_cmp_nlt_f16_e32 vcc_lo, v3.l, v2.l
; GFX1170-TRUE16-NEXT: v_cndmask_b16 v0.l, v1.l, v0.l, s0
; GFX1170-TRUE16-NEXT: v_cndmask_b16 v0.h, v2.l, v3.l, vcc_lo
@@ -1604,7 +1604,7 @@ define <2 x half> @v_test_fmax_legacy_uge_v2f16_nnan_flag(<2 x half> %a, <2 x ha
; GFX12-TRUE16-NEXT: v_lshrrev_b32_e32 v2, 16, v1
; GFX12-TRUE16-NEXT: v_lshrrev_b32_e32 v3, 16, v0
; GFX12-TRUE16-NEXT: v_cmp_nlt_f16_e64 s0, v0.l, v1.l
-; GFX12-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_2)
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX12-TRUE16-NEXT: v_cmp_nlt_f16_e32 vcc_lo, v3.l, v2.l
; GFX12-TRUE16-NEXT: s_wait_alu depctr_va_sdst(0)
; GFX12-TRUE16-NEXT: v_cndmask_b16 v0.l, v1.l, v0.l, s0
@@ -1674,7 +1674,7 @@ define <2 x half> @v_test_fmax_legacy_uge_v2f16_nsz_flag(<2 x half> %a, <2 x hal
; GFX1170-TRUE16-NEXT: v_lshrrev_b32_e32 v2, 16, v1
; GFX1170-TRUE16-NEXT: v_lshrrev_b32_e32 v3, 16, v0
; GFX1170-TRUE16-NEXT: v_cmp_nlt_f16_e64 s0, v0.l, v1.l
-; GFX1170-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX1170-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX1170-TRUE16-NEXT: v_cmp_nlt_f16_e32 vcc_lo, v3.l, v2.l
; GFX1170-TRUE16-NEXT: v_cndmask_b16 v0.l, v1.l, v0.l, s0
; GFX1170-TRUE16-NEXT: v_cndmask_b16 v0.h, v2.l, v3.l, vcc_lo
@@ -1703,7 +1703,7 @@ define <2 x half> @v_test_fmax_legacy_uge_v2f16_nsz_flag(<2 x half> %a, <2 x hal
; GFX12-TRUE16-NEXT: v_lshrrev_b32_e32 v2, 16, v1
; GFX12-TRUE16-NEXT: v_lshrrev_b32_e32 v3, 16, v0
; GFX12-TRUE16-NEXT: v_cmp_nlt_f16_e64 s0, v0.l, v1.l
-; GFX12-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_2)
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX12-TRUE16-NEXT: v_cmp_nlt_f16_e32 vcc_lo, v3.l, v2.l
; GFX12-TRUE16-NEXT: s_wait_alu depctr_va_sdst(0)
; GFX12-TRUE16-NEXT: v_cndmask_b16 v0.l, v1.l, v0.l, s0
@@ -1841,11 +1841,10 @@ define <4 x half> @v_test_fmin_legacy_ule_v4f16_safe(<4 x half> %a, <4 x half> %
; GFX1170-TRUE16-NEXT: v_cmp_ngt_f16_e32 vcc_lo, v1.l, v3.l
; GFX1170-TRUE16-NEXT: v_cmp_ngt_f16_e64 s0, v0.l, v2.l
; GFX1170-TRUE16-NEXT: v_cmp_ngt_f16_e64 s1, v5.l, v4.l
-; GFX1170-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(SKIP_1) | instid1(VALU_DEP_4)
+; GFX1170-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_4)
; GFX1170-TRUE16-NEXT: v_cmp_ngt_f16_e64 s2, v7.l, v6.l
; GFX1170-TRUE16-NEXT: v_cndmask_b16 v1.l, v3.l, v1.l, vcc_lo
; GFX1170-TRUE16-NEXT: v_cndmask_b16 v0.l, v2.l, v0.l, s0
-; GFX1170-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
; GFX1170-TRUE16-NEXT: v_cndmask_b16 v0.h, v4.l, v5.l, s1
; GFX1170-TRUE16-NEXT: v_cndmask_b16 v1.h, v6.l, v7.l, s2
; GFX1170-TRUE16-NEXT: s_setpc_b64 s[30:31]
@@ -1990,11 +1989,10 @@ define <4 x half> @v_test_fmin_legacy_ule_v4f16_nnan_flag(<4 x half> %a, <4 x ha
; GFX1170-TRUE16-NEXT: v_cmp_ngt_f16_e32 vcc_lo, v1.l, v3.l
; GFX1170-TRUE16-NEXT: v_cmp_ngt_f16_e64 s0, v0.l, v2.l
; GFX1170-TRUE16-NEXT: v_cmp_ngt_f16_e64 s1, v5.l, v4.l
-; GFX1170-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(SKIP_1) | instid1(VALU_DEP_4)
+; GFX1170-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_4)
; GFX1170-TRUE16-NEXT: v_cmp_ngt_f16_e64 s2, v7.l, v6.l
; GFX1170-TRUE16-NEXT: v_cndmask_b16 v1.l, v3.l, v1.l, vcc_lo
; GFX1170-TRUE16-NEXT: v_cndmask_b16 v0.l, v2.l, v0.l, s0
-; GFX1170-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
; GFX1170-TRUE16-NEXT: v_cndmask_b16 v0.h, v4.l, v5.l, s1
; GFX1170-TRUE16-NEXT: v_cndmask_b16 v1.h, v6.l, v7.l, s2
; GFX1170-TRUE16-NEXT: s_setpc_b64 s[30:31]
@@ -2139,11 +2137,10 @@ define <4 x half> @v_test_fmin_legacy_ule_v4f16_nsz_flag(<4 x half> %a, <4 x hal
; GFX1170-TRUE16-NEXT: v_cmp_ngt_f16_e32 vcc_lo, v1.l, v3.l
; GFX1170-TRUE16-NEXT: v_cmp_ngt_f16_e64 s0, v0.l, v2.l
; GFX1170-TRUE16-NEXT: v_cmp_ngt_f16_e64 s1, v5.l, v4.l
-; GFX1170-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(SKIP_1) | instid1(VALU_DEP_4)
+; GFX1170-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_4)
; GFX1170-TRUE16-NEXT: v_cmp_ngt_f16_e64 s2, v7.l, v6.l
; GFX1170-TRUE16-NEXT: v_cndmask_b16 v1.l, v3.l, v1.l, vcc_lo
; GFX1170-TRUE16-NEXT: v_cndmask_b16 v0.l, v2.l, v0.l, s0
-; GFX1170-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
; GFX1170-TRUE16-NEXT: v_cndmask_b16 v0.h, v4.l, v5.l, s1
; GFX1170-TRUE16-NEXT: v_cndmask_b16 v1.h, v6.l, v7.l, s2
; GFX1170-TRUE16-NEXT: s_setpc_b64 s[30:31]
@@ -2347,11 +2344,10 @@ define <4 x half> @v_test_fmax_legacy_uge_v4f16_safe(<4 x half> %a, <4 x half> %
; GFX1170-TRUE16-NEXT: v_cmp_nlt_f16_e32 vcc_lo, v1.l, v3.l
; GFX1170-TRUE16-NEXT: v_cmp_nlt_f16_e64 s0, v0.l, v2.l
; GFX1170-TRUE16-NEXT: v_cmp_nlt_f16_e64 s1, v5.l, v4.l
-; GFX1170-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(SKIP_1) | instid1(VALU_DEP_4)
+; GFX1170-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_4)
; GFX1170-TRUE16-NEXT: v_cmp_nlt_f16_e64 s2, v7.l, v6.l
; GFX1170-TRUE16-NEXT: v_cndmask_b16 v1.l, v3.l, v1.l, vcc_lo
; GFX1170-TRUE16-NEXT: v_cndmask_b16 v0.l, v2.l, v0.l, s0
-; GFX1170-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
; GFX1170-TRUE16-NEXT: v_cndmask_b16 v0.h, v4.l, v5.l, s1
; GFX1170-TRUE16-NEXT: v_cndmask_b16 v1.h, v6.l, v7.l, s2
; GFX1170-TRUE16-NEXT: s_setpc_b64 s[30:31]
@@ -2496,11 +2492,10 @@ define <4 x half> @v_test_fmax_legacy_uge_v4f16_nnan_flag(<4 x half> %a, <4 x ha
; GFX1170-TRUE16-NEXT: v_cmp_nlt_f16_e32 vcc_lo, v1.l, v3.l
; GFX1170-TRUE16-NEXT: v_cmp_nlt_f16_e64 s0, v0.l, v2.l
; GFX1170-TRUE16-NEXT: v_cmp_nlt_f16_e64 s1, v5.l, v4.l
-; GFX1170-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(SKIP_1) | instid1(VALU_DEP_4)
+; GFX1170-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_4)
; GFX1170-TRUE16-NEXT: v_cmp_nlt_f16_e64 s2, v7.l, v6.l
; GFX1170-TRUE16-NEXT: v_cndmask_b16 v1.l, v3.l, v1.l, vcc_lo
; GFX1170-TRUE16-NEXT: v_cndmask_b16 v0.l, v2.l, v0.l, s0
-; GFX1170-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
; GFX1170-TRUE16-NEXT: v_cndmask_b16 v0.h, v4.l, v5.l, s1
; GFX1170-TRUE16-NEXT: v_cndmask_b16 v1.h, v6.l, v7.l, s2
; GFX1170-TRUE16-NEXT: s_setpc_b64 s[30:31]
@@ -2645,11 +2640,10 @@ define <4 x half> @v_test_fmax_legacy_uge_v4f16_nsz_flag(<4 x half> %a, <4 x hal
; GFX1170-TRUE16-NEXT: v_cmp_nlt_f16_e32 vcc_lo, v1.l, v3.l
; GFX1170-TRUE16-NEXT: v_cmp_nlt_f16_e64 s0, v0.l, v2.l
; GFX1170-TRUE16-NEXT: v_cmp_nlt_f16_e64 s1, v5.l, v4.l
-; GFX1170-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(SKIP_1) | instid1(VALU_DEP_4)
+; GFX1170-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_4)
; GFX1170-TRUE16-NEXT: v_cmp_nlt_f16_e64 s2, v7.l, v6.l
; GFX1170-TRUE16-NEXT: v_cndmask_b16 v1.l, v3.l, v1.l, vcc_lo
; GFX1170-TRUE16-NEXT: v_cndmask_b16 v0.l, v2.l, v0.l, s0
-; GFX1170-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
; GFX1170-TRUE16-NEXT: v_cndmask_b16 v0.h, v4.l, v5.l, s1
; GFX1170-TRUE16-NEXT: v_cndmask_b16 v1.h, v6.l, v7.l, s2
; GFX1170-TRUE16-NEXT: s_setpc_b64 s[30:31]
diff --git a/llvm/test/CodeGen/AMDGPU/select.f16.ll b/llvm/test/CodeGen/AMDGPU/select.f16.ll
index d4ff8ed56b7c66..5f6e91913fa0fb 100644
--- a/llvm/test/CodeGen/AMDGPU/select.f16.ll
+++ b/llvm/test/CodeGen/AMDGPU/select.f16.ll
@@ -1370,7 +1370,7 @@ define amdgpu_kernel void @select_v2f16_imm_c(
; GFX11-TRUE16-NEXT: v_cmp_nlt_f16_e32 vcc_lo, v1.l, v0.l
; GFX11-TRUE16-NEXT: s_waitcnt vmcnt(0)
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v0, 16, v2
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_1) | instid1(VALU_DEP_2)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_1) | instid1(VALU_DEP_3)
; GFX11-TRUE16-NEXT: v_cmp_nlt_f16_e64 s0, v4.l, v3.l
; GFX11-TRUE16-NEXT: v_cndmask_b16 v1.l, 0x3800, v2.l, vcc_lo
; GFX11-TRUE16-NEXT: v_cndmask_b16 v1.h, 0x3900, v0.l, s0
@@ -1542,7 +1542,7 @@ define amdgpu_kernel void @select_v2f16_imm_d(
; GFX11-TRUE16-NEXT: v_cmp_lt_f16_e32 vcc_lo, v1.l, v0.l
; GFX11-TRUE16-NEXT: s_waitcnt vmcnt(0)
; GFX11-TRUE16-NEXT: v_lshrrev_b32_e32 v0, 16, v2
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_1) | instid1(VALU_DEP_2)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_1) | instid1(VALU_DEP_3)
; GFX11-TRUE16-NEXT: v_cmp_lt_f16_e64 s0, v4.l, v3.l
; GFX11-TRUE16-NEXT: v_cndmask_b16 v1.l, 0x3800, v2.l, vcc_lo
; GFX11-TRUE16-NEXT: v_cndmask_b16 v1.h, 0x3900, v0.l, s0
@@ -1681,10 +1681,9 @@ define <4 x half> @v_vselect_v4f16(<4 x half> %a, <4 x half> %b, <4 x i32> %cond
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e64 s1, 0, v4
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e64 s2, 0, v6
; GFX11-TRUE16-NEXT: v_cndmask_b16 v1.h, v7.l, v5.l, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_4)
; GFX11-TRUE16-NEXT: v_cndmask_b16 v0.h, v9.l, v8.l, vcc_lo
; GFX11-TRUE16-NEXT: v_cndmask_b16 v0.l, v2.l, v0.l, s1
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_4)
; GFX11-TRUE16-NEXT: v_cndmask_b16 v1.l, v3.l, v1.l, s2
; GFX11-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -2094,7 +2093,6 @@ define <16 x half> @v_vselect_v16f16(<16 x half> %a, <16 x half> %b, <16 x i32>
; GFX11-TRUE16-NEXT: v_cndmask_b16 v3.l, v11.l, v3.l, s2
; GFX11-TRUE16-NEXT: s_waitcnt vmcnt(0)
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e64 s3, 0, v31
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_cndmask_b16 v7.h, v17.l, v16.l, s3
; GFX11-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
diff --git a/llvm/test/CodeGen/AMDGPU/si-pre-emit-peephole-redundant-mode-writes.ll b/llvm/test/CodeGen/AMDGPU/si-pre-emit-peephole-redundant-mode-writes.ll
index 4d859f38512765..26b9e771357c30 100644
--- a/llvm/test/CodeGen/AMDGPU/si-pre-emit-peephole-redundant-mode-writes.ll
+++ b/llvm/test/CodeGen/AMDGPU/si-pre-emit-peephole-redundant-mode-writes.ll
@@ -82,10 +82,10 @@ define amdgpu_kernel void @dead_round_and_denorm_writes(ptr addrspace(1) %out, f
; CHECK: ; %bb.0:
; CHECK-NEXT: s_load_b128 s[0:3], s[4:5], 0x24
; CHECK-NEXT: s_round_mode 0x0
+; CHECK-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_3)
; CHECK-NEXT: s_denorm_mode 15
; CHECK-NEXT: s_wait_kmcnt 0x0
; CHECK-NEXT: s_add_f32 s2, s2, s3
-; CHECK-NEXT: s_delay_alu instid0(SALU_CYCLE_3)
; CHECK-NEXT: v_dual_mov_b32 v0, 0 :: v_dual_mov_b32 v1, s2
; CHECK-NEXT: global_store_b32 v0, v1, s[0:1]
; CHECK-NEXT: s_endpgm
@@ -101,10 +101,10 @@ define amdgpu_kernel void @dead_round_write_across_denorm_write(ptr addrspace(1)
; CHECK: ; %bb.0:
; CHECK-NEXT: s_load_b128 s[0:3], s[4:5], 0x24
; CHECK-NEXT: s_denorm_mode 15
+; CHECK-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_3)
; CHECK-NEXT: s_round_mode 0x0
; CHECK-NEXT: s_wait_kmcnt 0x0
; CHECK-NEXT: s_add_f32 s2, s2, s3
-; CHECK-NEXT: s_delay_alu instid0(SALU_CYCLE_3)
; CHECK-NEXT: v_dual_mov_b32 v0, 0 :: v_dual_mov_b32 v1, s2
; CHECK-NEXT: global_store_b32 v0, v1, s[0:1]
; CHECK-NEXT: s_endpgm
@@ -121,10 +121,10 @@ define amdgpu_kernel void @dead_denorm_write_across_round_write(ptr addrspace(1)
; CHECK: ; %bb.0:
; CHECK-NEXT: s_load_b128 s[0:3], s[4:5], 0x24
; CHECK-NEXT: s_round_mode 0x0
+; CHECK-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_3)
; CHECK-NEXT: s_denorm_mode 15
; CHECK-NEXT: s_wait_kmcnt 0x0
; CHECK-NEXT: s_add_f32 s2, s2, s3
-; CHECK-NEXT: s_delay_alu instid0(SALU_CYCLE_3)
; CHECK-NEXT: v_dual_mov_b32 v0, 0 :: v_dual_mov_b32 v1, s2
; CHECK-NEXT: global_store_b32 v0, v1, s[0:1]
; CHECK-NEXT: s_endpgm
@@ -148,6 +148,7 @@ define void @round_write_kept_across_inline_asm(ptr addrspace(1) %out, float %a,
; CHECK-NEXT: ;;#ASMSTART
; CHECK-NEXT: ;;#ASMEND
; CHECK-NEXT: s_round_mode 0x0
+; CHECK-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; CHECK-NEXT: v_add_f32_e32 v2, v2, v3
; CHECK-NEXT: global_store_b32 v[0:1], v2, off
; CHECK-NEXT: s_setpc_b64 s[30:31]
@@ -215,8 +216,10 @@ define void @round_write_kept_across_setreg(ptr addrspace(1) %out, float %a, flo
; CHECK-NEXT: s_wait_bvhcnt 0x0
; CHECK-NEXT: s_wait_kmcnt 0x0
; CHECK-NEXT: s_round_mode 0x0
+; CHECK-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; CHECK-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 8, 4), 1
; CHECK-NEXT: s_round_mode 0x0
+; CHECK-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; CHECK-NEXT: v_add_f32_e32 v2, v2, v3
; CHECK-NEXT: global_store_b32 v[0:1], v2, off
; CHECK-NEXT: s_setpc_b64 s[30:31]
@@ -239,6 +242,7 @@ define void @round_write_kept_across_side_effect(ptr addrspace(1) %out, float %a
; CHECK-NEXT: s_round_mode 0x0
; CHECK-NEXT: s_nop 0
; CHECK-NEXT: s_round_mode 0xf
+; CHECK-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; CHECK-NEXT: v_add_f32_e32 v2, v2, v3
; CHECK-NEXT: global_store_b32 v[0:1], v2, off
; CHECK-NEXT: s_setpc_b64 s[30:31]
@@ -259,8 +263,8 @@ define void @dead_round_write_then_restore(ptr addrspace(1) %out, float %a, floa
; CHECK-NEXT: s_wait_bvhcnt 0x0
; CHECK-NEXT: s_wait_kmcnt 0x0
; CHECK-NEXT: s_round_mode 0xf
+; CHECK-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; CHECK-NEXT: v_add_f32_e32 v2, v2, v3
-; CHECK-NEXT: s_delay_alu instid0(VALU_DEP_1)
; CHECK-NEXT: v_mul_f32_e32 v2, v2, v3
; CHECK-NEXT: global_store_b32 v[0:1], v2, off
; CHECK-NEXT: s_setpc_b64 s[30:31]
@@ -282,9 +286,10 @@ define void @round_write_kept_across_fp_reader(ptr addrspace(1) %out, float %a,
; CHECK-NEXT: s_wait_bvhcnt 0x0
; CHECK-NEXT: s_wait_kmcnt 0x0
; CHECK-NEXT: s_round_mode 0x0
+; CHECK-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; CHECK-NEXT: v_add_f32_e32 v2, v2, v3
; CHECK-NEXT: s_round_mode 0xf
-; CHECK-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; CHECK-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; CHECK-NEXT: v_mul_f32_e32 v2, v2, v3
; CHECK-NEXT: global_store_b32 v[0:1], v2, off
; CHECK-NEXT: s_setpc_b64 s[30:31]
@@ -306,7 +311,7 @@ define void @round_write_live_out(ptr addrspace(1) %out, float %a, float %b, i1
; CHECK-NEXT: s_wait_kmcnt 0x0
; CHECK-NEXT: v_and_b32_e32 v4, 1, v4
; CHECK-NEXT: s_mov_b32 s0, exec_lo
-; CHECK-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; CHECK-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; CHECK-NEXT: v_cmpx_eq_u32_e32 1, v4
; CHECK-NEXT: s_cbranch_execz .LBB13_2
; CHECK-NEXT: ; %bb.1: ; %then
@@ -315,6 +320,7 @@ define void @round_write_live_out(ptr addrspace(1) %out, float %a, float %b, i1
; CHECK-NEXT: s_wait_alu depctr_sa_sdst(0)
; CHECK-NEXT: s_or_b32 exec_lo, exec_lo, s0
; CHECK-NEXT: s_round_mode 0xf
+; CHECK-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; CHECK-NEXT: v_add_f32_e32 v2, v2, v3
; CHECK-NEXT: global_store_b32 v[0:1], v2, off
; CHECK-NEXT: s_setpc_b64 s[30:31]
diff --git a/llvm/test/CodeGen/AMDGPU/simulated-trap-pseudo-expand.ll b/llvm/test/CodeGen/AMDGPU/simulated-trap-pseudo-expand.ll
index 3eaea0035fe1cc..57087a02015638 100644
--- a/llvm/test/CodeGen/AMDGPU/simulated-trap-pseudo-expand.ll
+++ b/llvm/test/CodeGen/AMDGPU/simulated-trap-pseudo-expand.ll
@@ -9,9 +9,10 @@ define amdgpu_kernel void @simulated_trap_pseudo_expand(i64 %offset, i1 %should_
; CHECK-NEXT: s_load_b32 s0, s[4:5], 0x8
; CHECK-NEXT: s_waitcnt lgkmcnt(0)
; CHECK-NEXT: s_bitcmp1_b32 s0, 0
+; CHECK-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; CHECK-NEXT: s_cselect_b32 s0, -1, 0
-; CHECK-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; CHECK-NEXT: s_and_b32 vcc_lo, exec_lo, s0
+; CHECK-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; CHECK-NEXT: s_cbranch_vccz .LBB0_2
; CHECK-NEXT: ; %bb.1: ; %normal_path
; CHECK-NEXT: s_clause 0x1
@@ -36,6 +37,7 @@ define amdgpu_kernel void @simulated_trap_pseudo_expand(i64 %offset, i1 %should_
; CHECK-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; CHECK-NEXT: s_bitset1_b32 s0, 10
; CHECK-NEXT: s_mov_b32 m0, s0
+; CHECK-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; CHECK-NEXT: s_sendmsg sendmsg(MSG_INTERRUPT)
; CHECK-NEXT: s_mov_b32 m0, ttmp2
; CHECK-NEXT: .LBB0_3: ; =>This Inner Loop Header: Depth=1
diff --git a/llvm/test/CodeGen/AMDGPU/skip-if-dead.ll b/llvm/test/CodeGen/AMDGPU/skip-if-dead.ll
index cb8974c0ff87c0..10069b0e9df695 100644
--- a/llvm/test/CodeGen/AMDGPU/skip-if-dead.ll
+++ b/llvm/test/CodeGen/AMDGPU/skip-if-dead.ll
@@ -125,6 +125,7 @@ define amdgpu_ps void @test_kill_depth_var(float %x) #0 {
; GFX11-LABEL: test_kill_depth_var:
; GFX11: ; %bb.0:
; GFX11-NEXT: v_cmp_ngt_f32_e32 vcc, 0, v0
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_and_not1_b64 exec, exec, vcc
; GFX11-NEXT: s_cbranch_scc0 .LBB3_1
; GFX11-NEXT: s_endpgm
@@ -194,11 +195,12 @@ define amdgpu_ps void @test_kill_depth_var_x2_same(float %x) #0 {
; GFX11: ; %bb.0:
; GFX11-NEXT: v_cmp_ngt_f32_e32 vcc, 0, v0
; GFX11-NEXT: s_mov_b64 s[0:1], exec
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b64 s[0:1], s[0:1], vcc
; GFX11-NEXT: s_cbranch_scc0 .LBB4_2
; GFX11-NEXT: ; %bb.1:
; GFX11-NEXT: s_and_not1_b64 exec, exec, vcc
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_cmp_ngt_f32_e32 vcc, 0, v0
; GFX11-NEXT: s_and_not1_b64 s[0:1], s[0:1], vcc
; GFX11-NEXT: s_cbranch_scc0 .LBB4_2
@@ -270,11 +272,12 @@ define amdgpu_ps void @test_kill_depth_var_x2(float %x, float %y) #0 {
; GFX11: ; %bb.0:
; GFX11-NEXT: v_cmp_ngt_f32_e32 vcc, 0, v0
; GFX11-NEXT: s_mov_b64 s[0:1], exec
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b64 s[0:1], s[0:1], vcc
; GFX11-NEXT: s_cbranch_scc0 .LBB5_2
; GFX11-NEXT: ; %bb.1:
; GFX11-NEXT: s_and_not1_b64 exec, exec, vcc
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_cmp_ngt_f32_e32 vcc, 0, v1
; GFX11-NEXT: s_and_not1_b64 s[0:1], s[0:1], vcc
; GFX11-NEXT: s_cbranch_scc0 .LBB5_2
@@ -355,7 +358,7 @@ define amdgpu_ps void @test_kill_depth_var_x2_instructions(float %x) #0 {
; GFX11: ; %bb.0:
; GFX11-NEXT: v_cmp_ngt_f32_e32 vcc, 0, v0
; GFX11-NEXT: s_mov_b64 s[0:1], exec
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b64 s[0:1], s[0:1], vcc
; GFX11-NEXT: s_cbranch_scc0 .LBB6_2
; GFX11-NEXT: ; %bb.1:
@@ -364,6 +367,7 @@ define amdgpu_ps void @test_kill_depth_var_x2_instructions(float %x) #0 {
; GFX11-NEXT: v_mov_b32_e64 v7, -1
; GFX11-NEXT: ;;#ASMEND
; GFX11-NEXT: v_cmp_ngt_f32_e32 vcc, 0, v7
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_and_not1_b64 s[0:1], s[0:1], vcc
; GFX11-NEXT: s_cbranch_scc0 .LBB6_2
; GFX11-NEXT: s_endpgm
@@ -489,6 +493,7 @@ define amdgpu_ps float @test_kill_control_flow(i32 inreg %arg) #0 {
; GFX11-LABEL: test_kill_control_flow:
; GFX11: ; %bb.0: ; %entry
; GFX11-NEXT: s_cmp_lg_u32 s0, 0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc0 .LBB7_2
; GFX11-NEXT: ; %bb.1: ; %exit
; GFX11-NEXT: v_mov_b32_e32 v0, 1.0
@@ -509,11 +514,12 @@ define amdgpu_ps float @test_kill_control_flow(i32 inreg %arg) #0 {
; GFX11-NEXT: ;;#ASMEND
; GFX11-NEXT: v_cmp_ngt_f32_e32 vcc, 0, v7
; GFX11-NEXT: s_mov_b64 s[2:3], exec
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b64 s[2:3], s[2:3], vcc
; GFX11-NEXT: s_cbranch_scc0 .LBB7_4
; GFX11-NEXT: ; %bb.3: ; %bb
; GFX11-NEXT: s_and_not1_b64 exec, exec, vcc
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: v_mov_b32_e32 v0, 1.0
; GFX11-NEXT: s_branch .LBB7_5
; GFX11-NEXT: .LBB7_4:
@@ -685,6 +691,7 @@ define amdgpu_ps void @test_kill_control_flow_remainder(i32 inreg %arg) #0 {
; GFX11: ; %bb.0: ; %entry
; GFX11-NEXT: v_mov_b32_e32 v9, 0
; GFX11-NEXT: s_cmp_lg_u32 s0, 0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc0 .LBB8_2
; GFX11-NEXT: ; %bb.1: ; %exit
; GFX11-NEXT: global_store_b32 v[0:1], v9, off
@@ -709,6 +716,7 @@ define amdgpu_ps void @test_kill_control_flow_remainder(i32 inreg %arg) #0 {
; GFX11-NEXT: ;;#ASMSTART
; GFX11-NEXT: v_mov_b32_e64 v8, -1
; GFX11-NEXT: ;;#ASMEND
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_and_not1_b64 s[2:3], s[2:3], vcc
; GFX11-NEXT: s_cbranch_scc0 .LBB8_4
; GFX11-NEXT: ; %bb.3: ; %bb
@@ -878,6 +886,7 @@ define amdgpu_ps float @test_kill_control_flow_return(i32 inreg %arg) #0 {
; GFX11-NEXT: s_cbranch_scc0 .LBB9_4
; GFX11-NEXT: ; %bb.1: ; %entry
; GFX11-NEXT: s_and_b64 exec, exec, s[2:3]
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: v_mov_b32_e32 v0, 0
; GFX11-NEXT: s_cmp_lg_u32 s0, 0
; GFX11-NEXT: s_cbranch_scc0 .LBB9_3
@@ -1069,6 +1078,7 @@ define amdgpu_ps void @test_kill_divergent_loop(i32 %arg) #0 {
; GFX11-NEXT: s_mov_b64 s[0:1], exec
; GFX11-NEXT: s_mov_b64 s[2:3], exec
; GFX11-NEXT: v_cmpx_eq_u32_e32 0, v0
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b64 s[2:3], exec, s[2:3]
; GFX11-NEXT: s_cbranch_execz .LBB10_3
; GFX11-NEXT: .LBB10_1: ; %bb
@@ -1087,6 +1097,7 @@ define amdgpu_ps void @test_kill_divergent_loop(i32 %arg) #0 {
; GFX11-NEXT: v_nop_e64
; GFX11-NEXT: ;;#ASMEND
; GFX11-NEXT: v_cmp_ngt_f32_e32 vcc, 0, v7
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_and_not1_b64 s[0:1], s[0:1], vcc
; GFX11-NEXT: s_cbranch_scc0 .LBB10_4
; GFX11-NEXT: ; %bb.2: ; %bb
@@ -1095,9 +1106,11 @@ define amdgpu_ps void @test_kill_divergent_loop(i32 %arg) #0 {
; GFX11-NEXT: global_load_b32 v0, v[0:1], off glc dlc
; GFX11-NEXT: s_waitcnt vmcnt(0)
; GFX11-NEXT: v_cmp_eq_u32_e32 vcc, 0, v0
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_cbranch_vccnz .LBB10_1
; GFX11-NEXT: .LBB10_3: ; %Flow1
; GFX11-NEXT: s_or_b64 exec, exec, s[2:3]
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: v_mov_b32_e32 v0, 8
; GFX11-NEXT: global_store_b32 v[0:1], v0, off dlc
; GFX11-NEXT: s_waitcnt_vscnt null, 0x0
@@ -1235,7 +1248,7 @@ define amdgpu_ps void @phi_use_def_before_kill(float inreg %x, i32 inreg %y) #0
; GFX11-LABEL: phi_use_def_before_kill:
; GFX11: ; %bb.0: ; %bb
; GFX11-NEXT: v_add_f32_e64 v1, s0, 1.0
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_3) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_cmp_lt_f32_e32 vcc, 0, v1
; GFX11-NEXT: v_cndmask_b32_e64 v0, 0, -1.0, vcc
; GFX11-NEXT: v_cmp_nlt_f32_e32 vcc, 0, v1
@@ -1245,6 +1258,7 @@ define amdgpu_ps void @phi_use_def_before_kill(float inreg %x, i32 inreg %y) #0
; GFX11-NEXT: ; %bb.1: ; %bb
; GFX11-NEXT: s_and_not1_b64 exec, exec, vcc
; GFX11-NEXT: s_cmp_lg_u32 s1, 0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc0 .LBB11_3
; GFX11-NEXT: ; %bb.2: ; %bb8
; GFX11-NEXT: v_mov_b32_e32 v1, 8
@@ -1253,6 +1267,7 @@ define amdgpu_ps void @phi_use_def_before_kill(float inreg %x, i32 inreg %y) #0
; GFX11-NEXT: s_waitcnt_vscnt null, 0x0
; GFX11-NEXT: .LBB11_3: ; %phibb
; GFX11-NEXT: v_cmp_eq_f32_e32 vcc, 0, v0
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_cbranch_vccz .LBB11_5
; GFX11-NEXT: ; %bb.4: ; %bb10
; GFX11-NEXT: v_mov_b32_e32 v0, 9
@@ -1356,6 +1371,7 @@ define amdgpu_ps void @no_skip_no_successors(float inreg %arg, float inreg %arg1
; GFX11: ; %bb.0: ; %bb
; GFX11-NEXT: v_cmp_nge_f32_e64 s[4:5], s1, 0
; GFX11-NEXT: s_and_b64 vcc, exec, s[4:5]
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_vccz .LBB12_3
; GFX11-NEXT: ; %bb.1: ; %bb6
; GFX11-NEXT: s_mov_b64 s[2:3], exec
@@ -1497,19 +1513,20 @@ define amdgpu_ps void @if_after_kill_block(float %arg, float %arg1, float %arg2,
; GFX11: ; %bb.0: ; %bb
; GFX11-NEXT: s_mov_b64 s[0:1], exec
; GFX11-NEXT: s_wqm_b64 exec, exec
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: s_mov_b64 s[2:3], exec
; GFX11-NEXT: v_cmpx_nle_f32_e32 0, v1
; GFX11-NEXT: s_xor_b64 s[2:3], exec, s[2:3]
; GFX11-NEXT: s_cbranch_execz .LBB13_3
; GFX11-NEXT: ; %bb.1: ; %bb3
; GFX11-NEXT: v_cmp_ngt_f32_e32 vcc, 0, v0
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_and_not1_b64 s[0:1], s[0:1], vcc
; GFX11-NEXT: s_cbranch_scc0 .LBB13_6
; GFX11-NEXT: ; %bb.2: ; %bb3
; GFX11-NEXT: s_and_not1_b64 exec, exec, vcc
; GFX11-NEXT: .LBB13_3: ; %bb4
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_4) | instid1(VALU_DEP_1)
; GFX11-NEXT: s_or_b64 exec, exec, s[2:3]
; GFX11-NEXT: image_sample_c v0, v[2:3], s[0:7], s[0:3] dmask:0x10 dim:SQ_RSRC_IMG_1D
; GFX11-NEXT: s_mov_b64 s[0:1], exec
@@ -1656,6 +1673,7 @@ define amdgpu_ps void @cbranch_kill(i32 inreg %0, float %val0, float %val1) {
; GFX11-NEXT: s_mov_b64 s[2:3], exec
; GFX11-NEXT: s_waitcnt vmcnt(0)
; GFX11-NEXT: v_cmpx_ge_f32_e32 0, v1
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b64 s[2:3], exec, s[2:3]
; GFX11-NEXT: s_cbranch_execz .LBB14_3
; GFX11-NEXT: ; %bb.1: ; %kill
@@ -1666,11 +1684,12 @@ define amdgpu_ps void @cbranch_kill(i32 inreg %0, float %val0, float %val1) {
; GFX11-NEXT: ; %bb.2: ; %kill
; GFX11-NEXT: s_mov_b64 exec, 0
; GFX11-NEXT: .LBB14_3: ; %Flow
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_or_saveexec_b64 s[0:1], s[2:3]
; GFX11-NEXT: ; implicit-def: $vgpr2
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_xor_b64 exec, exec, s[0:1]
; GFX11-NEXT: ; %bb.4: ; %live
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: v_mul_f32_e32 v2, v0, v1
; GFX11-NEXT: ; %bb.5: ; %export
; GFX11-NEXT: s_or_b64 exec, exec, s[0:1]
@@ -1839,6 +1858,7 @@ define amdgpu_ps void @complex_loop(i32 inreg %cmpa, i32 %cmpb, i32 %cmpc) {
; GFX11-LABEL: complex_loop:
; GFX11: ; %bb.0: ; %.entry
; GFX11-NEXT: s_cmp_lt_i32 s0, 1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB15_7
; GFX11-NEXT: ; %bb.1: ; %.lr.ph
; GFX11-NEXT: s_mov_b64 s[2:3], exec
@@ -1849,16 +1869,18 @@ define amdgpu_ps void @complex_loop(i32 inreg %cmpa, i32 %cmpb, i32 %cmpc) {
; GFX11-NEXT: ; in Loop: Header=BB15_3 Depth=1
; GFX11-NEXT: s_or_b64 exec, exec, s[4:5]
; GFX11-NEXT: s_add_i32 s6, s6, 1
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX11-NEXT: v_cmp_ge_i32_e32 vcc, s6, v1
; GFX11-NEXT: v_mov_b32_e32 v2, s6
; GFX11-NEXT: s_or_b64 s[0:1], vcc, s[0:1]
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_not1_b64 exec, exec, s[0:1]
; GFX11-NEXT: s_cbranch_execz .LBB15_6
; GFX11-NEXT: .LBB15_3: ; %hdr
; GFX11-NEXT: ; =>This Inner Loop Header: Depth=1
; GFX11-NEXT: s_mov_b64 s[4:5], exec
; GFX11-NEXT: v_cmpx_gt_u32_e32 s6, v0
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_xor_b64 s[4:5], exec, s[4:5]
; GFX11-NEXT: s_cbranch_execz .LBB15_2
; GFX11-NEXT: ; %bb.4: ; %kill
@@ -1939,6 +1961,7 @@ define void @skip_mode_switch(i32 %arg) {
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: s_mov_b64 s[0:1], exec
; GFX11-NEXT: v_cmpx_eq_u32_e32 0, v0
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_cbranch_execz .LBB16_2
; GFX11-NEXT: ; %bb.1: ; %bb.0
; GFX11-NEXT: s_setreg_imm32_b32 hwreg(HW_REG_MODE, 0, 2), 3
@@ -2063,7 +2086,7 @@ define amdgpu_ps void @scc_use_after_kill_inst(float inreg %x, i32 inreg %y) #0
; GFX11-NEXT: v_add_f32_e64 v1, s0, 1.0
; GFX11-NEXT: s_mov_b64 s[2:3], exec
; GFX11-NEXT: s_cmp_lg_u32 s1, 0
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_3) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_cmp_lt_f32_e32 vcc, 0, v1
; GFX11-NEXT: v_cndmask_b32_e64 v0, 0, -1.0, vcc
; GFX11-NEXT: v_cmp_nlt_f32_e32 vcc, 0, v1
@@ -2080,6 +2103,7 @@ define amdgpu_ps void @scc_use_after_kill_inst(float inreg %x, i32 inreg %y) #0
; GFX11-NEXT: s_waitcnt_vscnt null, 0x0
; GFX11-NEXT: .LBB17_3: ; %phibb
; GFX11-NEXT: v_cmp_eq_f32_e32 vcc, 0, v0
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_cbranch_vccz .LBB17_5
; GFX11-NEXT: ; %bb.4: ; %bb10
; GFX11-NEXT: v_mov_b32_e32 v0, 9
diff --git a/llvm/test/CodeGen/AMDGPU/ssubo.ll b/llvm/test/CodeGen/AMDGPU/ssubo.ll
index 7d08ac6222ecc1..572f49ba19e072 100644
--- a/llvm/test/CodeGen/AMDGPU/ssubo.ll
+++ b/llvm/test/CodeGen/AMDGPU/ssubo.ll
@@ -114,16 +114,16 @@ define amdgpu_kernel void @ssubo_i64_zext(ptr addrspace(1) %out, i64 %a, i64 %b)
; GFX11-NEXT: v_cmp_lt_i64_e64 s6, s[2:3], s[4:5]
; GFX11-NEXT: s_sub_u32 s4, s2, s4
; GFX11-NEXT: s_subb_u32 s5, s3, s5
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_cmp_lt_i32 s5, 0
; GFX11-NEXT: s_cselect_b32 s2, -1, 0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_xor_b32 s2, s6, s2
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_b32 s2, s2, exec_lo
; GFX11-NEXT: s_cselect_b64 s[2:3], 1, 0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_add_u32 s2, s4, s2
; GFX11-NEXT: s_addc_u32 s3, s5, s3
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: v_dual_mov_b32 v2, 0 :: v_dual_mov_b32 v1, s3
; GFX11-NEXT: v_mov_b32_e32 v0, s2
; GFX11-NEXT: global_store_b64 v2, v[0:1], s[0:1]
@@ -470,14 +470,15 @@ define amdgpu_kernel void @s_ssubo_i64(ptr addrspace(1) %out, ptr addrspace(1) %
; GFX11-NEXT: v_cmp_lt_i64_e64 s8, s[4:5], s[6:7]
; GFX11-NEXT: s_sub_u32 s4, s4, s6
; GFX11-NEXT: s_subb_u32 s5, s5, s7
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: v_dual_mov_b32 v2, 0 :: v_dual_mov_b32 v1, s5
; GFX11-NEXT: s_cmp_lt_i32 s5, 0
; GFX11-NEXT: s_cselect_b32 s6, -1, 0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_xor_b32 s6, s8, s6
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_b32 s6, s6, exec_lo
; GFX11-NEXT: s_cselect_b32 s6, 1, 0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: v_dual_mov_b32 v0, s4 :: v_dual_mov_b32 v3, s6
; GFX11-NEXT: s_clause 0x1
; GFX11-NEXT: global_store_b64 v2, v[0:1], s[0:1]
@@ -604,14 +605,14 @@ define amdgpu_kernel void @v_ssubo_i64(ptr addrspace(1) %out, ptr addrspace(1) %
; GFX11-NEXT: global_load_b64 v[2:3], v6, s[10:11]
; GFX11-NEXT: s_waitcnt vmcnt(0)
; GFX11-NEXT: v_sub_co_u32 v4, vcc_lo, v0, v2
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX11-NEXT: v_sub_co_ci_u32_e64 v5, null, v1, v3, vcc_lo
; GFX11-NEXT: v_cmp_lt_i64_e32 vcc_lo, v[0:1], v[2:3]
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: v_cmp_gt_i32_e64 s0, 0, v5
; GFX11-NEXT: s_xor_b32 s0, vcc_lo, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_and_b32 s0, s0, exec_lo
; GFX11-NEXT: s_cselect_b32 s0, 1, 0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: v_mov_b32_e32 v0, s0
; GFX11-NEXT: s_clause 0x1
; GFX11-NEXT: global_store_b64 v6, v[4:5], s[4:5]
diff --git a/llvm/test/CodeGen/AMDGPU/store-atomic-flat.ll b/llvm/test/CodeGen/AMDGPU/store-atomic-flat.ll
index 855091bb8a0053..7c019a7cd8a0b6 100644
--- a/llvm/test/CodeGen/AMDGPU/store-atomic-flat.ll
+++ b/llvm/test/CodeGen/AMDGPU/store-atomic-flat.ll
@@ -146,7 +146,6 @@ define amdgpu_cs void @atomic_store_i16x4_monotonic_agent_offset_min(<4 x i16> %
; GFX11-LABEL: atomic_store_i16x4_monotonic_agent_offset_min:
; GFX11: ; %bb.0:
; GFX11-NEXT: v_add_co_u32 v2, vcc_lo, 0xfffff000, v2
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v3, null, -1, v3, vcc_lo
; GFX11-NEXT: flat_store_b64 v[2:3], v[0:1]
; GFX11-NEXT: s_endpgm
diff --git a/llvm/test/CodeGen/AMDGPU/strict_fptrunc_bf16.ll b/llvm/test/CodeGen/AMDGPU/strict_fptrunc_bf16.ll
index 0eaa01d2685620..16d673fcdb4442 100644
--- a/llvm/test/CodeGen/AMDGPU/strict_fptrunc_bf16.ll
+++ b/llvm/test/CodeGen/AMDGPU/strict_fptrunc_bf16.ll
@@ -54,12 +54,12 @@ define amdgpu_ps void @strict_fptrunc_f64_to_bf16(double %a, ptr %out) #0 {
; GFX1250-NEXT: v_cvt_f64_f32_e32 v[4:5], v6
; GFX1250-NEXT: v_cmp_gt_f64_e64 s0, |v[0:1]|, |v[4:5]|
; GFX1250-NEXT: v_cmp_nlg_f64_e32 vcc_lo, v[0:1], v[4:5]
-; GFX1250-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1250-NEXT: v_cndmask_b32_e64 v0, -1, 1, s0
+; GFX1250-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1250-NEXT: v_dual_add_nc_u32 v0, v6, v0 :: v_dual_bitop2_b32 v7, 1, v6 bitop3:0x40
-; GFX1250-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_2) | instid1(VALU_DEP_1)
; GFX1250-NEXT: v_cmp_eq_u32_e64 s0, 1, v7
; GFX1250-NEXT: s_or_b32 vcc_lo, vcc_lo, s0
+; GFX1250-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1250-NEXT: v_cndmask_b32_e32 v0, v0, v6, vcc_lo
; GFX1250-NEXT: v_cvt_pk_bf16_f32 v0, v0, s0
; GFX1250-NEXT: flat_store_b16 v[2:3], v0
@@ -132,17 +132,17 @@ define amdgpu_ps void @strict_fptrunc_v2f64_to_v2bf16(<2 x double> %a, ptr %out)
; GFX1250-NEXT: v_cmp_gt_f64_e64 s1, |v[2:3]|, |v[6:7]|
; GFX1250-NEXT: v_cmp_nlg_f64_e32 vcc_lo, v[2:3], v[6:7]
; GFX1250-NEXT: v_cmp_nlg_f64_e64 s0, v[0:1], v[8:9]
-; GFX1250-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX1250-NEXT: v_cndmask_b32_e64 v2, -1, 1, s1
; GFX1250-NEXT: v_cmp_gt_f64_e64 s1, |v[0:1]|, |v[8:9]|
+; GFX1250-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1250-NEXT: v_dual_add_nc_u32 v1, v10, v2 :: v_dual_bitop2_b32 v13, 1, v11 bitop3:0x40
-; GFX1250-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_3)
; GFX1250-NEXT: v_cmp_ne_u32_e64 s2, 0, v13
; GFX1250-NEXT: v_cndmask_b32_e64 v0, -1, 1, s1
; GFX1250-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1250-NEXT: v_dual_add_nc_u32 v0, v11, v0 :: v_dual_bitop2_b32 v12, 1, v10 bitop3:0x40
; GFX1250-NEXT: v_cmp_ne_u32_e64 s1, 0, v12
; GFX1250-NEXT: s_or_b32 vcc_lo, s1, vcc_lo
+; GFX1250-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX1250-NEXT: v_cndmask_b32_e32 v1, v1, v10, vcc_lo
; GFX1250-NEXT: s_or_b32 vcc_lo, s2, s0
; GFX1250-NEXT: v_cndmask_b32_e32 v0, v0, v11, vcc_lo
diff --git a/llvm/test/CodeGen/AMDGPU/sub.ll b/llvm/test/CodeGen/AMDGPU/sub.ll
index 96e1a70d94f19b..d5cc6f53955fc2 100644
--- a/llvm/test/CodeGen/AMDGPU/sub.ll
+++ b/llvm/test/CodeGen/AMDGPU/sub.ll
@@ -780,7 +780,6 @@ define amdgpu_kernel void @v_sub_i64(ptr addrspace(1) noalias %out, ptr addrspac
; GFX12-NEXT: global_load_b64 v[2:3], v2, s[4:5]
; GFX12-NEXT: s_wait_loadcnt 0x0
; GFX12-NEXT: v_sub_co_u32 v0, vcc_lo, v0, v2
-; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-NEXT: v_sub_co_ci_u32_e64 v1, null, v1, v3, vcc_lo
; GFX12-NEXT: global_store_b64 v4, v[0:1], s[0:1]
; GFX12-NEXT: s_endpgm
@@ -876,7 +875,6 @@ define amdgpu_kernel void @v_test_sub_v2i64(ptr addrspace(1) %out, ptr addrspace
; GFX12-NEXT: global_load_b128 v[4:7], v4, s[4:5]
; GFX12-NEXT: s_wait_loadcnt 0x0
; GFX12-NEXT: v_sub_co_u32 v2, vcc_lo, v2, v6
-; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-NEXT: v_sub_co_ci_u32_e64 v3, null, v3, v7, vcc_lo
; GFX12-NEXT: v_sub_co_u32 v0, vcc_lo, v0, v4
; GFX12-NEXT: s_wait_alu depctr_va_vcc(0)
@@ -1009,7 +1007,6 @@ define amdgpu_kernel void @v_test_sub_v4i64(ptr addrspace(1) %out, ptr addrspace
; GFX12-NEXT: global_load_b128 v[12:15], v12, s[6:7] offset:16
; GFX12-NEXT: s_wait_loadcnt 0x2
; GFX12-NEXT: v_sub_co_u32 v2, vcc_lo, v6, v2
-; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-NEXT: v_sub_co_ci_u32_e64 v3, null, v7, v3, vcc_lo
; GFX12-NEXT: s_wait_loadcnt 0x0
; GFX12-NEXT: v_sub_co_u32 v10, vcc_lo, v10, v14
diff --git a/llvm/test/CodeGen/AMDGPU/sub_i1.ll b/llvm/test/CodeGen/AMDGPU/sub_i1.ll
index f817b19cae0a46..e8fba4bc973328 100644
--- a/llvm/test/CodeGen/AMDGPU/sub_i1.ll
+++ b/llvm/test/CodeGen/AMDGPU/sub_i1.ll
@@ -201,9 +201,10 @@ define amdgpu_kernel void @sub_i1_cf(ptr addrspace(1) %out, ptr addrspace(1) %a,
; GFX11-NEXT: v_and_b32_e32 v0, 0x3ff, v0
; GFX11-NEXT: s_mov_b32 s7, exec_lo
; GFX11-NEXT: ; implicit-def: $sgpr6
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_cmpx_lt_u32_e32 15, v0
; GFX11-NEXT: s_xor_b32 s7, exec_lo, s7
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: s_cbranch_execz .LBB2_2
; GFX11-NEXT: ; %bb.1: ; %else
; GFX11-NEXT: v_mov_b32_e32 v0, 0
diff --git a/llvm/test/CodeGen/AMDGPU/sub_u64.ll b/llvm/test/CodeGen/AMDGPU/sub_u64.ll
index 6ec6d9fbdde062..fca93f8ee9f32b 100644
--- a/llvm/test/CodeGen/AMDGPU/sub_u64.ll
+++ b/llvm/test/CodeGen/AMDGPU/sub_u64.ll
@@ -7,7 +7,6 @@ define amdgpu_ps <2 x float> @test_sub_u64_vv(i64 %a, i64 %b) {
; GFX12-LABEL: test_sub_u64_vv:
; GFX12: ; %bb.0:
; GFX12-NEXT: v_sub_co_u32 v0, vcc_lo, v0, v2
-; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-NEXT: v_sub_co_ci_u32_e64 v1, null, v1, v3, vcc_lo
; GFX12-NEXT: ; return to shader part epilog
;
@@ -28,7 +27,6 @@ define amdgpu_ps <2 x float> @test_sub_u64_vs(i64 %a, i64 inreg %b) {
; GFX12-LABEL: test_sub_u64_vs:
; GFX12: ; %bb.0:
; GFX12-NEXT: v_sub_co_u32 v0, vcc_lo, v0, s0
-; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-NEXT: v_subrev_co_ci_u32_e64 v1, null, s1, v1, vcc_lo
; GFX12-NEXT: ; return to shader part epilog
;
@@ -49,7 +47,6 @@ define amdgpu_ps <2 x float> @test_sub_u64_sv(i64 inreg %a, i64 %b) {
; GFX12-LABEL: test_sub_u64_sv:
; GFX12: ; %bb.0:
; GFX12-NEXT: v_sub_co_u32 v0, vcc_lo, s0, v0
-; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-NEXT: v_sub_co_ci_u32_e64 v1, null, s1, v1, vcc_lo
; GFX12-NEXT: ; return to shader part epilog
;
@@ -93,7 +90,6 @@ define amdgpu_ps <2 x float> @test_sub_u64_inline_lit_v(i64 %a) {
; GFX12-LABEL: test_sub_u64_inline_lit_v:
; GFX12: ; %bb.0:
; GFX12-NEXT: v_sub_co_u32 v0, vcc_lo, 5, v0
-; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-NEXT: v_sub_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX12-NEXT: ; return to shader part epilog
;
@@ -114,7 +110,6 @@ define amdgpu_ps <2 x float> @test_sub_u64_v_inline_lit(i64 %a) {
; GFX12-LABEL: test_sub_u64_v_inline_lit:
; GFX12: ; %bb.0:
; GFX12-NEXT: v_add_co_u32 v0, vcc_lo, v0, -5
-; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-NEXT: v_add_co_ci_u32_e64 v1, null, -1, v1, vcc_lo
; GFX12-NEXT: ; return to shader part epilog
;
@@ -135,7 +130,6 @@ define amdgpu_ps <2 x float> @test_sub_u64_small_imm_v(i64 %a) {
; GFX12-LABEL: test_sub_u64_small_imm_v:
; GFX12: ; %bb.0:
; GFX12-NEXT: v_sub_co_u32 v0, vcc_lo, 0x1f4, v0
-; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-NEXT: v_sub_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
; GFX12-NEXT: ; return to shader part epilog
;
@@ -156,7 +150,6 @@ define amdgpu_ps <2 x float> @test_sub_u64_64bit_imm_v(i64 %a) {
; GFX12-LABEL: test_sub_u64_64bit_imm_v:
; GFX12: ; %bb.0:
; GFX12-NEXT: v_sub_co_u32 v0, vcc_lo, 0x3b9ac9ff, v0
-; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-NEXT: v_sub_co_ci_u32_e64 v1, null, 1, v1, vcc_lo
; GFX12-NEXT: ; return to shader part epilog
;
diff --git a/llvm/test/CodeGen/AMDGPU/trap-abis.ll b/llvm/test/CodeGen/AMDGPU/trap-abis.ll
index 5a359a5b2f3ccc..b6afe4f283a520 100644
--- a/llvm/test/CodeGen/AMDGPU/trap-abis.ll
+++ b/llvm/test/CodeGen/AMDGPU/trap-abis.ll
@@ -70,6 +70,7 @@ define amdgpu_kernel void @trap(ptr addrspace(1) nocapture readonly %arg0) {
; HSA-TRAP-GFX1100-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; HSA-TRAP-GFX1100-NEXT: s_bitset1_b32 s0, 10
; HSA-TRAP-GFX1100-NEXT: s_mov_b32 m0, s0
+; HSA-TRAP-GFX1100-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; HSA-TRAP-GFX1100-NEXT: s_sendmsg sendmsg(MSG_INTERRUPT)
; HSA-TRAP-GFX1100-NEXT: s_mov_b32 m0, ttmp2
; HSA-TRAP-GFX1100-NEXT: .LBB0_1: ; =>This Inner Loop Header: Depth=1
@@ -204,6 +205,7 @@ define amdgpu_kernel void @non_entry_trap(ptr addrspace(1) nocapture readonly %a
; HSA-TRAP-GFX1100-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; HSA-TRAP-GFX1100-NEXT: s_bitset1_b32 s0, 10
; HSA-TRAP-GFX1100-NEXT: s_mov_b32 m0, s0
+; HSA-TRAP-GFX1100-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; HSA-TRAP-GFX1100-NEXT: s_sendmsg sendmsg(MSG_INTERRUPT)
; HSA-TRAP-GFX1100-NEXT: s_mov_b32 m0, ttmp2
; HSA-TRAP-GFX1100-NEXT: .LBB1_3: ; =>This Inner Loop Header: Depth=1
@@ -351,6 +353,7 @@ define amdgpu_kernel void @trap_with_use_after(ptr addrspace(1) %arg0, ptr addrs
; HSA-TRAP-GFX1100-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; HSA-TRAP-GFX1100-NEXT: s_bitset1_b32 s0, 10
; HSA-TRAP-GFX1100-NEXT: s_mov_b32 m0, s0
+; HSA-TRAP-GFX1100-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; HSA-TRAP-GFX1100-NEXT: s_sendmsg sendmsg(MSG_INTERRUPT)
; HSA-TRAP-GFX1100-NEXT: s_mov_b32 m0, ttmp2
; HSA-TRAP-GFX1100-NEXT: .LBB2_3: ; =>This Inner Loop Header: Depth=1
diff --git a/llvm/test/CodeGen/AMDGPU/uaddo.ll b/llvm/test/CodeGen/AMDGPU/uaddo.ll
index b2326ee18d8067..b929b7c3135d20 100644
--- a/llvm/test/CodeGen/AMDGPU/uaddo.ll
+++ b/llvm/test/CodeGen/AMDGPU/uaddo.ll
@@ -1037,6 +1037,7 @@ define amdgpu_kernel void @s_uaddo_clamp_bit(ptr addrspace(1) %out, ptr addrspac
; GFX11-NEXT: s_waitcnt lgkmcnt(0)
; GFX11-NEXT: v_add_co_u32 v0, s1, s2, s3
; GFX11-NEXT: s_cmp_eq_u32 s2, s3
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB8_2
; GFX11-NEXT: ; %bb.1: ; %if
; GFX11-NEXT: s_xor_b32 s0, s1, -1
diff --git a/llvm/test/CodeGen/AMDGPU/ucmp.ll b/llvm/test/CodeGen/AMDGPU/ucmp.ll
index 2eef0cd806efb0..ec0153be3d6362 100644
--- a/llvm/test/CodeGen/AMDGPU/ucmp.ll
+++ b/llvm/test/CodeGen/AMDGPU/ucmp.ll
@@ -93,19 +93,17 @@ define i32 @ucmp_i128(i128 %a, i128 %b) {
; GFX12-GISEL-NEXT: v_cndmask_b32_e64 v8, 0, 1, vcc_lo
; GFX12-GISEL-NEXT: v_cmp_gt_u64_e32 vcc_lo, v[2:3], v[6:7]
; GFX12-GISEL-NEXT: s_wait_alu depctr_va_sdst(0)
-; GFX12-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX12-GISEL-NEXT: v_cndmask_b32_e64 v0, 0, 1, s0
; GFX12-GISEL-NEXT: v_cmp_lt_u64_e64 s0, v[2:3], v[6:7]
; GFX12-GISEL-NEXT: s_wait_alu depctr_va_vcc(0)
; GFX12-GISEL-NEXT: v_cndmask_b32_e64 v9, 0, 1, vcc_lo
; GFX12-GISEL-NEXT: v_cmp_eq_u64_e32 vcc_lo, v[2:3], v[6:7]
; GFX12-GISEL-NEXT: s_wait_alu depctr_va_sdst(0)
-; GFX12-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_2) | instid1(VALU_DEP_2)
; GFX12-GISEL-NEXT: v_cndmask_b32_e64 v1, 0, 1, s0
; GFX12-GISEL-NEXT: s_wait_alu depctr_va_vcc(0)
; GFX12-GISEL-NEXT: v_cndmask_b32_e32 v2, v9, v8, vcc_lo
+; GFX12-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX12-GISEL-NEXT: v_cndmask_b32_e32 v0, v1, v0, vcc_lo
-; GFX12-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-GISEL-NEXT: v_sub_nc_u32_e32 v0, v2, v0
; GFX12-GISEL-NEXT: s_setpc_b64 s[30:31]
%r = call i32 @llvm.ucmp.i32.i128(i128 %a, i128 %b)
@@ -699,19 +697,23 @@ define i32 @ucmp_i128_uniform(i128 inreg %a, i128 inreg %b) {
; GFX12-GISEL-NEXT: v_cmp_lt_u64_e64 s0, s[0:1], s[16:17]
; GFX12-GISEL-NEXT: v_cmp_lt_u64_e64 s6, s[2:3], s[18:19]
; GFX12-GISEL-NEXT: s_cmp_lg_u32 s4, 0
+; GFX12-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX12-GISEL-NEXT: s_cselect_b32 s4, 1, 0
; GFX12-GISEL-NEXT: s_cmp_lg_u32 s5, 0
; GFX12-GISEL-NEXT: s_cselect_b32 s5, 1, 0
; GFX12-GISEL-NEXT: s_cmp_eq_u64 s[2:3], s[18:19]
+; GFX12-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX12-GISEL-NEXT: s_cselect_b32 s1, 1, 0
; GFX12-GISEL-NEXT: s_cmp_lg_u32 s0, 0
; GFX12-GISEL-NEXT: s_cselect_b32 s0, 1, 0
; GFX12-GISEL-NEXT: s_cmp_lg_u32 s6, 0
+; GFX12-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX12-GISEL-NEXT: s_cselect_b32 s6, 1, 0
; GFX12-GISEL-NEXT: s_cmp_eq_u64 s[2:3], s[18:19]
; GFX12-GISEL-NEXT: s_cselect_b32 s2, 1, 0
; GFX12-GISEL-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-GISEL-NEXT: s_cmp_lg_u32 s1, 0
+; GFX12-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX12-GISEL-NEXT: s_cselect_b32 s1, s4, s5
; GFX12-GISEL-NEXT: s_cmp_lg_u32 s2, 0
; GFX12-GISEL-NEXT: s_cselect_b32 s0, s0, s6
@@ -786,11 +788,13 @@ define i32 @ucmp_i64_uniform(i64 inreg %a, i64 inreg %b) {
; GFX12-GISEL-NEXT: v_cmp_gt_u64_e64 s4, s[0:1], s[2:3]
; GFX12-GISEL-NEXT: v_cmp_lt_u64_e64 s0, s[0:1], s[2:3]
; GFX12-GISEL-NEXT: s_cmp_lg_u32 s4, 0
+; GFX12-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX12-GISEL-NEXT: s_cselect_b32 s1, 1, 0
; GFX12-GISEL-NEXT: s_cmp_lg_u32 s0, 0
; GFX12-GISEL-NEXT: s_cselect_b32 s0, 1, 0
; GFX12-GISEL-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-GISEL-NEXT: s_cmp_lg_u32 s1, 0
+; GFX12-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX12-GISEL-NEXT: s_cselect_b32 s1, 1, 0
; GFX12-GISEL-NEXT: s_cmp_lg_u32 s0, 0
; GFX12-GISEL-NEXT: s_cselect_b32 s0, 1, 0
@@ -837,6 +841,7 @@ define i32 @ucmp_i32_uniform(i32 inreg %a, i32 inreg %b) {
; GFX12-SDAG-NEXT: s_wait_bvhcnt 0x0
; GFX12-SDAG-NEXT: s_wait_kmcnt 0x0
; GFX12-SDAG-NEXT: s_cmp_gt_u32 s0, s1
+; GFX12-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-SDAG-NEXT: s_cselect_b32 s2, 1, 0
; GFX12-SDAG-NEXT: s_cmp_ge_u32 s0, s1
; GFX12-SDAG-NEXT: s_wait_alu depctr_sa_sdst(0)
@@ -853,11 +858,13 @@ define i32 @ucmp_i32_uniform(i32 inreg %a, i32 inreg %b) {
; GFX12-GISEL-NEXT: s_wait_bvhcnt 0x0
; GFX12-GISEL-NEXT: s_wait_kmcnt 0x0
; GFX12-GISEL-NEXT: s_cmp_gt_u32 s0, s1
+; GFX12-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX12-GISEL-NEXT: s_cselect_b32 s2, 1, 0
; GFX12-GISEL-NEXT: s_cmp_lt_u32 s0, s1
; GFX12-GISEL-NEXT: s_cselect_b32 s0, 1, 0
; GFX12-GISEL-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-GISEL-NEXT: s_cmp_lg_u32 s2, 0
+; GFX12-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX12-GISEL-NEXT: s_cselect_b32 s1, 1, 0
; GFX12-GISEL-NEXT: s_cmp_lg_u32 s0, 0
; GFX12-GISEL-NEXT: s_cselect_b32 s0, 1, 0
@@ -911,6 +918,7 @@ define i32 @ucmp_i16_uniform(i16 inreg %a, i16 inreg %b) {
; GFX12-SDAG-NEXT: s_and_b32 s0, 0xffff, s0
; GFX12-SDAG-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-SDAG-NEXT: s_cmp_gt_u32 s0, s1
+; GFX12-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-SDAG-NEXT: s_cselect_b32 s2, 1, 0
; GFX12-SDAG-NEXT: s_cmp_ge_u32 s0, s1
; GFX12-SDAG-NEXT: s_wait_alu depctr_sa_sdst(0)
@@ -930,11 +938,13 @@ define i32 @ucmp_i16_uniform(i16 inreg %a, i16 inreg %b) {
; GFX12-GISEL-NEXT: s_and_b32 s1, 0xffff, s1
; GFX12-GISEL-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-GISEL-NEXT: s_cmp_gt_u32 s0, s1
+; GFX12-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX12-GISEL-NEXT: s_cselect_b32 s2, 1, 0
; GFX12-GISEL-NEXT: s_cmp_lt_u32 s0, s1
; GFX12-GISEL-NEXT: s_cselect_b32 s0, 1, 0
; GFX12-GISEL-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-GISEL-NEXT: s_cmp_lg_u32 s2, 0
+; GFX12-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX12-GISEL-NEXT: s_cselect_b32 s1, 1, 0
; GFX12-GISEL-NEXT: s_cmp_lg_u32 s0, 0
; GFX12-GISEL-NEXT: s_cselect_b32 s0, 1, 0
@@ -988,6 +998,7 @@ define i32 @ucmp_i8_uniform(i8 inreg %a, i8 inreg %b) {
; GFX12-SDAG-NEXT: s_and_b32 s0, s0, 0xff
; GFX12-SDAG-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-SDAG-NEXT: s_cmp_gt_u32 s0, s1
+; GFX12-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-SDAG-NEXT: s_cselect_b32 s2, 1, 0
; GFX12-SDAG-NEXT: s_cmp_ge_u32 s0, s1
; GFX12-SDAG-NEXT: s_wait_alu depctr_sa_sdst(0)
@@ -1007,11 +1018,13 @@ define i32 @ucmp_i8_uniform(i8 inreg %a, i8 inreg %b) {
; GFX12-GISEL-NEXT: s_and_b32 s1, s1, 0xff
; GFX12-GISEL-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-GISEL-NEXT: s_cmp_gt_u32 s0, s1
+; GFX12-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX12-GISEL-NEXT: s_cselect_b32 s2, 1, 0
; GFX12-GISEL-NEXT: s_cmp_lt_u32 s0, s1
; GFX12-GISEL-NEXT: s_cselect_b32 s0, 1, 0
; GFX12-GISEL-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-GISEL-NEXT: s_cmp_lg_u32 s2, 0
+; GFX12-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX12-GISEL-NEXT: s_cselect_b32 s1, 1, 0
; GFX12-GISEL-NEXT: s_cmp_lg_u32 s0, 0
; GFX12-GISEL-NEXT: s_cselect_b32 s0, 1, 0
@@ -1084,6 +1097,7 @@ define <2 x i32> @ucmp_v2i16_uniform(<2 x i16> inreg %a, <2 x i16> inreg %b) {
; GFX12-SDAG-NEXT: s_and_b32 s3, s0, 0xffff
; GFX12-SDAG-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-SDAG-NEXT: s_cmp_gt_u32 s3, s2
+; GFX12-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-SDAG-NEXT: s_cselect_b32 s4, 1, 0
; GFX12-SDAG-NEXT: s_cmp_ge_u32 s3, s2
; GFX12-SDAG-NEXT: s_wait_alu depctr_sa_sdst(0)
@@ -1092,6 +1106,7 @@ define <2 x i32> @ucmp_v2i16_uniform(<2 x i16> inreg %a, <2 x i16> inreg %b) {
; GFX12-SDAG-NEXT: s_lshr_b32 s0, s0, 16
; GFX12-SDAG-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-SDAG-NEXT: s_cmp_gt_u32 s0, s1
+; GFX12-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-SDAG-NEXT: s_cselect_b32 s3, 1, 0
; GFX12-SDAG-NEXT: s_cmp_ge_u32 s0, s1
; GFX12-SDAG-NEXT: s_wait_alu depctr_sa_sdst(0)
@@ -1113,19 +1128,23 @@ define <2 x i32> @ucmp_v2i16_uniform(<2 x i16> inreg %a, <2 x i16> inreg %b) {
; GFX12-GISEL-NEXT: s_lshr_b32 s1, s1, 16
; GFX12-GISEL-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-GISEL-NEXT: s_cmp_gt_u32 s0, s3
+; GFX12-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX12-GISEL-NEXT: s_cselect_b32 s4, 1, 0
; GFX12-GISEL-NEXT: s_cmp_gt_u32 s2, s1
; GFX12-GISEL-NEXT: s_cselect_b32 s5, 1, 0
; GFX12-GISEL-NEXT: s_cmp_lt_u32 s0, s3
+; GFX12-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX12-GISEL-NEXT: s_cselect_b32 s0, 1, 0
; GFX12-GISEL-NEXT: s_cmp_lt_u32 s2, s1
; GFX12-GISEL-NEXT: s_cselect_b32 s1, 1, 0
; GFX12-GISEL-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-GISEL-NEXT: s_cmp_lg_u32 s4, 0
+; GFX12-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX12-GISEL-NEXT: s_cselect_b32 s2, 1, 0
; GFX12-GISEL-NEXT: s_cmp_lg_u32 s5, 0
; GFX12-GISEL-NEXT: s_cselect_b32 s3, 1, 0
; GFX12-GISEL-NEXT: s_cmp_lg_u32 s0, 0
+; GFX12-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX12-GISEL-NEXT: s_cselect_b32 s0, 1, 0
; GFX12-GISEL-NEXT: s_cmp_lg_u32 s1, 0
; GFX12-GISEL-NEXT: s_cselect_b32 s1, 1, 0
@@ -1243,6 +1262,7 @@ define <4 x i32> @ucmp_v4i8_uniform(<4 x i8> inreg %a, <4 x i8> inreg %b) {
; GFX12-SDAG-NEXT: s_and_b32 s1, s1, 0xff
; GFX12-SDAG-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-SDAG-NEXT: s_cmp_gt_u32 s0, s7
+; GFX12-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_4) | instid1(SALU_CYCLE_1)
; GFX12-SDAG-NEXT: s_cselect_b32 s8, 1, 0
; GFX12-SDAG-NEXT: s_cmp_ge_u32 s0, s7
; GFX12-SDAG-NEXT: s_wait_alu depctr_sa_sdst(0)
@@ -1260,6 +1280,7 @@ define <4 x i32> @ucmp_v4i8_uniform(<4 x i8> inreg %a, <4 x i8> inreg %b) {
; GFX12-SDAG-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-SDAG-NEXT: s_cselect_b32 s2, s6, -1
; GFX12-SDAG-NEXT: s_cmp_gt_u32 s3, s4
+; GFX12-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-SDAG-NEXT: s_cselect_b32 s5, 1, 0
; GFX12-SDAG-NEXT: s_cmp_ge_u32 s3, s4
; GFX12-SDAG-NEXT: s_wait_alu depctr_sa_sdst(0)
@@ -1279,6 +1300,7 @@ define <4 x i32> @ucmp_v4i8_uniform(<4 x i8> inreg %a, <4 x i8> inreg %b) {
; GFX12-GISEL-NEXT: s_and_b32 s4, s16, 0xff
; GFX12-GISEL-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-GISEL-NEXT: s_cmp_gt_u32 s0, s4
+; GFX12-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_4) | instid1(SALU_CYCLE_1)
; GFX12-GISEL-NEXT: s_cselect_b32 s5, 1, 0
; GFX12-GISEL-NEXT: s_and_b32 s1, s1, 0xff
; GFX12-GISEL-NEXT: s_and_b32 s6, s17, 0xff
@@ -1289,6 +1311,7 @@ define <4 x i32> @ucmp_v4i8_uniform(<4 x i8> inreg %a, <4 x i8> inreg %b) {
; GFX12-GISEL-NEXT: s_and_b32 s8, s18, 0xff
; GFX12-GISEL-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-GISEL-NEXT: s_cmp_gt_u32 s2, s8
+; GFX12-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_4) | instid1(SALU_CYCLE_1)
; GFX12-GISEL-NEXT: s_cselect_b32 s9, 1, 0
; GFX12-GISEL-NEXT: s_and_b32 s3, s3, 0xff
; GFX12-GISEL-NEXT: s_and_b32 s10, s19, 0xff
@@ -1296,27 +1319,33 @@ define <4 x i32> @ucmp_v4i8_uniform(<4 x i8> inreg %a, <4 x i8> inreg %b) {
; GFX12-GISEL-NEXT: s_cmp_gt_u32 s3, s10
; GFX12-GISEL-NEXT: s_cselect_b32 s11, 1, 0
; GFX12-GISEL-NEXT: s_cmp_lt_u32 s0, s4
+; GFX12-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX12-GISEL-NEXT: s_cselect_b32 s0, 1, 0
; GFX12-GISEL-NEXT: s_cmp_lt_u32 s1, s6
; GFX12-GISEL-NEXT: s_cselect_b32 s1, 1, 0
; GFX12-GISEL-NEXT: s_cmp_lt_u32 s2, s8
+; GFX12-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX12-GISEL-NEXT: s_cselect_b32 s2, 1, 0
; GFX12-GISEL-NEXT: s_cmp_lt_u32 s3, s10
; GFX12-GISEL-NEXT: s_cselect_b32 s3, 1, 0
; GFX12-GISEL-NEXT: s_cmp_lg_u32 s5, 0
+; GFX12-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX12-GISEL-NEXT: s_cselect_b32 s4, 1, 0
; GFX12-GISEL-NEXT: s_cmp_lg_u32 s7, 0
; GFX12-GISEL-NEXT: s_cselect_b32 s5, 1, 0
; GFX12-GISEL-NEXT: s_cmp_lg_u32 s9, 0
+; GFX12-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
; GFX12-GISEL-NEXT: s_cselect_b32 s6, 1, 0
; GFX12-GISEL-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-GISEL-NEXT: s_cmp_lg_u32 s11, 0
; GFX12-GISEL-NEXT: s_cselect_b32 s7, 1, 0
; GFX12-GISEL-NEXT: s_cmp_lg_u32 s0, 0
+; GFX12-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX12-GISEL-NEXT: s_cselect_b32 s0, 1, 0
; GFX12-GISEL-NEXT: s_cmp_lg_u32 s1, 0
; GFX12-GISEL-NEXT: s_cselect_b32 s1, 1, 0
; GFX12-GISEL-NEXT: s_cmp_lg_u32 s2, 0
+; GFX12-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX12-GISEL-NEXT: s_cselect_b32 s2, 1, 0
; GFX12-GISEL-NEXT: s_cmp_lg_u32 s3, 0
; GFX12-GISEL-NEXT: s_cselect_b32 s3, 1, 0
diff --git a/llvm/test/CodeGen/AMDGPU/umin-sub-to-usubo-select-combine.ll b/llvm/test/CodeGen/AMDGPU/umin-sub-to-usubo-select-combine.ll
index 81a4d41ef7388a..ed86b57d749ad3 100644
--- a/llvm/test/CodeGen/AMDGPU/umin-sub-to-usubo-select-combine.ll
+++ b/llvm/test/CodeGen/AMDGPU/umin-sub-to-usubo-select-combine.ll
@@ -101,8 +101,8 @@ define i64 @v_underflow_compare_fold_i64(i64 %a, i64 %b) #0 {
; GFX11: ; %bb.0:
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: v_sub_co_u32 v2, vcc_lo, v0, v2
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_sub_co_ci_u32_e64 v3, null, v1, v3, vcc_lo
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_cmp_lt_u64_e32 vcc_lo, v[2:3], v[0:1]
; GFX11-NEXT: v_dual_cndmask_b32 v0, v0, v2 :: v_dual_cndmask_b32 v1, v1, v3
; GFX11-NEXT: s_setpc_b64 s[30:31]
@@ -126,8 +126,8 @@ define i64 @v_underflow_compare_fold_i64_commute(i64 %a, i64 %b) #0 {
; GFX11: ; %bb.0:
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: v_sub_co_u32 v2, vcc_lo, v0, v2
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_sub_co_ci_u32_e64 v3, null, v1, v3, vcc_lo
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_cmp_lt_u64_e32 vcc_lo, v[0:1], v[2:3]
; GFX11-NEXT: v_dual_cndmask_b32 v0, v2, v0 :: v_dual_cndmask_b32 v1, v3, v1
; GFX11-NEXT: s_setpc_b64 s[30:31]
@@ -153,8 +153,8 @@ define i64 @v_underflow_compare_fold_i64_multi_use(i64 %a, i64 %b, ptr addrspace
; GFX11: ; %bb.0:
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: v_sub_co_u32 v2, vcc_lo, v0, v2
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_sub_co_ci_u32_e64 v3, null, v1, v3, vcc_lo
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_cmp_lt_u64_e32 vcc_lo, v[2:3], v[0:1]
; GFX11-NEXT: global_store_b64 v[4:5], v[2:3], off
; GFX11-NEXT: v_dual_cndmask_b32 v0, v0, v2 :: v_dual_cndmask_b32 v1, v1, v3
diff --git a/llvm/test/CodeGen/AMDGPU/uniform-inside-divergent-cfg.ll b/llvm/test/CodeGen/AMDGPU/uniform-inside-divergent-cfg.ll
index a2e272d9127e4a..db67b54297c761 100644
--- a/llvm/test/CodeGen/AMDGPU/uniform-inside-divergent-cfg.ll
+++ b/llvm/test/CodeGen/AMDGPU/uniform-inside-divergent-cfg.ll
@@ -16,15 +16,18 @@ define amdgpu_kernel void @div_unif_div(ptr addrspace(1) %out, float %ubeta, i32
; CHECK-NEXT: v_cmp_eq_f32_e64 s0, s0, 0
; CHECK-NEXT: s_mov_b32 s1, exec_lo
; CHECK-NEXT: s_and_b32 vcc_lo, exec_lo, s0
+; CHECK-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; CHECK-NEXT: s_cbranch_vccnz .LBB0_7
; CHECK-NEXT: ; %bb.2: ; %B
; CHECK-NEXT: s_mov_b32 s0, exec_lo
; CHECK-NEXT: ; implicit-def: $vgpr1
; CHECK-NEXT: v_cmpx_lt_u32_e32 2, v0
+; CHECK-NEXT: s_delay_alu instid0(VALU_DEP_1)
; CHECK-NEXT: s_xor_b32 s0, exec_lo, s0
; CHECK-NEXT: ; %bb.3: ; %B2
; CHECK-NEXT: v_mul_u32_u24_e32 v1, 3, v0
; CHECK-NEXT: ; %bb.4: ; %Flow
+; CHECK-NEXT: s_delay_alu instid0(VALU_DEP_2)
; CHECK-NEXT: s_and_not1_saveexec_b32 s0, s0
; CHECK-NEXT: ; %bb.5: ; %B1
; CHECK-NEXT: v_add_nc_u32_e32 v1, 7, v0
@@ -37,6 +40,7 @@ define amdgpu_kernel void @div_unif_div(ptr addrspace(1) %out, float %ubeta, i32
; CHECK-NEXT: v_mov_b32_e32 v1, 1
; CHECK-NEXT: .LBB0_8: ; %join
; CHECK-NEXT: s_or_b32 exec_lo, exec_lo, s2
+; CHECK-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; CHECK-NEXT: s_and_saveexec_b32 s0, s1
; CHECK-NEXT: s_cbranch_execz .LBB0_10
; CHECK-NEXT: ; %bb.9: ; %do.store
@@ -96,10 +100,12 @@ define amdgpu_kernel void @unif_div(ptr addrspace(1) %out, float %u1, float %u2,
; CHECK-NEXT: ; %bb.1: ; %A
; CHECK-NEXT: v_cmp_eq_f32_e64 s0, s0, 0
; CHECK-NEXT: s_and_b32 vcc_lo, exec_lo, s0
+; CHECK-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; CHECK-NEXT: s_cbranch_vccnz .LBB1_4
; CHECK-NEXT: ; %bb.2: ; %B
; CHECK-NEXT: v_cmp_eq_f32_e64 s0, s1, 0
; CHECK-NEXT: s_and_b32 vcc_lo, exec_lo, s0
+; CHECK-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; CHECK-NEXT: s_cbranch_vccnz .LBB1_5
; CHECK-NEXT: ; %bb.3: ; %C
; CHECK-NEXT: v_add_nc_u32_e32 v1, 7, v0
@@ -114,6 +120,7 @@ define amdgpu_kernel void @unif_div(ptr addrspace(1) %out, float %u1, float %u2,
; CHECK-NEXT: s_or_b32 s3, exec_lo, exec_lo
; CHECK-NEXT: .LBB1_7: ; %join
; CHECK-NEXT: s_or_b32 exec_lo, exec_lo, s2
+; CHECK-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; CHECK-NEXT: s_and_saveexec_b32 s0, s3
; CHECK-NEXT: s_cbranch_execz .LBB1_9
; CHECK-NEXT: ; %bb.8: ; %do.store
diff --git a/llvm/test/CodeGen/AMDGPU/uniform-select.ll b/llvm/test/CodeGen/AMDGPU/uniform-select.ll
index d5119e6f2bdd3d..b50c88e64631c9 100644
--- a/llvm/test/CodeGen/AMDGPU/uniform-select.ll
+++ b/llvm/test/CodeGen/AMDGPU/uniform-select.ll
@@ -163,39 +163,42 @@ define amdgpu_kernel void @test_insert_extract(i32 %p, i32 %q) {
; GFX1100-NEXT: ; =>This Inner Loop Header: Depth=1
; GFX1100-NEXT: s_waitcnt lgkmcnt(0)
; GFX1100-NEXT: s_cmp_eq_u32 s1, 1
+; GFX1100-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1100-NEXT: s_cselect_b32 s7, -1, 0
-; GFX1100-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_3) | instid1(SALU_CYCLE_1)
; GFX1100-NEXT: s_and_b32 s7, s7, exec_lo
; GFX1100-NEXT: s_cselect_b32 s7, s3, s2
; GFX1100-NEXT: s_cmp_eq_u32 s1, 2
+; GFX1100-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1100-NEXT: s_cselect_b32 s8, -1, 0
; GFX1100-NEXT: s_and_b32 s8, s8, exec_lo
; GFX1100-NEXT: s_cselect_b32 s7, s4, s7
; GFX1100-NEXT: s_cmp_eq_u32 s1, 3
+; GFX1100-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1100-NEXT: s_cselect_b32 s8, -1, 0
-; GFX1100-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX1100-NEXT: s_and_b32 s8, s8, exec_lo
; GFX1100-NEXT: s_cselect_b32 s7, s5, s7
+; GFX1100-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX1100-NEXT: s_or_b32 s7, s7, s0
; GFX1100-NEXT: s_cmp_eq_u32 s1, 1
; GFX1100-NEXT: s_cselect_b32 s8, -1, 0
-; GFX1100-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_3) | instid1(SALU_CYCLE_1)
+; GFX1100-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
; GFX1100-NEXT: s_and_b32 s9, s8, exec_lo
; GFX1100-NEXT: s_cselect_b32 s3, s7, s3
; GFX1100-NEXT: s_cmp_eq_u32 s1, 3
; GFX1100-NEXT: s_cselect_b32 s9, -1, 0
+; GFX1100-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
; GFX1100-NEXT: s_and_b32 s10, s9, exec_lo
; GFX1100-NEXT: s_cselect_b32 s5, s7, s5
; GFX1100-NEXT: s_cmp_eq_u32 s1, 2
; GFX1100-NEXT: s_cselect_b32 s10, -1, 0
-; GFX1100-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_4) | instid1(SALU_CYCLE_1)
+; GFX1100-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
; GFX1100-NEXT: s_and_b32 s11, s10, exec_lo
; GFX1100-NEXT: s_cselect_b32 s4, s7, s4
; GFX1100-NEXT: s_cmp_eq_u32 s1, 0
; GFX1100-NEXT: s_cselect_b32 s2, s7, s2
; GFX1100-NEXT: s_or_b32 s7, s10, s8
+; GFX1100-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1100-NEXT: s_or_b32 s7, s9, s7
-; GFX1100-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1100-NEXT: s_and_b32 s7, s7, exec_lo
; GFX1100-NEXT: s_cselect_b32 s6, 0, s6
; GFX1100-NEXT: s_cbranch_vccnz .LBB0_1
diff --git a/llvm/test/CodeGen/AMDGPU/uniform-vgpr-to-sgpr-return.ll b/llvm/test/CodeGen/AMDGPU/uniform-vgpr-to-sgpr-return.ll
index a6a134478889ea..3e492e6e375f35 100644
--- a/llvm/test/CodeGen/AMDGPU/uniform-vgpr-to-sgpr-return.ll
+++ b/llvm/test/CodeGen/AMDGPU/uniform-vgpr-to-sgpr-return.ll
@@ -21,7 +21,7 @@ define amdgpu_ps i64 @uniform_v_to_s_i64(double inreg %a, double inreg %b) {
; GFX11: ; %bb.0:
; GFX11-NEXT: v_max_f64 v[0:1], s[0:1], s[2:3]
; GFX11-NEXT: v_cmp_u_f64_e64 s0, s[0:1], s[2:3]
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_3)
; GFX11-NEXT: v_cndmask_b32_e64 v1, v1, 0x7ff80000, s0
; GFX11-NEXT: v_cndmask_b32_e64 v0, v0, 0, s0
; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
@@ -79,7 +79,7 @@ define amdgpu_ps double @uniform_v_to_s_double(double inreg %a, double inreg %b)
; GFX11: ; %bb.0:
; GFX11-NEXT: v_max_f64 v[0:1], s[0:1], s[2:3]
; GFX11-NEXT: v_cmp_u_f64_e64 s0, s[0:1], s[2:3]
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_3)
; GFX11-NEXT: v_cndmask_b32_e64 v1, v1, 0x7ff80000, s0
; GFX11-NEXT: v_cndmask_b32_e64 v0, v0, 0, s0
; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
diff --git a/llvm/test/CodeGen/AMDGPU/usubo.ll b/llvm/test/CodeGen/AMDGPU/usubo.ll
index d3579bc2983cb6..143a305f16f336 100644
--- a/llvm/test/CodeGen/AMDGPU/usubo.ll
+++ b/llvm/test/CodeGen/AMDGPU/usubo.ll
@@ -1036,6 +1036,7 @@ define amdgpu_kernel void @s_usubo_clamp_bit(ptr addrspace(1) %out, ptr addrspac
; GFX11-NEXT: s_waitcnt lgkmcnt(0)
; GFX11-NEXT: v_sub_co_u32 v0, s1, s2, s3
; GFX11-NEXT: s_cmp_eq_u32 s2, s3
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cbranch_scc1 .LBB8_2
; GFX11-NEXT: ; %bb.1: ; %if
; GFX11-NEXT: s_xor_b32 s0, s1, -1
diff --git a/llvm/test/CodeGen/AMDGPU/v_cndmask.ll b/llvm/test/CodeGen/AMDGPU/v_cndmask.ll
index 5b578ce4b89ddc..f10a4c314a1c81 100644
--- a/llvm/test/CodeGen/AMDGPU/v_cndmask.ll
+++ b/llvm/test/CodeGen/AMDGPU/v_cndmask.ll
@@ -89,6 +89,7 @@ define amdgpu_kernel void @v_cnd_nan_nosgpr(ptr addrspace(1) %out, i32 %c, ptr a
; GFX11-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
; GFX11-NEXT: s_waitcnt lgkmcnt(0)
; GFX11-NEXT: s_cmp_eq_u32 s2, 0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cselect_b64 vcc, -1, 0
; GFX11-NEXT: s_waitcnt vmcnt(0)
; GFX11-NEXT: v_cndmask_b32_e32 v0, -1, v0, vcc
@@ -107,6 +108,7 @@ define amdgpu_kernel void @v_cnd_nan_nosgpr(ptr addrspace(1) %out, i32 %c, ptr a
; GFX12-NEXT: s_load_b96 s[0:2], s[4:5], 0x24
; GFX12-NEXT: s_wait_kmcnt 0x0
; GFX12-NEXT: s_cmp_eq_u32 s2, 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b64 vcc, -1, 0
; GFX12-NEXT: s_wait_loadcnt 0x0
; GFX12-NEXT: v_cndmask_b32_e32 v0, -1, v0, vcc
@@ -172,8 +174,8 @@ define amdgpu_kernel void @v_cnd_nan(ptr addrspace(1) %out, i32 %c, float %f) #0
; GFX11-NEXT: v_mov_b32_e32 v0, 0
; GFX11-NEXT: s_waitcnt lgkmcnt(0)
; GFX11-NEXT: s_cmp_eq_u32 s2, 0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_cselect_b64 s[4:5], -1, 0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: v_cndmask_b32_e64 v1, -1, s3, s[4:5]
; GFX11-NEXT: global_store_b32 v0, v1, s[0:1]
; GFX11-NEXT: s_endpgm
@@ -184,8 +186,8 @@ define amdgpu_kernel void @v_cnd_nan(ptr addrspace(1) %out, i32 %c, float %f) #0
; GFX12-NEXT: v_mov_b32_e32 v0, 0
; GFX12-NEXT: s_wait_kmcnt 0x0
; GFX12-NEXT: s_cmp_eq_u32 s2, 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b32 s2, s3, -1
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: v_mov_b32_e32 v1, s2
; GFX12-NEXT: global_store_b32 v0, v1, s[0:1]
; GFX12-NEXT: s_endpgm
@@ -247,7 +249,7 @@ define amdgpu_kernel void @fcmp_sgprX_k0_select_k1_sgprZ_f32(ptr addrspace(1) %o
; GFX11-NEXT: s_load_b64 s[0:1], s[4:5], 0x4c
; GFX11-NEXT: s_load_b64 s[2:3], s[4:5], 0x24
; GFX11-NEXT: v_and_b32_e32 v0, 0x3ff, v0
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_2) | instid1(VALU_DEP_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_lshlrev_b32_e32 v0, 2, v0
; GFX11-NEXT: s_waitcnt lgkmcnt(0)
; GFX11-NEXT: v_cmp_nlg_f32_e64 s[4:5], s0, 0
@@ -261,11 +263,12 @@ define amdgpu_kernel void @fcmp_sgprX_k0_select_k1_sgprZ_f32(ptr addrspace(1) %o
; GFX12-NEXT: s_load_b64 s[0:1], s[4:5], 0x4c
; GFX12-NEXT: s_load_b64 s[2:3], s[4:5], 0x24
; GFX12-NEXT: v_and_b32_e32 v0, 0x3ff, v0
-; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_3) | instid1(SALU_CYCLE_1)
+; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_3)
; GFX12-NEXT: v_lshlrev_b32_e32 v0, 2, v0
; GFX12-NEXT: s_wait_kmcnt 0x0
; GFX12-NEXT: s_cmp_nlg_f32 s0, 0
; GFX12-NEXT: s_cselect_b32 s0, s1, 1.0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: v_mov_b32_e32 v1, s0
; GFX12-NEXT: global_store_b32 v0, v1, s[2:3]
; GFX12-NEXT: s_endpgm
@@ -327,7 +330,7 @@ define amdgpu_kernel void @fcmp_sgprX_k0_select_k1_sgprX_f32(ptr addrspace(1) %o
; GFX11-NEXT: s_load_b32 s6, s[4:5], 0x2c
; GFX11-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
; GFX11-NEXT: v_and_b32_e32 v0, 0x3ff, v0
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_2) | instid1(VALU_DEP_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_lshlrev_b32_e32 v0, 2, v0
; GFX11-NEXT: s_waitcnt lgkmcnt(0)
; GFX11-NEXT: v_cmp_nlg_f32_e64 s[2:3], s6, 0
@@ -339,11 +342,12 @@ define amdgpu_kernel void @fcmp_sgprX_k0_select_k1_sgprX_f32(ptr addrspace(1) %o
; GFX12: ; %bb.0:
; GFX12-NEXT: s_load_b96 s[0:2], s[4:5], 0x24
; GFX12-NEXT: v_and_b32_e32 v0, 0x3ff, v0
-; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_3) | instid1(SALU_CYCLE_1)
+; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_3)
; GFX12-NEXT: v_lshlrev_b32_e32 v0, 2, v0
; GFX12-NEXT: s_wait_kmcnt 0x0
; GFX12-NEXT: s_cmp_nlg_f32 s2, 0
; GFX12-NEXT: s_cselect_b32 s2, s2, 1.0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: v_mov_b32_e32 v1, s2
; GFX12-NEXT: global_store_b32 v0, v1, s[0:1]
; GFX12-NEXT: s_endpgm
@@ -405,7 +409,7 @@ define amdgpu_kernel void @fcmp_sgprX_k0_select_k0_sgprZ_f32(ptr addrspace(1) %o
; GFX11-NEXT: s_load_b64 s[0:1], s[4:5], 0x4c
; GFX11-NEXT: s_load_b64 s[2:3], s[4:5], 0x24
; GFX11-NEXT: v_and_b32_e32 v0, 0x3ff, v0
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_2) | instid1(VALU_DEP_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_lshlrev_b32_e32 v0, 2, v0
; GFX11-NEXT: s_waitcnt lgkmcnt(0)
; GFX11-NEXT: v_cmp_nlg_f32_e64 s[4:5], s0, 0
@@ -419,11 +423,12 @@ define amdgpu_kernel void @fcmp_sgprX_k0_select_k0_sgprZ_f32(ptr addrspace(1) %o
; GFX12-NEXT: s_load_b64 s[0:1], s[4:5], 0x4c
; GFX12-NEXT: s_load_b64 s[2:3], s[4:5], 0x24
; GFX12-NEXT: v_and_b32_e32 v0, 0x3ff, v0
-; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_3) | instid1(SALU_CYCLE_1)
+; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_3)
; GFX12-NEXT: v_lshlrev_b32_e32 v0, 2, v0
; GFX12-NEXT: s_wait_kmcnt 0x0
; GFX12-NEXT: s_cmp_nlg_f32 s0, 0
; GFX12-NEXT: s_cselect_b32 s0, s1, 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: v_mov_b32_e32 v1, s0
; GFX12-NEXT: global_store_b32 v0, v1, s[2:3]
; GFX12-NEXT: s_endpgm
@@ -485,7 +490,7 @@ define amdgpu_kernel void @fcmp_sgprX_k0_select_k0_sgprX_f32(ptr addrspace(1) %o
; GFX11-NEXT: s_load_b32 s6, s[4:5], 0x2c
; GFX11-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
; GFX11-NEXT: v_and_b32_e32 v0, 0x3ff, v0
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_2) | instid1(VALU_DEP_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_lshlrev_b32_e32 v0, 2, v0
; GFX11-NEXT: s_waitcnt lgkmcnt(0)
; GFX11-NEXT: v_cmp_nlg_f32_e64 s[2:3], s6, 0
@@ -497,11 +502,12 @@ define amdgpu_kernel void @fcmp_sgprX_k0_select_k0_sgprX_f32(ptr addrspace(1) %o
; GFX12: ; %bb.0:
; GFX12-NEXT: s_load_b96 s[0:2], s[4:5], 0x24
; GFX12-NEXT: v_and_b32_e32 v0, 0x3ff, v0
-; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_3) | instid1(SALU_CYCLE_1)
+; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_3)
; GFX12-NEXT: v_lshlrev_b32_e32 v0, 2, v0
; GFX12-NEXT: s_wait_kmcnt 0x0
; GFX12-NEXT: s_cmp_nlg_f32 s2, 0
; GFX12-NEXT: s_cselect_b32 s2, s2, 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: v_mov_b32_e32 v1, s2
; GFX12-NEXT: global_store_b32 v0, v1, s[0:1]
; GFX12-NEXT: s_endpgm
@@ -600,6 +606,7 @@ define amdgpu_kernel void @fcmp_sgprX_k0_select_k0_vgprZ_f32(ptr addrspace(1) %o
; GFX12-NEXT: s_load_b96 s[0:2], s[4:5], 0x24
; GFX12-NEXT: s_wait_kmcnt 0x0
; GFX12-NEXT: s_cmp_nlg_f32 s2, 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_3)
; GFX12-NEXT: s_cselect_b64 vcc, -1, 0
; GFX12-NEXT: s_wait_loadcnt 0x0
; GFX12-NEXT: v_cndmask_b32_e32 v1, 0, v1, vcc
@@ -702,6 +709,7 @@ define amdgpu_kernel void @fcmp_sgprX_k0_select_k1_vgprZ_f32(ptr addrspace(1) %o
; GFX12-NEXT: s_load_b96 s[0:2], s[4:5], 0x24
; GFX12-NEXT: s_wait_kmcnt 0x0
; GFX12-NEXT: s_cmp_nlg_f32 s2, 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_3)
; GFX12-NEXT: s_cselect_b64 vcc, -1, 0
; GFX12-NEXT: s_wait_loadcnt 0x0
; GFX12-NEXT: v_cndmask_b32_e32 v1, 1.0, v1, vcc
@@ -2242,11 +2250,11 @@ define amdgpu_kernel void @v_cndmask_abs_neg_f16(ptr addrspace(1) %out, i32 %c,
; GFX11-TRUE16-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
; GFX11-TRUE16-NEXT: s_waitcnt lgkmcnt(0)
; GFX11-TRUE16-NEXT: s_cmp_lg_u32 s2, 0
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_3) | instid1(VALU_DEP_1)
; GFX11-TRUE16-NEXT: s_cselect_b64 s[2:3], -1, 0
; GFX11-TRUE16-NEXT: s_waitcnt vmcnt(0)
; GFX11-TRUE16-NEXT: v_and_b16 v0.h, 0x7fff, v0.l
; GFX11-TRUE16-NEXT: v_xor_b16 v0.l, 0x8000, v0.l
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_cndmask_b16 v0.l, v0.l, v0.h, s[2:3]
; GFX11-TRUE16-NEXT: global_store_b16 v1, v0, s[0:1]
; GFX11-TRUE16-NEXT: s_endpgm
@@ -2265,11 +2273,11 @@ define amdgpu_kernel void @v_cndmask_abs_neg_f16(ptr addrspace(1) %out, i32 %c,
; GFX11-FAKE16-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
; GFX11-FAKE16-NEXT: s_waitcnt lgkmcnt(0)
; GFX11-FAKE16-NEXT: s_cmp_lg_u32 s2, 0
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_3) | instid1(VALU_DEP_1)
; GFX11-FAKE16-NEXT: s_cselect_b64 vcc, -1, 0
; GFX11-FAKE16-NEXT: s_waitcnt vmcnt(0)
; GFX11-FAKE16-NEXT: v_and_b32_e32 v1, 0x7fff, v0
; GFX11-FAKE16-NEXT: v_xor_b32_e32 v0, 0x8000, v0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_cndmask_b32_e32 v0, v0, v1, vcc
; GFX11-FAKE16-NEXT: global_store_b16 v2, v0, s[0:1]
; GFX11-FAKE16-NEXT: s_endpgm
@@ -2286,11 +2294,11 @@ define amdgpu_kernel void @v_cndmask_abs_neg_f16(ptr addrspace(1) %out, i32 %c,
; GFX12-TRUE16-NEXT: s_load_b96 s[0:2], s[4:5], 0x24
; GFX12-TRUE16-NEXT: s_wait_kmcnt 0x0
; GFX12-TRUE16-NEXT: s_cmp_lg_u32 s2, 0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_3) | instid1(VALU_DEP_1)
; GFX12-TRUE16-NEXT: s_cselect_b64 s[2:3], -1, 0
; GFX12-TRUE16-NEXT: s_wait_loadcnt 0x0
; GFX12-TRUE16-NEXT: v_and_b16 v0.h, 0x7fff, v0.l
; GFX12-TRUE16-NEXT: v_xor_b16 v0.l, 0x8000, v0.l
-; GFX12-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-TRUE16-NEXT: v_cndmask_b16 v0.l, v0.l, v0.h, s[2:3]
; GFX12-TRUE16-NEXT: global_store_b16 v1, v0, s[0:1]
; GFX12-TRUE16-NEXT: s_endpgm
@@ -2307,11 +2315,11 @@ define amdgpu_kernel void @v_cndmask_abs_neg_f16(ptr addrspace(1) %out, i32 %c,
; GFX12-FAKE16-NEXT: s_load_b96 s[0:2], s[4:5], 0x24
; GFX12-FAKE16-NEXT: s_wait_kmcnt 0x0
; GFX12-FAKE16-NEXT: s_cmp_lg_u32 s2, 0
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_3) | instid1(VALU_DEP_1)
; GFX12-FAKE16-NEXT: s_cselect_b64 vcc, -1, 0
; GFX12-FAKE16-NEXT: s_wait_loadcnt 0x0
; GFX12-FAKE16-NEXT: v_and_b32_e32 v1, 0x7fff, v0
; GFX12-FAKE16-NEXT: v_xor_b32_e32 v0, 0x8000, v0
-; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-FAKE16-NEXT: v_cndmask_b32_e32 v0, v0, v1, vcc
; GFX12-FAKE16-NEXT: global_store_b16 v2, v0, s[0:1]
; GFX12-FAKE16-NEXT: s_endpgm
@@ -2402,6 +2410,7 @@ define amdgpu_kernel void @v_cndmask_abs_neg_f32(ptr addrspace(1) %out, i32 %c,
; GFX11-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
; GFX11-NEXT: s_waitcnt lgkmcnt(0)
; GFX11-NEXT: s_cmp_lg_u32 s2, 0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: s_cselect_b64 s[2:3], -1, 0
; GFX11-NEXT: s_waitcnt vmcnt(0)
; GFX11-NEXT: v_cndmask_b32_e64 v0, -v0, |v0|, s[2:3]
@@ -2420,6 +2429,7 @@ define amdgpu_kernel void @v_cndmask_abs_neg_f32(ptr addrspace(1) %out, i32 %c,
; GFX12-NEXT: s_load_b96 s[0:2], s[4:5], 0x24
; GFX12-NEXT: s_wait_kmcnt 0x0
; GFX12-NEXT: s_cmp_lg_u32 s2, 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: s_cselect_b64 s[2:3], -1, 0
; GFX12-NEXT: s_wait_loadcnt 0x0
; GFX12-NEXT: v_cndmask_b32_e64 v0, -v0, |v0|, s[2:3]
@@ -2518,11 +2528,11 @@ define amdgpu_kernel void @v_cndmask_abs_neg_f64(ptr addrspace(1) %out, i32 %c,
; GFX11-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
; GFX11-NEXT: s_waitcnt lgkmcnt(0)
; GFX11-NEXT: s_cmp_lg_u32 s2, 0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_3) | instid1(VALU_DEP_1)
; GFX11-NEXT: s_cselect_b64 vcc, -1, 0
; GFX11-NEXT: s_waitcnt vmcnt(0)
; GFX11-NEXT: v_and_b32_e32 v2, 0x7fffffff, v1
; GFX11-NEXT: v_xor_b32_e32 v1, 0x80000000, v1
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_cndmask_b32_e32 v1, v1, v2, vcc
; GFX11-NEXT: global_store_b64 v3, v[0:1], s[0:1]
; GFX11-NEXT: s_endpgm
@@ -2539,11 +2549,11 @@ define amdgpu_kernel void @v_cndmask_abs_neg_f64(ptr addrspace(1) %out, i32 %c,
; GFX12-NEXT: s_load_b96 s[0:2], s[4:5], 0x24
; GFX12-NEXT: s_wait_kmcnt 0x0
; GFX12-NEXT: s_cmp_lg_u32 s2, 0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_3) | instid1(VALU_DEP_1)
; GFX12-NEXT: s_cselect_b64 vcc, -1, 0
; GFX12-NEXT: s_wait_loadcnt 0x0
; GFX12-NEXT: v_and_b32_e32 v2, 0x7fffffff, v1
; GFX12-NEXT: v_xor_b32_e32 v1, 0x80000000, v1
-; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-NEXT: v_cndmask_b32_e32 v1, v1, v2, vcc
; GFX12-NEXT: global_store_b64 v3, v[0:1], s[0:1]
; GFX12-NEXT: s_endpgm
diff --git a/llvm/test/CodeGen/AMDGPU/v_swap_b16.ll b/llvm/test/CodeGen/AMDGPU/v_swap_b16.ll
index 297d43357c3761..967da57bb127a0 100644
--- a/llvm/test/CodeGen/AMDGPU/v_swap_b16.ll
+++ b/llvm/test/CodeGen/AMDGPU/v_swap_b16.ll
@@ -19,7 +19,7 @@ define half @swap(half %a, half %b, i32 %i) {
; GFX11-TRUE16-NEXT: v_swap_b16 v0.l, v0.h
; GFX11-TRUE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, 0, v2
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB0_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %ret
@@ -37,11 +37,12 @@ define half @swap(half %a, half %b, i32 %i) {
; GFX11-FAKE16-NEXT: v_swap_b32 v1, v0
; GFX11-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, 0, v2
; GFX11-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB0_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %ret
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v0, v1
; GFX11-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -66,6 +67,7 @@ define half @swap(half %a, half %b, i32 %i) {
; GFX12-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_cbranch_execnz .LBB0_1
; GFX12-TRUE16-NEXT: ; %bb.2: ; %ret
; GFX12-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
@@ -83,7 +85,7 @@ define half @swap(half %a, half %b, i32 %i) {
; GFX12-FAKE16-NEXT: ; =>This Inner Loop Header: Depth=1
; GFX12-FAKE16-NEXT: v_dual_mov_b32 v3, v1 :: v_dual_add_nc_u32 v2, -1, v2
; GFX12-FAKE16-NEXT: v_swap_b32 v1, v0
-; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2)
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_4) | instid1(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: v_cmp_eq_u32_e32 vcc_lo, 0, v2
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
@@ -92,6 +94,7 @@ define half @swap(half %a, half %b, i32 %i) {
; GFX12-FAKE16-NEXT: s_cbranch_execnz .LBB0_1
; GFX12-FAKE16-NEXT: ; %bb.2: ; %ret
; GFX12-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: v_mov_b32_e32 v0, v1
; GFX12-FAKE16-NEXT: s_setpc_b64 s[30:31]
entry:
@@ -136,11 +139,12 @@ define half @swap_B(half %a, half %b, half %c, i32 %i) {
; GFX11-TRUE16-NEXT: ;;#ASMEND
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v2.l, v1.l
; GFX11-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
; GFX11-TRUE16-NEXT: s_cbranch_execnz .LBB1_1
; GFX11-TRUE16-NEXT: ; %bb.2: ; %ret
; GFX11-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-TRUE16-NEXT: v_mov_b16_e32 v0.l, v0.h
; GFX11-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -166,9 +170,11 @@ define half @swap_B(half %a, half %b, half %c, i32 %i) {
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v2, v4
; GFX11-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX11-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: s_cbranch_execnz .LBB1_1
; GFX11-FAKE16-NEXT: ; %bb.2: ; %ret
; GFX11-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v0, v1
; GFX11-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -202,9 +208,11 @@ define half @swap_B(half %a, half %b, half %c, i32 %i) {
; GFX12-TRUE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-TRUE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-TRUE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: s_cbranch_execnz .LBB1_1
; GFX12-TRUE16-NEXT: ; %bb.2: ; %ret
; GFX12-TRUE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-TRUE16-NEXT: v_mov_b16_e32 v0.l, v0.h
; GFX12-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
@@ -236,9 +244,11 @@ define half @swap_B(half %a, half %b, half %c, i32 %i) {
; GFX12-FAKE16-NEXT: s_or_b32 s0, vcc_lo, s0
; GFX12-FAKE16-NEXT: s_wait_alu depctr_sa_sdst(0)
; GFX12-FAKE16-NEXT: s_and_not1_b32 exec_lo, exec_lo, s0
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: s_cbranch_execnz .LBB1_1
; GFX12-FAKE16-NEXT: ; %bb.2: ; %ret
; GFX12-FAKE16-NEXT: s_or_b32 exec_lo, exec_lo, s0
+; GFX12-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-FAKE16-NEXT: v_mov_b32_e32 v0, v1
; GFX12-FAKE16-NEXT: s_setpc_b64 s[30:31]
entry:
diff --git a/llvm/test/CodeGen/AMDGPU/vector-reduce-add.ll b/llvm/test/CodeGen/AMDGPU/vector-reduce-add.ll
index 01b7293dcd7ab4..2ebe33397c9c44 100644
--- a/llvm/test/CodeGen/AMDGPU/vector-reduce-add.ll
+++ b/llvm/test/CodeGen/AMDGPU/vector-reduce-add.ll
@@ -2724,7 +2724,6 @@ define i64 @test_vector_reduce_add_v2i64(<2 x i64> %v) {
; GFX11: ; %bb.0: ; %entry
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, v1, v3, vcc_lo
; GFX11-NEXT: s_setpc_b64 s[30:31]
;
@@ -2794,10 +2793,9 @@ define i64 @test_vector_reduce_add_v3i64(<3 x i64> %v) {
; GFX11: ; %bb.0: ; %entry
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, v1, v3, vcc_lo
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-NEXT: v_add_co_u32 v0, vcc_lo, v0, v4
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_add_co_ci_u32_e64 v1, null, v1, v5, vcc_lo
; GFX11-NEXT: s_setpc_b64 s[30:31]
;
@@ -2914,11 +2912,10 @@ define i64 @test_vector_reduce_add_v4i64(<4 x i64> %v) {
; GFX11-SDAG: ; %bb.0: ; %entry
; GFX11-SDAG-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-SDAG-NEXT: v_add_co_u32 v2, vcc_lo, v2, v6
-; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-SDAG-NEXT: v_add_co_ci_u32_e64 v3, null, v3, v7, vcc_lo
; GFX11-SDAG-NEXT: v_add_co_u32 v0, vcc_lo, v0, v4
; GFX11-SDAG-NEXT: v_add_co_ci_u32_e64 v1, null, v1, v5, vcc_lo
-; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-SDAG-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
; GFX11-SDAG-NEXT: v_add_co_ci_u32_e64 v1, null, v1, v3, vcc_lo
; GFX11-SDAG-NEXT: s_setpc_b64 s[30:31]
@@ -2927,11 +2924,10 @@ define i64 @test_vector_reduce_add_v4i64(<4 x i64> %v) {
; GFX11-GISEL: ; %bb.0: ; %entry
; GFX11-GISEL-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, v0, v4
-; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, v1, v5, vcc_lo
; GFX11-GISEL-NEXT: v_add_co_u32 v2, vcc_lo, v2, v6
; GFX11-GISEL-NEXT: v_add_co_ci_u32_e64 v3, null, v3, v7, vcc_lo
-; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
; GFX11-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, v1, v3, vcc_lo
; GFX11-GISEL-NEXT: s_setpc_b64 s[30:31]
@@ -3135,22 +3131,20 @@ define i64 @test_vector_reduce_add_v8i64(<8 x i64> %v) {
; GFX11-SDAG: ; %bb.0: ; %entry
; GFX11-SDAG-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-SDAG-NEXT: v_add_co_u32 v4, vcc_lo, v4, v12
-; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-SDAG-NEXT: v_add_co_ci_u32_e64 v5, null, v5, v13, vcc_lo
; GFX11-SDAG-NEXT: v_add_co_u32 v6, vcc_lo, v6, v14
; GFX11-SDAG-NEXT: v_add_co_ci_u32_e64 v7, null, v7, v15, vcc_lo
; GFX11-SDAG-NEXT: v_add_co_u32 v2, vcc_lo, v2, v10
-; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-SDAG-NEXT: v_add_co_ci_u32_e64 v3, null, v3, v11, vcc_lo
; GFX11-SDAG-NEXT: v_add_co_u32 v0, vcc_lo, v0, v8
; GFX11-SDAG-NEXT: v_add_co_ci_u32_e64 v1, null, v1, v9, vcc_lo
-; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
; GFX11-SDAG-NEXT: v_add_co_u32 v2, vcc_lo, v2, v6
; GFX11-SDAG-NEXT: v_add_co_ci_u32_e64 v3, null, v3, v7, vcc_lo
-; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
; GFX11-SDAG-NEXT: v_add_co_u32 v0, vcc_lo, v0, v4
; GFX11-SDAG-NEXT: v_add_co_ci_u32_e64 v1, null, v1, v5, vcc_lo
-; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-SDAG-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
; GFX11-SDAG-NEXT: v_add_co_ci_u32_e64 v1, null, v1, v3, vcc_lo
; GFX11-SDAG-NEXT: s_setpc_b64 s[30:31]
@@ -3159,22 +3153,20 @@ define i64 @test_vector_reduce_add_v8i64(<8 x i64> %v) {
; GFX11-GISEL: ; %bb.0: ; %entry
; GFX11-GISEL-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, v0, v8
-; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, v1, v9, vcc_lo
; GFX11-GISEL-NEXT: v_add_co_u32 v2, vcc_lo, v2, v10
; GFX11-GISEL-NEXT: v_add_co_ci_u32_e64 v3, null, v3, v11, vcc_lo
; GFX11-GISEL-NEXT: v_add_co_u32 v4, vcc_lo, v4, v12
-; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-GISEL-NEXT: v_add_co_ci_u32_e64 v5, null, v5, v13, vcc_lo
; GFX11-GISEL-NEXT: v_add_co_u32 v6, vcc_lo, v6, v14
; GFX11-GISEL-NEXT: v_add_co_ci_u32_e64 v7, null, v7, v15, vcc_lo
-; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
; GFX11-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, v0, v4
; GFX11-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, v1, v5, vcc_lo
-; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
; GFX11-GISEL-NEXT: v_add_co_u32 v2, vcc_lo, v2, v6
; GFX11-GISEL-NEXT: v_add_co_ci_u32_e64 v3, null, v3, v7, vcc_lo
-; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
; GFX11-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, v1, v3, vcc_lo
; GFX11-GISEL-NEXT: s_setpc_b64 s[30:31]
@@ -3547,25 +3539,21 @@ define i64 @test_vector_reduce_add_v16i64(<16 x i64> %v) {
; GFX11-SDAG-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-SDAG-NEXT: scratch_load_b32 v31, off, s32
; GFX11-SDAG-NEXT: v_add_co_u32 v10, vcc_lo, v10, v26
-; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-SDAG-NEXT: v_add_co_ci_u32_e64 v11, null, v11, v27, vcc_lo
; GFX11-SDAG-NEXT: v_add_co_u32 v2, vcc_lo, v2, v18
; GFX11-SDAG-NEXT: v_add_co_ci_u32_e64 v3, null, v3, v19, vcc_lo
; GFX11-SDAG-NEXT: v_add_co_u32 v6, vcc_lo, v6, v22
-; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-SDAG-NEXT: v_add_co_ci_u32_e64 v7, null, v7, v23, vcc_lo
; GFX11-SDAG-NEXT: v_add_co_u32 v8, vcc_lo, v8, v24
; GFX11-SDAG-NEXT: v_add_co_ci_u32_e64 v9, null, v9, v25, vcc_lo
; GFX11-SDAG-NEXT: v_add_co_u32 v12, vcc_lo, v12, v28
-; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-SDAG-NEXT: v_add_co_ci_u32_e64 v13, null, v13, v29, vcc_lo
; GFX11-SDAG-NEXT: v_add_co_u32 v4, vcc_lo, v4, v20
; GFX11-SDAG-NEXT: v_add_co_ci_u32_e64 v5, null, v5, v21, vcc_lo
; GFX11-SDAG-NEXT: v_add_co_u32 v0, vcc_lo, v0, v16
-; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_4)
; GFX11-SDAG-NEXT: v_add_co_ci_u32_e64 v1, null, v1, v17, vcc_lo
+; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
; GFX11-SDAG-NEXT: v_add_co_u32 v4, vcc_lo, v4, v12
-; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_4) | instid1(VALU_DEP_1)
; GFX11-SDAG-NEXT: v_add_co_ci_u32_e64 v5, null, v5, v13, vcc_lo
; GFX11-SDAG-NEXT: v_add_co_u32 v12, vcc_lo, v14, v30
; GFX11-SDAG-NEXT: s_waitcnt vmcnt(0)
@@ -3573,18 +3561,18 @@ define i64 @test_vector_reduce_add_v16i64(<16 x i64> %v) {
; GFX11-SDAG-NEXT: v_add_co_u32 v2, vcc_lo, v2, v10
; GFX11-SDAG-NEXT: v_add_co_ci_u32_e64 v3, null, v3, v11, vcc_lo
; GFX11-SDAG-NEXT: v_add_co_u32 v6, vcc_lo, v6, v12
-; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
+; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(SKIP_2) | instid1(VALU_DEP_4)
; GFX11-SDAG-NEXT: v_add_co_ci_u32_e64 v7, null, v7, v13, vcc_lo
; GFX11-SDAG-NEXT: v_add_co_u32 v0, vcc_lo, v0, v8
; GFX11-SDAG-NEXT: v_add_co_ci_u32_e64 v1, null, v1, v9, vcc_lo
-; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-SDAG-NEXT: v_add_co_u32 v2, vcc_lo, v2, v6
+; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
; GFX11-SDAG-NEXT: v_add_co_ci_u32_e64 v3, null, v3, v7, vcc_lo
-; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-SDAG-NEXT: v_add_co_u32 v0, vcc_lo, v0, v4
+; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-SDAG-NEXT: v_add_co_ci_u32_e64 v1, null, v1, v5, vcc_lo
-; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-SDAG-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
+; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX11-SDAG-NEXT: v_add_co_ci_u32_e64 v1, null, v1, v3, vcc_lo
; GFX11-SDAG-NEXT: s_setpc_b64 s[30:31]
;
@@ -3593,27 +3581,22 @@ define i64 @test_vector_reduce_add_v16i64(<16 x i64> %v) {
; GFX11-GISEL-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX11-GISEL-NEXT: scratch_load_b32 v31, off, s32
; GFX11-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, v0, v16
-; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, v1, v17, vcc_lo
; GFX11-GISEL-NEXT: v_add_co_u32 v2, vcc_lo, v2, v18
; GFX11-GISEL-NEXT: v_add_co_ci_u32_e64 v3, null, v3, v19, vcc_lo
; GFX11-GISEL-NEXT: v_add_co_u32 v4, vcc_lo, v4, v20
-; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-GISEL-NEXT: v_add_co_ci_u32_e64 v5, null, v5, v21, vcc_lo
; GFX11-GISEL-NEXT: v_add_co_u32 v6, vcc_lo, v6, v22
; GFX11-GISEL-NEXT: v_add_co_ci_u32_e64 v7, null, v7, v23, vcc_lo
; GFX11-GISEL-NEXT: v_add_co_u32 v8, vcc_lo, v8, v24
-; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-GISEL-NEXT: v_add_co_ci_u32_e64 v9, null, v9, v25, vcc_lo
; GFX11-GISEL-NEXT: v_add_co_u32 v10, vcc_lo, v10, v26
; GFX11-GISEL-NEXT: v_add_co_ci_u32_e64 v11, null, v11, v27, vcc_lo
; GFX11-GISEL-NEXT: v_add_co_u32 v12, vcc_lo, v12, v28
-; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-GISEL-NEXT: v_add_co_ci_u32_e64 v13, null, v13, v29, vcc_lo
; GFX11-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, v0, v8
; GFX11-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, v1, v9, vcc_lo
; GFX11-GISEL-NEXT: v_add_co_u32 v2, vcc_lo, v2, v10
-; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_4) | instid1(VALU_DEP_1)
; GFX11-GISEL-NEXT: v_add_co_ci_u32_e64 v3, null, v3, v11, vcc_lo
; GFX11-GISEL-NEXT: v_add_co_u32 v8, vcc_lo, v14, v30
; GFX11-GISEL-NEXT: s_waitcnt vmcnt(0)
@@ -3621,16 +3604,16 @@ define i64 @test_vector_reduce_add_v16i64(<16 x i64> %v) {
; GFX11-GISEL-NEXT: v_add_co_u32 v4, vcc_lo, v4, v12
; GFX11-GISEL-NEXT: v_add_co_ci_u32_e64 v5, null, v5, v13, vcc_lo
; GFX11-GISEL-NEXT: v_add_co_u32 v6, vcc_lo, v6, v8
-; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_4)
+; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
; GFX11-GISEL-NEXT: v_add_co_ci_u32_e64 v7, null, v7, v9, vcc_lo
; GFX11-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, v0, v4
-; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_4)
+; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
; GFX11-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, v1, v5, vcc_lo
; GFX11-GISEL-NEXT: v_add_co_u32 v2, vcc_lo, v2, v6
-; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-GISEL-NEXT: v_add_co_ci_u32_e64 v3, null, v3, v7, vcc_lo
; GFX11-GISEL-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
-; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX11-GISEL-NEXT: v_add_co_ci_u32_e64 v1, null, v1, v3, vcc_lo
; GFX11-GISEL-NEXT: s_setpc_b64 s[30:31]
;
diff --git a/llvm/test/CodeGen/AMDGPU/vector-reduce-fmaximum.ll b/llvm/test/CodeGen/AMDGPU/vector-reduce-fmaximum.ll
index 3ac13782301dc5..bdc5526c745939 100644
--- a/llvm/test/CodeGen/AMDGPU/vector-reduce-fmaximum.ll
+++ b/llvm/test/CodeGen/AMDGPU/vector-reduce-fmaximum.ll
@@ -2189,7 +2189,7 @@ define double @test_vector_reduce_fmaximum_v4double(<4 x double> %v) {
; GFX11-NEXT: v_cmp_u_f64_e32 vcc_lo, v[2:3], v[6:7]
; GFX11-NEXT: v_max_f64 v[2:3], v[0:1], v[4:5]
; GFX11-NEXT: v_cmp_u_f64_e64 s0, v[0:1], v[4:5]
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(SKIP_1) | instid1(VALU_DEP_3)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(SKIP_1) | instid1(VALU_DEP_4)
; GFX11-NEXT: v_cndmask_b32_e64 v1, v9, 0x7ff80000, vcc_lo
; GFX11-NEXT: v_cndmask_b32_e64 v0, v8, 0, vcc_lo
; GFX11-NEXT: v_cndmask_b32_e64 v3, v3, 0x7ff80000, s0
@@ -2395,7 +2395,7 @@ define double @test_vector_reduce_fmaximum_v8double(<8 x double> %v) {
; GFX11-NEXT: v_cmp_u_f64_e32 vcc_lo, v[6:7], v[4:5]
; GFX11-NEXT: v_max_f64 v[4:5], v[2:3], v[0:1]
; GFX11-NEXT: v_cmp_u_f64_e64 s0, v[2:3], v[0:1]
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(SKIP_1) | instid1(VALU_DEP_3)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(SKIP_1) | instid1(VALU_DEP_4)
; GFX11-NEXT: v_cndmask_b32_e64 v1, v9, 0x7ff80000, vcc_lo
; GFX11-NEXT: v_cndmask_b32_e64 v0, v8, 0, vcc_lo
; GFX11-NEXT: v_cndmask_b32_e64 v3, v5, 0x7ff80000, s0
@@ -2786,7 +2786,7 @@ define double @test_vector_reduce_fmaximum_v16double(<16 x double> %v) {
; GFX11-NEXT: v_cmp_u_f64_e64 s0, v[4:5], v[2:3]
; GFX11-NEXT: v_cndmask_b32_e64 v3, v9, 0x7ff80000, vcc_lo
; GFX11-NEXT: v_cndmask_b32_e64 v2, v8, 0, vcc_lo
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_1) | instid1(VALU_DEP_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_cndmask_b32_e64 v1, v1, 0x7ff80000, s0
; GFX11-NEXT: v_cndmask_b32_e64 v0, v0, 0, s0
; GFX11-NEXT: v_max_f64 v[4:5], v[2:3], v[0:1]
diff --git a/llvm/test/CodeGen/AMDGPU/vector-reduce-fminimum.ll b/llvm/test/CodeGen/AMDGPU/vector-reduce-fminimum.ll
index 1ad17d8b89b5a5..4712a221c1b71b 100644
--- a/llvm/test/CodeGen/AMDGPU/vector-reduce-fminimum.ll
+++ b/llvm/test/CodeGen/AMDGPU/vector-reduce-fminimum.ll
@@ -2533,7 +2533,7 @@ define double @test_vector_reduce_fminimum_v4double(<4 x double> %v) {
; GFX11-NEXT: v_cmp_u_f64_e32 vcc_lo, v[2:3], v[6:7]
; GFX11-NEXT: v_min_f64 v[2:3], v[0:1], v[4:5]
; GFX11-NEXT: v_cmp_u_f64_e64 s0, v[0:1], v[4:5]
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(SKIP_1) | instid1(VALU_DEP_3)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(SKIP_1) | instid1(VALU_DEP_4)
; GFX11-NEXT: v_cndmask_b32_e64 v1, v9, 0x7ff80000, vcc_lo
; GFX11-NEXT: v_cndmask_b32_e64 v0, v8, 0, vcc_lo
; GFX11-NEXT: v_cndmask_b32_e64 v3, v3, 0x7ff80000, s0
@@ -2761,7 +2761,7 @@ define double @test_vector_reduce_fminimum_v8double(<8 x double> %v) {
; GFX11-NEXT: v_cmp_u_f64_e32 vcc_lo, v[6:7], v[4:5]
; GFX11-NEXT: v_min_f64 v[4:5], v[2:3], v[0:1]
; GFX11-NEXT: v_cmp_u_f64_e64 s0, v[2:3], v[0:1]
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(SKIP_1) | instid1(VALU_DEP_3)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(SKIP_1) | instid1(VALU_DEP_4)
; GFX11-NEXT: v_cndmask_b32_e64 v1, v9, 0x7ff80000, vcc_lo
; GFX11-NEXT: v_cndmask_b32_e64 v0, v8, 0, vcc_lo
; GFX11-NEXT: v_cndmask_b32_e64 v3, v5, 0x7ff80000, s0
@@ -3184,7 +3184,7 @@ define double @test_vector_reduce_fminimum_v16double(<16 x double> %v) {
; GFX11-NEXT: v_cmp_u_f64_e64 s0, v[4:5], v[2:3]
; GFX11-NEXT: v_cndmask_b32_e64 v3, v9, 0x7ff80000, vcc_lo
; GFX11-NEXT: v_cndmask_b32_e64 v2, v8, 0, vcc_lo
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_1) | instid1(VALU_DEP_1)
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_cndmask_b32_e64 v1, v1, 0x7ff80000, s0
; GFX11-NEXT: v_cndmask_b32_e64 v0, v0, 0, s0
; GFX11-NEXT: v_min_f64 v[4:5], v[2:3], v[0:1]
diff --git a/llvm/test/CodeGen/AMDGPU/vector-reduce-smax.ll b/llvm/test/CodeGen/AMDGPU/vector-reduce-smax.ll
index 219961c35071d6..9b9b0f35867f95 100644
--- a/llvm/test/CodeGen/AMDGPU/vector-reduce-smax.ll
+++ b/llvm/test/CodeGen/AMDGPU/vector-reduce-smax.ll
@@ -3333,9 +3333,9 @@ define i64 @test_vector_reduce_smax_v4i64(<4 x i64> %v) {
; GFX11-SDAG-NEXT: v_cmp_gt_i64_e32 vcc_lo, v[2:3], v[6:7]
; GFX11-SDAG-NEXT: v_cmp_gt_i64_e64 s0, v[0:1], v[4:5]
; GFX11-SDAG-NEXT: v_dual_cndmask_b32 v3, v7, v3 :: v_dual_cndmask_b32 v2, v6, v2
-; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-SDAG-NEXT: v_cndmask_b32_e64 v1, v5, v1, s0
; GFX11-SDAG-NEXT: v_cndmask_b32_e64 v0, v4, v0, s0
+; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-SDAG-NEXT: v_cmp_gt_i64_e32 vcc_lo, v[0:1], v[2:3]
; GFX11-SDAG-NEXT: v_dual_cndmask_b32 v0, v2, v0 :: v_dual_cndmask_b32 v1, v3, v1
; GFX11-SDAG-NEXT: s_setpc_b64 s[30:31]
@@ -3346,9 +3346,9 @@ define i64 @test_vector_reduce_smax_v4i64(<4 x i64> %v) {
; GFX11-GISEL-NEXT: v_cmp_gt_i64_e32 vcc_lo, v[0:1], v[4:5]
; GFX11-GISEL-NEXT: v_cmp_gt_i64_e64 s0, v[2:3], v[6:7]
; GFX11-GISEL-NEXT: v_dual_cndmask_b32 v0, v4, v0 :: v_dual_cndmask_b32 v1, v5, v1
-; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-GISEL-NEXT: v_cndmask_b32_e64 v2, v6, v2, s0
; GFX11-GISEL-NEXT: v_cndmask_b32_e64 v3, v7, v3, s0
+; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-GISEL-NEXT: v_cmp_gt_i64_e32 vcc_lo, v[0:1], v[2:3]
; GFX11-GISEL-NEXT: v_dual_cndmask_b32 v0, v2, v0 :: v_dual_cndmask_b32 v1, v3, v1
; GFX11-GISEL-NEXT: s_setpc_b64 s[30:31]
@@ -3365,9 +3365,9 @@ define i64 @test_vector_reduce_smax_v4i64(<4 x i64> %v) {
; GFX12-SDAG-NEXT: s_wait_alu depctr_va_vcc(0)
; GFX12-SDAG-NEXT: v_dual_cndmask_b32 v3, v7, v3 :: v_dual_cndmask_b32 v2, v6, v2
; GFX12-SDAG-NEXT: s_wait_alu depctr_va_sdst(0)
-; GFX12-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX12-SDAG-NEXT: v_cndmask_b32_e64 v1, v5, v1, s0
; GFX12-SDAG-NEXT: v_cndmask_b32_e64 v0, v4, v0, s0
+; GFX12-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-SDAG-NEXT: v_cmp_gt_i64_e32 vcc_lo, v[0:1], v[2:3]
; GFX12-SDAG-NEXT: s_wait_alu depctr_va_vcc(0)
; GFX12-SDAG-NEXT: v_dual_cndmask_b32 v0, v2, v0 :: v_dual_cndmask_b32 v1, v3, v1
@@ -3385,9 +3385,9 @@ define i64 @test_vector_reduce_smax_v4i64(<4 x i64> %v) {
; GFX12-GISEL-NEXT: s_wait_alu depctr_va_vcc(0)
; GFX12-GISEL-NEXT: v_dual_cndmask_b32 v0, v4, v0 :: v_dual_cndmask_b32 v1, v5, v1
; GFX12-GISEL-NEXT: s_wait_alu depctr_va_sdst(0)
-; GFX12-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX12-GISEL-NEXT: v_cndmask_b32_e64 v2, v6, v2, s0
; GFX12-GISEL-NEXT: v_cndmask_b32_e64 v3, v7, v3, s0
+; GFX12-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-GISEL-NEXT: v_cmp_gt_i64_e32 vcc_lo, v[0:1], v[2:3]
; GFX12-GISEL-NEXT: s_wait_alu depctr_va_vcc(0)
; GFX12-GISEL-NEXT: v_dual_cndmask_b32 v0, v2, v0 :: v_dual_cndmask_b32 v1, v3, v1
@@ -3628,14 +3628,13 @@ define i64 @test_vector_reduce_smax_v8i64(<8 x i64> %v) {
; GFX11-SDAG-NEXT: v_cmp_gt_i64_e64 s1, v[2:3], v[10:11]
; GFX11-SDAG-NEXT: v_dual_cndmask_b32 v5, v13, v5 :: v_dual_cndmask_b32 v4, v12, v4
; GFX11-SDAG-NEXT: v_cmp_gt_i64_e32 vcc_lo, v[0:1], v[8:9]
-; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(SKIP_4) | instid1(VALU_DEP_2)
; GFX11-SDAG-NEXT: v_cndmask_b32_e64 v7, v15, v7, s0
; GFX11-SDAG-NEXT: v_cndmask_b32_e64 v6, v14, v6, s0
; GFX11-SDAG-NEXT: v_cndmask_b32_e64 v3, v11, v3, s1
; GFX11-SDAG-NEXT: v_cndmask_b32_e64 v2, v10, v2, s1
; GFX11-SDAG-NEXT: v_dual_cndmask_b32 v1, v9, v1 :: v_dual_cndmask_b32 v0, v8, v0
+; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-SDAG-NEXT: v_cmp_gt_i64_e32 vcc_lo, v[2:3], v[6:7]
-; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX11-SDAG-NEXT: v_cmp_gt_i64_e64 s0, v[0:1], v[4:5]
; GFX11-SDAG-NEXT: v_dual_cndmask_b32 v3, v7, v3 :: v_dual_cndmask_b32 v2, v6, v2
; GFX11-SDAG-NEXT: v_cndmask_b32_e64 v1, v5, v1, s0
@@ -3653,14 +3652,13 @@ define i64 @test_vector_reduce_smax_v8i64(<8 x i64> %v) {
; GFX11-GISEL-NEXT: v_cmp_gt_i64_e64 s1, v[4:5], v[12:13]
; GFX11-GISEL-NEXT: v_dual_cndmask_b32 v0, v8, v0 :: v_dual_cndmask_b32 v1, v9, v1
; GFX11-GISEL-NEXT: v_cmp_gt_i64_e32 vcc_lo, v[6:7], v[14:15]
-; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(SKIP_4) | instid1(VALU_DEP_2)
; GFX11-GISEL-NEXT: v_cndmask_b32_e64 v2, v10, v2, s0
; GFX11-GISEL-NEXT: v_cndmask_b32_e64 v3, v11, v3, s0
; GFX11-GISEL-NEXT: v_cndmask_b32_e64 v4, v12, v4, s1
; GFX11-GISEL-NEXT: v_cndmask_b32_e64 v5, v13, v5, s1
; GFX11-GISEL-NEXT: v_dual_cndmask_b32 v6, v14, v6 :: v_dual_cndmask_b32 v7, v15, v7
+; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-GISEL-NEXT: v_cmp_gt_i64_e32 vcc_lo, v[0:1], v[4:5]
-; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX11-GISEL-NEXT: v_cmp_gt_i64_e64 s0, v[2:3], v[6:7]
; GFX11-GISEL-NEXT: v_dual_cndmask_b32 v0, v4, v0 :: v_dual_cndmask_b32 v1, v5, v1
; GFX11-GISEL-NEXT: v_cndmask_b32_e64 v2, v6, v2, s0
@@ -3696,9 +3694,9 @@ define i64 @test_vector_reduce_smax_v8i64(<8 x i64> %v) {
; GFX12-SDAG-NEXT: s_wait_alu depctr_va_vcc(0)
; GFX12-SDAG-NEXT: v_dual_cndmask_b32 v3, v7, v3 :: v_dual_cndmask_b32 v2, v6, v2
; GFX12-SDAG-NEXT: s_wait_alu depctr_va_sdst(0)
-; GFX12-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX12-SDAG-NEXT: v_cndmask_b32_e64 v1, v5, v1, s0
; GFX12-SDAG-NEXT: v_cndmask_b32_e64 v0, v4, v0, s0
+; GFX12-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-SDAG-NEXT: v_cmp_gt_i64_e32 vcc_lo, v[0:1], v[2:3]
; GFX12-SDAG-NEXT: s_wait_alu depctr_va_vcc(0)
; GFX12-SDAG-NEXT: v_dual_cndmask_b32 v0, v2, v0 :: v_dual_cndmask_b32 v1, v3, v1
@@ -3730,9 +3728,9 @@ define i64 @test_vector_reduce_smax_v8i64(<8 x i64> %v) {
; GFX12-GISEL-NEXT: s_wait_alu depctr_va_vcc(0)
; GFX12-GISEL-NEXT: v_dual_cndmask_b32 v0, v4, v0 :: v_dual_cndmask_b32 v1, v5, v1
; GFX12-GISEL-NEXT: s_wait_alu depctr_va_sdst(0)
-; GFX12-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX12-GISEL-NEXT: v_cndmask_b32_e64 v2, v6, v2, s0
; GFX12-GISEL-NEXT: v_cndmask_b32_e64 v3, v7, v3, s0
+; GFX12-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-GISEL-NEXT: v_cmp_gt_i64_e32 vcc_lo, v[0:1], v[2:3]
; GFX12-GISEL-NEXT: s_wait_alu depctr_va_vcc(0)
; GFX12-GISEL-NEXT: v_dual_cndmask_b32 v0, v2, v0 :: v_dual_cndmask_b32 v1, v3, v1
@@ -4176,7 +4174,6 @@ define i64 @test_vector_reduce_smax_v16i64(<16 x i64> %v) {
; GFX11-SDAG-NEXT: v_cmp_gt_i64_e64 s1, v[12:13], v[28:29]
; GFX11-SDAG-NEXT: v_cmp_gt_i64_e64 s2, v[6:7], v[22:23]
; GFX11-SDAG-NEXT: v_dual_cndmask_b32 v9, v25, v9 :: v_dual_cndmask_b32 v8, v24, v8
-; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_4)
; GFX11-SDAG-NEXT: v_cndmask_b32_e64 v1, v17, v1, s0
; GFX11-SDAG-NEXT: v_cndmask_b32_e64 v0, v16, v0, s0
; GFX11-SDAG-NEXT: v_cmp_gt_i64_e32 vcc_lo, v[10:11], v[26:27]
@@ -4193,7 +4190,6 @@ define i64 @test_vector_reduce_smax_v16i64(<16 x i64> %v) {
; GFX11-SDAG-NEXT: v_cndmask_b32_e64 v6, v22, v6, s2
; GFX11-SDAG-NEXT: v_cmp_gt_i64_e64 s1, v[0:1], v[8:9]
; GFX11-SDAG-NEXT: v_cmp_gt_i64_e64 s0, v[2:3], v[10:11]
-; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-SDAG-NEXT: v_cndmask_b32_e64 v1, v9, v1, s1
; GFX11-SDAG-NEXT: v_cndmask_b32_e64 v3, v11, v3, s0
; GFX11-SDAG-NEXT: v_cndmask_b32_e64 v2, v10, v2, s0
@@ -4202,16 +4198,15 @@ define i64 @test_vector_reduce_smax_v16i64(<16 x i64> %v) {
; GFX11-SDAG-NEXT: v_cmp_gt_i64_e32 vcc_lo, v[14:15], v[30:31]
; GFX11-SDAG-NEXT: v_dual_cndmask_b32 v15, v31, v15 :: v_dual_cndmask_b32 v14, v30, v14
; GFX11-SDAG-NEXT: v_cmp_gt_i64_e32 vcc_lo, v[4:5], v[12:13]
-; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_2)
+; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_3) | instid1(VALU_DEP_3)
; GFX11-SDAG-NEXT: v_cmp_gt_i64_e64 s0, v[6:7], v[14:15]
; GFX11-SDAG-NEXT: v_dual_cndmask_b32 v5, v13, v5 :: v_dual_cndmask_b32 v4, v12, v4
; GFX11-SDAG-NEXT: v_cndmask_b32_e64 v7, v15, v7, s0
; GFX11-SDAG-NEXT: v_cndmask_b32_e64 v6, v14, v6, s0
-; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-SDAG-NEXT: v_cmp_gt_i64_e32 vcc_lo, v[0:1], v[4:5]
+; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_3) | instid1(VALU_DEP_1)
; GFX11-SDAG-NEXT: v_cmp_gt_i64_e64 s0, v[2:3], v[6:7]
; GFX11-SDAG-NEXT: v_dual_cndmask_b32 v1, v5, v1 :: v_dual_cndmask_b32 v0, v4, v0
-; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-SDAG-NEXT: v_cndmask_b32_e64 v2, v6, v2, s0
; GFX11-SDAG-NEXT: v_cndmask_b32_e64 v3, v7, v3, s0
; GFX11-SDAG-NEXT: v_cmp_gt_i64_e32 vcc_lo, v[0:1], v[2:3]
@@ -4227,7 +4222,6 @@ define i64 @test_vector_reduce_smax_v16i64(<16 x i64> %v) {
; GFX11-GISEL-NEXT: v_cmp_gt_i64_e64 s1, v[4:5], v[20:21]
; GFX11-GISEL-NEXT: v_dual_cndmask_b32 v0, v16, v0 :: v_dual_cndmask_b32 v1, v17, v1
; GFX11-GISEL-NEXT: v_cmp_gt_i64_e32 vcc_lo, v[6:7], v[22:23]
-; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_4)
; GFX11-GISEL-NEXT: v_cndmask_b32_e64 v2, v18, v2, s0
; GFX11-GISEL-NEXT: v_cndmask_b32_e64 v3, v19, v3, s0
; GFX11-GISEL-NEXT: v_cmp_gt_i64_e64 s0, v[8:9], v[24:25]
@@ -4239,32 +4233,30 @@ define i64 @test_vector_reduce_smax_v16i64(<16 x i64> %v) {
; GFX11-GISEL-NEXT: v_cmp_gt_i64_e64 s1, v[12:13], v[28:29]
; GFX11-GISEL-NEXT: v_cndmask_b32_e64 v9, v25, v9, s0
; GFX11-GISEL-NEXT: v_dual_cndmask_b32 v10, v26, v10 :: v_dual_cndmask_b32 v11, v27, v11
-; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_4)
+; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_3) | instid1(VALU_DEP_2)
; GFX11-GISEL-NEXT: v_cmp_gt_i64_e32 vcc_lo, v[0:1], v[8:9]
; GFX11-GISEL-NEXT: v_cndmask_b32_e64 v12, v28, v12, s1
; GFX11-GISEL-NEXT: v_cndmask_b32_e64 v13, v29, v13, s1
; GFX11-GISEL-NEXT: v_dual_cndmask_b32 v0, v8, v0 :: v_dual_cndmask_b32 v1, v9, v1
-; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-GISEL-NEXT: v_cmp_gt_i64_e64 s1, v[4:5], v[12:13]
; GFX11-GISEL-NEXT: v_cndmask_b32_e64 v4, v12, v4, s1
; GFX11-GISEL-NEXT: v_cndmask_b32_e64 v5, v13, v5, s1
; GFX11-GISEL-NEXT: s_waitcnt vmcnt(0)
; GFX11-GISEL-NEXT: v_cmp_gt_i64_e64 s0, v[14:15], v[30:31]
-; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_2) | instid1(VALU_DEP_2)
; GFX11-GISEL-NEXT: v_cndmask_b32_e64 v14, v30, v14, s0
; GFX11-GISEL-NEXT: v_cndmask_b32_e64 v15, v31, v15, s0
; GFX11-GISEL-NEXT: v_cmp_gt_i64_e64 s0, v[2:3], v[10:11]
+; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_4) | instid1(VALU_DEP_2)
; GFX11-GISEL-NEXT: v_cmp_gt_i64_e32 vcc_lo, v[6:7], v[14:15]
-; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_3) | instid1(VALU_DEP_2)
; GFX11-GISEL-NEXT: v_cndmask_b32_e64 v2, v10, v2, s0
; GFX11-GISEL-NEXT: v_cndmask_b32_e64 v3, v11, v3, s0
; GFX11-GISEL-NEXT: v_dual_cndmask_b32 v6, v14, v6 :: v_dual_cndmask_b32 v7, v15, v7
; GFX11-GISEL-NEXT: v_cmp_gt_i64_e32 vcc_lo, v[0:1], v[4:5]
; GFX11-GISEL-NEXT: v_cmp_gt_i64_e64 s0, v[2:3], v[6:7]
; GFX11-GISEL-NEXT: v_dual_cndmask_b32 v0, v4, v0 :: v_dual_cndmask_b32 v1, v5, v1
-; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-GISEL-NEXT: v_cndmask_b32_e64 v2, v6, v2, s0
; GFX11-GISEL-NEXT: v_cndmask_b32_e64 v3, v7, v3, s0
+; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-GISEL-NEXT: v_cmp_gt_i64_e32 vcc_lo, v[0:1], v[2:3]
; GFX11-GISEL-NEXT: v_dual_cndmask_b32 v0, v2, v0 :: v_dual_cndmask_b32 v1, v3, v1
; GFX11-GISEL-NEXT: s_setpc_b64 s[30:31]
@@ -4303,7 +4295,6 @@ define i64 @test_vector_reduce_smax_v16i64(<16 x i64> %v) {
; GFX12-SDAG-NEXT: v_cmp_gt_i64_e64 s1, v[0:1], v[8:9]
; GFX12-SDAG-NEXT: v_cmp_gt_i64_e64 s0, v[2:3], v[10:11]
; GFX12-SDAG-NEXT: s_wait_alu depctr_va_sdst(0)
-; GFX12-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX12-SDAG-NEXT: v_cndmask_b32_e64 v1, v9, v1, s1
; GFX12-SDAG-NEXT: v_cndmask_b32_e64 v3, v11, v3, s0
; GFX12-SDAG-NEXT: v_cndmask_b32_e64 v2, v10, v2, s0
@@ -4313,7 +4304,7 @@ define i64 @test_vector_reduce_smax_v16i64(<16 x i64> %v) {
; GFX12-SDAG-NEXT: s_wait_alu depctr_va_vcc(0)
; GFX12-SDAG-NEXT: v_dual_cndmask_b32 v15, v31, v15 :: v_dual_cndmask_b32 v14, v30, v14
; GFX12-SDAG-NEXT: v_cmp_gt_i64_e32 vcc_lo, v[4:5], v[12:13]
-; GFX12-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_3) | instid1(VALU_DEP_2)
+; GFX12-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX12-SDAG-NEXT: v_cmp_gt_i64_e64 s0, v[6:7], v[14:15]
; GFX12-SDAG-NEXT: s_wait_alu depctr_va_vcc(0)
; GFX12-SDAG-NEXT: v_dual_cndmask_b32 v5, v13, v5 :: v_dual_cndmask_b32 v4, v12, v4
@@ -4326,9 +4317,9 @@ define i64 @test_vector_reduce_smax_v16i64(<16 x i64> %v) {
; GFX12-SDAG-NEXT: s_wait_alu depctr_va_vcc(0)
; GFX12-SDAG-NEXT: v_dual_cndmask_b32 v1, v5, v1 :: v_dual_cndmask_b32 v0, v4, v0
; GFX12-SDAG-NEXT: s_wait_alu depctr_va_sdst(0)
-; GFX12-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX12-SDAG-NEXT: v_cndmask_b32_e64 v2, v6, v2, s0
; GFX12-SDAG-NEXT: v_cndmask_b32_e64 v3, v7, v3, s0
+; GFX12-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-SDAG-NEXT: v_cmp_gt_i64_e32 vcc_lo, v[0:1], v[2:3]
; GFX12-SDAG-NEXT: s_wait_alu depctr_va_vcc(0)
; GFX12-SDAG-NEXT: v_dual_cndmask_b32 v0, v2, v0 :: v_dual_cndmask_b32 v1, v3, v1
@@ -4370,7 +4361,7 @@ define i64 @test_vector_reduce_smax_v16i64(<16 x i64> %v) {
; GFX12-GISEL-NEXT: v_cndmask_b32_e64 v13, v29, v13, s1
; GFX12-GISEL-NEXT: s_wait_alu depctr_va_vcc(0)
; GFX12-GISEL-NEXT: v_dual_cndmask_b32 v0, v8, v0 :: v_dual_cndmask_b32 v1, v9, v1
-; GFX12-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_1)
+; GFX12-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX12-GISEL-NEXT: v_cmp_gt_i64_e64 s1, v[4:5], v[12:13]
; GFX12-GISEL-NEXT: s_wait_alu depctr_va_sdst(0)
; GFX12-GISEL-NEXT: v_cndmask_b32_e64 v4, v12, v4, s1
@@ -4378,25 +4369,25 @@ define i64 @test_vector_reduce_smax_v16i64(<16 x i64> %v) {
; GFX12-GISEL-NEXT: s_wait_loadcnt 0x0
; GFX12-GISEL-NEXT: v_cmp_gt_i64_e64 s0, v[14:15], v[30:31]
; GFX12-GISEL-NEXT: s_wait_alu depctr_va_sdst(0)
-; GFX12-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_2) | instid1(VALU_DEP_2)
; GFX12-GISEL-NEXT: v_cndmask_b32_e64 v14, v30, v14, s0
; GFX12-GISEL-NEXT: v_cndmask_b32_e64 v15, v31, v15, s0
; GFX12-GISEL-NEXT: v_cmp_gt_i64_e64 s0, v[2:3], v[10:11]
+; GFX12-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX12-GISEL-NEXT: v_cmp_gt_i64_e32 vcc_lo, v[6:7], v[14:15]
; GFX12-GISEL-NEXT: s_wait_alu depctr_va_sdst(0)
-; GFX12-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_4) | instid1(VALU_DEP_2)
; GFX12-GISEL-NEXT: v_cndmask_b32_e64 v2, v10, v2, s0
; GFX12-GISEL-NEXT: v_cndmask_b32_e64 v3, v11, v3, s0
; GFX12-GISEL-NEXT: s_wait_alu depctr_va_vcc(0)
; GFX12-GISEL-NEXT: v_dual_cndmask_b32 v6, v14, v6 :: v_dual_cndmask_b32 v7, v15, v7
; GFX12-GISEL-NEXT: v_cmp_gt_i64_e32 vcc_lo, v[0:1], v[4:5]
+; GFX12-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX12-GISEL-NEXT: v_cmp_gt_i64_e64 s0, v[2:3], v[6:7]
; GFX12-GISEL-NEXT: s_wait_alu depctr_va_vcc(0)
; GFX12-GISEL-NEXT: v_dual_cndmask_b32 v0, v4, v0 :: v_dual_cndmask_b32 v1, v5, v1
; GFX12-GISEL-NEXT: s_wait_alu depctr_va_sdst(0)
-; GFX12-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX12-GISEL-NEXT: v_cndmask_b32_e64 v2, v6, v2, s0
; GFX12-GISEL-NEXT: v_cndmask_b32_e64 v3, v7, v3, s0
+; GFX12-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-GISEL-NEXT: v_cmp_gt_i64_e32 vcc_lo, v[0:1], v[2:3]
; GFX12-GISEL-NEXT: s_wait_alu depctr_va_vcc(0)
; GFX12-GISEL-NEXT: v_dual_cndmask_b32 v0, v2, v0 :: v_dual_cndmask_b32 v1, v3, v1
diff --git a/llvm/test/CodeGen/AMDGPU/vector-reduce-smin.ll b/llvm/test/CodeGen/AMDGPU/vector-reduce-smin.ll
index 376a64bb862719..e4e3304fc6bb9f 100644
--- a/llvm/test/CodeGen/AMDGPU/vector-reduce-smin.ll
+++ b/llvm/test/CodeGen/AMDGPU/vector-reduce-smin.ll
@@ -3332,9 +3332,9 @@ define i64 @test_vector_reduce_smin_v4i64(<4 x i64> %v) {
; GFX11-SDAG-NEXT: v_cmp_lt_i64_e32 vcc_lo, v[2:3], v[6:7]
; GFX11-SDAG-NEXT: v_cmp_lt_i64_e64 s0, v[0:1], v[4:5]
; GFX11-SDAG-NEXT: v_dual_cndmask_b32 v3, v7, v3 :: v_dual_cndmask_b32 v2, v6, v2
-; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-SDAG-NEXT: v_cndmask_b32_e64 v1, v5, v1, s0
; GFX11-SDAG-NEXT: v_cndmask_b32_e64 v0, v4, v0, s0
+; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-SDAG-NEXT: v_cmp_lt_i64_e32 vcc_lo, v[0:1], v[2:3]
; GFX11-SDAG-NEXT: v_dual_cndmask_b32 v0, v2, v0 :: v_dual_cndmask_b32 v1, v3, v1
; GFX11-SDAG-NEXT: s_setpc_b64 s[30:31]
@@ -3345,9 +3345,9 @@ define i64 @test_vector_reduce_smin_v4i64(<4 x i64> %v) {
; GFX11-GISEL-NEXT: v_cmp_lt_i64_e32 vcc_lo, v[0:1], v[4:5]
; GFX11-GISEL-NEXT: v_cmp_lt_i64_e64 s0, v[2:3], v[6:7]
; GFX11-GISEL-NEXT: v_dual_cndmask_b32 v0, v4, v0 :: v_dual_cndmask_b32 v1, v5, v1
-; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-GISEL-NEXT: v_cndmask_b32_e64 v2, v6, v2, s0
; GFX11-GISEL-NEXT: v_cndmask_b32_e64 v3, v7, v3, s0
+; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-GISEL-NEXT: v_cmp_lt_i64_e32 vcc_lo, v[0:1], v[2:3]
; GFX11-GISEL-NEXT: v_dual_cndmask_b32 v0, v2, v0 :: v_dual_cndmask_b32 v1, v3, v1
; GFX11-GISEL-NEXT: s_setpc_b64 s[30:31]
@@ -3364,9 +3364,9 @@ define i64 @test_vector_reduce_smin_v4i64(<4 x i64> %v) {
; GFX12-SDAG-NEXT: s_wait_alu depctr_va_vcc(0)
; GFX12-SDAG-NEXT: v_dual_cndmask_b32 v3, v7, v3 :: v_dual_cndmask_b32 v2, v6, v2
; GFX12-SDAG-NEXT: s_wait_alu depctr_va_sdst(0)
-; GFX12-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX12-SDAG-NEXT: v_cndmask_b32_e64 v1, v5, v1, s0
; GFX12-SDAG-NEXT: v_cndmask_b32_e64 v0, v4, v0, s0
+; GFX12-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-SDAG-NEXT: v_cmp_lt_i64_e32 vcc_lo, v[0:1], v[2:3]
; GFX12-SDAG-NEXT: s_wait_alu depctr_va_vcc(0)
; GFX12-SDAG-NEXT: v_dual_cndmask_b32 v0, v2, v0 :: v_dual_cndmask_b32 v1, v3, v1
@@ -3384,9 +3384,9 @@ define i64 @test_vector_reduce_smin_v4i64(<4 x i64> %v) {
; GFX12-GISEL-NEXT: s_wait_alu depctr_va_vcc(0)
; GFX12-GISEL-NEXT: v_dual_cndmask_b32 v0, v4, v0 :: v_dual_cndmask_b32 v1, v5, v1
; GFX12-GISEL-NEXT: s_wait_alu depctr_va_sdst(0)
-; GFX12-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX12-GISEL-NEXT: v_cndmask_b32_e64 v2, v6, v2, s0
; GFX12-GISEL-NEXT: v_cndmask_b32_e64 v3, v7, v3, s0
+; GFX12-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-GISEL-NEXT: v_cmp_lt_i64_e32 vcc_lo, v[0:1], v[2:3]
; GFX12-GISEL-NEXT: s_wait_alu depctr_va_vcc(0)
; GFX12-GISEL-NEXT: v_dual_cndmask_b32 v0, v2, v0 :: v_dual_cndmask_b32 v1, v3, v1
@@ -3627,14 +3627,13 @@ define i64 @test_vector_reduce_smin_v8i64(<8 x i64> %v) {
; GFX11-SDAG-NEXT: v_cmp_lt_i64_e64 s1, v[2:3], v[10:11]
; GFX11-SDAG-NEXT: v_dual_cndmask_b32 v5, v13, v5 :: v_dual_cndmask_b32 v4, v12, v4
; GFX11-SDAG-NEXT: v_cmp_lt_i64_e32 vcc_lo, v[0:1], v[8:9]
-; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(SKIP_4) | instid1(VALU_DEP_2)
; GFX11-SDAG-NEXT: v_cndmask_b32_e64 v7, v15, v7, s0
; GFX11-SDAG-NEXT: v_cndmask_b32_e64 v6, v14, v6, s0
; GFX11-SDAG-NEXT: v_cndmask_b32_e64 v3, v11, v3, s1
; GFX11-SDAG-NEXT: v_cndmask_b32_e64 v2, v10, v2, s1
; GFX11-SDAG-NEXT: v_dual_cndmask_b32 v1, v9, v1 :: v_dual_cndmask_b32 v0, v8, v0
+; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-SDAG-NEXT: v_cmp_lt_i64_e32 vcc_lo, v[2:3], v[6:7]
-; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX11-SDAG-NEXT: v_cmp_lt_i64_e64 s0, v[0:1], v[4:5]
; GFX11-SDAG-NEXT: v_dual_cndmask_b32 v3, v7, v3 :: v_dual_cndmask_b32 v2, v6, v2
; GFX11-SDAG-NEXT: v_cndmask_b32_e64 v1, v5, v1, s0
@@ -3652,14 +3651,13 @@ define i64 @test_vector_reduce_smin_v8i64(<8 x i64> %v) {
; GFX11-GISEL-NEXT: v_cmp_lt_i64_e64 s1, v[4:5], v[12:13]
; GFX11-GISEL-NEXT: v_dual_cndmask_b32 v0, v8, v0 :: v_dual_cndmask_b32 v1, v9, v1
; GFX11-GISEL-NEXT: v_cmp_lt_i64_e32 vcc_lo, v[6:7], v[14:15]
-; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(SKIP_4) | instid1(VALU_DEP_2)
; GFX11-GISEL-NEXT: v_cndmask_b32_e64 v2, v10, v2, s0
; GFX11-GISEL-NEXT: v_cndmask_b32_e64 v3, v11, v3, s0
; GFX11-GISEL-NEXT: v_cndmask_b32_e64 v4, v12, v4, s1
; GFX11-GISEL-NEXT: v_cndmask_b32_e64 v5, v13, v5, s1
; GFX11-GISEL-NEXT: v_dual_cndmask_b32 v6, v14, v6 :: v_dual_cndmask_b32 v7, v15, v7
+; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-GISEL-NEXT: v_cmp_lt_i64_e32 vcc_lo, v[0:1], v[4:5]
-; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX11-GISEL-NEXT: v_cmp_lt_i64_e64 s0, v[2:3], v[6:7]
; GFX11-GISEL-NEXT: v_dual_cndmask_b32 v0, v4, v0 :: v_dual_cndmask_b32 v1, v5, v1
; GFX11-GISEL-NEXT: v_cndmask_b32_e64 v2, v6, v2, s0
@@ -3695,9 +3693,9 @@ define i64 @test_vector_reduce_smin_v8i64(<8 x i64> %v) {
; GFX12-SDAG-NEXT: s_wait_alu depctr_va_vcc(0)
; GFX12-SDAG-NEXT: v_dual_cndmask_b32 v3, v7, v3 :: v_dual_cndmask_b32 v2, v6, v2
; GFX12-SDAG-NEXT: s_wait_alu depctr_va_sdst(0)
-; GFX12-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX12-SDAG-NEXT: v_cndmask_b32_e64 v1, v5, v1, s0
; GFX12-SDAG-NEXT: v_cndmask_b32_e64 v0, v4, v0, s0
+; GFX12-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-SDAG-NEXT: v_cmp_lt_i64_e32 vcc_lo, v[0:1], v[2:3]
; GFX12-SDAG-NEXT: s_wait_alu depctr_va_vcc(0)
; GFX12-SDAG-NEXT: v_dual_cndmask_b32 v0, v2, v0 :: v_dual_cndmask_b32 v1, v3, v1
@@ -3729,9 +3727,9 @@ define i64 @test_vector_reduce_smin_v8i64(<8 x i64> %v) {
; GFX12-GISEL-NEXT: s_wait_alu depctr_va_vcc(0)
; GFX12-GISEL-NEXT: v_dual_cndmask_b32 v0, v4, v0 :: v_dual_cndmask_b32 v1, v5, v1
; GFX12-GISEL-NEXT: s_wait_alu depctr_va_sdst(0)
-; GFX12-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX12-GISEL-NEXT: v_cndmask_b32_e64 v2, v6, v2, s0
; GFX12-GISEL-NEXT: v_cndmask_b32_e64 v3, v7, v3, s0
+; GFX12-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-GISEL-NEXT: v_cmp_lt_i64_e32 vcc_lo, v[0:1], v[2:3]
; GFX12-GISEL-NEXT: s_wait_alu depctr_va_vcc(0)
; GFX12-GISEL-NEXT: v_dual_cndmask_b32 v0, v2, v0 :: v_dual_cndmask_b32 v1, v3, v1
@@ -4175,7 +4173,6 @@ define i64 @test_vector_reduce_smin_v16i64(<16 x i64> %v) {
; GFX11-SDAG-NEXT: v_cmp_lt_i64_e64 s1, v[12:13], v[28:29]
; GFX11-SDAG-NEXT: v_cmp_lt_i64_e64 s2, v[6:7], v[22:23]
; GFX11-SDAG-NEXT: v_dual_cndmask_b32 v9, v25, v9 :: v_dual_cndmask_b32 v8, v24, v8
-; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_4)
; GFX11-SDAG-NEXT: v_cndmask_b32_e64 v1, v17, v1, s0
; GFX11-SDAG-NEXT: v_cndmask_b32_e64 v0, v16, v0, s0
; GFX11-SDAG-NEXT: v_cmp_lt_i64_e32 vcc_lo, v[10:11], v[26:27]
@@ -4192,7 +4189,6 @@ define i64 @test_vector_reduce_smin_v16i64(<16 x i64> %v) {
; GFX11-SDAG-NEXT: v_cndmask_b32_e64 v6, v22, v6, s2
; GFX11-SDAG-NEXT: v_cmp_lt_i64_e64 s1, v[0:1], v[8:9]
; GFX11-SDAG-NEXT: v_cmp_lt_i64_e64 s0, v[2:3], v[10:11]
-; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-SDAG-NEXT: v_cndmask_b32_e64 v1, v9, v1, s1
; GFX11-SDAG-NEXT: v_cndmask_b32_e64 v3, v11, v3, s0
; GFX11-SDAG-NEXT: v_cndmask_b32_e64 v2, v10, v2, s0
@@ -4201,16 +4197,15 @@ define i64 @test_vector_reduce_smin_v16i64(<16 x i64> %v) {
; GFX11-SDAG-NEXT: v_cmp_lt_i64_e32 vcc_lo, v[14:15], v[30:31]
; GFX11-SDAG-NEXT: v_dual_cndmask_b32 v15, v31, v15 :: v_dual_cndmask_b32 v14, v30, v14
; GFX11-SDAG-NEXT: v_cmp_lt_i64_e32 vcc_lo, v[4:5], v[12:13]
-; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_2)
+; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_3) | instid1(VALU_DEP_3)
; GFX11-SDAG-NEXT: v_cmp_lt_i64_e64 s0, v[6:7], v[14:15]
; GFX11-SDAG-NEXT: v_dual_cndmask_b32 v5, v13, v5 :: v_dual_cndmask_b32 v4, v12, v4
; GFX11-SDAG-NEXT: v_cndmask_b32_e64 v7, v15, v7, s0
; GFX11-SDAG-NEXT: v_cndmask_b32_e64 v6, v14, v6, s0
-; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-SDAG-NEXT: v_cmp_lt_i64_e32 vcc_lo, v[0:1], v[4:5]
+; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_3) | instid1(VALU_DEP_1)
; GFX11-SDAG-NEXT: v_cmp_lt_i64_e64 s0, v[2:3], v[6:7]
; GFX11-SDAG-NEXT: v_dual_cndmask_b32 v1, v5, v1 :: v_dual_cndmask_b32 v0, v4, v0
-; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-SDAG-NEXT: v_cndmask_b32_e64 v2, v6, v2, s0
; GFX11-SDAG-NEXT: v_cndmask_b32_e64 v3, v7, v3, s0
; GFX11-SDAG-NEXT: v_cmp_lt_i64_e32 vcc_lo, v[0:1], v[2:3]
@@ -4226,7 +4221,6 @@ define i64 @test_vector_reduce_smin_v16i64(<16 x i64> %v) {
; GFX11-GISEL-NEXT: v_cmp_lt_i64_e64 s1, v[4:5], v[20:21]
; GFX11-GISEL-NEXT: v_dual_cndmask_b32 v0, v16, v0 :: v_dual_cndmask_b32 v1, v17, v1
; GFX11-GISEL-NEXT: v_cmp_lt_i64_e32 vcc_lo, v[6:7], v[22:23]
-; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_4)
; GFX11-GISEL-NEXT: v_cndmask_b32_e64 v2, v18, v2, s0
; GFX11-GISEL-NEXT: v_cndmask_b32_e64 v3, v19, v3, s0
; GFX11-GISEL-NEXT: v_cmp_lt_i64_e64 s0, v[8:9], v[24:25]
@@ -4238,32 +4232,30 @@ define i64 @test_vector_reduce_smin_v16i64(<16 x i64> %v) {
; GFX11-GISEL-NEXT: v_cmp_lt_i64_e64 s1, v[12:13], v[28:29]
; GFX11-GISEL-NEXT: v_cndmask_b32_e64 v9, v25, v9, s0
; GFX11-GISEL-NEXT: v_dual_cndmask_b32 v10, v26, v10 :: v_dual_cndmask_b32 v11, v27, v11
-; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_4)
+; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_3) | instid1(VALU_DEP_2)
; GFX11-GISEL-NEXT: v_cmp_lt_i64_e32 vcc_lo, v[0:1], v[8:9]
; GFX11-GISEL-NEXT: v_cndmask_b32_e64 v12, v28, v12, s1
; GFX11-GISEL-NEXT: v_cndmask_b32_e64 v13, v29, v13, s1
; GFX11-GISEL-NEXT: v_dual_cndmask_b32 v0, v8, v0 :: v_dual_cndmask_b32 v1, v9, v1
-; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-GISEL-NEXT: v_cmp_lt_i64_e64 s1, v[4:5], v[12:13]
; GFX11-GISEL-NEXT: v_cndmask_b32_e64 v4, v12, v4, s1
; GFX11-GISEL-NEXT: v_cndmask_b32_e64 v5, v13, v5, s1
; GFX11-GISEL-NEXT: s_waitcnt vmcnt(0)
; GFX11-GISEL-NEXT: v_cmp_lt_i64_e64 s0, v[14:15], v[30:31]
-; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_2) | instid1(VALU_DEP_2)
; GFX11-GISEL-NEXT: v_cndmask_b32_e64 v14, v30, v14, s0
; GFX11-GISEL-NEXT: v_cndmask_b32_e64 v15, v31, v15, s0
; GFX11-GISEL-NEXT: v_cmp_lt_i64_e64 s0, v[2:3], v[10:11]
+; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_4) | instid1(VALU_DEP_2)
; GFX11-GISEL-NEXT: v_cmp_lt_i64_e32 vcc_lo, v[6:7], v[14:15]
-; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_3) | instid1(VALU_DEP_2)
; GFX11-GISEL-NEXT: v_cndmask_b32_e64 v2, v10, v2, s0
; GFX11-GISEL-NEXT: v_cndmask_b32_e64 v3, v11, v3, s0
; GFX11-GISEL-NEXT: v_dual_cndmask_b32 v6, v14, v6 :: v_dual_cndmask_b32 v7, v15, v7
; GFX11-GISEL-NEXT: v_cmp_lt_i64_e32 vcc_lo, v[0:1], v[4:5]
; GFX11-GISEL-NEXT: v_cmp_lt_i64_e64 s0, v[2:3], v[6:7]
; GFX11-GISEL-NEXT: v_dual_cndmask_b32 v0, v4, v0 :: v_dual_cndmask_b32 v1, v5, v1
-; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-GISEL-NEXT: v_cndmask_b32_e64 v2, v6, v2, s0
; GFX11-GISEL-NEXT: v_cndmask_b32_e64 v3, v7, v3, s0
+; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-GISEL-NEXT: v_cmp_lt_i64_e32 vcc_lo, v[0:1], v[2:3]
; GFX11-GISEL-NEXT: v_dual_cndmask_b32 v0, v2, v0 :: v_dual_cndmask_b32 v1, v3, v1
; GFX11-GISEL-NEXT: s_setpc_b64 s[30:31]
@@ -4302,7 +4294,6 @@ define i64 @test_vector_reduce_smin_v16i64(<16 x i64> %v) {
; GFX12-SDAG-NEXT: v_cmp_lt_i64_e64 s1, v[0:1], v[8:9]
; GFX12-SDAG-NEXT: v_cmp_lt_i64_e64 s0, v[2:3], v[10:11]
; GFX12-SDAG-NEXT: s_wait_alu depctr_va_sdst(0)
-; GFX12-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX12-SDAG-NEXT: v_cndmask_b32_e64 v1, v9, v1, s1
; GFX12-SDAG-NEXT: v_cndmask_b32_e64 v3, v11, v3, s0
; GFX12-SDAG-NEXT: v_cndmask_b32_e64 v2, v10, v2, s0
@@ -4312,7 +4303,7 @@ define i64 @test_vector_reduce_smin_v16i64(<16 x i64> %v) {
; GFX12-SDAG-NEXT: s_wait_alu depctr_va_vcc(0)
; GFX12-SDAG-NEXT: v_dual_cndmask_b32 v15, v31, v15 :: v_dual_cndmask_b32 v14, v30, v14
; GFX12-SDAG-NEXT: v_cmp_lt_i64_e32 vcc_lo, v[4:5], v[12:13]
-; GFX12-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_3) | instid1(VALU_DEP_2)
+; GFX12-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX12-SDAG-NEXT: v_cmp_lt_i64_e64 s0, v[6:7], v[14:15]
; GFX12-SDAG-NEXT: s_wait_alu depctr_va_vcc(0)
; GFX12-SDAG-NEXT: v_dual_cndmask_b32 v5, v13, v5 :: v_dual_cndmask_b32 v4, v12, v4
@@ -4325,9 +4316,9 @@ define i64 @test_vector_reduce_smin_v16i64(<16 x i64> %v) {
; GFX12-SDAG-NEXT: s_wait_alu depctr_va_vcc(0)
; GFX12-SDAG-NEXT: v_dual_cndmask_b32 v1, v5, v1 :: v_dual_cndmask_b32 v0, v4, v0
; GFX12-SDAG-NEXT: s_wait_alu depctr_va_sdst(0)
-; GFX12-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX12-SDAG-NEXT: v_cndmask_b32_e64 v2, v6, v2, s0
; GFX12-SDAG-NEXT: v_cndmask_b32_e64 v3, v7, v3, s0
+; GFX12-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-SDAG-NEXT: v_cmp_lt_i64_e32 vcc_lo, v[0:1], v[2:3]
; GFX12-SDAG-NEXT: s_wait_alu depctr_va_vcc(0)
; GFX12-SDAG-NEXT: v_dual_cndmask_b32 v0, v2, v0 :: v_dual_cndmask_b32 v1, v3, v1
@@ -4369,7 +4360,7 @@ define i64 @test_vector_reduce_smin_v16i64(<16 x i64> %v) {
; GFX12-GISEL-NEXT: v_cndmask_b32_e64 v13, v29, v13, s1
; GFX12-GISEL-NEXT: s_wait_alu depctr_va_vcc(0)
; GFX12-GISEL-NEXT: v_dual_cndmask_b32 v0, v8, v0 :: v_dual_cndmask_b32 v1, v9, v1
-; GFX12-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_1)
+; GFX12-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX12-GISEL-NEXT: v_cmp_lt_i64_e64 s1, v[4:5], v[12:13]
; GFX12-GISEL-NEXT: s_wait_alu depctr_va_sdst(0)
; GFX12-GISEL-NEXT: v_cndmask_b32_e64 v4, v12, v4, s1
@@ -4377,25 +4368,25 @@ define i64 @test_vector_reduce_smin_v16i64(<16 x i64> %v) {
; GFX12-GISEL-NEXT: s_wait_loadcnt 0x0
; GFX12-GISEL-NEXT: v_cmp_lt_i64_e64 s0, v[14:15], v[30:31]
; GFX12-GISEL-NEXT: s_wait_alu depctr_va_sdst(0)
-; GFX12-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_2) | instid1(VALU_DEP_2)
; GFX12-GISEL-NEXT: v_cndmask_b32_e64 v14, v30, v14, s0
; GFX12-GISEL-NEXT: v_cndmask_b32_e64 v15, v31, v15, s0
; GFX12-GISEL-NEXT: v_cmp_lt_i64_e64 s0, v[2:3], v[10:11]
+; GFX12-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX12-GISEL-NEXT: v_cmp_lt_i64_e32 vcc_lo, v[6:7], v[14:15]
; GFX12-GISEL-NEXT: s_wait_alu depctr_va_sdst(0)
-; GFX12-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_4) | instid1(VALU_DEP_2)
; GFX12-GISEL-NEXT: v_cndmask_b32_e64 v2, v10, v2, s0
; GFX12-GISEL-NEXT: v_cndmask_b32_e64 v3, v11, v3, s0
; GFX12-GISEL-NEXT: s_wait_alu depctr_va_vcc(0)
; GFX12-GISEL-NEXT: v_dual_cndmask_b32 v6, v14, v6 :: v_dual_cndmask_b32 v7, v15, v7
; GFX12-GISEL-NEXT: v_cmp_lt_i64_e32 vcc_lo, v[0:1], v[4:5]
+; GFX12-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX12-GISEL-NEXT: v_cmp_lt_i64_e64 s0, v[2:3], v[6:7]
; GFX12-GISEL-NEXT: s_wait_alu depctr_va_vcc(0)
; GFX12-GISEL-NEXT: v_dual_cndmask_b32 v0, v4, v0 :: v_dual_cndmask_b32 v1, v5, v1
; GFX12-GISEL-NEXT: s_wait_alu depctr_va_sdst(0)
-; GFX12-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX12-GISEL-NEXT: v_cndmask_b32_e64 v2, v6, v2, s0
; GFX12-GISEL-NEXT: v_cndmask_b32_e64 v3, v7, v3, s0
+; GFX12-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-GISEL-NEXT: v_cmp_lt_i64_e32 vcc_lo, v[0:1], v[2:3]
; GFX12-GISEL-NEXT: s_wait_alu depctr_va_vcc(0)
; GFX12-GISEL-NEXT: v_dual_cndmask_b32 v0, v2, v0 :: v_dual_cndmask_b32 v1, v3, v1
diff --git a/llvm/test/CodeGen/AMDGPU/vector-reduce-umax.ll b/llvm/test/CodeGen/AMDGPU/vector-reduce-umax.ll
index 087832601598a9..fbcd5cff955434 100644
--- a/llvm/test/CodeGen/AMDGPU/vector-reduce-umax.ll
+++ b/llvm/test/CodeGen/AMDGPU/vector-reduce-umax.ll
@@ -3209,9 +3209,9 @@ define i64 @test_vector_reduce_umax_v4i64(<4 x i64> %v) {
; GFX11-SDAG-NEXT: v_cmp_gt_u64_e32 vcc_lo, v[2:3], v[6:7]
; GFX11-SDAG-NEXT: v_cmp_gt_u64_e64 s0, v[0:1], v[4:5]
; GFX11-SDAG-NEXT: v_dual_cndmask_b32 v3, v7, v3 :: v_dual_cndmask_b32 v2, v6, v2
-; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-SDAG-NEXT: v_cndmask_b32_e64 v1, v5, v1, s0
; GFX11-SDAG-NEXT: v_cndmask_b32_e64 v0, v4, v0, s0
+; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-SDAG-NEXT: v_cmp_gt_u64_e32 vcc_lo, v[0:1], v[2:3]
; GFX11-SDAG-NEXT: v_dual_cndmask_b32 v0, v2, v0 :: v_dual_cndmask_b32 v1, v3, v1
; GFX11-SDAG-NEXT: s_setpc_b64 s[30:31]
@@ -3222,9 +3222,9 @@ define i64 @test_vector_reduce_umax_v4i64(<4 x i64> %v) {
; GFX11-GISEL-NEXT: v_cmp_gt_u64_e32 vcc_lo, v[0:1], v[4:5]
; GFX11-GISEL-NEXT: v_cmp_gt_u64_e64 s0, v[2:3], v[6:7]
; GFX11-GISEL-NEXT: v_dual_cndmask_b32 v0, v4, v0 :: v_dual_cndmask_b32 v1, v5, v1
-; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-GISEL-NEXT: v_cndmask_b32_e64 v2, v6, v2, s0
; GFX11-GISEL-NEXT: v_cndmask_b32_e64 v3, v7, v3, s0
+; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-GISEL-NEXT: v_cmp_gt_u64_e32 vcc_lo, v[0:1], v[2:3]
; GFX11-GISEL-NEXT: v_dual_cndmask_b32 v0, v2, v0 :: v_dual_cndmask_b32 v1, v3, v1
; GFX11-GISEL-NEXT: s_setpc_b64 s[30:31]
@@ -3241,9 +3241,9 @@ define i64 @test_vector_reduce_umax_v4i64(<4 x i64> %v) {
; GFX12-SDAG-NEXT: s_wait_alu depctr_va_vcc(0)
; GFX12-SDAG-NEXT: v_dual_cndmask_b32 v3, v7, v3 :: v_dual_cndmask_b32 v2, v6, v2
; GFX12-SDAG-NEXT: s_wait_alu depctr_va_sdst(0)
-; GFX12-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX12-SDAG-NEXT: v_cndmask_b32_e64 v1, v5, v1, s0
; GFX12-SDAG-NEXT: v_cndmask_b32_e64 v0, v4, v0, s0
+; GFX12-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-SDAG-NEXT: v_cmp_gt_u64_e32 vcc_lo, v[0:1], v[2:3]
; GFX12-SDAG-NEXT: s_wait_alu depctr_va_vcc(0)
; GFX12-SDAG-NEXT: v_dual_cndmask_b32 v0, v2, v0 :: v_dual_cndmask_b32 v1, v3, v1
@@ -3261,9 +3261,9 @@ define i64 @test_vector_reduce_umax_v4i64(<4 x i64> %v) {
; GFX12-GISEL-NEXT: s_wait_alu depctr_va_vcc(0)
; GFX12-GISEL-NEXT: v_dual_cndmask_b32 v0, v4, v0 :: v_dual_cndmask_b32 v1, v5, v1
; GFX12-GISEL-NEXT: s_wait_alu depctr_va_sdst(0)
-; GFX12-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX12-GISEL-NEXT: v_cndmask_b32_e64 v2, v6, v2, s0
; GFX12-GISEL-NEXT: v_cndmask_b32_e64 v3, v7, v3, s0
+; GFX12-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-GISEL-NEXT: v_cmp_gt_u64_e32 vcc_lo, v[0:1], v[2:3]
; GFX12-GISEL-NEXT: s_wait_alu depctr_va_vcc(0)
; GFX12-GISEL-NEXT: v_dual_cndmask_b32 v0, v2, v0 :: v_dual_cndmask_b32 v1, v3, v1
@@ -3504,14 +3504,13 @@ define i64 @test_vector_reduce_umax_v8i64(<8 x i64> %v) {
; GFX11-SDAG-NEXT: v_cmp_gt_u64_e64 s1, v[2:3], v[10:11]
; GFX11-SDAG-NEXT: v_dual_cndmask_b32 v5, v13, v5 :: v_dual_cndmask_b32 v4, v12, v4
; GFX11-SDAG-NEXT: v_cmp_gt_u64_e32 vcc_lo, v[0:1], v[8:9]
-; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(SKIP_4) | instid1(VALU_DEP_2)
; GFX11-SDAG-NEXT: v_cndmask_b32_e64 v7, v15, v7, s0
; GFX11-SDAG-NEXT: v_cndmask_b32_e64 v6, v14, v6, s0
; GFX11-SDAG-NEXT: v_cndmask_b32_e64 v3, v11, v3, s1
; GFX11-SDAG-NEXT: v_cndmask_b32_e64 v2, v10, v2, s1
; GFX11-SDAG-NEXT: v_dual_cndmask_b32 v1, v9, v1 :: v_dual_cndmask_b32 v0, v8, v0
+; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-SDAG-NEXT: v_cmp_gt_u64_e32 vcc_lo, v[2:3], v[6:7]
-; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX11-SDAG-NEXT: v_cmp_gt_u64_e64 s0, v[0:1], v[4:5]
; GFX11-SDAG-NEXT: v_dual_cndmask_b32 v3, v7, v3 :: v_dual_cndmask_b32 v2, v6, v2
; GFX11-SDAG-NEXT: v_cndmask_b32_e64 v1, v5, v1, s0
@@ -3529,14 +3528,13 @@ define i64 @test_vector_reduce_umax_v8i64(<8 x i64> %v) {
; GFX11-GISEL-NEXT: v_cmp_gt_u64_e64 s1, v[4:5], v[12:13]
; GFX11-GISEL-NEXT: v_dual_cndmask_b32 v0, v8, v0 :: v_dual_cndmask_b32 v1, v9, v1
; GFX11-GISEL-NEXT: v_cmp_gt_u64_e32 vcc_lo, v[6:7], v[14:15]
-; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(SKIP_4) | instid1(VALU_DEP_2)
; GFX11-GISEL-NEXT: v_cndmask_b32_e64 v2, v10, v2, s0
; GFX11-GISEL-NEXT: v_cndmask_b32_e64 v3, v11, v3, s0
; GFX11-GISEL-NEXT: v_cndmask_b32_e64 v4, v12, v4, s1
; GFX11-GISEL-NEXT: v_cndmask_b32_e64 v5, v13, v5, s1
; GFX11-GISEL-NEXT: v_dual_cndmask_b32 v6, v14, v6 :: v_dual_cndmask_b32 v7, v15, v7
+; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-GISEL-NEXT: v_cmp_gt_u64_e32 vcc_lo, v[0:1], v[4:5]
-; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX11-GISEL-NEXT: v_cmp_gt_u64_e64 s0, v[2:3], v[6:7]
; GFX11-GISEL-NEXT: v_dual_cndmask_b32 v0, v4, v0 :: v_dual_cndmask_b32 v1, v5, v1
; GFX11-GISEL-NEXT: v_cndmask_b32_e64 v2, v6, v2, s0
@@ -3572,9 +3570,9 @@ define i64 @test_vector_reduce_umax_v8i64(<8 x i64> %v) {
; GFX12-SDAG-NEXT: s_wait_alu depctr_va_vcc(0)
; GFX12-SDAG-NEXT: v_dual_cndmask_b32 v3, v7, v3 :: v_dual_cndmask_b32 v2, v6, v2
; GFX12-SDAG-NEXT: s_wait_alu depctr_va_sdst(0)
-; GFX12-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX12-SDAG-NEXT: v_cndmask_b32_e64 v1, v5, v1, s0
; GFX12-SDAG-NEXT: v_cndmask_b32_e64 v0, v4, v0, s0
+; GFX12-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-SDAG-NEXT: v_cmp_gt_u64_e32 vcc_lo, v[0:1], v[2:3]
; GFX12-SDAG-NEXT: s_wait_alu depctr_va_vcc(0)
; GFX12-SDAG-NEXT: v_dual_cndmask_b32 v0, v2, v0 :: v_dual_cndmask_b32 v1, v3, v1
@@ -3606,9 +3604,9 @@ define i64 @test_vector_reduce_umax_v8i64(<8 x i64> %v) {
; GFX12-GISEL-NEXT: s_wait_alu depctr_va_vcc(0)
; GFX12-GISEL-NEXT: v_dual_cndmask_b32 v0, v4, v0 :: v_dual_cndmask_b32 v1, v5, v1
; GFX12-GISEL-NEXT: s_wait_alu depctr_va_sdst(0)
-; GFX12-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX12-GISEL-NEXT: v_cndmask_b32_e64 v2, v6, v2, s0
; GFX12-GISEL-NEXT: v_cndmask_b32_e64 v3, v7, v3, s0
+; GFX12-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-GISEL-NEXT: v_cmp_gt_u64_e32 vcc_lo, v[0:1], v[2:3]
; GFX12-GISEL-NEXT: s_wait_alu depctr_va_vcc(0)
; GFX12-GISEL-NEXT: v_dual_cndmask_b32 v0, v2, v0 :: v_dual_cndmask_b32 v1, v3, v1
@@ -4052,7 +4050,6 @@ define i64 @test_vector_reduce_umax_v16i64(<16 x i64> %v) {
; GFX11-SDAG-NEXT: v_cmp_gt_u64_e64 s1, v[12:13], v[28:29]
; GFX11-SDAG-NEXT: v_cmp_gt_u64_e64 s2, v[6:7], v[22:23]
; GFX11-SDAG-NEXT: v_dual_cndmask_b32 v9, v25, v9 :: v_dual_cndmask_b32 v8, v24, v8
-; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_4)
; GFX11-SDAG-NEXT: v_cndmask_b32_e64 v1, v17, v1, s0
; GFX11-SDAG-NEXT: v_cndmask_b32_e64 v0, v16, v0, s0
; GFX11-SDAG-NEXT: v_cmp_gt_u64_e32 vcc_lo, v[10:11], v[26:27]
@@ -4069,7 +4066,6 @@ define i64 @test_vector_reduce_umax_v16i64(<16 x i64> %v) {
; GFX11-SDAG-NEXT: v_cndmask_b32_e64 v6, v22, v6, s2
; GFX11-SDAG-NEXT: v_cmp_gt_u64_e64 s1, v[0:1], v[8:9]
; GFX11-SDAG-NEXT: v_cmp_gt_u64_e64 s0, v[2:3], v[10:11]
-; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-SDAG-NEXT: v_cndmask_b32_e64 v1, v9, v1, s1
; GFX11-SDAG-NEXT: v_cndmask_b32_e64 v3, v11, v3, s0
; GFX11-SDAG-NEXT: v_cndmask_b32_e64 v2, v10, v2, s0
@@ -4078,16 +4074,15 @@ define i64 @test_vector_reduce_umax_v16i64(<16 x i64> %v) {
; GFX11-SDAG-NEXT: v_cmp_gt_u64_e32 vcc_lo, v[14:15], v[30:31]
; GFX11-SDAG-NEXT: v_dual_cndmask_b32 v15, v31, v15 :: v_dual_cndmask_b32 v14, v30, v14
; GFX11-SDAG-NEXT: v_cmp_gt_u64_e32 vcc_lo, v[4:5], v[12:13]
-; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_2)
+; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_3) | instid1(VALU_DEP_3)
; GFX11-SDAG-NEXT: v_cmp_gt_u64_e64 s0, v[6:7], v[14:15]
; GFX11-SDAG-NEXT: v_dual_cndmask_b32 v5, v13, v5 :: v_dual_cndmask_b32 v4, v12, v4
; GFX11-SDAG-NEXT: v_cndmask_b32_e64 v7, v15, v7, s0
; GFX11-SDAG-NEXT: v_cndmask_b32_e64 v6, v14, v6, s0
-; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-SDAG-NEXT: v_cmp_gt_u64_e32 vcc_lo, v[0:1], v[4:5]
+; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_3) | instid1(VALU_DEP_1)
; GFX11-SDAG-NEXT: v_cmp_gt_u64_e64 s0, v[2:3], v[6:7]
; GFX11-SDAG-NEXT: v_dual_cndmask_b32 v1, v5, v1 :: v_dual_cndmask_b32 v0, v4, v0
-; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-SDAG-NEXT: v_cndmask_b32_e64 v2, v6, v2, s0
; GFX11-SDAG-NEXT: v_cndmask_b32_e64 v3, v7, v3, s0
; GFX11-SDAG-NEXT: v_cmp_gt_u64_e32 vcc_lo, v[0:1], v[2:3]
@@ -4103,7 +4098,6 @@ define i64 @test_vector_reduce_umax_v16i64(<16 x i64> %v) {
; GFX11-GISEL-NEXT: v_cmp_gt_u64_e64 s1, v[4:5], v[20:21]
; GFX11-GISEL-NEXT: v_dual_cndmask_b32 v0, v16, v0 :: v_dual_cndmask_b32 v1, v17, v1
; GFX11-GISEL-NEXT: v_cmp_gt_u64_e32 vcc_lo, v[6:7], v[22:23]
-; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_4)
; GFX11-GISEL-NEXT: v_cndmask_b32_e64 v2, v18, v2, s0
; GFX11-GISEL-NEXT: v_cndmask_b32_e64 v3, v19, v3, s0
; GFX11-GISEL-NEXT: v_cmp_gt_u64_e64 s0, v[8:9], v[24:25]
@@ -4115,32 +4109,30 @@ define i64 @test_vector_reduce_umax_v16i64(<16 x i64> %v) {
; GFX11-GISEL-NEXT: v_cmp_gt_u64_e64 s1, v[12:13], v[28:29]
; GFX11-GISEL-NEXT: v_cndmask_b32_e64 v9, v25, v9, s0
; GFX11-GISEL-NEXT: v_dual_cndmask_b32 v10, v26, v10 :: v_dual_cndmask_b32 v11, v27, v11
-; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_4)
+; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_3) | instid1(VALU_DEP_2)
; GFX11-GISEL-NEXT: v_cmp_gt_u64_e32 vcc_lo, v[0:1], v[8:9]
; GFX11-GISEL-NEXT: v_cndmask_b32_e64 v12, v28, v12, s1
; GFX11-GISEL-NEXT: v_cndmask_b32_e64 v13, v29, v13, s1
; GFX11-GISEL-NEXT: v_dual_cndmask_b32 v0, v8, v0 :: v_dual_cndmask_b32 v1, v9, v1
-; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-GISEL-NEXT: v_cmp_gt_u64_e64 s1, v[4:5], v[12:13]
; GFX11-GISEL-NEXT: v_cndmask_b32_e64 v4, v12, v4, s1
; GFX11-GISEL-NEXT: v_cndmask_b32_e64 v5, v13, v5, s1
; GFX11-GISEL-NEXT: s_waitcnt vmcnt(0)
; GFX11-GISEL-NEXT: v_cmp_gt_u64_e64 s0, v[14:15], v[30:31]
-; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_2) | instid1(VALU_DEP_2)
; GFX11-GISEL-NEXT: v_cndmask_b32_e64 v14, v30, v14, s0
; GFX11-GISEL-NEXT: v_cndmask_b32_e64 v15, v31, v15, s0
; GFX11-GISEL-NEXT: v_cmp_gt_u64_e64 s0, v[2:3], v[10:11]
+; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_4) | instid1(VALU_DEP_2)
; GFX11-GISEL-NEXT: v_cmp_gt_u64_e32 vcc_lo, v[6:7], v[14:15]
-; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_3) | instid1(VALU_DEP_2)
; GFX11-GISEL-NEXT: v_cndmask_b32_e64 v2, v10, v2, s0
; GFX11-GISEL-NEXT: v_cndmask_b32_e64 v3, v11, v3, s0
; GFX11-GISEL-NEXT: v_dual_cndmask_b32 v6, v14, v6 :: v_dual_cndmask_b32 v7, v15, v7
; GFX11-GISEL-NEXT: v_cmp_gt_u64_e32 vcc_lo, v[0:1], v[4:5]
; GFX11-GISEL-NEXT: v_cmp_gt_u64_e64 s0, v[2:3], v[6:7]
; GFX11-GISEL-NEXT: v_dual_cndmask_b32 v0, v4, v0 :: v_dual_cndmask_b32 v1, v5, v1
-; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-GISEL-NEXT: v_cndmask_b32_e64 v2, v6, v2, s0
; GFX11-GISEL-NEXT: v_cndmask_b32_e64 v3, v7, v3, s0
+; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-GISEL-NEXT: v_cmp_gt_u64_e32 vcc_lo, v[0:1], v[2:3]
; GFX11-GISEL-NEXT: v_dual_cndmask_b32 v0, v2, v0 :: v_dual_cndmask_b32 v1, v3, v1
; GFX11-GISEL-NEXT: s_setpc_b64 s[30:31]
@@ -4179,7 +4171,6 @@ define i64 @test_vector_reduce_umax_v16i64(<16 x i64> %v) {
; GFX12-SDAG-NEXT: v_cmp_gt_u64_e64 s1, v[0:1], v[8:9]
; GFX12-SDAG-NEXT: v_cmp_gt_u64_e64 s0, v[2:3], v[10:11]
; GFX12-SDAG-NEXT: s_wait_alu depctr_va_sdst(0)
-; GFX12-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX12-SDAG-NEXT: v_cndmask_b32_e64 v1, v9, v1, s1
; GFX12-SDAG-NEXT: v_cndmask_b32_e64 v3, v11, v3, s0
; GFX12-SDAG-NEXT: v_cndmask_b32_e64 v2, v10, v2, s0
@@ -4189,7 +4180,7 @@ define i64 @test_vector_reduce_umax_v16i64(<16 x i64> %v) {
; GFX12-SDAG-NEXT: s_wait_alu depctr_va_vcc(0)
; GFX12-SDAG-NEXT: v_dual_cndmask_b32 v15, v31, v15 :: v_dual_cndmask_b32 v14, v30, v14
; GFX12-SDAG-NEXT: v_cmp_gt_u64_e32 vcc_lo, v[4:5], v[12:13]
-; GFX12-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_3) | instid1(VALU_DEP_2)
+; GFX12-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX12-SDAG-NEXT: v_cmp_gt_u64_e64 s0, v[6:7], v[14:15]
; GFX12-SDAG-NEXT: s_wait_alu depctr_va_vcc(0)
; GFX12-SDAG-NEXT: v_dual_cndmask_b32 v5, v13, v5 :: v_dual_cndmask_b32 v4, v12, v4
@@ -4202,9 +4193,9 @@ define i64 @test_vector_reduce_umax_v16i64(<16 x i64> %v) {
; GFX12-SDAG-NEXT: s_wait_alu depctr_va_vcc(0)
; GFX12-SDAG-NEXT: v_dual_cndmask_b32 v1, v5, v1 :: v_dual_cndmask_b32 v0, v4, v0
; GFX12-SDAG-NEXT: s_wait_alu depctr_va_sdst(0)
-; GFX12-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX12-SDAG-NEXT: v_cndmask_b32_e64 v2, v6, v2, s0
; GFX12-SDAG-NEXT: v_cndmask_b32_e64 v3, v7, v3, s0
+; GFX12-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-SDAG-NEXT: v_cmp_gt_u64_e32 vcc_lo, v[0:1], v[2:3]
; GFX12-SDAG-NEXT: s_wait_alu depctr_va_vcc(0)
; GFX12-SDAG-NEXT: v_dual_cndmask_b32 v0, v2, v0 :: v_dual_cndmask_b32 v1, v3, v1
@@ -4246,7 +4237,7 @@ define i64 @test_vector_reduce_umax_v16i64(<16 x i64> %v) {
; GFX12-GISEL-NEXT: v_cndmask_b32_e64 v13, v29, v13, s1
; GFX12-GISEL-NEXT: s_wait_alu depctr_va_vcc(0)
; GFX12-GISEL-NEXT: v_dual_cndmask_b32 v0, v8, v0 :: v_dual_cndmask_b32 v1, v9, v1
-; GFX12-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_1)
+; GFX12-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX12-GISEL-NEXT: v_cmp_gt_u64_e64 s1, v[4:5], v[12:13]
; GFX12-GISEL-NEXT: s_wait_alu depctr_va_sdst(0)
; GFX12-GISEL-NEXT: v_cndmask_b32_e64 v4, v12, v4, s1
@@ -4254,25 +4245,25 @@ define i64 @test_vector_reduce_umax_v16i64(<16 x i64> %v) {
; GFX12-GISEL-NEXT: s_wait_loadcnt 0x0
; GFX12-GISEL-NEXT: v_cmp_gt_u64_e64 s0, v[14:15], v[30:31]
; GFX12-GISEL-NEXT: s_wait_alu depctr_va_sdst(0)
-; GFX12-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_2) | instid1(VALU_DEP_2)
; GFX12-GISEL-NEXT: v_cndmask_b32_e64 v14, v30, v14, s0
; GFX12-GISEL-NEXT: v_cndmask_b32_e64 v15, v31, v15, s0
; GFX12-GISEL-NEXT: v_cmp_gt_u64_e64 s0, v[2:3], v[10:11]
+; GFX12-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX12-GISEL-NEXT: v_cmp_gt_u64_e32 vcc_lo, v[6:7], v[14:15]
; GFX12-GISEL-NEXT: s_wait_alu depctr_va_sdst(0)
-; GFX12-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_4) | instid1(VALU_DEP_2)
; GFX12-GISEL-NEXT: v_cndmask_b32_e64 v2, v10, v2, s0
; GFX12-GISEL-NEXT: v_cndmask_b32_e64 v3, v11, v3, s0
; GFX12-GISEL-NEXT: s_wait_alu depctr_va_vcc(0)
; GFX12-GISEL-NEXT: v_dual_cndmask_b32 v6, v14, v6 :: v_dual_cndmask_b32 v7, v15, v7
; GFX12-GISEL-NEXT: v_cmp_gt_u64_e32 vcc_lo, v[0:1], v[4:5]
+; GFX12-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX12-GISEL-NEXT: v_cmp_gt_u64_e64 s0, v[2:3], v[6:7]
; GFX12-GISEL-NEXT: s_wait_alu depctr_va_vcc(0)
; GFX12-GISEL-NEXT: v_dual_cndmask_b32 v0, v4, v0 :: v_dual_cndmask_b32 v1, v5, v1
; GFX12-GISEL-NEXT: s_wait_alu depctr_va_sdst(0)
-; GFX12-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX12-GISEL-NEXT: v_cndmask_b32_e64 v2, v6, v2, s0
; GFX12-GISEL-NEXT: v_cndmask_b32_e64 v3, v7, v3, s0
+; GFX12-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-GISEL-NEXT: v_cmp_gt_u64_e32 vcc_lo, v[0:1], v[2:3]
; GFX12-GISEL-NEXT: s_wait_alu depctr_va_vcc(0)
; GFX12-GISEL-NEXT: v_dual_cndmask_b32 v0, v2, v0 :: v_dual_cndmask_b32 v1, v3, v1
diff --git a/llvm/test/CodeGen/AMDGPU/vector-reduce-umin.ll b/llvm/test/CodeGen/AMDGPU/vector-reduce-umin.ll
index 5443cce424a5c0..c591ec1f537e20 100644
--- a/llvm/test/CodeGen/AMDGPU/vector-reduce-umin.ll
+++ b/llvm/test/CodeGen/AMDGPU/vector-reduce-umin.ll
@@ -2982,9 +2982,9 @@ define i64 @test_vector_reduce_umin_v4i64(<4 x i64> %v) {
; GFX11-SDAG-NEXT: v_cmp_lt_u64_e32 vcc_lo, v[2:3], v[6:7]
; GFX11-SDAG-NEXT: v_cmp_lt_u64_e64 s0, v[0:1], v[4:5]
; GFX11-SDAG-NEXT: v_dual_cndmask_b32 v3, v7, v3 :: v_dual_cndmask_b32 v2, v6, v2
-; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-SDAG-NEXT: v_cndmask_b32_e64 v1, v5, v1, s0
; GFX11-SDAG-NEXT: v_cndmask_b32_e64 v0, v4, v0, s0
+; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-SDAG-NEXT: v_cmp_lt_u64_e32 vcc_lo, v[0:1], v[2:3]
; GFX11-SDAG-NEXT: v_dual_cndmask_b32 v0, v2, v0 :: v_dual_cndmask_b32 v1, v3, v1
; GFX11-SDAG-NEXT: s_setpc_b64 s[30:31]
@@ -2995,9 +2995,9 @@ define i64 @test_vector_reduce_umin_v4i64(<4 x i64> %v) {
; GFX11-GISEL-NEXT: v_cmp_lt_u64_e32 vcc_lo, v[0:1], v[4:5]
; GFX11-GISEL-NEXT: v_cmp_lt_u64_e64 s0, v[2:3], v[6:7]
; GFX11-GISEL-NEXT: v_dual_cndmask_b32 v0, v4, v0 :: v_dual_cndmask_b32 v1, v5, v1
-; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-GISEL-NEXT: v_cndmask_b32_e64 v2, v6, v2, s0
; GFX11-GISEL-NEXT: v_cndmask_b32_e64 v3, v7, v3, s0
+; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-GISEL-NEXT: v_cmp_lt_u64_e32 vcc_lo, v[0:1], v[2:3]
; GFX11-GISEL-NEXT: v_dual_cndmask_b32 v0, v2, v0 :: v_dual_cndmask_b32 v1, v3, v1
; GFX11-GISEL-NEXT: s_setpc_b64 s[30:31]
@@ -3014,9 +3014,9 @@ define i64 @test_vector_reduce_umin_v4i64(<4 x i64> %v) {
; GFX12-SDAG-NEXT: s_wait_alu depctr_va_vcc(0)
; GFX12-SDAG-NEXT: v_dual_cndmask_b32 v3, v7, v3 :: v_dual_cndmask_b32 v2, v6, v2
; GFX12-SDAG-NEXT: s_wait_alu depctr_va_sdst(0)
-; GFX12-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX12-SDAG-NEXT: v_cndmask_b32_e64 v1, v5, v1, s0
; GFX12-SDAG-NEXT: v_cndmask_b32_e64 v0, v4, v0, s0
+; GFX12-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-SDAG-NEXT: v_cmp_lt_u64_e32 vcc_lo, v[0:1], v[2:3]
; GFX12-SDAG-NEXT: s_wait_alu depctr_va_vcc(0)
; GFX12-SDAG-NEXT: v_dual_cndmask_b32 v0, v2, v0 :: v_dual_cndmask_b32 v1, v3, v1
@@ -3034,9 +3034,9 @@ define i64 @test_vector_reduce_umin_v4i64(<4 x i64> %v) {
; GFX12-GISEL-NEXT: s_wait_alu depctr_va_vcc(0)
; GFX12-GISEL-NEXT: v_dual_cndmask_b32 v0, v4, v0 :: v_dual_cndmask_b32 v1, v5, v1
; GFX12-GISEL-NEXT: s_wait_alu depctr_va_sdst(0)
-; GFX12-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX12-GISEL-NEXT: v_cndmask_b32_e64 v2, v6, v2, s0
; GFX12-GISEL-NEXT: v_cndmask_b32_e64 v3, v7, v3, s0
+; GFX12-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-GISEL-NEXT: v_cmp_lt_u64_e32 vcc_lo, v[0:1], v[2:3]
; GFX12-GISEL-NEXT: s_wait_alu depctr_va_vcc(0)
; GFX12-GISEL-NEXT: v_dual_cndmask_b32 v0, v2, v0 :: v_dual_cndmask_b32 v1, v3, v1
@@ -3277,14 +3277,13 @@ define i64 @test_vector_reduce_umin_v8i64(<8 x i64> %v) {
; GFX11-SDAG-NEXT: v_cmp_lt_u64_e64 s1, v[2:3], v[10:11]
; GFX11-SDAG-NEXT: v_dual_cndmask_b32 v5, v13, v5 :: v_dual_cndmask_b32 v4, v12, v4
; GFX11-SDAG-NEXT: v_cmp_lt_u64_e32 vcc_lo, v[0:1], v[8:9]
-; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(SKIP_4) | instid1(VALU_DEP_2)
; GFX11-SDAG-NEXT: v_cndmask_b32_e64 v7, v15, v7, s0
; GFX11-SDAG-NEXT: v_cndmask_b32_e64 v6, v14, v6, s0
; GFX11-SDAG-NEXT: v_cndmask_b32_e64 v3, v11, v3, s1
; GFX11-SDAG-NEXT: v_cndmask_b32_e64 v2, v10, v2, s1
; GFX11-SDAG-NEXT: v_dual_cndmask_b32 v1, v9, v1 :: v_dual_cndmask_b32 v0, v8, v0
+; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-SDAG-NEXT: v_cmp_lt_u64_e32 vcc_lo, v[2:3], v[6:7]
-; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX11-SDAG-NEXT: v_cmp_lt_u64_e64 s0, v[0:1], v[4:5]
; GFX11-SDAG-NEXT: v_dual_cndmask_b32 v3, v7, v3 :: v_dual_cndmask_b32 v2, v6, v2
; GFX11-SDAG-NEXT: v_cndmask_b32_e64 v1, v5, v1, s0
@@ -3302,14 +3301,13 @@ define i64 @test_vector_reduce_umin_v8i64(<8 x i64> %v) {
; GFX11-GISEL-NEXT: v_cmp_lt_u64_e64 s1, v[4:5], v[12:13]
; GFX11-GISEL-NEXT: v_dual_cndmask_b32 v0, v8, v0 :: v_dual_cndmask_b32 v1, v9, v1
; GFX11-GISEL-NEXT: v_cmp_lt_u64_e32 vcc_lo, v[6:7], v[14:15]
-; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_4) | instskip(SKIP_4) | instid1(VALU_DEP_2)
; GFX11-GISEL-NEXT: v_cndmask_b32_e64 v2, v10, v2, s0
; GFX11-GISEL-NEXT: v_cndmask_b32_e64 v3, v11, v3, s0
; GFX11-GISEL-NEXT: v_cndmask_b32_e64 v4, v12, v4, s1
; GFX11-GISEL-NEXT: v_cndmask_b32_e64 v5, v13, v5, s1
; GFX11-GISEL-NEXT: v_dual_cndmask_b32 v6, v14, v6 :: v_dual_cndmask_b32 v7, v15, v7
+; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-GISEL-NEXT: v_cmp_lt_u64_e32 vcc_lo, v[0:1], v[4:5]
-; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_2)
; GFX11-GISEL-NEXT: v_cmp_lt_u64_e64 s0, v[2:3], v[6:7]
; GFX11-GISEL-NEXT: v_dual_cndmask_b32 v0, v4, v0 :: v_dual_cndmask_b32 v1, v5, v1
; GFX11-GISEL-NEXT: v_cndmask_b32_e64 v2, v6, v2, s0
@@ -3345,9 +3343,9 @@ define i64 @test_vector_reduce_umin_v8i64(<8 x i64> %v) {
; GFX12-SDAG-NEXT: s_wait_alu depctr_va_vcc(0)
; GFX12-SDAG-NEXT: v_dual_cndmask_b32 v3, v7, v3 :: v_dual_cndmask_b32 v2, v6, v2
; GFX12-SDAG-NEXT: s_wait_alu depctr_va_sdst(0)
-; GFX12-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX12-SDAG-NEXT: v_cndmask_b32_e64 v1, v5, v1, s0
; GFX12-SDAG-NEXT: v_cndmask_b32_e64 v0, v4, v0, s0
+; GFX12-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-SDAG-NEXT: v_cmp_lt_u64_e32 vcc_lo, v[0:1], v[2:3]
; GFX12-SDAG-NEXT: s_wait_alu depctr_va_vcc(0)
; GFX12-SDAG-NEXT: v_dual_cndmask_b32 v0, v2, v0 :: v_dual_cndmask_b32 v1, v3, v1
@@ -3379,9 +3377,9 @@ define i64 @test_vector_reduce_umin_v8i64(<8 x i64> %v) {
; GFX12-GISEL-NEXT: s_wait_alu depctr_va_vcc(0)
; GFX12-GISEL-NEXT: v_dual_cndmask_b32 v0, v4, v0 :: v_dual_cndmask_b32 v1, v5, v1
; GFX12-GISEL-NEXT: s_wait_alu depctr_va_sdst(0)
-; GFX12-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX12-GISEL-NEXT: v_cndmask_b32_e64 v2, v6, v2, s0
; GFX12-GISEL-NEXT: v_cndmask_b32_e64 v3, v7, v3, s0
+; GFX12-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-GISEL-NEXT: v_cmp_lt_u64_e32 vcc_lo, v[0:1], v[2:3]
; GFX12-GISEL-NEXT: s_wait_alu depctr_va_vcc(0)
; GFX12-GISEL-NEXT: v_dual_cndmask_b32 v0, v2, v0 :: v_dual_cndmask_b32 v1, v3, v1
@@ -3825,7 +3823,6 @@ define i64 @test_vector_reduce_umin_v16i64(<16 x i64> %v) {
; GFX11-SDAG-NEXT: v_cmp_lt_u64_e64 s1, v[12:13], v[28:29]
; GFX11-SDAG-NEXT: v_cmp_lt_u64_e64 s2, v[6:7], v[22:23]
; GFX11-SDAG-NEXT: v_dual_cndmask_b32 v9, v25, v9 :: v_dual_cndmask_b32 v8, v24, v8
-; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_4)
; GFX11-SDAG-NEXT: v_cndmask_b32_e64 v1, v17, v1, s0
; GFX11-SDAG-NEXT: v_cndmask_b32_e64 v0, v16, v0, s0
; GFX11-SDAG-NEXT: v_cmp_lt_u64_e32 vcc_lo, v[10:11], v[26:27]
@@ -3842,7 +3839,6 @@ define i64 @test_vector_reduce_umin_v16i64(<16 x i64> %v) {
; GFX11-SDAG-NEXT: v_cndmask_b32_e64 v6, v22, v6, s2
; GFX11-SDAG-NEXT: v_cmp_lt_u64_e64 s1, v[0:1], v[8:9]
; GFX11-SDAG-NEXT: v_cmp_lt_u64_e64 s0, v[2:3], v[10:11]
-; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-SDAG-NEXT: v_cndmask_b32_e64 v1, v9, v1, s1
; GFX11-SDAG-NEXT: v_cndmask_b32_e64 v3, v11, v3, s0
; GFX11-SDAG-NEXT: v_cndmask_b32_e64 v2, v10, v2, s0
@@ -3851,16 +3847,15 @@ define i64 @test_vector_reduce_umin_v16i64(<16 x i64> %v) {
; GFX11-SDAG-NEXT: v_cmp_lt_u64_e32 vcc_lo, v[14:15], v[30:31]
; GFX11-SDAG-NEXT: v_dual_cndmask_b32 v15, v31, v15 :: v_dual_cndmask_b32 v14, v30, v14
; GFX11-SDAG-NEXT: v_cmp_lt_u64_e32 vcc_lo, v[4:5], v[12:13]
-; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_2)
+; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_3) | instid1(VALU_DEP_3)
; GFX11-SDAG-NEXT: v_cmp_lt_u64_e64 s0, v[6:7], v[14:15]
; GFX11-SDAG-NEXT: v_dual_cndmask_b32 v5, v13, v5 :: v_dual_cndmask_b32 v4, v12, v4
; GFX11-SDAG-NEXT: v_cndmask_b32_e64 v7, v15, v7, s0
; GFX11-SDAG-NEXT: v_cndmask_b32_e64 v6, v14, v6, s0
-; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX11-SDAG-NEXT: v_cmp_lt_u64_e32 vcc_lo, v[0:1], v[4:5]
+; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_3) | instid1(VALU_DEP_1)
; GFX11-SDAG-NEXT: v_cmp_lt_u64_e64 s0, v[2:3], v[6:7]
; GFX11-SDAG-NEXT: v_dual_cndmask_b32 v1, v5, v1 :: v_dual_cndmask_b32 v0, v4, v0
-; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-SDAG-NEXT: v_cndmask_b32_e64 v2, v6, v2, s0
; GFX11-SDAG-NEXT: v_cndmask_b32_e64 v3, v7, v3, s0
; GFX11-SDAG-NEXT: v_cmp_lt_u64_e32 vcc_lo, v[0:1], v[2:3]
@@ -3876,7 +3871,6 @@ define i64 @test_vector_reduce_umin_v16i64(<16 x i64> %v) {
; GFX11-GISEL-NEXT: v_cmp_lt_u64_e64 s1, v[4:5], v[20:21]
; GFX11-GISEL-NEXT: v_dual_cndmask_b32 v0, v16, v0 :: v_dual_cndmask_b32 v1, v17, v1
; GFX11-GISEL-NEXT: v_cmp_lt_u64_e32 vcc_lo, v[6:7], v[22:23]
-; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_4)
; GFX11-GISEL-NEXT: v_cndmask_b32_e64 v2, v18, v2, s0
; GFX11-GISEL-NEXT: v_cndmask_b32_e64 v3, v19, v3, s0
; GFX11-GISEL-NEXT: v_cmp_lt_u64_e64 s0, v[8:9], v[24:25]
@@ -3888,32 +3882,30 @@ define i64 @test_vector_reduce_umin_v16i64(<16 x i64> %v) {
; GFX11-GISEL-NEXT: v_cmp_lt_u64_e64 s1, v[12:13], v[28:29]
; GFX11-GISEL-NEXT: v_cndmask_b32_e64 v9, v25, v9, s0
; GFX11-GISEL-NEXT: v_dual_cndmask_b32 v10, v26, v10 :: v_dual_cndmask_b32 v11, v27, v11
-; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_4)
+; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_3) | instid1(VALU_DEP_2)
; GFX11-GISEL-NEXT: v_cmp_lt_u64_e32 vcc_lo, v[0:1], v[8:9]
; GFX11-GISEL-NEXT: v_cndmask_b32_e64 v12, v28, v12, s1
; GFX11-GISEL-NEXT: v_cndmask_b32_e64 v13, v29, v13, s1
; GFX11-GISEL-NEXT: v_dual_cndmask_b32 v0, v8, v0 :: v_dual_cndmask_b32 v1, v9, v1
-; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-GISEL-NEXT: v_cmp_lt_u64_e64 s1, v[4:5], v[12:13]
; GFX11-GISEL-NEXT: v_cndmask_b32_e64 v4, v12, v4, s1
; GFX11-GISEL-NEXT: v_cndmask_b32_e64 v5, v13, v5, s1
; GFX11-GISEL-NEXT: s_waitcnt vmcnt(0)
; GFX11-GISEL-NEXT: v_cmp_lt_u64_e64 s0, v[14:15], v[30:31]
-; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_2) | instid1(VALU_DEP_2)
; GFX11-GISEL-NEXT: v_cndmask_b32_e64 v14, v30, v14, s0
; GFX11-GISEL-NEXT: v_cndmask_b32_e64 v15, v31, v15, s0
; GFX11-GISEL-NEXT: v_cmp_lt_u64_e64 s0, v[2:3], v[10:11]
+; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_4) | instid1(VALU_DEP_2)
; GFX11-GISEL-NEXT: v_cmp_lt_u64_e32 vcc_lo, v[6:7], v[14:15]
-; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_3) | instid1(VALU_DEP_2)
; GFX11-GISEL-NEXT: v_cndmask_b32_e64 v2, v10, v2, s0
; GFX11-GISEL-NEXT: v_cndmask_b32_e64 v3, v11, v3, s0
; GFX11-GISEL-NEXT: v_dual_cndmask_b32 v6, v14, v6 :: v_dual_cndmask_b32 v7, v15, v7
; GFX11-GISEL-NEXT: v_cmp_lt_u64_e32 vcc_lo, v[0:1], v[4:5]
; GFX11-GISEL-NEXT: v_cmp_lt_u64_e64 s0, v[2:3], v[6:7]
; GFX11-GISEL-NEXT: v_dual_cndmask_b32 v0, v4, v0 :: v_dual_cndmask_b32 v1, v5, v1
-; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-GISEL-NEXT: v_cndmask_b32_e64 v2, v6, v2, s0
; GFX11-GISEL-NEXT: v_cndmask_b32_e64 v3, v7, v3, s0
+; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-GISEL-NEXT: v_cmp_lt_u64_e32 vcc_lo, v[0:1], v[2:3]
; GFX11-GISEL-NEXT: v_dual_cndmask_b32 v0, v2, v0 :: v_dual_cndmask_b32 v1, v3, v1
; GFX11-GISEL-NEXT: s_setpc_b64 s[30:31]
@@ -3952,7 +3944,6 @@ define i64 @test_vector_reduce_umin_v16i64(<16 x i64> %v) {
; GFX12-SDAG-NEXT: v_cmp_lt_u64_e64 s1, v[0:1], v[8:9]
; GFX12-SDAG-NEXT: v_cmp_lt_u64_e64 s0, v[2:3], v[10:11]
; GFX12-SDAG-NEXT: s_wait_alu depctr_va_sdst(0)
-; GFX12-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX12-SDAG-NEXT: v_cndmask_b32_e64 v1, v9, v1, s1
; GFX12-SDAG-NEXT: v_cndmask_b32_e64 v3, v11, v3, s0
; GFX12-SDAG-NEXT: v_cndmask_b32_e64 v2, v10, v2, s0
@@ -3962,7 +3953,7 @@ define i64 @test_vector_reduce_umin_v16i64(<16 x i64> %v) {
; GFX12-SDAG-NEXT: s_wait_alu depctr_va_vcc(0)
; GFX12-SDAG-NEXT: v_dual_cndmask_b32 v15, v31, v15 :: v_dual_cndmask_b32 v14, v30, v14
; GFX12-SDAG-NEXT: v_cmp_lt_u64_e32 vcc_lo, v[4:5], v[12:13]
-; GFX12-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_3) | instid1(VALU_DEP_2)
+; GFX12-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX12-SDAG-NEXT: v_cmp_lt_u64_e64 s0, v[6:7], v[14:15]
; GFX12-SDAG-NEXT: s_wait_alu depctr_va_vcc(0)
; GFX12-SDAG-NEXT: v_dual_cndmask_b32 v5, v13, v5 :: v_dual_cndmask_b32 v4, v12, v4
@@ -3975,9 +3966,9 @@ define i64 @test_vector_reduce_umin_v16i64(<16 x i64> %v) {
; GFX12-SDAG-NEXT: s_wait_alu depctr_va_vcc(0)
; GFX12-SDAG-NEXT: v_dual_cndmask_b32 v1, v5, v1 :: v_dual_cndmask_b32 v0, v4, v0
; GFX12-SDAG-NEXT: s_wait_alu depctr_va_sdst(0)
-; GFX12-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX12-SDAG-NEXT: v_cndmask_b32_e64 v2, v6, v2, s0
; GFX12-SDAG-NEXT: v_cndmask_b32_e64 v3, v7, v3, s0
+; GFX12-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-SDAG-NEXT: v_cmp_lt_u64_e32 vcc_lo, v[0:1], v[2:3]
; GFX12-SDAG-NEXT: s_wait_alu depctr_va_vcc(0)
; GFX12-SDAG-NEXT: v_dual_cndmask_b32 v0, v2, v0 :: v_dual_cndmask_b32 v1, v3, v1
@@ -4019,7 +4010,7 @@ define i64 @test_vector_reduce_umin_v16i64(<16 x i64> %v) {
; GFX12-GISEL-NEXT: v_cndmask_b32_e64 v13, v29, v13, s1
; GFX12-GISEL-NEXT: s_wait_alu depctr_va_vcc(0)
; GFX12-GISEL-NEXT: v_dual_cndmask_b32 v0, v8, v0 :: v_dual_cndmask_b32 v1, v9, v1
-; GFX12-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_1)
+; GFX12-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX12-GISEL-NEXT: v_cmp_lt_u64_e64 s1, v[4:5], v[12:13]
; GFX12-GISEL-NEXT: s_wait_alu depctr_va_sdst(0)
; GFX12-GISEL-NEXT: v_cndmask_b32_e64 v4, v12, v4, s1
@@ -4027,25 +4018,25 @@ define i64 @test_vector_reduce_umin_v16i64(<16 x i64> %v) {
; GFX12-GISEL-NEXT: s_wait_loadcnt 0x0
; GFX12-GISEL-NEXT: v_cmp_lt_u64_e64 s0, v[14:15], v[30:31]
; GFX12-GISEL-NEXT: s_wait_alu depctr_va_sdst(0)
-; GFX12-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_2) | instid1(VALU_DEP_2)
; GFX12-GISEL-NEXT: v_cndmask_b32_e64 v14, v30, v14, s0
; GFX12-GISEL-NEXT: v_cndmask_b32_e64 v15, v31, v15, s0
; GFX12-GISEL-NEXT: v_cmp_lt_u64_e64 s0, v[2:3], v[10:11]
+; GFX12-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX12-GISEL-NEXT: v_cmp_lt_u64_e32 vcc_lo, v[6:7], v[14:15]
; GFX12-GISEL-NEXT: s_wait_alu depctr_va_sdst(0)
-; GFX12-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_4) | instid1(VALU_DEP_2)
; GFX12-GISEL-NEXT: v_cndmask_b32_e64 v2, v10, v2, s0
; GFX12-GISEL-NEXT: v_cndmask_b32_e64 v3, v11, v3, s0
; GFX12-GISEL-NEXT: s_wait_alu depctr_va_vcc(0)
; GFX12-GISEL-NEXT: v_dual_cndmask_b32 v6, v14, v6 :: v_dual_cndmask_b32 v7, v15, v7
; GFX12-GISEL-NEXT: v_cmp_lt_u64_e32 vcc_lo, v[0:1], v[4:5]
+; GFX12-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX12-GISEL-NEXT: v_cmp_lt_u64_e64 s0, v[2:3], v[6:7]
; GFX12-GISEL-NEXT: s_wait_alu depctr_va_vcc(0)
; GFX12-GISEL-NEXT: v_dual_cndmask_b32 v0, v4, v0 :: v_dual_cndmask_b32 v1, v5, v1
; GFX12-GISEL-NEXT: s_wait_alu depctr_va_sdst(0)
-; GFX12-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX12-GISEL-NEXT: v_cndmask_b32_e64 v2, v6, v2, s0
; GFX12-GISEL-NEXT: v_cndmask_b32_e64 v3, v7, v3, s0
+; GFX12-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-GISEL-NEXT: v_cmp_lt_u64_e32 vcc_lo, v[0:1], v[2:3]
; GFX12-GISEL-NEXT: s_wait_alu depctr_va_vcc(0)
; GFX12-GISEL-NEXT: v_dual_cndmask_b32 v0, v2, v0 :: v_dual_cndmask_b32 v1, v3, v1
diff --git a/llvm/test/CodeGen/AMDGPU/vgpr-lowering-gfx1250-t16.mir b/llvm/test/CodeGen/AMDGPU/vgpr-lowering-gfx1250-t16.mir
index e258e5b9d23c7c..23c0d893d59782 100644
--- a/llvm/test/CodeGen/AMDGPU/vgpr-lowering-gfx1250-t16.mir
+++ b/llvm/test/CodeGen/AMDGPU/vgpr-lowering-gfx1250-t16.mir
@@ -24,6 +24,7 @@ body: |
; GCN-NEXT: s_set_vgpr_msb 0x45
; ASM-SAME: ; msbs: dst=1 src0=1 src1=1 src2=0
+ ; GCN-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GCN-NEXT: v_add_f16_e64 v0.h /*v256.h*/, v1.h /*v257.h*/, v2.h /*v258.h*/
$vgpr256_hi16 = V_ADD_F16_t16_e64 0, undef $vgpr257_hi16, 0, undef $vgpr258_hi16, 0, 0, 0, implicit $exec, implicit $mode
@@ -38,6 +39,7 @@ body: |
; GCN-NEXT: s_set_vgpr_msb 0x458a
; ASM-SAME: ; msbs: dst=2 src0=2 src1=2 src2=0
+ ; GCN-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GCN-NEXT: v_add_f16_e64 v0.h /*v512.h*/, v1.h /*v513.h*/, v2.h /*v514.h*/
$vgpr512_hi16 = V_ADD_F16_t16_e64 0, undef $vgpr513_hi16, 0, undef $vgpr514_hi16, 0, 0, 0, implicit $exec, implicit $mode
@@ -52,6 +54,7 @@ body: |
; GCN-NEXT: s_set_vgpr_msb 0x8acf
; ASM-SAME: ; msbs: dst=3 src0=3 src1=3 src2=0
+ ; GCN-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GCN-NEXT: v_add_f16_e64 v0.h /*v768.h*/, v1.h /*v769.h*/, v2.h /*v770.h*/
$vgpr768_hi16 = V_ADD_F16_t16_e64 0, undef $vgpr769_hi16, 0, undef $vgpr770_hi16, 0, 0, 0, implicit $exec, implicit $mode
@@ -77,10 +80,12 @@ body: |
; We use an extra instruction to set the MSB, and then we expect it to be reset to 0 (lower 16-bit).
; GCN: s_set_vgpr_msb 0xcf
+ ; GCN-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GCN-NEXT: v_add_f16_e64 v0.h /*v768.h*/, v1.h /*v769.h*/, v2.h /*v770.h*/
$vgpr768_hi16 = V_ADD_F16_t16_e64 0, undef $vgpr769_hi16, 0, undef $vgpr770_hi16, 0, 0, 0, implicit $exec, implicit $mode
; GCN-NEXT: s_set_vgpr_msb 0xcf00
+ ; GCN-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GCN-NEXT: v_fmamk_f16 v26.l, v56.l, 0x1, v58.l
$vgpr26_lo16 = V_FMAMK_F16_t16 undef $vgpr56_lo16, 1, undef $vgpr58_lo16, implicit $exec, implicit $mode
diff --git a/llvm/test/CodeGen/AMDGPU/vgpr-lowering-gfx1250.mir b/llvm/test/CodeGen/AMDGPU/vgpr-lowering-gfx1250.mir
index 89c3084160e75a..f31a7c95e124a9 100644
--- a/llvm/test/CodeGen/AMDGPU/vgpr-lowering-gfx1250.mir
+++ b/llvm/test/CodeGen/AMDGPU/vgpr-lowering-gfx1250.mir
@@ -24,12 +24,14 @@ body: |
; Single bit change
; GCN-NEXT: s_set_vgpr_msb 0x4101
; ASM-SAME: ; msbs: dst=0 src0=1 src1=0 src2=0
+ ; GCN-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GCN-NEXT: v_rcp_f32_e64 v255, v2 /*v258*/
$vgpr255 = V_RCP_F32_e64 0, undef $vgpr258, 0, 0, implicit $exec, implicit $mode
; Reset
; GCN-NEXT: s_set_vgpr_msb 0x100
; ASM-SAME: ; msbs: dst=0 src0=0 src1=0 src2=0
+ ; GCN-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GCN-NEXT: v_rcp_f32_e64 v255, v1
$vgpr255 = V_RCP_F32_e64 0, undef $vgpr1, 0, 0, implicit $exec, implicit $mode
@@ -42,7 +44,7 @@ body: |
; GCN-NEXT: s_set_vgpr_msb 0x544
; ASM-SAME: ; msbs: dst=1 src0=0 src1=1 src2=0
- ; GCN-NEXT: s_delay_alu instid0(VALU_DEP_1)
+ ; GCN-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GCN-NEXT: v_add_f32_e64 v2 /*v258*/, v0, v251 /*v507*/
$vgpr258 = V_ADD_F32_e64 0, $vgpr0, 0, undef $vgpr507, 0, 0, implicit $exec, implicit $mode
@@ -50,6 +52,7 @@ body: |
; GCN-NEXT: s_set_vgpr_msb 0x4455
; ASM-SAME: ; msbs: dst=1 src0=1 src1=1 src2=1
+ ; GCN-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GCN-NEXT: v_fma_f32 v3 /*v259*/, v4 /*v260*/, v5 /*v261*/, v6 /*v262*/
$vgpr259 = V_FMA_F32_e64 0, undef $vgpr260, 0, undef $vgpr261, 0, undef $vgpr262, 0, 0, implicit $exec, implicit $mode
@@ -190,11 +193,13 @@ body: |
; GCN-NEXT: s_set_vgpr_msb 0x41aa
; ASM-SAME: ; msbs: dst=2 src0=2 src1=2 src2=2
+ ; GCN-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GCN-NEXT: v_fma_f32 v0 /*v512*/, v1 /*v513*/, v2 /*v514*/, v3 /*v515*/
$vgpr512 = V_FMA_F32_e64 0, undef $vgpr513, 0, undef $vgpr514, 0, undef $vgpr515, 0, 0, implicit $exec, implicit $mode
; GCN-NEXT: s_set_vgpr_msb 0xaaab
; ASM-SAME: ; msbs: dst=2 src0=3 src1=2 src2=2
+ ; GCN-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GCN-NEXT: v_fma_f32 v0 /*v512*/, v0 /*v768*/, v2 /*v514*/, v3 /*v515*/
$vgpr512 = V_FMA_F32_e64 0, undef $vgpr768, 0, undef $vgpr514, 0, undef $vgpr515, 0, 0, implicit $exec, implicit $mode
@@ -204,16 +209,19 @@ body: |
; GCN-NEXT: s_set_vgpr_msb 0xabba
; ASM-SAME: ; msbs: dst=2 src0=2 src1=2 src2=3
+ ; GCN-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GCN-NEXT: v_fma_f32 v0 /*v512*/, v1 /*v513*/, v2 /*v514*/, v3 /*v771*/
$vgpr512 = V_FMA_F32_e64 0, undef $vgpr513, 0, undef $vgpr514, 0, undef $vgpr771, 0, 0, implicit $exec, implicit $mode
; GCN-NEXT: s_set_vgpr_msb 0xbaea
; ASM-SAME: ; msbs: dst=3 src0=2 src1=2 src2=2
+ ; GCN-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GCN-NEXT: v_fma_f32 v255 /*v1023*/, v1 /*v513*/, v2 /*v514*/, v3 /*v515*/
$vgpr1023 = V_FMA_F32_e64 0, undef $vgpr513, 0, undef $vgpr514, 0, undef $vgpr515, 0, 0, implicit $exec, implicit $mode
; GCN-NEXT: s_set_vgpr_msb 0xeaff
; ASM-SAME: ; msbs: dst=3 src0=3 src1=3 src2=3
+ ; GCN-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GCN-NEXT: v_fma_f32 v0 /*v768*/, v1 /*v769*/, v2 /*v770*/, v3 /*v771*/
$vgpr768 = V_FMA_F32_e64 0, undef $vgpr769, 0, undef $vgpr770, 0, undef $vgpr771, 0, 0, implicit $exec, implicit $mode
@@ -226,6 +234,7 @@ body: |
; GCN-NEXT: s_set_vgpr_msb 0x4200
; ASM-SAME: ; msbs: dst=0 src0=0 src1=0 src2=0
+ ; GCN-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GCN-NEXT: v_fma_f32 v0, v1, v2, v3
$vgpr0 = V_FMA_F32_e64 0, undef $vgpr1, 0, undef $vgpr2, 0, undef $vgpr3, 0, 0, implicit $exec, implicit $mode
@@ -265,61 +274,73 @@ body: |
; GCN-NEXT: s_set_vgpr_msb 64
; ASM-SAME: ; msbs: dst=1 src0=0 src1=0 src2=0
+ ; GCN-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GCN-NEXT: v_dual_sub_f32 v244 /*v500*/, v1, v2 :: v_dual_mul_f32 v0 /*v256*/, v3, v4
$vgpr500, $vgpr256 = V_DUAL_SUB_F32_e32_X_MUL_F32_e32_gfx1250 undef $vgpr1, undef $vgpr2, undef $vgpr3, undef $vgpr4, implicit $mode, implicit $exec
; GCN-NEXT: s_set_vgpr_msb 0x4041
; ASM-SAME: ; msbs: dst=1 src0=1 src1=0 src2=0
+ ; GCN-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GCN-NEXT: v_dual_sub_f32 v244 /*v500*/, s1, v2 :: v_dual_mul_f32 v0 /*v256*/, v44 /*v300*/, v4
$vgpr500, $vgpr256 = V_DUAL_SUB_F32_e32_X_MUL_F32_e32_gfx1250 undef $sgpr1, undef $vgpr2, undef $vgpr300, undef $vgpr4, implicit $mode, implicit $exec
; GCN-NEXT: s_set_vgpr_msb 0x4104
; ASM-SAME: ; msbs: dst=0 src0=0 src1=1 src2=0
+ ; GCN-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GCN-NEXT: v_dual_sub_f32 v255, v1, v44 /*v300*/ :: v_dual_mul_f32 v6, v0, v1 /*v257*/
$vgpr255, $vgpr6 = V_DUAL_SUB_F32_e32_X_MUL_F32_e32_gfx1250 undef $vgpr1, undef $vgpr300, undef $vgpr0, $vgpr257, implicit $mode, implicit $exec
; GCN-NEXT: s_set_vgpr_msb 0x401
; ASM-SAME: ; msbs: dst=0 src0=1 src1=0 src2=0
+ ; GCN-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GCN-NEXT: v_dual_sub_f32 v255, 0, v1 :: v_dual_mul_f32 v6, v44 /*v300*/, v3
$vgpr255, $vgpr6 = V_DUAL_SUB_F32_e32_X_MUL_F32_e32_gfx1250 0, undef $vgpr1, undef $vgpr300, undef $vgpr3, implicit $mode, implicit $exec
; GCN-NEXT: s_set_vgpr_msb 0x140
; ASM-SAME: ; msbs: dst=1 src0=0 src1=0 src2=0
+ ; GCN-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GCN-NEXT: v_dual_fmamk_f32 v243 /*v499*/, v0, 0xa, v3 :: v_dual_fmac_f32 v0 /*v256*/, v1, v1
$vgpr499, $vgpr256 = V_DUAL_FMAMK_F32_X_FMAC_F32_e32_gfx1250 undef $vgpr0, 10, undef $vgpr3, undef $vgpr1, undef $vgpr1, $vgpr256, implicit $mode, implicit $exec
; GCN-NEXT: s_set_vgpr_msb 0x4005
; ASM-SAME: ; msbs: dst=0 src0=1 src1=1 src2=0
+ ; GCN-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GCN-NEXT: v_dual_mov_b32 v2, v3 /*v259*/ :: v_dual_add_f32 v3, v1 /*v257*/, v2 /*v258*/
$vgpr2, $vgpr3 = V_DUAL_MOV_B32_e32_X_ADD_F32_e32_gfx1250 undef $vgpr259, undef $vgpr257, undef $vgpr258, implicit $exec, implicit $mode
; GCN-NEXT: s_set_vgpr_msb 0x554
; ASM-SAME: ; msbs: dst=1 src0=0 src1=1 src2=1
+ ; GCN-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GCN-NEXT: v_dual_fmamk_f32 v244 /*v500*/, v0, 0xa, v44 /*v300*/ :: v_dual_fmac_f32 v3 /*v259*/, v1, v1 /*v257*/
$vgpr500, $vgpr259 = V_DUAL_FMAMK_F32_X_FMAC_F32_e32_gfx1250 undef $vgpr0, 10, undef $vgpr300, undef $vgpr1, undef $vgpr257, $vgpr259, implicit $mode, implicit $exec
; GCN-NEXT: s_set_vgpr_msb 0x5410
; ASM-SAME: ; msbs: dst=0 src0=0 src1=0 src2=1
+ ; GCN-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GCN-NEXT: v_dual_fma_f32 v0, v6, v6, v44 /*v300*/ :: v_dual_fma_f32 v1, v4, v5, v45 /*v301*/
$vgpr0, $vgpr1 = V_DUAL_FMA_F32_e64_X_FMA_F32_e64_e96_gfx1250 0, undef $vgpr6, 0, undef $vgpr6, 0, undef $vgpr300, 0, undef $vgpr4, 0, undef $vgpr5, 0, undef $vgpr301, implicit $mode, implicit $exec
; GCN-NEXT: s_set_vgpr_msb 0x1000
; ASM-SAME: ; msbs: dst=0 src0=0 src1=0 src2=0
+ ; GCN-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GCN-NEXT: v_dual_fmac_f32 v2, v6, v6 :: v_dual_fma_f32 v3, v4, v5, v3
$vgpr2, $vgpr3 = V_DUAL_FMAC_F32_e32_X_FMA_F32_e64_e96_gfx1250 0, undef $vgpr6, 0, undef $vgpr6, undef $vgpr2, 0, undef $vgpr4, 0, undef $vgpr5, 0, $vgpr3, implicit $mode, implicit $exec
; GCN-NEXT: s_set_vgpr_msb 64
; ASM-SAME: ; msbs: dst=1 src0=0 src1=0 src2=0
+ ; GCN-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GCN-NEXT: v_dual_fma_f32 v244 /*v500*/, v6, v7, v8 :: v_dual_add_f32 v3 /*v259*/, v4, v5
$vgpr500, $vgpr259 = V_DUAL_FMA_F32_e64_X_ADD_F32_e32_e96_gfx1250 0, undef $vgpr6, 0, undef $vgpr7, 0, undef $vgpr8, 0, undef $vgpr4, 0, undef $vgpr5, implicit $mode, implicit $exec
; GCN-NEXT: s_set_vgpr_msb 0x40ae
; ASM-SAME: ; msbs: dst=2 src0=2 src1=3 src2=2
+ ; GCN-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GCN-NEXT: v_dual_fmac_f32 v2 /*v514*/, v6 /*v518*/, v8 /*v776*/ :: v_dual_fma_f32 v3 /*v515*/, v4 /*v516*/, v7 /*v775*/, v3 /*v515*/
$vgpr514, $vgpr515 = V_DUAL_FMAC_F32_e32_X_FMA_F32_e64_e96_gfx1250 0, undef $vgpr518, 0, undef $vgpr776, undef $vgpr514, 0, undef $vgpr516, 0, undef $vgpr775, 0, $vgpr515, implicit $mode, implicit $exec
; GCN-NEXT: s_set_vgpr_msb 0xae54
; ASM-SAME: ; msbs: dst=1 src0=0 src1=1 src2=1
+ ; GCN-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GCN-NEXT: v_dual_fmac_f32 v7 /*v263*/, v1, v1 /*v257*/ :: v_dual_fmamk_f32 v244 /*v500*/, v0, 0xa, v44 /*v300*/
$vgpr263, $vgpr500 = V_DUAL_FMAC_F32_e32_X_FMAMK_F32_gfx1250 undef $vgpr1, undef $vgpr257, $vgpr263, undef $vgpr0, 10, undef $vgpr300, implicit $mode, implicit $exec
@@ -338,11 +359,13 @@ body: |
; GCN-NEXT: s_set_vgpr_msb 0x45
; ASM-SAME: ; msbs: dst=1 src0=1 src1=1 src2=0
+ ; GCN-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GCN-NEXT: v_fmaak_f32 v0 /*v256*/, v1 /*v257*/, v2 /*v258*/, 0x1
$vgpr256 = V_FMAAK_F32 undef $vgpr257, undef $vgpr258, 1, implicit $exec, implicit $mode
; GCN-NEXT: s_set_vgpr_msb 0x4505
; ASM-SAME: ; msbs: dst=0 src0=1 src1=1 src2=0
+ ; GCN-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GCN-NEXT: v_fmaak_f32 v0, v1 /*v257*/, v2 /*v258*/, 0x1
$vgpr0 = V_FMAAK_F32 undef $vgpr257, undef $vgpr258, 1, implicit $exec, implicit $mode
@@ -350,6 +373,7 @@ body: |
; is still allowed, so src2=1 is piggybacked here.
; GCN-NEXT: s_set_vgpr_msb 0x551
; ASM-SAME: ; msbs: dst=1 src0=1 src1=0 src2=1
+ ; GCN-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GCN-NEXT: v_fmaak_f32 v0 /*v256*/, v1 /*v257*/, v2, 0x1
$vgpr256 = V_FMAAK_F32 undef $vgpr257, undef $vgpr2, 1, implicit $exec, implicit $mode
@@ -365,41 +389,49 @@ body: |
; GCN-NEXT: s_set_vgpr_msb 0x5111
; ASM-SAME: ; msbs: dst=0 src0=1 src1=0 src2=1
+ ; GCN-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GCN-NEXT: v_fmamk_f32 v0, v1 /*v257*/, 0x1, v2 /*v258*/
$vgpr0 = V_FMAMK_F32 undef $vgpr257, 1, undef $vgpr258, implicit $exec, implicit $mode
; GCN-NEXT: s_set_vgpr_msb 0x1141
; ASM-SAME: ; msbs: dst=1 src0=1 src1=0 src2=0
+ ; GCN-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GCN-NEXT: v_fmamk_f32 v0 /*v256*/, v1 /*v257*/, 0x1, v2
$vgpr256 = V_FMAMK_F32 undef $vgpr257, 1, undef $vgpr2, implicit $exec, implicit $mode
; GCN-NEXT: s_set_vgpr_msb 0x4150
; ASM-SAME: ; msbs: dst=1 src0=0 src1=0 src2=1
+ ; GCN-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GCN-NEXT: v_fmamk_f32 v0 /*v256*/, v1, 0x1, v2 /*v258*/
$vgpr256 = V_FMAMK_F32 undef $vgpr1, 1, undef $vgpr258, implicit $exec, implicit $mode
; GCN-NEXT: s_set_vgpr_msb 0x5051
; ASM-SAME: ; msbs: dst=1 src0=1 src1=0 src2=1
+ ; GCN-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GCN-NEXT: v_fmamk_f64 v[4:5] /*v[260:261]*/, v[100:101] /*v[356:357]*/, 0x1, v[2:3] /*v[258:259]*/
$vgpr260_vgpr261 = V_FMAMK_F64 undef $vgpr356_vgpr357, 1, undef $vgpr258_vgpr259, implicit $exec, implicit $mode
; GCN-NEXT: s_set_vgpr_msb 0x5101
; ASM-SAME: ; msbs: dst=0 src0=1 src1=0 src2=0
+ ; GCN-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GCN-NEXT: v_fmamk_f64 v[0:1], v[100:101] /*v[356:357]*/, 0x1, v[2:3]
$vgpr0_vgpr1 = V_FMAMK_F64 undef $vgpr356_vgpr357, 1, undef $vgpr2_vgpr3, implicit $exec, implicit $mode
; GCN-NEXT: s_set_vgpr_msb 0x110
; ASM-SAME: ; msbs: dst=0 src0=0 src1=0 src2=1
+ ; GCN-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GCN-NEXT: v_fmamk_f64 v[0:1], v[2:3], 0x1, v[100:101] /*v[356:357]*/
$vgpr0_vgpr1 = V_FMAMK_F64 undef $vgpr2_vgpr3, 1, undef $vgpr356_vgpr357, implicit $exec, implicit $mode
; GCN-NEXT: s_set_vgpr_msb 0x1040
; ASM-SAME: ; msbs: dst=1 src0=0 src1=0 src2=0
+ ; GCN-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GCN-NEXT: v_fmamk_f64 v[0:1] /*v[256:257]*/, v[2:3], 0x1, v[4:5]
$vgpr256_vgpr257 = V_FMAMK_F64 undef $vgpr2_vgpr3, 1, undef $vgpr4_vgpr5, implicit $exec, implicit $mode
; GCN-NEXT: s_set_vgpr_msb 0x4000
; ASM-SAME: ; msbs: dst=0 src0=0 src1=0 src2=0
+ ; GCN-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GCN-NEXT: v_fmamk_f16 v26, v56, 0x1, v58
$vgpr26 = V_FMAMK_F16_fake16 undef $vgpr56, 1, undef $vgpr58, implicit $exec, implicit $mode
@@ -428,6 +460,7 @@ body: |
; Accumulation instructions apply DST to both the destination and one of the source VGPRs
; GCN-NEXT: s_set_vgpr_msb 64
; ASM-SAME: ; msbs: dst=1 src0=0 src1=0 src2=0
+ ; GCN-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GCN-NEXT: v_fmac_f32_e64 v0 /*v256*/, |v0|, |v1| clamp mul:4
$vgpr256 = V_FMAC_F32_e64 2, undef $vgpr0, 2, undef $vgpr1, 2, undef $vgpr256, 1, 2, implicit $mode, implicit $exec
@@ -495,6 +528,7 @@ body: |
; GCN-NEXT: s_set_vgpr_msb 0x55
; ASM-SAME: ; msbs: dst=1 src0=1 src1=1 src2=1
+ ; GCN-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GCN-NEXT: v_fma_f32 v3 /*v259*/, v4 /*v260*/, v5 /*v261*/, v6 /*v262*/
$vgpr259 = V_FMA_F32_e64 0, undef $vgpr260, 0, undef $vgpr261, 0, undef $vgpr262, 0, 0, implicit $exec, implicit $mode
@@ -516,11 +550,13 @@ body: |
; GCN-NEXT: s_set_vgpr_msb 0x4000
; ASM-SAME: ; msbs: dst=0 src0=0 src1=0 src2=0
+ ; GCN-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GCN-NEXT: v_fma_f32 v3, v4, v5, s2
$vgpr3 = V_FMA_F32_e64 0, undef $vgpr4, 0, undef $vgpr5, 0, undef $sgpr2, 0, 0, implicit $exec, implicit $mode
; GCN-NEXT: s_set_vgpr_msb 1
; ASM-SAME: ; msbs: dst=0 src0=1 src1=0 src2=0
+ ; GCN-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GCN-NEXT: v_fma_f32 v3, v4 /*v260*/, v5, 1
$vgpr3 = V_FMA_F32_e64 0, undef $vgpr260, 0, undef $vgpr5, 0, 1, 0, 0, implicit $exec, implicit $mode
@@ -1002,6 +1038,7 @@ body: |
; s_set_vgpr_msb, changing src1 from 0 to 2 and breaking step 2.
; GCN-NEXT: s_set_vgpr_msb 0x424a
; ASM-SAME: ; msbs: dst=1 src0=2 src1=2 src2=0
+ ; GCN-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GCN-NEXT: v_add_f32_e64 v0 /*v256*/, v0 /*v512*/, v1 /*v513*/
$vgpr256 = V_ADD_F32_e64 0, undef $vgpr512, 0, undef $vgpr513, 0, 0, implicit $exec, implicit $mode
@@ -1032,6 +1069,7 @@ body: |
; GCN-NEXT: s_set_vgpr_msb 4
; ASM-SAME: ; msbs: dst=0 src0=0 src1=1 src2=0
+ ; GCN-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GCN-NEXT: v_add_f32_e32 v0, 1, v0 /*v256*/
$vgpr0 = V_ADD_F32_e32 1, $vgpr256, implicit $mode, implicit $exec
diff --git a/llvm/test/CodeGen/AMDGPU/vgpr-mark-last-scratch-load.ll b/llvm/test/CodeGen/AMDGPU/vgpr-mark-last-scratch-load.ll
index 735b8e0f740893..be970a3c9d10c6 100644
--- a/llvm/test/CodeGen/AMDGPU/vgpr-mark-last-scratch-load.ll
+++ b/llvm/test/CodeGen/AMDGPU/vgpr-mark-last-scratch-load.ll
@@ -10,7 +10,7 @@ define amdgpu_cs void @max_6_vgprs(ptr addrspace(1) %p) "amdgpu-num-vgpr"="6" no
; CHECK-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; CHECK-NEXT: v_lshlrev_b64_e32 v[2:3], 2, v[2:3]
; CHECK-NEXT: v_add_co_u32 v0, vcc_lo, v0, v2
-; CHECK-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; CHECK-NEXT: s_delay_alu instid0(VALU_DEP_2)
; CHECK-NEXT: v_add_co_ci_u32_e64 v1, null, v1, v3, vcc_lo
; CHECK-NEXT: global_load_b32 v5, v[0:1], off scope:SCOPE_SYS
; CHECK-NEXT: s_wait_loadcnt 0x0
@@ -78,7 +78,7 @@ define amdgpu_cs void @max_11_vgprs_branch(ptr addrspace(1) %p, i32 %tmp) "amdgp
; CHECK-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; CHECK-NEXT: v_lshlrev_b64_e32 v[3:4], 2, v[3:4]
; CHECK-NEXT: v_add_co_u32 v0, vcc_lo, v0, v3
-; CHECK-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; CHECK-NEXT: s_delay_alu instid0(VALU_DEP_2)
; CHECK-NEXT: v_add_co_ci_u32_e64 v1, null, v1, v4, vcc_lo
; CHECK-NEXT: global_load_b32 v3, v[0:1], off offset:336 scope:SCOPE_SYS
; CHECK-NEXT: s_wait_loadcnt 0x0
@@ -93,6 +93,7 @@ define amdgpu_cs void @max_11_vgprs_branch(ptr addrspace(1) %p, i32 %tmp) "amdgp
; CHECK-NEXT: s_wait_loadcnt 0x0
; CHECK-NEXT: scratch_store_b32 off, v3, off offset:4 ; 4-byte Folded Spill
; CHECK-NEXT: v_cmpx_eq_u32_e32 0, v2
+; CHECK-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; CHECK-NEXT: s_xor_b32 s0, exec_lo, s0
; CHECK-NEXT: s_cbranch_execz .LBB1_2
; CHECK-NEXT: ; %bb.1: ; %.false
diff --git a/llvm/test/CodeGen/AMDGPU/whole-wave-functions.ll b/llvm/test/CodeGen/AMDGPU/whole-wave-functions.ll
index ac540c39215030..03f32f1f00cb32 100644
--- a/llvm/test/CodeGen/AMDGPU/whole-wave-functions.ll
+++ b/llvm/test/CodeGen/AMDGPU/whole-wave-functions.ll
@@ -118,8 +118,8 @@ define amdgpu_gfx_whole_wave i32 @basic_test(i1 %active, i32 %a, i32 %b) #0 {
; GFX1250-DAGISEL-NEXT: scratch_store_b32 off, v1, s32 offset:4 nv
; GFX1250-DAGISEL-NEXT: s_wait_xcnt 0x0
; GFX1250-DAGISEL-NEXT: s_mov_b32 exec_lo, -1
+; GFX1250-DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1250-DAGISEL-NEXT: v_dual_cndmask_b32 v0, 5, v0 :: v_dual_cndmask_b32 v1, 3, v1
-; GFX1250-DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1250-DAGISEL-NEXT: v_mov_b32_dpp v0, v1 quad_perm:[1,0,0,0] row_mask:0x1 bank_mask:0x1
; GFX1250-DAGISEL-NEXT: s_xor_b32 exec_lo, vcc_lo, -1
; GFX1250-DAGISEL-NEXT: s_clause 0x1 ; 8-byte Folded Reload
@@ -243,8 +243,8 @@ define amdgpu_gfx_whole_wave i32 @single_use_of_active(i1 %active, i32 %a, i32 %
; GFX1250-DAGISEL-NEXT: scratch_store_b32 off, v1, s32 offset:4 nv
; GFX1250-DAGISEL-NEXT: s_wait_xcnt 0x0
; GFX1250-DAGISEL-NEXT: s_mov_b32 exec_lo, -1
+; GFX1250-DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1250-DAGISEL-NEXT: v_cndmask_b32_e32 v1, 17, v1, vcc_lo
-; GFX1250-DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1250-DAGISEL-NEXT: v_mov_b32_dpp v0, v1 quad_perm:[1,0,0,0] row_mask:0x1 bank_mask:0x1
; GFX1250-DAGISEL-NEXT: s_xor_b32 exec_lo, vcc_lo, -1
; GFX1250-DAGISEL-NEXT: s_clause 0x1 ; 8-byte Folded Reload
@@ -271,6 +271,7 @@ define amdgpu_gfx_whole_wave i32 @unused_active(i1 %active, i32 %a, i32 %b) #0 {
; DAGISEL-NEXT: s_xor_saveexec_b32 s0, -1
; DAGISEL-NEXT: scratch_store_b32 off, v0, s32 ; 4-byte Folded Spill
; DAGISEL-NEXT: s_mov_b32 exec_lo, -1
+; DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; DAGISEL-NEXT: v_mov_b32_e32 v0, 14
; DAGISEL-NEXT: s_wait_alu depctr_sa_sdst(0)
; DAGISEL-NEXT: s_xor_b32 exec_lo, s0, -1
@@ -289,6 +290,7 @@ define amdgpu_gfx_whole_wave i32 @unused_active(i1 %active, i32 %a, i32 %b) #0 {
; GISEL-NEXT: s_xor_saveexec_b32 s0, -1
; GISEL-NEXT: scratch_store_b32 off, v0, s32 ; 4-byte Folded Spill
; GISEL-NEXT: s_mov_b32 exec_lo, -1
+; GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GISEL-NEXT: v_mov_b32_e32 v0, 14
; GISEL-NEXT: s_wait_alu depctr_sa_sdst(0)
; GISEL-NEXT: s_xor_b32 exec_lo, s0, -1
@@ -307,6 +309,7 @@ define amdgpu_gfx_whole_wave i32 @unused_active(i1 %active, i32 %a, i32 %b) #0 {
; DAGISEL64-NEXT: s_xor_saveexec_b64 s[0:1], -1
; DAGISEL64-NEXT: scratch_store_b32 off, v0, s32 ; 4-byte Folded Spill
; DAGISEL64-NEXT: s_mov_b64 exec, -1
+; DAGISEL64-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; DAGISEL64-NEXT: v_mov_b32_e32 v0, 14
; DAGISEL64-NEXT: s_wait_alu depctr_sa_sdst(0)
; DAGISEL64-NEXT: s_xor_b64 exec, s[0:1], -1
@@ -325,6 +328,7 @@ define amdgpu_gfx_whole_wave i32 @unused_active(i1 %active, i32 %a, i32 %b) #0 {
; GISEL64-NEXT: s_xor_saveexec_b64 s[0:1], -1
; GISEL64-NEXT: scratch_store_b32 off, v0, s32 ; 4-byte Folded Spill
; GISEL64-NEXT: s_mov_b64 exec, -1
+; GISEL64-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GISEL64-NEXT: v_mov_b32_e32 v0, 14
; GISEL64-NEXT: s_wait_alu depctr_sa_sdst(0)
; GISEL64-NEXT: s_xor_b64 exec, s[0:1], -1
@@ -341,6 +345,7 @@ define amdgpu_gfx_whole_wave i32 @unused_active(i1 %active, i32 %a, i32 %b) #0 {
; GFX1250-DAGISEL-NEXT: scratch_store_b32 off, v0, s32 nv ; 4-byte Folded Spill
; GFX1250-DAGISEL-NEXT: s_wait_xcnt 0x0
; GFX1250-DAGISEL-NEXT: s_mov_b32 exec_lo, -1
+; GFX1250-DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1250-DAGISEL-NEXT: v_mov_b32_e32 v0, 14
; GFX1250-DAGISEL-NEXT: s_xor_b32 exec_lo, s0, -1
; GFX1250-DAGISEL-NEXT: scratch_load_b32 v0, off, s32 nv ; 4-byte Folded Reload
@@ -917,8 +922,9 @@ define amdgpu_gfx_whole_wave i32 @multiple_blocks(i1 %active, i32 %a, i32 %b) #0
; DAGISEL-NEXT: v_add_nc_u32_e32 v1, v0, v1
; DAGISEL-NEXT: ; %bb.2: ; %if.end
; DAGISEL-NEXT: s_wait_alu depctr_sa_sdst(0)
+; DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2)
; DAGISEL-NEXT: s_or_b32 exec_lo, exec_lo, s1
-; DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; DAGISEL-NEXT: v_cndmask_b32_e32 v0, v1, v0, vcc_lo
; DAGISEL-NEXT: s_xor_b32 exec_lo, vcc_lo, -1
; DAGISEL-NEXT: s_clause 0x1 ; 8-byte Folded Reload
@@ -947,8 +953,9 @@ define amdgpu_gfx_whole_wave i32 @multiple_blocks(i1 %active, i32 %a, i32 %b) #0
; GISEL-NEXT: v_add_nc_u32_e32 v1, v0, v1
; GISEL-NEXT: ; %bb.2: ; %if.end
; GISEL-NEXT: s_wait_alu depctr_sa_sdst(0)
+; GISEL-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GISEL-NEXT: s_or_b32 exec_lo, exec_lo, s1
-; GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GISEL-NEXT: v_cndmask_b32_e32 v0, v1, v0, vcc_lo
; GISEL-NEXT: s_xor_b32 exec_lo, vcc_lo, -1
; GISEL-NEXT: s_clause 0x1 ; 8-byte Folded Reload
@@ -977,8 +984,9 @@ define amdgpu_gfx_whole_wave i32 @multiple_blocks(i1 %active, i32 %a, i32 %b) #0
; DAGISEL64-NEXT: v_add_nc_u32_e32 v1, v0, v1
; DAGISEL64-NEXT: ; %bb.2: ; %if.end
; DAGISEL64-NEXT: s_wait_alu depctr_sa_sdst(0)
+; DAGISEL64-NEXT: s_delay_alu instid0(VALU_DEP_2)
; DAGISEL64-NEXT: s_or_b64 exec, exec, s[2:3]
-; DAGISEL64-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; DAGISEL64-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; DAGISEL64-NEXT: v_cndmask_b32_e32 v0, v1, v0, vcc
; DAGISEL64-NEXT: s_xor_b64 exec, vcc, -1
; DAGISEL64-NEXT: s_clause 0x1 ; 8-byte Folded Reload
@@ -1007,8 +1015,9 @@ define amdgpu_gfx_whole_wave i32 @multiple_blocks(i1 %active, i32 %a, i32 %b) #0
; GISEL64-NEXT: v_add_nc_u32_e32 v1, v0, v1
; GISEL64-NEXT: ; %bb.2: ; %if.end
; GISEL64-NEXT: s_wait_alu depctr_sa_sdst(0)
+; GISEL64-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GISEL64-NEXT: s_or_b64 exec, exec, s[2:3]
-; GISEL64-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GISEL64-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GISEL64-NEXT: v_cndmask_b32_e32 v0, v1, v0, vcc
; GISEL64-NEXT: s_xor_b64 exec, vcc, -1
; GISEL64-NEXT: s_clause 0x1 ; 8-byte Folded Reload
@@ -1034,8 +1043,9 @@ define amdgpu_gfx_whole_wave i32 @multiple_blocks(i1 %active, i32 %a, i32 %b) #0
; GFX1250-DAGISEL-NEXT: ; %bb.1: ; %if.then
; GFX1250-DAGISEL-NEXT: v_add_nc_u32_e32 v1, v0, v1
; GFX1250-DAGISEL-NEXT: ; %bb.2: ; %if.end
+; GFX1250-DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX1250-DAGISEL-NEXT: s_or_b32 exec_lo, exec_lo, s1
-; GFX1250-DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1250-DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
; GFX1250-DAGISEL-NEXT: v_cndmask_b32_e32 v0, v1, v0, vcc_lo
; GFX1250-DAGISEL-NEXT: s_xor_b32 exec_lo, vcc_lo, -1
; GFX1250-DAGISEL-NEXT: s_clause 0x1 ; 8-byte Folded Reload
@@ -1195,10 +1205,11 @@ define amdgpu_gfx_whole_wave i64 @ret_64(i1 %active, i64 %a, i64 %b) #0 {
; GFX1250-DAGISEL-NEXT: scratch_store_b32 off, v3, s32 offset:12 nv
; GFX1250-DAGISEL-NEXT: s_wait_xcnt 0x0
; GFX1250-DAGISEL-NEXT: s_mov_b32 exec_lo, -1
+; GFX1250-DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX1250-DAGISEL-NEXT: v_dual_cndmask_b32 v1, 0, v1 :: v_dual_cndmask_b32 v0, 5, v0
; GFX1250-DAGISEL-NEXT: v_dual_cndmask_b32 v2, 3, v2 :: v_dual_cndmask_b32 v3, 0, v3
-; GFX1250-DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
; GFX1250-DAGISEL-NEXT: v_mov_b32_dpp v0, v2 quad_perm:[1,0,0,0] row_mask:0x1 bank_mask:0x1
+; GFX1250-DAGISEL-NEXT: s_delay_alu instid0(VALU_DEP_2)
; GFX1250-DAGISEL-NEXT: v_mov_b32_dpp v1, v3 quad_perm:[1,0,0,0] row_mask:0x1 bank_mask:0x1
; GFX1250-DAGISEL-NEXT: s_xor_b32 exec_lo, vcc_lo, -1
; GFX1250-DAGISEL-NEXT: s_clause 0x3 ; 16-byte Folded Reload
@@ -1233,6 +1244,7 @@ define amdgpu_gfx_whole_wave void @inreg_args(i1 %active, i32 inreg %i32, <4 x i
; DAGISEL-NEXT: scratch_store_b32 off, v4, s32 offset:16
; DAGISEL-NEXT: scratch_store_b32 off, v5, s32 offset:20
; DAGISEL-NEXT: s_mov_b32 exec_lo, -1
+; DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; DAGISEL-NEXT: v_dual_mov_b32 v4, s4 :: v_dual_mov_b32 v5, s9
; DAGISEL-NEXT: v_dual_mov_b32 v0, s5 :: v_dual_mov_b32 v1, s6
; DAGISEL-NEXT: v_dual_mov_b32 v2, s7 :: v_dual_mov_b32 v3, s8
@@ -1309,6 +1321,7 @@ define amdgpu_gfx_whole_wave void @inreg_args(i1 %active, i32 inreg %i32, <4 x i
; DAGISEL64-NEXT: scratch_store_b32 off, v4, s32 offset:16
; DAGISEL64-NEXT: scratch_store_b32 off, v5, s32 offset:20
; DAGISEL64-NEXT: s_mov_b64 exec, -1
+; DAGISEL64-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; DAGISEL64-NEXT: v_mov_b32_e32 v4, s4
; DAGISEL64-NEXT: v_mov_b32_e32 v0, s5
; DAGISEL64-NEXT: v_mov_b32_e32 v1, s6
@@ -1389,6 +1402,7 @@ define amdgpu_gfx_whole_wave void @inreg_args(i1 %active, i32 inreg %i32, <4 x i
; GFX1250-DAGISEL-NEXT: scratch_store_b32 off, v5, s32 offset:20 nv
; GFX1250-DAGISEL-NEXT: s_wait_xcnt 0x0
; GFX1250-DAGISEL-NEXT: s_mov_b32 exec_lo, -1
+; GFX1250-DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1250-DAGISEL-NEXT: v_dual_mov_b32 v4, s4 :: v_dual_mov_b32 v5, s9
; GFX1250-DAGISEL-NEXT: v_dual_mov_b32 v0, s5 :: v_dual_mov_b32 v1, s6
; GFX1250-DAGISEL-NEXT: v_dual_mov_b32 v2, s7 :: v_dual_mov_b32 v3, s8
@@ -4822,6 +4836,7 @@ define amdgpu_gfx_whole_wave <2 x half> @tail_call_gfx_from_whole_wave(i1 %activ
; DAGISEL-NEXT: scratch_store_b32 off, v246, s32 offset:568
; DAGISEL-NEXT: scratch_store_b32 off, v247, s32 offset:572
; DAGISEL-NEXT: s_mov_b32 exec_lo, -1
+; DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; DAGISEL-NEXT: v_mov_b32_e32 v2, v0
; DAGISEL-NEXT: s_mov_b32 s37, gfx_callee at abs32@hi
; DAGISEL-NEXT: s_mov_b32 s36, gfx_callee at abs32@lo
@@ -5138,6 +5153,7 @@ define amdgpu_gfx_whole_wave <2 x half> @tail_call_gfx_from_whole_wave(i1 %activ
; GISEL-NEXT: scratch_store_b32 off, v246, s32 offset:568
; GISEL-NEXT: scratch_store_b32 off, v247, s32 offset:572
; GISEL-NEXT: s_mov_b32 exec_lo, -1
+; GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GISEL-NEXT: v_mov_b32_e32 v2, v0
; GISEL-NEXT: s_mov_b32 s36, gfx_callee at abs32@lo
; GISEL-NEXT: s_mov_b32 s37, gfx_callee at abs32@hi
@@ -5454,6 +5470,7 @@ define amdgpu_gfx_whole_wave <2 x half> @tail_call_gfx_from_whole_wave(i1 %activ
; DAGISEL64-NEXT: scratch_store_b32 off, v246, s32 offset:568
; DAGISEL64-NEXT: scratch_store_b32 off, v247, s32 offset:572
; DAGISEL64-NEXT: s_mov_b64 exec, -1
+; DAGISEL64-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; DAGISEL64-NEXT: v_mov_b32_e32 v2, v0
; DAGISEL64-NEXT: s_mov_b32 s37, gfx_callee at abs32@hi
; DAGISEL64-NEXT: s_mov_b32 s36, gfx_callee at abs32@lo
@@ -5770,6 +5787,7 @@ define amdgpu_gfx_whole_wave <2 x half> @tail_call_gfx_from_whole_wave(i1 %activ
; GISEL64-NEXT: scratch_store_b32 off, v246, s32 offset:568
; GISEL64-NEXT: scratch_store_b32 off, v247, s32 offset:572
; GISEL64-NEXT: s_mov_b64 exec, -1
+; GISEL64-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GISEL64-NEXT: v_mov_b32_e32 v2, v0
; GISEL64-NEXT: s_mov_b32 s36, gfx_callee at abs32@lo
; GISEL64-NEXT: s_mov_b32 s37, gfx_callee at abs32@hi
@@ -6865,6 +6883,7 @@ define amdgpu_gfx_whole_wave <2 x half> @tail_call_gfx_from_whole_wave(i1 %activ
; GFX1250-DAGISEL-NEXT: scratch_store_b32 off, v255 /*v1023*/, s32 offset:3644 nv
; GFX1250-DAGISEL-NEXT: s_wait_xcnt 0x0
; GFX1250-DAGISEL-NEXT: s_mov_b32 exec_lo, -1
+; GFX1250-DAGISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1250-DAGISEL-NEXT: v_mov_b32_e32 v2, v0
; GFX1250-DAGISEL-NEXT: s_mov_b64 s[36:37], gfx_callee at abs64
; GFX1250-DAGISEL-NEXT: s_set_vgpr_msb 0xc00 ; msbs: dst=0 src0=0 src1=0 src2=0
diff --git a/llvm/test/CodeGen/AMDGPU/workgroup-id-in-arch-sgprs.ll b/llvm/test/CodeGen/AMDGPU/workgroup-id-in-arch-sgprs.ll
index ac54854ff6bfe4..7e2cac2594219e 100644
--- a/llvm/test/CodeGen/AMDGPU/workgroup-id-in-arch-sgprs.ll
+++ b/llvm/test/CodeGen/AMDGPU/workgroup-id-in-arch-sgprs.ll
@@ -38,10 +38,11 @@ define amdgpu_kernel void @workgroup_id_x(ptr addrspace(1) %ptrx) {
; GFX1250-SDAG-NEXT: s_wait_xcnt 0x0
; GFX1250-SDAG-NEXT: s_getreg_b32 s4, hwreg(HW_REG_IB_STS2, 6, 4)
; GFX1250-SDAG-NEXT: s_mul_i32 s2, ttmp9, s2
-; GFX1250-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
+; GFX1250-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX1250-SDAG-NEXT: s_add_co_i32 s3, s3, s2
; GFX1250-SDAG-NEXT: s_cmp_eq_u32 s4, 0
; GFX1250-SDAG-NEXT: s_cselect_b32 s2, ttmp9, s3
+; GFX1250-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1250-SDAG-NEXT: v_dual_mov_b32 v0, 0 :: v_dual_mov_b32 v1, s2
; GFX1250-SDAG-NEXT: s_wait_kmcnt 0x0
; GFX1250-SDAG-NEXT: global_store_b32 v0, v1, s[0:1]
@@ -63,8 +64,8 @@ define amdgpu_kernel void @workgroup_id_x(ptr addrspace(1) %ptrx) {
; GFX1250-GISEL-NEXT: v_mov_b32_e32 v1, 0
; GFX1250-GISEL-NEXT: s_add_co_i32 s3, s3, s2
; GFX1250-GISEL-NEXT: s_cmp_eq_u32 s4, 0
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1250-GISEL-NEXT: s_cselect_b32 s2, ttmp9, s3
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1250-GISEL-NEXT: v_mov_b32_e32 v0, s2
; GFX1250-GISEL-NEXT: s_wait_kmcnt 0x0
; GFX1250-GISEL-NEXT: global_store_b32 v1, v0, s[0:1]
@@ -128,6 +129,7 @@ define amdgpu_kernel void @workgroup_id_xy(ptr addrspace(1) %ptrx, ptr addrspace
; GFX1250-SDAG-NEXT: s_getreg_b32 s8, hwreg(HW_REG_IB_STS2, 6, 4)
; GFX1250-SDAG-NEXT: s_add_co_i32 s5, s5, s7
; GFX1250-SDAG-NEXT: s_cmp_eq_u32 s8, 0
+; GFX1250-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1250-SDAG-NEXT: s_cselect_b32 s5, ttmp9, s5
; GFX1250-SDAG-NEXT: s_cselect_b32 s4, s4, s6
; GFX1250-SDAG-NEXT: v_dual_mov_b32 v0, 0 :: v_dual_mov_b32 v1, s5
@@ -163,8 +165,8 @@ define amdgpu_kernel void @workgroup_id_xy(ptr addrspace(1) %ptrx, ptr addrspace
; GFX1250-GISEL-NEXT: v_mov_b32_e32 v0, s4
; GFX1250-GISEL-NEXT: s_add_co_i32 s8, s8, s5
; GFX1250-GISEL-NEXT: s_cmp_eq_u32 s6, 0
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1250-GISEL-NEXT: s_cselect_b32 s4, s7, s8
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1250-GISEL-NEXT: v_mov_b32_e32 v2, s4
; GFX1250-GISEL-NEXT: s_wait_kmcnt 0x0
; GFX1250-GISEL-NEXT: s_clause 0x1
@@ -253,11 +255,12 @@ define amdgpu_kernel void @workgroup_id_xyz(ptr addrspace(1) %ptrx, ptr addrspac
; GFX1250-SDAG-NEXT: s_getreg_b32 s12, hwreg(HW_REG_IB_STS2, 6, 4)
; GFX1250-SDAG-NEXT: s_add_co_i32 s4, s4, s11
; GFX1250-SDAG-NEXT: s_cmp_eq_u32 s12, 0
+; GFX1250-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1250-SDAG-NEXT: s_cselect_b32 s4, ttmp9, s4
-; GFX1250-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
; GFX1250-SDAG-NEXT: v_dual_mov_b32 v0, 0 :: v_dual_mov_b32 v1, s4
; GFX1250-SDAG-NEXT: s_cselect_b32 s4, s10, s9
; GFX1250-SDAG-NEXT: s_cselect_b32 s5, s8, s5
+; GFX1250-SDAG-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1250-SDAG-NEXT: v_dual_mov_b32 v2, s4 :: v_dual_mov_b32 v3, s5
; GFX1250-SDAG-NEXT: s_wait_kmcnt 0x0
; GFX1250-SDAG-NEXT: s_clause 0x2
@@ -280,6 +283,7 @@ define amdgpu_kernel void @workgroup_id_xyz(ptr addrspace(1) %ptrx, ptr addrspac
; GFX1250-GISEL-NEXT: v_mov_b32_e32 v1, 0
; GFX1250-GISEL-NEXT: s_add_co_i32 s1, s1, s0
; GFX1250-GISEL-NEXT: s_cmp_eq_u32 s8, 0
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1250-GISEL-NEXT: s_cselect_b32 s9, ttmp9, s1
; GFX1250-GISEL-NEXT: s_bfe_u32 s0, ttmp6, 0x40010
; GFX1250-GISEL-NEXT: s_and_b32 s10, ttmp7, 0xffff
@@ -299,10 +303,11 @@ define amdgpu_kernel void @workgroup_id_xyz(ptr addrspace(1) %ptrx, ptr addrspac
; GFX1250-GISEL-NEXT: s_add_co_i32 s5, s5, 1
; GFX1250-GISEL-NEXT: s_bfe_u32 s11, ttmp6, 0x40008
; GFX1250-GISEL-NEXT: s_mul_i32 s5, s10, s5
-; GFX1250-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
; GFX1250-GISEL-NEXT: s_add_co_i32 s11, s11, s5
; GFX1250-GISEL-NEXT: s_cmp_eq_u32 s8, 0
; GFX1250-GISEL-NEXT: s_cselect_b32 s5, s10, s11
+; GFX1250-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1250-GISEL-NEXT: v_dual_mov_b32 v2, s4 :: v_dual_mov_b32 v3, s5
; GFX1250-GISEL-NEXT: s_wait_kmcnt 0x0
; GFX1250-GISEL-NEXT: s_clause 0x2
diff --git a/llvm/test/CodeGen/AMDGPU/workitem-intrinsic-opts.ll b/llvm/test/CodeGen/AMDGPU/workitem-intrinsic-opts.ll
index 07937b347a6224..aa769a59a2f2dc 100644
--- a/llvm/test/CodeGen/AMDGPU/workitem-intrinsic-opts.ll
+++ b/llvm/test/CodeGen/AMDGPU/workitem-intrinsic-opts.ll
@@ -246,6 +246,7 @@ define i1 @workgroup_zero() {
; GISEL-GFX12-NEXT: s_or_b32 s0, s0, s1
; GISEL-GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GISEL-GFX12-NEXT: s_cmp_eq_u32 s0, 0
+; GISEL-GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GISEL-GFX12-NEXT: s_cselect_b32 s0, 1, 0
; GISEL-GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GISEL-GFX12-NEXT: v_mov_b32_e32 v0, s0
>From 55cf92a1fecd60c650afed242398a2c70826992c Mon Sep 17 00:00:00 2001
From: Reem Elkhouly <reem.elkhouly at amd.com>
Date: Mon, 14 Sep 2026 11:12:00 +0900
Subject: [PATCH 3/3] Use the target subclass in target code
---
.../Target/AMDGPU/AMDGPUInsertDelayAlu.cpp | 24 ++++++-------------
1 file changed, 7 insertions(+), 17 deletions(-)
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUInsertDelayAlu.cpp b/llvm/lib/Target/AMDGPU/AMDGPUInsertDelayAlu.cpp
index 76de039e07e950..883c9cebad2fe6 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUInsertDelayAlu.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUInsertDelayAlu.cpp
@@ -351,15 +351,12 @@ class AMDGPUInsertDelayAlu {
return (Imm & 0x780) ? nullptr : DelayAlu;
}
- static bool isFastForwardProducer(const MachineInstr &MI,
- const MachineOperand &MO) {
+ bool isFastForwardProducer(const MachineInstr &MI, const MachineOperand &MO) {
assert((MO.isReg() && MO.isDef()) && "Expected a register definition");
if (!SIInstrInfo::isVALU(MI, /*AllowLDSDMA=*/false))
return false;
- const TargetRegisterInfo *TRI =
- MI.getMF()->getSubtarget().getRegisterInfo();
Register Reg = MO.getReg();
if (AMDGPU::isSGPR(Reg, TRI)) {
switch (MI.getOpcode()) {
@@ -388,20 +385,16 @@ class AMDGPUInsertDelayAlu {
return false;
}
- static unsigned int getVOPDComponentOpCode(const MachineInstr &MI,
- unsigned OpNo,
- const MachineOperand &MO) {
+ unsigned int getVOPDComponentOpCode(const MachineInstr &MI, unsigned OpNo,
+ const MachineOperand &MO) {
Register Reg = MO.getReg();
auto MIOpCode = MI.getOpcode();
// Get the component instruction descriptors for VOPD.
auto [OpX, OpY] = AMDGPU::getVOPDComponents(MIOpCode);
- const MCInstrInfo *MCII = MI.getMF()->getTarget().getMCInstrInfo();
if (MO.isImplicit()) {
- const TargetRegisterInfo *TRI =
- MI.getMF()->getSubtarget().getRegisterInfo();
for (unsigned CompOp : {OpX, OpY}) {
- const MCInstrDesc &CompDesc = MCII->get(CompOp);
+ const MCInstrDesc &CompDesc = SII->get(CompOp);
bool UsesReg = any_of(CompDesc.implicit_uses(), [&](MCPhysReg R) {
return TRI->regsOverlap(R, Reg);
});
@@ -414,7 +407,7 @@ class AMDGPUInsertDelayAlu {
// Explicit source: identify which component/source THIS operand is by
// matching its operand index (OpNo). The same register can appear in
// multiple slots, so matching by register value is not reliable.
- const auto &InstInfo = AMDGPU::getVOPDInstInfo(MIOpCode, MCII);
+ const auto &InstInfo = AMDGPU::getVOPDInstInfo(MIOpCode, SII);
// Map a parsed source index (0/1/2) to the matching named operand for a
// standalone component opcode.
@@ -450,16 +443,13 @@ class AMDGPUInsertDelayAlu {
return MIOpCode;
}
- static bool isFastForwardConsumer(const MachineInstr &MI,
- const MachineOperand &MO, Register VccReg,
- Register ExecReg, unsigned OpNo) {
+ bool isFastForwardConsumer(const MachineInstr &MI, const MachineOperand &MO,
+ Register VccReg, Register ExecReg, unsigned OpNo) {
assert((MO.isReg() && MO.isUse()) && "Expected a register use");
if (!SIInstrInfo::isVALU(MI, /*AllowLDSDMA=*/false))
return false;
- const TargetRegisterInfo *TRI =
- MI.getMF()->getSubtarget().getRegisterInfo();
Register Reg = MO.getReg();
auto MIOpCode = MI.getOpcode();
More information about the llvm-commits
mailing list