[llvm] [AMDGPU][GFX12/GFX13] Add cdbg branch support and lower llvm.is.debugging.enabled (PR #219178)
Alexander Hück via llvm-commits
llvm-commits at lists.llvm.org
Thu Aug 27 04:11:55 PDT 2026
https://github.com/ahueck created https://github.com/llvm/llvm-project/pull/219178
This is PR 3 of 4 in a series adding `llvm.is.debugging.enabled` intrinsic to LLVM IR and lowering this intrinsic for AMDGPU targets.
An RFC post will follow.
## Changes of this PR
- Opt AMDGPU into lowering `llvm.is.debugging.enabled`. Unsupported subtargets replace the intrinsic with false.
- Add cdbg conditional branch instruction support for GFX12 and GFX13.
- On supported subtargets, SelectionDAG and GlobalISel lowers the intrinsic by either reading the generation-specific debugging-state bits with `s_getreg` or fuse safe single-branch uses into `s_cbranch_cdbgsys_or_user`. This includes negated branch conditions.
- Mark materialized and fused observations NoMerge to keep calls distinct.
PR 4 uses this support to implement AMDPAL lowering of `llvm.debugtrap`.
>From a0a8085981d024e89ad8c1ebad2b8730ab707f98 Mon Sep 17 00:00:00 2001
From: =?UTF-8?q?Alexander=20H=C3=BCck?= <alexander.huck at amd.com>
Date: Mon, 17 Aug 2026 07:28:30 -0400
Subject: [PATCH 1/5] [AMDGPU][GlobalISel] Make control-flow intrinsic matching
non-mutating
Refactor AMDGPULegalizer's control-flow intrinsic handling to separate
structural matching from MIR mutation. Replace verifyCFIntrinsic with an
explicit match-and-commit flow, and share branch fallthrough redirection
across the affected legalization paths.
---
.../lib/Target/AMDGPU/AMDGPULegalizerInfo.cpp | 134 ++++++++++--------
.../GlobalISel/legalize-amdgcn.if-invalid.mir | 27 +++-
2 files changed, 97 insertions(+), 64 deletions(-)
diff --git a/llvm/lib/Target/AMDGPU/AMDGPULegalizerInfo.cpp b/llvm/lib/Target/AMDGPU/AMDGPULegalizerInfo.cpp
index 7c1a26f761c96..0f0812a5cac36 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPULegalizerInfo.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPULegalizerInfo.cpp
@@ -4804,47 +4804,77 @@ static bool isNot(const MachineRegisterInfo &MRI, const MachineInstr &MI) {
return ConstVal == -1;
}
-// Return the use branch instruction, otherwise null if the usage is invalid.
-static MachineInstr *
-verifyCFIntrinsic(MachineInstr &MI, MachineRegisterInfo &MRI, MachineInstr *&Br,
- MachineBasicBlock *&UncondBrTarget, bool &Negated) {
+namespace {
+struct CFIntrinsicBranchMatch {
+ MachineInstr *Negation = nullptr;
+ MachineInstr *CondBr = nullptr;
+ MachineInstr *UncondBr = nullptr;
+
+ MachineBasicBlock *ConditionTrueTarget = nullptr;
+ MachineBasicBlock *ConditionFalseTarget = nullptr;
+
+ bool isNegated() const { return Negation != nullptr; }
+
+ // Retarget the explicit branch for the fallthrough edge, or materialize
+ // the branch when that edge was represented by layout fallthrough. The
+ // builder must be positioned at the replacement branch.
+ void redirectFallthroughEdge(MachineIRBuilder &B,
+ MachineBasicBlock &Target) const {
+ if (UncondBr)
+ UncondBr->getOperand(0).setMBB(&Target);
+ else
+ B.buildBr(Target);
+ }
+
+ void eraseDeadNegation(MachineRegisterInfo &MRI) {
+ if (Negation)
+ eraseInstr(*Negation, MRI);
+ }
+};
+} // namespace
+
+static std::optional<CFIntrinsicBranchMatch>
+matchCFIntrinsicBranchUse(MachineInstr &MI, MachineRegisterInfo &MRI) {
Register CondDef = MI.getOperand(0).getReg();
if (!MRI.hasOneNonDBGUse(CondDef))
- return nullptr;
+ return std::nullopt;
MachineBasicBlock *Parent = MI.getParent();
MachineInstr *UseMI = &*MRI.use_instr_nodbg_begin(CondDef);
+ MachineInstr *Negation = nullptr;
if (isNot(MRI, *UseMI)) {
+ Negation = UseMI;
Register NegatedCond = UseMI->getOperand(0).getReg();
if (!MRI.hasOneNonDBGUse(NegatedCond))
- return nullptr;
-
- // We're deleting the def of this value, so we need to remove it.
- eraseInstr(*UseMI, MRI);
+ return std::nullopt;
UseMI = &*MRI.use_instr_nodbg_begin(NegatedCond);
- Negated = true;
}
if (UseMI->getParent() != Parent || UseMI->getOpcode() != AMDGPU::G_BRCOND)
- return nullptr;
+ return std::nullopt;
// Make sure the cond br is followed by a G_BR, or is the last instruction.
+ MachineInstr *UncondBr = nullptr;
+ MachineBasicBlock *OtherTarget = nullptr;
MachineBasicBlock::iterator Next = std::next(UseMI->getIterator());
if (Next == Parent->end()) {
MachineFunction::iterator NextMBB = std::next(Parent->getIterator());
if (NextMBB == Parent->getParent()->end()) // Illegal intrinsic use.
- return nullptr;
- UncondBrTarget = &*NextMBB;
+ return std::nullopt;
+ OtherTarget = &*NextMBB;
} else {
if (Next->getOpcode() != AMDGPU::G_BR)
- return nullptr;
- Br = &*Next;
- UncondBrTarget = Br->getOperand(0).getMBB();
+ return std::nullopt;
+ UncondBr = &*Next;
+ OtherTarget = UncondBr->getOperand(0).getMBB();
}
- return UseMI;
+ MachineBasicBlock *TakenTarget = UseMI->getOperand(1).getMBB();
+ return CFIntrinsicBranchMatch{Negation, UseMI, UncondBr,
+ Negation ? OtherTarget : TakenTarget,
+ Negation ? TakenTarget : OtherTarget};
}
void AMDGPULegalizerInfo::buildLoadInputValue(Register DstReg,
@@ -8230,80 +8260,58 @@ bool AMDGPULegalizerInfo::legalizeIntrinsic(LegalizerHelper &Helper,
return true;
case Intrinsic::amdgcn_if:
case Intrinsic::amdgcn_else: {
- MachineInstr *Br = nullptr;
- MachineBasicBlock *UncondBrTarget = nullptr;
- bool Negated = false;
- if (MachineInstr *BrCond =
- verifyCFIntrinsic(MI, MRI, Br, UncondBrTarget, Negated)) {
- const SIRegisterInfo *TRI
- = static_cast<const SIRegisterInfo *>(MRI.getTargetRegisterInfo());
+ if (auto Match = matchCFIntrinsicBranchUse(MI, MRI)) {
+ const SIRegisterInfo *TRI =
+ static_cast<const SIRegisterInfo *>(MRI.getTargetRegisterInfo());
Register Def = MI.getOperand(1).getReg();
Register Use = MI.getOperand(3).getReg();
- MachineBasicBlock *CondBrTarget = BrCond->getOperand(1).getMBB();
-
- if (Negated)
- std::swap(CondBrTarget, UncondBrTarget);
-
- B.setInsertPt(B.getMBB(), BrCond->getIterator());
+ B.setInsertPt(B.getMBB(), Match->CondBr->getIterator());
if (IntrID == Intrinsic::amdgcn_if) {
B.buildInstr(AMDGPU::SI_IF)
- .addDef(Def)
- .addUse(Use)
- .addMBB(UncondBrTarget);
+ .addDef(Def)
+ .addUse(Use)
+ .addMBB(Match->ConditionFalseTarget);
} else {
B.buildInstr(AMDGPU::SI_ELSE)
.addDef(Def)
.addUse(Use)
- .addMBB(UncondBrTarget);
+ .addMBB(Match->ConditionFalseTarget);
}
- if (Br) {
- Br->getOperand(0).setMBB(CondBrTarget);
- } else {
- // The IRTranslator skips inserting the G_BR for fallthrough cases, but
- // since we're swapping branch targets it needs to be reinserted.
- // FIXME: IRTranslator should probably not do this
- B.buildBr(*CondBrTarget);
- }
+ // The IRTranslator skips inserting the G_BR for fallthrough cases, but
+ // since we're swapping branch targets it needs to be reinserted.
+ // FIXME: IRTranslator should probably not do this
+ Match->redirectFallthroughEdge(B, *Match->ConditionTrueTarget);
MRI.setRegClass(Def, TRI->getWaveMaskRegClass());
MRI.setRegClass(Use, TRI->getWaveMaskRegClass());
+ Match->eraseDeadNegation(MRI);
MI.eraseFromParent();
- BrCond->eraseFromParent();
+ Match->CondBr->eraseFromParent();
return true;
}
return false;
}
case Intrinsic::amdgcn_loop: {
- MachineInstr *Br = nullptr;
- MachineBasicBlock *UncondBrTarget = nullptr;
- bool Negated = false;
- if (MachineInstr *BrCond =
- verifyCFIntrinsic(MI, MRI, Br, UncondBrTarget, Negated)) {
- const SIRegisterInfo *TRI
- = static_cast<const SIRegisterInfo *>(MRI.getTargetRegisterInfo());
-
- MachineBasicBlock *CondBrTarget = BrCond->getOperand(1).getMBB();
- Register Reg = MI.getOperand(2).getReg();
+ if (auto Match = matchCFIntrinsicBranchUse(MI, MRI)) {
+ const SIRegisterInfo *TRI =
+ static_cast<const SIRegisterInfo *>(MRI.getTargetRegisterInfo());
- if (Negated)
- std::swap(CondBrTarget, UncondBrTarget);
+ Register Reg = MI.getOperand(2).getReg();
- B.setInsertPt(B.getMBB(), BrCond->getIterator());
+ B.setInsertPt(B.getMBB(), Match->CondBr->getIterator());
B.buildInstr(AMDGPU::SI_LOOP)
- .addUse(Reg)
- .addMBB(UncondBrTarget);
+ .addUse(Reg)
+ .addMBB(Match->ConditionFalseTarget);
- if (Br)
- Br->getOperand(0).setMBB(CondBrTarget);
- else
- B.buildBr(*CondBrTarget);
+ Match->redirectFallthroughEdge(B, *Match->ConditionTrueTarget);
+ Match->eraseDeadNegation(MRI);
MI.eraseFromParent();
- BrCond->eraseFromParent();
+ Match->CondBr->eraseFromParent();
MRI.setRegClass(Reg, TRI->getWaveMaskRegClass());
return true;
}
diff --git a/llvm/test/CodeGen/AMDGPU/GlobalISel/legalize-amdgcn.if-invalid.mir b/llvm/test/CodeGen/AMDGPU/GlobalISel/legalize-amdgcn.if-invalid.mir
index d351eb6a57446..77040216daca8 100644
--- a/llvm/test/CodeGen/AMDGPU/GlobalISel/legalize-amdgcn.if-invalid.mir
+++ b/llvm/test/CodeGen/AMDGPU/GlobalISel/legalize-amdgcn.if-invalid.mir
@@ -1,4 +1,5 @@
# RUN: llc -mtriple=amdgpu8.03-mesa-mesa3d -O0 -run-pass=legalizer -global-isel-abort=2 -pass-remarks-missed='gisel*' -filetype=null %s 2>&1 | FileCheck -check-prefix=ERR %s
+# RUN: llc -mtriple=amdgpu8.03-mesa-mesa3d -O0 -run-pass=legalizer -global-isel-abort=0 %s -o - | FileCheck -check-prefix=MIR %s
# Make sure incorrect usage of control flow intrinsics fails to select in case some transform separated the intrinsic from its branch.
@@ -8,7 +9,7 @@
# ERR-NEXT: remark: <unknown>:0:0: unable to legalize instruction: %3:_(s1), %4:_(s64) = G_INTRINSIC_W_SIDE_EFFECTS intrinsic(@llvm.amdgcn.if), %2:_(s1) (in function: brcond_si_if_xor_0)
# ERR-NEXT: remark: <unknown>:0:0: unable to legalize instruction: %3:_(s1), %4:_(s64) = G_INTRINSIC_W_SIDE_EFFECTS intrinsic(@llvm.amdgcn.if), %2:_(s1) (in function: brcond_si_if_or_neg1)
# ERR-NEXT: remark: <unknown>:0:0: unable to legalize instruction: %3:_(s1), %4:_(s64) = G_INTRINSIC_W_SIDE_EFFECTS intrinsic(@llvm.amdgcn.if), %2:_(s1) (in function: brcond_si_if_negated_multi_use)
-
+# ERR-NEXT: remark: <unknown>:0:0: unable to legalize instruction: %3:_(s1), %4:_(s64) = G_INTRINSIC_W_SIDE_EFFECTS intrinsic(@llvm.amdgcn.if), %2:_(s1) (in function: si_if_negated_invalid_branch_layout)
---
name: brcond_si_if_different_block
@@ -134,3 +135,27 @@ body: |
bb.3:
S_NOP 2
...
+
+# A failed match after recognizing the negation must not modify the MIR.
+# MIR-LABEL: name: si_if_negated_invalid_branch_layout
+# MIR: [[NOT:%[0-9]+]]:_(s1) = G_XOR
+# MIR-NEXT: G_BRCOND [[NOT]](s1), %bb.1
+# MIR-NEXT: S_ENDPGM 0
+---
+name: si_if_negated_invalid_branch_layout
+body: |
+ bb.0:
+ successors: %bb.1
+ liveins: $vgpr0, $vgpr1
+ %0:_(s32) = COPY $vgpr0
+ %1:_(s32) = COPY $vgpr1
+ %2:_(s1) = G_ICMP intpred(ne), %0, %1
+ %3:_(s1), %4:_(s64) = G_INTRINSIC_W_SIDE_EFFECTS intrinsic(@llvm.amdgcn.if), %2
+ %5:_(s1) = G_CONSTANT i1 true
+ %6:_(s1) = G_XOR %3, %5
+ G_BRCOND %6, %bb.1
+ S_ENDPGM 0
+
+ bb.1:
+ S_ENDPGM 0
+...
>From 61e8be4e9f5dfb6ce64cf9ee1ab6e78af9a7715c Mon Sep 17 00:00:00 2001
From: =?UTF-8?q?Alexander=20H=C3=BCck?= <alexander.huck at amd.com>
Date: Mon, 17 Aug 2026 11:58:25 -0400
Subject: [PATCH 2/5] [AMDGPU][SelectionDAG] Refactor control-flow branch
matching
Introduce BRCONDMatch to normalize wrapped and negated branch conditions
and record their true and false targets. Use the match in LowerBRCOND to
centralize fallthrough redirection and separate branch matching from
control-flow intrinsic lowering.
---
llvm/lib/Target/AMDGPU/SIISelLowering.cpp | 124 ++++++++++++++--------
1 file changed, 81 insertions(+), 43 deletions(-)
diff --git a/llvm/lib/Target/AMDGPU/SIISelLowering.cpp b/llvm/lib/Target/AMDGPU/SIISelLowering.cpp
index d9810e3d9fbd9..86cfd885c4f8f 100644
--- a/llvm/lib/Target/AMDGPU/SIISelLowering.cpp
+++ b/llvm/lib/Target/AMDGPU/SIISelLowering.cpp
@@ -8584,6 +8584,72 @@ static SDNode *findUser(SDValue Value, unsigned Opcode) {
return nullptr;
}
+namespace {
+struct BRCONDMatch {
+ SDValue CondBr;
+ SDValue Condition;
+ SDValue ConditionWrapper;
+ SDNode *UncondBr = nullptr;
+
+ SDValue ConditionTrueTarget;
+ SDValue ConditionFalseTarget;
+
+ void redirectFallthroughEdge(SelectionDAG &DAG, SDValue Target) const {
+ if (UncondBr->getOperand(1) == Target)
+ return;
+
+ SDValue NewBr = DAG.getNode(ISD::BR, SDLoc(CondBr), UncondBr->getVTList(),
+ {UncondBr->getOperand(0), Target});
+ DAG.ReplaceAllUsesWith(UncondBr, NewBr.getNode());
+ }
+};
+} // namespace
+
+static std::optional<BRCONDMatch> matchBRCOND(SDValue CondBr) {
+ if (CondBr.getOpcode() != ISD::BRCOND)
+ return std::nullopt;
+
+ SDValue Condition = CondBr.getOperand(1);
+ SDValue ConditionWrapper;
+ bool IsNegated = false;
+
+ switch (Condition.getOpcode()) {
+ case ISD::SETCC: {
+ ISD::CondCode CC = cast<CondCodeSDNode>(Condition.getOperand(2))->get();
+ if (auto *C = dyn_cast<ConstantSDNode>(Condition.getOperand(1));
+ C && (CC == ISD::SETEQ || CC == ISD::SETNE)) {
+ IsNegated = (CC == ISD::SETEQ) == (C->getZExtValue() == 0);
+ ConditionWrapper = Condition;
+ Condition = Condition.getOperand(0);
+ }
+ break;
+ }
+ case ISD::XOR:
+ if (auto *C = dyn_cast<ConstantSDNode>(Condition.getOperand(1));
+ C && C->getZExtValue()) {
+ ConditionWrapper = Condition;
+ Condition = Condition.getOperand(0);
+ IsNegated = true;
+ }
+ break;
+ default:
+ break;
+ }
+
+ SDNode *UncondBr = findUser(CondBr, ISD::BR);
+ if (!UncondBr)
+ return std::nullopt;
+
+ SDValue TakenTarget = CondBr.getOperand(2);
+ SDValue OtherTarget = UncondBr->getOperand(1);
+ return BRCONDMatch{CondBr,
+ Condition,
+ ConditionWrapper,
+ UncondBr,
+ IsNegated ? OtherTarget : TakenTarget,
+ IsNegated ? TakenTarget : OtherTarget};
+}
+
unsigned SITargetLowering::isCFIntrinsic(const SDNode *Intr) const {
if (Intr->getOpcode() == ISD::INTRINSIC_W_CHAIN) {
switch (Intr->getConstantOperandVal(1)) {
@@ -8650,37 +8716,12 @@ bool SITargetLowering::shouldUseLDSConstAddress(const GlobalValue *GV) const {
/// This transforms the control flow intrinsics to get the branch destination as
/// last parameter, also switches branch target with BR if the need arise
SDValue SITargetLowering::LowerBRCOND(SDValue BRCOND, SelectionDAG &DAG) const {
- SDLoc DL(BRCOND);
+ auto Match = matchBRCOND(BRCOND);
+ if (!Match)
+ return BRCOND;
- SDNode *Intr = BRCOND.getOperand(1).getNode();
- SDValue Target = BRCOND.getOperand(2);
- SDNode *BR = nullptr;
- SDNode *SetCC = nullptr;
-
- switch (Intr->getOpcode()) {
- case ISD::SETCC: {
- // As long as we negate the condition everything is fine
- SetCC = Intr;
- Intr = SetCC->getOperand(0).getNode();
- break;
- }
- case ISD::XOR: {
- // Similar to SETCC, if we have (xor c, -1), we will be fine.
- SDValue LHS = Intr->getOperand(0);
- SDValue RHS = Intr->getOperand(1);
- if (auto *C = dyn_cast<ConstantSDNode>(RHS); C && C->getZExtValue()) {
- Intr = LHS.getNode();
- break;
- }
- [[fallthrough]];
- }
- default: {
- // Get the target from BR if we don't negate the condition
- BR = findUser(BRCOND, ISD::BR);
- assert(BR && "brcond missing unconditional branch user");
- Target = BR->getOperand(1);
- }
- }
+ SDLoc DL(Match->CondBr);
+ SDNode *Intr = Match->Condition.getNode();
unsigned CFNode = isCFIntrinsic(Intr);
if (CFNode == 0) {
@@ -8691,18 +8732,20 @@ SDValue SITargetLowering::LowerBRCOND(SDValue BRCOND, SelectionDAG &DAG) const {
bool HaveChain = Intr->getOpcode() == ISD::INTRINSIC_VOID ||
Intr->getOpcode() == ISD::INTRINSIC_W_CHAIN;
- assert(!SetCC ||
- (SetCC->getConstantOperandVal(1) == 1 &&
- cast<CondCodeSDNode>(SetCC->getOperand(2).getNode())->get() ==
- ISD::SETNE));
+ assert((!Match->ConditionWrapper ||
+ Match->ConditionWrapper.getOpcode() != ISD::SETCC ||
+ (Match->ConditionWrapper.getConstantOperandVal(1) == 1 &&
+ cast<CondCodeSDNode>(Match->ConditionWrapper.getOperand(2).getNode())
+ ->get() == ISD::SETNE)) &&
+ "unexpected control flow intrinsic condition wrapper");
// operands of the new intrinsic call
SmallVector<SDValue, 4> Ops;
if (HaveChain)
- Ops.push_back(BRCOND.getOperand(0));
+ Ops.push_back(Match->CondBr.getOperand(0));
Ops.append(Intr->op_begin() + (HaveChain ? 2 : 1), Intr->op_end());
- Ops.push_back(Target);
+ Ops.push_back(Match->ConditionFalseTarget);
ArrayRef<EVT> Res(Intr->value_begin() + 1, Intr->value_end());
@@ -8710,17 +8753,12 @@ SDValue SITargetLowering::LowerBRCOND(SDValue BRCOND, SelectionDAG &DAG) const {
SDNode *Result = DAG.getNode(CFNode, DL, DAG.getVTList(Res), Ops).getNode();
if (!HaveChain) {
- SDValue Ops[] = {SDValue(Result, 0), BRCOND.getOperand(0)};
+ SDValue Ops[] = {SDValue(Result, 0), Match->CondBr.getOperand(0)};
Result = DAG.getMergeValues(Ops, DL).getNode();
}
- if (BR) {
- // Give the branch instruction our target
- SDValue Ops[] = {BR->getOperand(0), BRCOND.getOperand(2)};
- SDValue NewBR = DAG.getNode(ISD::BR, DL, BR->getVTList(), Ops);
- DAG.ReplaceAllUsesWith(BR, NewBR.getNode());
- }
+ Match->redirectFallthroughEdge(DAG, Match->ConditionTrueTarget);
SDValue Chain = SDValue(Result, Result->getNumValues() - 1);
>From 6bbc717745a25b39c94b373715f0791f183dbdde Mon Sep 17 00:00:00 2001
From: =?UTF-8?q?Alexander=20H=C3=BCck?= <alexander.huck at amd.com>
Date: Mon, 24 Aug 2026 08:47:41 +0000
Subject: [PATCH 3/5] [IR][CodeGen] Add llvm.is.debugging.enabled
Add an intrinsic reporting whether debugging is enabled for the
target-defined execution context. Each call is a distinct observation, so
calls must not be removed, commoned, or reordered with one another.
This is modeled with inaccessiblemem readwrite and nomerge.
Add TargetMachine::canLowerIsDebuggingEnabled(), defaulting to false.
Pre-ISel intrinsic lowering replaces the intrinsic with false unless the
target opts in. A target that opts in is responsible for all of its
subtargets.
---
llvm/docs/LangRef.md | 30 ++++++++++
llvm/include/llvm/IR/Intrinsics.td | 6 ++
llvm/include/llvm/Target/TargetMachine.h | 7 +++
llvm/lib/CodeGen/PreISelIntrinsicLowering.cpp | 8 +++
llvm/test/Assembler/is-debugging-enabled.ll | 11 ++++
.../CodeGen/NVPTX/is-debugging-enabled.ll | 11 ++++
llvm/test/CodeGen/X86/is-debugging-enabled.ll | 12 ++++
.../Transforms/DCE/is-debugging-enabled.ll | 12 ++++
.../EarlyCSE/is-debugging-enabled.ll | 17 ++++++
.../Transforms/LICM/is-debugging-enabled.ll | 25 ++++++++
.../LoopUnroll/is-debugging-enabled.ll | 30 ++++++++++
.../is-debugging-enabled.ll | 13 +++++
.../SimplifyCFG/is-debugging-enabled.ll | 58 +++++++++++++++++++
llvm/test/Verifier/is-debugging-enabled.ll | 15 +++++
14 files changed, 255 insertions(+)
create mode 100644 llvm/test/Assembler/is-debugging-enabled.ll
create mode 100644 llvm/test/CodeGen/NVPTX/is-debugging-enabled.ll
create mode 100644 llvm/test/CodeGen/X86/is-debugging-enabled.ll
create mode 100644 llvm/test/Transforms/DCE/is-debugging-enabled.ll
create mode 100644 llvm/test/Transforms/EarlyCSE/is-debugging-enabled.ll
create mode 100644 llvm/test/Transforms/LICM/is-debugging-enabled.ll
create mode 100644 llvm/test/Transforms/LoopUnroll/is-debugging-enabled.ll
create mode 100644 llvm/test/Transforms/PreISelIntrinsicLowering/is-debugging-enabled.ll
create mode 100644 llvm/test/Transforms/SimplifyCFG/is-debugging-enabled.ll
create mode 100644 llvm/test/Verifier/is-debugging-enabled.ll
diff --git a/llvm/docs/LangRef.md b/llvm/docs/LangRef.md
index ee1600b96f1dd..9f2ab993f3a80 100644
--- a/llvm/docs/LangRef.md
+++ b/llvm/docs/LangRef.md
@@ -26263,6 +26263,36 @@ This intrinsic is lowered to code which is intended to cause an
execution trap with the intention of requesting the attention of a
debugger.
+(llvm.is.debugging.enabled)=
+
+#### '`llvm.is.debugging.enabled`' Intrinsic
+
+##### Syntax:
+
+```llvm
+declare noundef i1 @llvm.is.debugging.enabled() nomerge memory(inaccessiblemem: readwrite)
+```
+
+##### Overview:
+
+The '`llvm.is.debugging.enabled`' intrinsic returns whether debugging is enabled
+for the target-defined execution context of the current invocation.
+
+##### Arguments:
+
+None.
+
+##### Semantics:
+
+Each call observes the current debugging-enabled state. Calls are distinct
+observations: LLVM must not assume separate calls return the same value, and a
+call may not be removed when its result is unused, commoned with another call,
+or reordered with respect to one.
+
+On targets that do not support querying debugging-enabled state, the intrinsic
+returns `false` and performs no observation.
+
+
(llvm.ubsantrap)=
#### '`llvm.ubsantrap`' Intrinsic
diff --git a/llvm/include/llvm/IR/Intrinsics.td b/llvm/include/llvm/IR/Intrinsics.td
index c20ea64c7eef4..b14375e277e7a 100644
--- a/llvm/include/llvm/IR/Intrinsics.td
+++ b/llvm/include/llvm/IR/Intrinsics.td
@@ -2059,6 +2059,12 @@ def int_trap : Intrinsic<[], [],
ClangBuiltin<"__builtin_trap">;
def int_debugtrap : Intrinsic<[]>,
ClangBuiltin<"__builtin_debugtrap">;
+// Query whether debugging is enabled for the current execution context.
+def int_is_debugging_enabled :
+ DefaultAttrsIntrinsic<[llvm_i1_ty], [],
+ [IntrInaccessibleMemOnly,
+ IntrNoMerge,
+ NoUndef<RetIndex>]>;
def int_ubsantrap : Intrinsic<[], [llvm_i8_ty],
[IntrNoReturn, IntrCold, ImmArg<ArgIndex<0>>,
IntrInaccessibleMemOnly, IntrWriteMem]>;
diff --git a/llvm/include/llvm/Target/TargetMachine.h b/llvm/include/llvm/Target/TargetMachine.h
index a6b73d636dc27..ea26be5e48bfb 100644
--- a/llvm/include/llvm/Target/TargetMachine.h
+++ b/llvm/include/llvm/Target/TargetMachine.h
@@ -567,6 +567,13 @@ class LLVM_ABI TargetMachine {
/// this function returns false, the intrinsic will be supported generically
/// but without loop detection support.
virtual bool canLowerCondLoop() const { return false; }
+
+ /// Returns whether this target takes responsibility for lowering
+ /// llvm.is.debugging.enabled. If false, generic lowering replaces the intrinsic
+ /// with false. If true, the target must handle every supported subtarget,
+ /// including replacing the intrinsic with false on subtargets without native
+ /// lowering.
+ virtual bool canLowerIsDebuggingEnabled() const { return false; }
};
} // end namespace llvm
diff --git a/llvm/lib/CodeGen/PreISelIntrinsicLowering.cpp b/llvm/lib/CodeGen/PreISelIntrinsicLowering.cpp
index 24788148b9f15..4ec866d7b973f 100644
--- a/llvm/lib/CodeGen/PreISelIntrinsicLowering.cpp
+++ b/llvm/lib/CodeGen/PreISelIntrinsicLowering.cpp
@@ -815,6 +815,14 @@ bool PreISelIntrinsicLowering::lowerIntrinsics(Module &M) const {
case Intrinsic::protected_field_ptr:
Changed |= expandProtectedFieldPtr(F);
break;
+ case Intrinsic::is_debugging_enabled:
+ if (!TM || !TM->canLowerIsDebuggingEnabled())
+ Changed |= forEachCall(F, [](CallInst *CI) {
+ CI->replaceAllUsesWith(ConstantInt::getFalse(CI->getContext()));
+ CI->eraseFromParent();
+ return true;
+ });
+ break;
case Intrinsic::cond_loop:
if (!TM->canLowerCondLoop())
Changed |= expandCondLoop(F);
diff --git a/llvm/test/Assembler/is-debugging-enabled.ll b/llvm/test/Assembler/is-debugging-enabled.ll
new file mode 100644
index 0000000000000..a48d002373e38
--- /dev/null
+++ b/llvm/test/Assembler/is-debugging-enabled.ll
@@ -0,0 +1,11 @@
+; RUN: llvm-as < %s | llvm-dis | FileCheck %s
+
+define i1 @query_debugging_enabled() {
+ %enabled = call i1 @llvm.is.debugging.enabled()
+ ret i1 %enabled
+}
+
+; CHECK: declare noundef i1 @llvm.is.debugging.enabled() #[[ATTRS:[0-9]+]]
+; CHECK: attributes #[[ATTRS]] = { nocallback nofree nomerge nosync nounwind willreturn memory(inaccessiblemem: readwrite) }
+
+declare i1 @llvm.is.debugging.enabled()
diff --git a/llvm/test/CodeGen/NVPTX/is-debugging-enabled.ll b/llvm/test/CodeGen/NVPTX/is-debugging-enabled.ll
new file mode 100644
index 0000000000000..d2794285b4466
--- /dev/null
+++ b/llvm/test/CodeGen/NVPTX/is-debugging-enabled.ll
@@ -0,0 +1,11 @@
+; RUN: llc -mtriple=nvptx64-nvidia-cuda < %s | FileCheck %s
+
+define i1 @query() {
+; CHECK-LABEL: query(
+; CHECK: st.param.b32 [func_retval0], 0;
+; CHECK-NEXT: ret;
+ %enabled = call i1 @llvm.is.debugging.enabled()
+ ret i1 %enabled
+}
+
+declare i1 @llvm.is.debugging.enabled()
diff --git a/llvm/test/CodeGen/X86/is-debugging-enabled.ll b/llvm/test/CodeGen/X86/is-debugging-enabled.ll
new file mode 100644
index 0000000000000..a2f8d98ce2081
--- /dev/null
+++ b/llvm/test/CodeGen/X86/is-debugging-enabled.ll
@@ -0,0 +1,12 @@
+; RUN: llc -mtriple=x86_64 < %s | FileCheck %s
+
+define i1 @query() {
+; CHECK-LABEL: query:
+; CHECK: # %bb.0:
+; CHECK-NEXT: xorl %eax, %eax
+; CHECK-NEXT: retq
+ %enabled = call i1 @llvm.is.debugging.enabled()
+ ret i1 %enabled
+}
+
+declare i1 @llvm.is.debugging.enabled()
diff --git a/llvm/test/Transforms/DCE/is-debugging-enabled.ll b/llvm/test/Transforms/DCE/is-debugging-enabled.ll
new file mode 100644
index 0000000000000..ddf9d80648e78
--- /dev/null
+++ b/llvm/test/Transforms/DCE/is-debugging-enabled.ll
@@ -0,0 +1,12 @@
+; RUN: opt -passes=dce -S < %s | FileCheck %s
+
+define void @unused_query_is_observable() {
+; CHECK-LABEL: define void @unused_query_is_observable() {
+; CHECK-NEXT: [[ENABLED:%.*]] = call i1 @llvm.is.debugging.enabled()
+; CHECK-NEXT: ret void
+;
+ %enabled = call i1 @llvm.is.debugging.enabled()
+ ret void
+}
+
+declare i1 @llvm.is.debugging.enabled()
diff --git a/llvm/test/Transforms/EarlyCSE/is-debugging-enabled.ll b/llvm/test/Transforms/EarlyCSE/is-debugging-enabled.ll
new file mode 100644
index 0000000000000..e9a4ba6d137e5
--- /dev/null
+++ b/llvm/test/Transforms/EarlyCSE/is-debugging-enabled.ll
@@ -0,0 +1,17 @@
+; RUN: opt -passes=early-cse -S < %s | FileCheck %s
+; RUN: opt -passes=gvn -S < %s | FileCheck %s
+
+define i1 @distinct_queries_are_not_commoned() {
+; CHECK-LABEL: define i1 @distinct_queries_are_not_commoned() {
+; CHECK-NEXT: [[FIRST:%.*]] = call i1 @llvm.is.debugging.enabled()
+; CHECK-NEXT: [[SECOND:%.*]] = call i1 @llvm.is.debugging.enabled()
+; CHECK-NEXT: [[DIFFER:%.*]] = xor i1 [[FIRST]], [[SECOND]]
+; CHECK-NEXT: ret i1 [[DIFFER]]
+;
+ %first = call i1 @llvm.is.debugging.enabled()
+ %second = call i1 @llvm.is.debugging.enabled()
+ %differ = xor i1 %first, %second
+ ret i1 %differ
+}
+
+declare i1 @llvm.is.debugging.enabled()
diff --git a/llvm/test/Transforms/LICM/is-debugging-enabled.ll b/llvm/test/Transforms/LICM/is-debugging-enabled.ll
new file mode 100644
index 0000000000000..e7725f716fe5a
--- /dev/null
+++ b/llvm/test/Transforms/LICM/is-debugging-enabled.ll
@@ -0,0 +1,25 @@
+; RUN: opt -passes=licm -S < %s | FileCheck %s
+
+define void @query_stays_in_loop(i32 %count) {
+; CHECK-LABEL: define void @query_stays_in_loop(
+; CHECK: loop:
+; CHECK: call i1 @llvm.is.debugging.enabled()
+; CHECK: br i1 {{.*}}, label %loop, label %exit
+; CHECK: exit:
+;
+entry:
+ %nonzero = icmp ne i32 %count, 0
+ br i1 %nonzero, label %loop, label %exit
+
+loop:
+ %index = phi i32 [ 0, %entry ], [ %next, %loop ]
+ %enabled = call i1 @llvm.is.debugging.enabled()
+ %next = add nuw i32 %index, 1
+ %more = icmp ult i32 %next, %count
+ br i1 %more, label %loop, label %exit
+
+exit:
+ ret void
+}
+
+declare i1 @llvm.is.debugging.enabled()
diff --git a/llvm/test/Transforms/LoopUnroll/is-debugging-enabled.ll b/llvm/test/Transforms/LoopUnroll/is-debugging-enabled.ll
new file mode 100644
index 0000000000000..de1d71edd6d9f
--- /dev/null
+++ b/llvm/test/Transforms/LoopUnroll/is-debugging-enabled.ll
@@ -0,0 +1,30 @@
+; RUN: opt -passes=loop-unroll -S < %s | FileCheck %s
+
+; Full unrolling may duplicate the static query site because it preserves the
+; three dynamically executed observations from the original loop.
+define void @query_allows_full_unrolling() {
+; CHECK-LABEL: define void @query_allows_full_unrolling() {
+; CHECK: loop:
+; CHECK-NEXT: call i1 @llvm.is.debugging.enabled()
+; CHECK-NEXT: call i1 @llvm.is.debugging.enabled()
+; CHECK-NEXT: call i1 @llvm.is.debugging.enabled()
+; CHECK-NEXT: ret void
+;
+entry:
+ br label %loop
+
+loop:
+ %index = phi i32 [ 0, %entry ], [ %next, %loop ]
+ %enabled = call i1 @llvm.is.debugging.enabled()
+ %next = add nuw nsw i32 %index, 1
+ %done = icmp eq i32 %next, 3
+ br i1 %done, label %exit, label %loop, !llvm.loop !0
+
+exit:
+ ret void
+}
+
+declare i1 @llvm.is.debugging.enabled()
+
+!0 = distinct !{!0, !1}
+!1 = !{!"llvm.loop.unroll.full"}
diff --git a/llvm/test/Transforms/PreISelIntrinsicLowering/is-debugging-enabled.ll b/llvm/test/Transforms/PreISelIntrinsicLowering/is-debugging-enabled.ll
new file mode 100644
index 0000000000000..2ce321558677b
--- /dev/null
+++ b/llvm/test/Transforms/PreISelIntrinsicLowering/is-debugging-enabled.ll
@@ -0,0 +1,13 @@
+; RUN: %if x86-registered-target %{ opt -mtriple=x86_64 -passes=pre-isel-intrinsic-lowering -S < %s | FileCheck %s --check-prefix=UNSUPPORTED %}
+; RUN: %if nvptx-registered-target %{ opt -mtriple=nvptx64-nvidia-cuda -passes=pre-isel-intrinsic-lowering -S < %s | FileCheck %s --check-prefix=UNSUPPORTED %}
+; RUN: %if amdgpu-registered-target %{ opt -mtriple=r600 -passes=pre-isel-intrinsic-lowering -S < %s | FileCheck %s --check-prefix=UNSUPPORTED %}
+; RUN: %if amdgpu-registered-target %{ opt -mtriple=amdgcn-amd-amdhsa -passes=pre-isel-intrinsic-lowering -S < %s | FileCheck %s --check-prefix=UNSUPPORTED %}
+
+define i1 @query() {
+; UNSUPPORTED-LABEL: define i1 @query() {
+; UNSUPPORTED-NEXT: ret i1 false
+ %enabled = call i1 @llvm.is.debugging.enabled()
+ ret i1 %enabled
+}
+
+declare i1 @llvm.is.debugging.enabled()
diff --git a/llvm/test/Transforms/SimplifyCFG/is-debugging-enabled.ll b/llvm/test/Transforms/SimplifyCFG/is-debugging-enabled.ll
new file mode 100644
index 0000000000000..299a97cdbb487
--- /dev/null
+++ b/llvm/test/Transforms/SimplifyCFG/is-debugging-enabled.ll
@@ -0,0 +1,58 @@
+; RUN: opt -passes='default<O1>' -S < %s | FileCheck %s
+
+; The three query sites must not be merged by CFG simplification.
+define void @preserve_query_sites(i32 %selector) {
+; CHECK-LABEL: define void @preserve_query_sites(
+; CHECK: first:
+; CHECK-NEXT: call i1 @llvm.is.debugging.enabled()
+; CHECK: second:
+; CHECK-NEXT: call i1 @llvm.is.debugging.enabled()
+; CHECK: join:
+; CHECK-NEXT: call i1 @llvm.is.debugging.enabled()
+;
+entry:
+ switch i32 %selector, label %join [
+ i32 5, label %first
+ i32 7, label %second
+ ]
+
+first:
+ %first.query = call i1 @llvm.is.debugging.enabled()
+ br label %join
+
+second:
+ %second.query = call i1 @llvm.is.debugging.enabled()
+ br label %join
+
+join:
+ %join.query = call i1 @llvm.is.debugging.enabled()
+ ret void
+}
+
+; The equivalent unannotated calls establish that the pipeline exercises the
+; merge opportunity used above.
+define void @mergeable_control(i32 %selector) {
+; CHECK-LABEL: define void @mergeable_control(
+; CHECK-COUNT-1: call i1 @ordinary.query()
+;
+entry:
+ switch i32 %selector, label %join [
+ i32 5, label %first
+ i32 7, label %second
+ ]
+
+first:
+ %first.query = call i1 @ordinary.query()
+ br label %join
+
+second:
+ %second.query = call i1 @ordinary.query()
+ br label %join
+
+join:
+ %join.query = call i1 @ordinary.query()
+ ret void
+}
+
+declare i1 @llvm.is.debugging.enabled()
+declare i1 @ordinary.query()
diff --git a/llvm/test/Verifier/is-debugging-enabled.ll b/llvm/test/Verifier/is-debugging-enabled.ll
new file mode 100644
index 0000000000000..640318f9f082e
--- /dev/null
+++ b/llvm/test/Verifier/is-debugging-enabled.ll
@@ -0,0 +1,15 @@
+; RUN: split-file %s %t
+; RUN: not opt -passes=verify -disable-output %t/wrong-return.ll 2>&1 | FileCheck %s --check-prefix=RETURN
+; RUN: not opt -passes=verify -disable-output %t/wrong-arity.ll 2>&1 | FileCheck %s --check-prefix=ARITY
+
+;--- wrong-return.ll
+
+; RETURN: intrinsic return type expected i1, but got i32
+; RETURN-NEXT: declare i32 @llvm.is.debugging.enabled()
+declare i32 @llvm.is.debugging.enabled()
+
+;--- wrong-arity.ll
+
+; ARITY: intrinsic has incorrect number of args. Expected 0, but got 1
+; ARITY-NEXT: declare i1 @llvm.is.debugging.enabled(i1)
+declare i1 @llvm.is.debugging.enabled(i1)
>From a34d49e154fcc324b0fef5cb01d01f7a06848f72 Mon Sep 17 00:00:00 2001
From: =?UTF-8?q?Alexander=20H=C3=BCck?= <alexander.huck at amd.com>
Date: Mon, 24 Aug 2026 09:52:55 +0000
Subject: [PATCH 4/5] [AMDGPU][GFX12/GFX13] Add cdbg conditional branch
instructions
---
llvm/lib/Target/AMDGPU/AMDGPU.td | 12 +++++++++++-
llvm/lib/Target/AMDGPU/SOPInstructions.td | 20 +++++++++++--------
llvm/test/MC/AMDGPU/gfx12_asm_sopp.s | 24 +++++++++++++++++++++++
llvm/test/MC/AMDGPU/gfx12_unsupported.s | 12 ------------
llvm/test/MC/AMDGPU/gfx13_asm_sopp.s | 24 +++++++++++++++++++++++
5 files changed, 71 insertions(+), 21 deletions(-)
diff --git a/llvm/lib/Target/AMDGPU/AMDGPU.td b/llvm/lib/Target/AMDGPU/AMDGPU.td
index ed2ee1ff70f4f..17b22beebcbb9 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPU.td
+++ b/llvm/lib/Target/AMDGPU/AMDGPU.td
@@ -149,6 +149,12 @@ defm TrapHandler: AMDGPUSubtargetFeature<"trap-handler",
/*GenPredicate=*/0, /*GenAssemblerPredicate=*/1, [], InlineIgnore
>;
+defm CDBGSysOrUserBranch : AMDGPUSubtargetFeature<
+ "cdbg-sys-or-user-branch",
+ "Support querying debugging state with s_cbranch_cdbgsys_or_user",
+ /*GenPredicate=*/1, /*GenAssemblerPredicate=*/0
+>;
+
defm UnalignedScratchAccess : AMDGPUSubtargetFeature<"unaligned-scratch-access",
"Support unaligned scratch loads and stores",
/*GenPredicate=*/1, /*GenAssemblerPredicate=*/1, [], InlineIgnore
@@ -2232,7 +2238,8 @@ def FeatureISAVersion11_0_3 : FeatureSet<
def FeatureISAVersion11_5_Common : FeatureSet<
!listconcat(FeatureISAVersion11_Common.Features,
- [FeatureSALUFloatInsts,
+ [FeatureCDBGSysOrUserBranch,
+ FeatureSALUFloatInsts,
FeatureDPPSrc1SGPR,
FeatureRequiredExportPriority,
FeatureDot5Insts,
@@ -2275,6 +2282,7 @@ def FeatureISAVersion11_7_Generic: FeatureSet<
def FeatureISAVersion12 : FeatureSet<
[FeatureGFX12,
FeatureSupportsWave64, FeatureSupportsWGP,
+ FeatureCDBGSysOrUserBranch,
FeatureBackOffBarrier,
FeatureAddressableLocalMemorySize65536,
FeatureHalfAddressablePhysicalLocalMemory,
@@ -2343,6 +2351,7 @@ def FeatureISAVersion12 : FeatureSet<
def FeatureISAVersion12_50_Common : FeatureSet<
[FeatureGFX12,
FeatureGFX1250Insts,
+ FeatureCDBGSysOrUserBranch,
FeatureBackOffBarrier,
FeatureRequiresAlignedVGPRs,
FeatureCuMode,
@@ -2536,6 +2545,7 @@ def FeatureISAVersion12_5_Generic: FeatureSet<
def FeatureISAVersion13 : FeatureSet<
[FeatureGFX13,
FeatureGFX1250Insts,
+ FeatureCDBGSysOrUserBranch,
FeatureAddressableLocalMemorySize196608,
Feature64BitLiterals,
FeatureLDSBankCount32,
diff --git a/llvm/lib/Target/AMDGPU/SOPInstructions.td b/llvm/lib/Target/AMDGPU/SOPInstructions.td
index 2c5a2b1f7d0d7..0d314b46901c4 100644
--- a/llvm/lib/Target/AMDGPU/SOPInstructions.td
+++ b/llvm/lib/Target/AMDGPU/SOPInstructions.td
@@ -2996,10 +2996,6 @@ multiclass SOPP_Real_With_Relaxation_gfx11<bits<7> op> {
}
defm S_INST_PREFETCH : SOPP_Real_32_gfx11<0x004, "s_set_inst_prefetch_distance">;
-defm S_CBRANCH_CDBGSYS : SOPP_Real_With_Relaxation_gfx11<0x027>;
-defm S_CBRANCH_CDBGUSER : SOPP_Real_With_Relaxation_gfx11<0x028>;
-defm S_CBRANCH_CDBGSYS_OR_USER : SOPP_Real_With_Relaxation_gfx11<0x029>;
-defm S_CBRANCH_CDBGSYS_AND_USER : SOPP_Real_With_Relaxation_gfx11<0x02a>;
defm S_ENDPGM_ORDERED_PS_DONE : SOPP_Real_32_gfx11<0x032>;
defm S_BARRIER : SOPP_Real_32_gfx11<0x03d>;
@@ -3191,10 +3187,14 @@ multiclass SOPP_Real_With_Relaxation_gfx6_gfx7_gfx8_gfx9_gfx10<bits<7> op> {
}
let isBranch = 1 in {
-defm S_CBRANCH_CDBGSYS : SOPP_Real_With_Relaxation_gfx6_gfx7_gfx8_gfx9_gfx10<0x017>;
-defm S_CBRANCH_CDBGUSER : SOPP_Real_With_Relaxation_gfx6_gfx7_gfx8_gfx9_gfx10<0x018>;
-defm S_CBRANCH_CDBGSYS_OR_USER : SOPP_Real_With_Relaxation_gfx6_gfx7_gfx8_gfx9_gfx10<0x019>;
-defm S_CBRANCH_CDBGSYS_AND_USER : SOPP_Real_With_Relaxation_gfx6_gfx7_gfx8_gfx9_gfx10<0x01A>;
+defm S_CBRANCH_CDBGSYS :
+ SOPP_Real_With_Relaxation_gfx6_gfx7_gfx8_gfx9_gfx10_gfx13<0x017>;
+defm S_CBRANCH_CDBGUSER :
+ SOPP_Real_With_Relaxation_gfx6_gfx7_gfx8_gfx9_gfx10_gfx13<0x018>;
+defm S_CBRANCH_CDBGSYS_OR_USER :
+ SOPP_Real_With_Relaxation_gfx6_gfx7_gfx8_gfx9_gfx10_gfx13<0x019>;
+defm S_CBRANCH_CDBGSYS_AND_USER :
+ SOPP_Real_With_Relaxation_gfx6_gfx7_gfx8_gfx9_gfx10_gfx13<0x01A>;
}
//===----------------------------------------------------------------------===//
@@ -3245,6 +3245,10 @@ defm S_CBRANCH_VCCZ : SOPP_Real_With_Relaxation_gfx11_gfx12<0x023>
defm S_CBRANCH_VCCNZ : SOPP_Real_With_Relaxation_gfx11_gfx12<0x024>;
defm S_CBRANCH_EXECZ : SOPP_Real_With_Relaxation_gfx11_gfx12<0x025>;
defm S_CBRANCH_EXECNZ : SOPP_Real_With_Relaxation_gfx11_gfx12<0x026>;
+defm S_CBRANCH_CDBGSYS : SOPP_Real_With_Relaxation_gfx11_gfx12<0x027>;
+defm S_CBRANCH_CDBGUSER : SOPP_Real_With_Relaxation_gfx11_gfx12<0x028>;
+defm S_CBRANCH_CDBGSYS_OR_USER : SOPP_Real_With_Relaxation_gfx11_gfx12<0x029>;
+defm S_CBRANCH_CDBGSYS_AND_USER : SOPP_Real_With_Relaxation_gfx11_gfx12<0x02a>;
defm S_ENDPGM : SOPP_Real_32_gfx11_gfx12<0x030>;
defm S_ENDPGM_SAVED : SOPP_Real_32_gfx11_gfx12<0x031>;
defm S_WAKEUP : SOPP_Real_32_gfx11_gfx12<0x034>;
diff --git a/llvm/test/MC/AMDGPU/gfx12_asm_sopp.s b/llvm/test/MC/AMDGPU/gfx12_asm_sopp.s
index 2bc0e8e83189f..59638beb8d4fe 100644
--- a/llvm/test/MC/AMDGPU/gfx12_asm_sopp.s
+++ b/llvm/test/MC/AMDGPU/gfx12_asm_sopp.s
@@ -35,6 +35,30 @@ s_cbranch_execnz 0x0
s_cbranch_execnz 0x1234
// GFX12: s_cbranch_execnz 4660 ; encoding: [0x34,0x12,0xa6,0xbf]
+s_cbranch_cdbgsys 0x0
+// GFX12: s_cbranch_cdbgsys 0 ; encoding: [0x00,0x00,0xa7,0xbf]
+
+s_cbranch_cdbgsys 0x1234
+// GFX12: s_cbranch_cdbgsys 4660 ; encoding: [0x34,0x12,0xa7,0xbf]
+
+s_cbranch_cdbguser 0x0
+// GFX12: s_cbranch_cdbguser 0 ; encoding: [0x00,0x00,0xa8,0xbf]
+
+s_cbranch_cdbguser 0x1234
+// GFX12: s_cbranch_cdbguser 4660 ; encoding: [0x34,0x12,0xa8,0xbf]
+
+s_cbranch_cdbgsys_or_user 0x0
+// GFX12: s_cbranch_cdbgsys_or_user 0 ; encoding: [0x00,0x00,0xa9,0xbf]
+
+s_cbranch_cdbgsys_or_user 0x1234
+// GFX12: s_cbranch_cdbgsys_or_user 4660 ; encoding: [0x34,0x12,0xa9,0xbf]
+
+s_cbranch_cdbgsys_and_user 0x0
+// GFX12: s_cbranch_cdbgsys_and_user 0 ; encoding: [0x00,0x00,0xaa,0xbf]
+
+s_cbranch_cdbgsys_and_user 0x1234
+// GFX12: s_cbranch_cdbgsys_and_user 4660 ; encoding: [0x34,0x12,0xaa,0xbf]
+
s_cbranch_execz 0x0
// GFX12: s_cbranch_execz 0 ; encoding: [0x00,0x00,0xa5,0xbf]
diff --git a/llvm/test/MC/AMDGPU/gfx12_unsupported.s b/llvm/test/MC/AMDGPU/gfx12_unsupported.s
index d6b56a1feb6fc..b2f418bb97e48 100644
--- a/llvm/test/MC/AMDGPU/gfx12_unsupported.s
+++ b/llvm/test/MC/AMDGPU/gfx12_unsupported.s
@@ -80,18 +80,6 @@ global_atomic_cmpswap_f32 v[5:6], off, s[96:99], s3
s_barrier
// CHECK: :[[@LINE-1]]:1: error: instruction not supported on this GPU (gfx1200): s_barrier
-s_cbranch_cdbgsys 0
-// CHECK: :[[@LINE-1]]:1: error: instruction not supported on this GPU (gfx1200): s_cbranch_cdbgsys
-
-s_cbranch_cdbgsys_and_user 0
-// CHECK: :[[@LINE-1]]:1: error: instruction not supported on this GPU (gfx1200): s_cbranch_cdbgsys_and_user
-
-s_cbranch_cdbgsys_or_user 0
-// CHECK: :[[@LINE-1]]:1: error: instruction not supported on this GPU (gfx1200): s_cbranch_cdbgsys_or_user
-
-s_cbranch_cdbguser 0
-// CHECK: :[[@LINE-1]]:1: error: instruction not supported on this GPU (gfx1200): s_cbranch_cdbguser
-
s_cmpk_eq_i32 s0, 0
// CHECK: :[[@LINE-1]]:1: error: instruction not supported on this GPU (gfx1200): s_cmpk_eq_i32
diff --git a/llvm/test/MC/AMDGPU/gfx13_asm_sopp.s b/llvm/test/MC/AMDGPU/gfx13_asm_sopp.s
index 8d6bf2e1bedfc..163fb1780d1f2 100644
--- a/llvm/test/MC/AMDGPU/gfx13_asm_sopp.s
+++ b/llvm/test/MC/AMDGPU/gfx13_asm_sopp.s
@@ -128,6 +128,30 @@ s_decperflevel 0x1234
s_ttracedata
// GFX13: s_ttracedata ; encoding: [0x00,0x00,0x96,0xbf]
+s_cbranch_cdbgsys 0
+// GFX13: s_cbranch_cdbgsys 0 ; encoding: [0x00,0x00,0x97,0xbf]
+
+s_cbranch_cdbgsys 0x1234
+// GFX13: s_cbranch_cdbgsys 4660 ; encoding: [0x34,0x12,0x97,0xbf]
+
+s_cbranch_cdbguser 0
+// GFX13: s_cbranch_cdbguser 0 ; encoding: [0x00,0x00,0x98,0xbf]
+
+s_cbranch_cdbguser 0x1234
+// GFX13: s_cbranch_cdbguser 4660 ; encoding: [0x34,0x12,0x98,0xbf]
+
+s_cbranch_cdbgsys_or_user 0
+// GFX13: s_cbranch_cdbgsys_or_user 0 ; encoding: [0x00,0x00,0x99,0xbf]
+
+s_cbranch_cdbgsys_or_user 0x1234
+// GFX13: s_cbranch_cdbgsys_or_user 4660 ; encoding: [0x34,0x12,0x99,0xbf]
+
+s_cbranch_cdbgsys_and_user 0
+// GFX13: s_cbranch_cdbgsys_and_user 0 ; encoding: [0x00,0x00,0x9a,0xbf]
+
+s_cbranch_cdbgsys_and_user 0x1234
+// GFX13: s_cbranch_cdbgsys_and_user 4660 ; encoding: [0x34,0x12,0x9a,0xbf]
+
s_endpgm_saved
// GFX13: s_endpgm_saved ; encoding: [0x00,0x00,0x9b,0xbf]
>From 0b7a46c97518350ba38d60b2524fc5f82d3d5c99 Mon Sep 17 00:00:00 2001
From: =?UTF-8?q?Alexander=20H=C3=BCck?= <alexander.huck at amd.com>
Date: Mon, 24 Aug 2026 10:11:05 +0000
Subject: [PATCH 5/5] [AMDGPU] Lower llvm.is.debugging.enabled
Opt AMDGPU into lowering llvm.is.debugging.enabled. Unsupported
subtargets replace the intrinsic with false.
On supported subtargets, SelectionDAG and GlobalISel either materialize
the query by reading the generation-specific COND_DBG_USER and
COND_DBG_SYS bits with s_getreg, or fuse safe single-branch uses into
s_cbranch_cdbgsys_or_user, including negated conditions.
Mark fused and materialized observations NoMerge to keep calls distinct.
---
llvm/docs/AMDGPUUsage.rst | 46 ++
.../lib/Target/AMDGPU/AMDGPULegalizerInfo.cpp | 38 ++
.../Target/AMDGPU/AMDGPULowerIntrinsics.cpp | 16 +
llvm/lib/Target/AMDGPU/AMDGPUTargetMachine.h | 2 +
llvm/lib/Target/AMDGPU/SIISelLowering.cpp | 51 ++
.../Target/AMDGPU/Utils/AMDGPUBaseInfo.cpp | 10 +
llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.h | 4 +
.../AMDGPU/always_uniform.ll | 28 ++
...-debugging-enabled-divergent-exec-guard.ll | 236 +++++++++
.../AMDGPU/is-debugging-enabled-shapes.ll | 449 ++++++++++++++++++
.../is-debugging-enabled-unsupported.ll | 32 ++
.../CodeGen/AMDGPU/is-debugging-enabled.ll | 257 ++++++++++
.../is-debugging-enabled.ll | 6 +-
13 files changed, 1174 insertions(+), 1 deletion(-)
create mode 100644 llvm/test/CodeGen/AMDGPU/is-debugging-enabled-divergent-exec-guard.ll
create mode 100644 llvm/test/CodeGen/AMDGPU/is-debugging-enabled-shapes.ll
create mode 100644 llvm/test/CodeGen/AMDGPU/is-debugging-enabled-unsupported.ll
create mode 100644 llvm/test/CodeGen/AMDGPU/is-debugging-enabled.ll
diff --git a/llvm/docs/AMDGPUUsage.rst b/llvm/docs/AMDGPUUsage.rst
index 40a2820b9e04f..e82af4d16c9df 100644
--- a/llvm/docs/AMDGPUUsage.rst
+++ b/llvm/docs/AMDGPUUsage.rst
@@ -1709,6 +1709,52 @@ The AMDGPU backend implements the following LLVM IR intrinsics.
The format is a 64-bit concatenation of the MODE and TRAPSTS registers.
:ref:`llvm.set.fpenv<int_set_fpenv>` Sets the floating point environment to the specified state.
+
+ :ref:`llvm.is.debugging.enabled <llvm.is.debugging.enabled>`
+ Supported on GFX11.5, GFX12, and GFX13 targets. Other
+ subtargets lower the result to ``false``.
+
+ The target-defined execution context is the current wave.
+ The result is uniform across the active lanes of that
+ wave, including when the intrinsic is executed in
+ divergent control flow. Each call remains a distinct
+ observation of the wave's debugging-enabled state.
+
+ A wave executes in debugging mode when either the
+ ``COND_DBG_SYS`` or ``COND_DBG_USER`` bit is set.
+ ``COND_DBG_SYS`` reflects a system-wide debugger
+ attach, while ``COND_DBG_USER`` reflects per-dispatch
+ launch control, configured through
+ :ref:`CDBG_USER <amdgpu-amdhsa-compute_pgm_rsrc1-gfx6-gfx13-table>`
+ in the kernel descriptor. The driver sets these
+ bits, and provides an interface allowing the user
+ mode runtime or an external debugger to control
+ the setting.
+
+ The intrinsic lowers in one of two forms.
+ When the query feeds a single conditional branch in the
+ same basic block, with no intervening observable
+ operations, it may fuse into a single
+ ``s_cbranch_cdbgsys_or_user``. Branch-hint and
+ negated-condition forms are supported; other uses of
+ the query value prevent fusion.
+
+ Any other use materializes an ``i1`` by reading the
+ adjacent ``COND_DBG_USER`` and ``COND_DBG_SYS`` bits
+ and testing them against zero. The register holding
+ them differs by generation:
+
+ .. code-block:: none
+
+ ; GFX11.5
+ s_getreg_b32 s0, hwreg(HW_REG_STATUS, 20, 2)
+ ; GFX12
+ s_getreg_b32 s0, hwreg(HW_REG_WAVE_STATE_PRIV, 16, 2)
+ ; GFX13
+ s_getreg_b32 s0, hwreg(HW_REG_WAVE_STATUS, 20, 2)
+ ; all generations
+ s_cmp_lg_u32 s0, 0
+
llvm.amdgcn.readfirstlane Provides direct access to v_readfirstlane_b32. Returns the value in
the lowest active lane of the input operand. Currently implemented
for i16, i32, float, half, bfloat, <2 x i16>, <2 x half>, <2 x bfloat>,
diff --git a/llvm/lib/Target/AMDGPU/AMDGPULegalizerInfo.cpp b/llvm/lib/Target/AMDGPU/AMDGPULegalizerInfo.cpp
index 0f0812a5cac36..c42060e69a274 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPULegalizerInfo.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPULegalizerInfo.cpp
@@ -8215,6 +8215,44 @@ bool AMDGPULegalizerInfo::legalizeIntrinsic(LegalizerHelper &Helper,
// Replace the use G_BRCOND with the exec manipulate and branch pseudos.
auto IntrID = cast<GIntrinsic>(MI).getIntrinsicID();
switch (IntrID) {
+ case Intrinsic::is_debugging_enabled: {
+ auto Match = matchCFIntrinsicBranchUse(MI, MRI);
+ bool CannotFuse =
+ !Match || any_of(make_range(std::next(MI.getIterator()),
+ Match->CondBr->getIterator()),
+ [](const MachineInstr &Between) {
+ return !Between.isMetaInstruction() &&
+ (Between.mayLoadOrStore() ||
+ Between.hasUnmodeledSideEffects());
+ });
+ if (CannotFuse) {
+ auto Bits =
+ B.buildIntrinsic(Intrinsic::amdgcn_s_getreg, {LLT::scalar(32)})
+ .addImm(AMDGPU::Hwreg::getDebuggingEnabledHwregImm(ST));
+ Bits->setFlag(MachineInstr::NoMerge);
+ B.buildICmp(CmpInst::ICMP_NE, MI.getOperand(0).getReg(), Bits.getReg(0),
+ B.buildConstant(LLT::scalar(32), 0));
+ MI.eraseFromParent();
+ return true;
+ }
+
+ B.setInsertPt(*Match->CondBr->getParent(), Match->CondBr->getIterator());
+ B.setDebugLoc(Match->CondBr->getDebugLoc());
+ MachineInstrBuilder CDBGBranch =
+ B.buildInstr(AMDGPU::S_CBRANCH_CDBGSYS_OR_USER)
+ .addMBB(Match->ConditionTrueTarget);
+ CDBGBranch->setFlag(MachineInstr::NoMerge);
+
+ if (Match->isNegated())
+ Match->redirectFallthroughEdge(B, *Match->ConditionFalseTarget);
+
+ Register Cond = MI.getOperand(0).getReg();
+ MRI.markUsesInDebugValueAsUndef(Cond);
+ Match->eraseDeadNegation(MRI);
+ MI.eraseFromParent();
+ Match->CondBr->eraseFromParent();
+ return true;
+ }
case Intrinsic::amdgcn_icmp: {
// amdgcn.icmp(i1 src0, i1 0, NE) -> ballot(src0)
// This is the only valid form of amdgcn.icmp with i1 inputs.
diff --git a/llvm/lib/Target/AMDGPU/AMDGPULowerIntrinsics.cpp b/llvm/lib/Target/AMDGPU/AMDGPULowerIntrinsics.cpp
index 123d9930fc49e..274cba75e5254 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPULowerIntrinsics.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPULowerIntrinsics.cpp
@@ -37,6 +37,7 @@ class AMDGPULowerIntrinsicsImpl {
bool run();
private:
+ bool visitIsDebuggingEnabled(IntrinsicInst &I);
bool visitBarrier(IntrinsicInst &I);
bool visitPtrSBufferLoad(IntrinsicInst &I);
};
@@ -63,6 +64,16 @@ template <class T> static void forEachCall(Function &Intrin, T Callback) {
} // anonymous namespace
+bool AMDGPULowerIntrinsicsImpl::visitIsDebuggingEnabled(IntrinsicInst &I) {
+ const GCNSubtarget &ST = TM.getSubtarget<GCNSubtarget>(*I.getFunction());
+ if (ST.hasCDBGSysOrUserBranch())
+ return false;
+
+ I.replaceAllUsesWith(ConstantInt::getFalse(I.getContext()));
+ I.eraseFromParent();
+ return true;
+}
+
bool AMDGPULowerIntrinsicsImpl::run() {
bool Changed = false;
@@ -70,6 +81,11 @@ bool AMDGPULowerIntrinsicsImpl::run() {
switch (F.getIntrinsicID()) {
default:
continue;
+ case Intrinsic::is_debugging_enabled:
+ forEachCall(F, [&](IntrinsicInst *II) {
+ Changed |= visitIsDebuggingEnabled(*II);
+ });
+ break;
case Intrinsic::amdgcn_s_barrier:
case Intrinsic::amdgcn_s_barrier_signal:
case Intrinsic::amdgcn_s_barrier_signal_isfirst:
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUTargetMachine.h b/llvm/lib/Target/AMDGPU/AMDGPUTargetMachine.h
index 189b6399dbf7b..86255b475d15d 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUTargetMachine.h
+++ b/llvm/lib/Target/AMDGPU/AMDGPUTargetMachine.h
@@ -99,6 +99,8 @@ class GCNTargetMachine final : public AMDGPUTargetMachine {
bool useIPRA() const override { return true; }
+ bool canLowerIsDebuggingEnabled() const override { return true; }
+
Error buildCodeGenPipeline(ModulePassManager &MPM, ModuleAnalysisManager &MAM,
raw_pwrite_stream &Out, raw_pwrite_stream *DwoOut,
CodeGenFileType FileType,
diff --git a/llvm/lib/Target/AMDGPU/SIISelLowering.cpp b/llvm/lib/Target/AMDGPU/SIISelLowering.cpp
index 86cfd885c4f8f..cca51d7eb3629 100644
--- a/llvm/lib/Target/AMDGPU/SIISelLowering.cpp
+++ b/llvm/lib/Target/AMDGPU/SIISelLowering.cpp
@@ -8713,6 +8713,39 @@ bool SITargetLowering::shouldUseLDSConstAddress(const GlobalValue *GV) const {
return OS == Triple::AMDHSA || OS == Triple::AMDPAL;
}
+/// Fuses a debugging-state query from \p Match into a single
+/// S_CBRANCH_CDBGSYS_OR_USER, which branches when debugging is enabled.
+/// Returns a null SDValue if the condition does not test such a query, or if
+/// something observable happens between the query and the branch.
+static SDValue lowerDebuggingEnabledBRCOND(const BRCONDMatch &Match,
+ SelectionDAG &DAG) {
+ if ((Match.ConditionWrapper && !Match.ConditionWrapper.hasOneUse()) ||
+ !Match.Condition.hasOneUse())
+ return SDValue();
+
+ SDValue Cond = Match.Condition;
+ if (Cond.getOpcode() != ISD::INTRINSIC_W_CHAIN ||
+ Cond.getConstantOperandVal(1) != Intrinsic::is_debugging_enabled)
+ return SDValue();
+
+ if (!Match.CondBr.getOperand(0).reachesChainWithoutSideEffects(
+ Cond.getValue(1)))
+ return SDValue();
+
+ assert(Cond->getNumValues() == 2 && "expected query value and chain");
+
+ Match.redirectFallthroughEdge(DAG, Match.ConditionFalseTarget);
+
+ DAG.ReplaceAllUsesOfValueWith(Cond.getValue(1), Cond.getOperand(0));
+
+ SDLoc DL(Match.CondBr);
+ MachineSDNode *CDBGBranch =
+ DAG.getMachineNode(AMDGPU::S_CBRANCH_CDBGSYS_OR_USER, DL, MVT::Other,
+ Match.ConditionTrueTarget, Match.CondBr.getOperand(0));
+ DAG.addNoMergeSiteInfo(CDBGBranch, true);
+ return SDValue(CDBGBranch, 0);
+}
+
/// This transforms the control flow intrinsics to get the branch destination as
/// last parameter, also switches branch target with BR if the need arise
SDValue SITargetLowering::LowerBRCOND(SDValue BRCOND, SelectionDAG &DAG) const {
@@ -8720,6 +8753,9 @@ SDValue SITargetLowering::LowerBRCOND(SDValue BRCOND, SelectionDAG &DAG) const {
if (!Match)
return BRCOND;
+ if (SDValue V = lowerDebuggingEnabledBRCOND(*Match, DAG))
+ return V;
+
SDLoc DL(Match->CondBr);
SDNode *Intr = Match->Condition.getNode();
@@ -12455,6 +12491,21 @@ SDValue SITargetLowering::LowerINTRINSIC_W_CHAIN(SDValue Op,
return DAG.getAtomicLoad(ISD::NON_EXTLOAD, DL, MII->getMemoryVT(), VT,
Chain, Ptr, MII->getMemOperand());
}
+ case Intrinsic::is_debugging_enabled: {
+ SDValue GetReg = DAG.getNode(
+ ISD::INTRINSIC_W_CHAIN, DL, DAG.getVTList(MVT::i32, MVT::Other),
+ Op.getOperand(0),
+ DAG.getTargetConstant(Intrinsic::amdgcn_s_getreg, DL, MVT::i32),
+ DAG.getTargetConstant(
+ AMDGPU::Hwreg::getDebuggingEnabledHwregImm(*Subtarget), DL,
+ MVT::i32));
+ DAG.addNoMergeSiteInfo(GetReg.getNode(), true);
+
+ SDValue Enabled =
+ DAG.getSetCC(DL, Op.getValueType(), GetReg,
+ DAG.getConstant(0, DL, MVT::i32), ISD::SETNE);
+ return DAG.getMergeValues({Enabled, GetReg.getValue(1)}, DL);
+ }
case Intrinsic::amdgcn_av_load_b128: {
MemIntrinsicSDNode *MII = cast<MemIntrinsicSDNode>(Op);
SDValue Chain = Op->getOperand(0);
diff --git a/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.cpp b/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.cpp
index a0648da67c06d..871c20f9aa177 100644
--- a/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.cpp
+++ b/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.cpp
@@ -2612,6 +2612,16 @@ bool isGFX13(const MCSubtargetInfo &STI) {
bool isGFX13Plus(const MCSubtargetInfo &STI) { return isGFX13(STI); }
+namespace Hwreg {
+
+unsigned getDebuggingEnabledHwregImm(const MCSubtargetInfo &STI) {
+ if (isGFX12(STI))
+ return HwregEncoding::encode(ID_STATE_PRIV, 16, 2);
+ return HwregEncoding::encode(ID_STATUS, 20, 2);
+}
+
+} // namespace Hwreg
+
bool supportsWGP(const MCSubtargetInfo &STI) {
if (isGFX1250(STI))
return false;
diff --git a/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.h b/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.h
index d429584b00999..26b679be43e07 100644
--- a/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.h
+++ b/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.h
@@ -1175,6 +1175,10 @@ struct HwregSize : EncodingField<15, 11, 32> {
using HwregEncoding = EncodingFields<HwregId, HwregOffset, HwregSize>;
+/// \returns the s_getreg immediate that reads the adjacent COND_DBG_USER and
+/// COND_DBG_SYS bits for \p STI.
+unsigned getDebuggingEnabledHwregImm(const MCSubtargetInfo &STI);
+
} // namespace Hwreg
namespace DepCtr {
diff --git a/llvm/test/Analysis/UniformityAnalysis/AMDGPU/always_uniform.ll b/llvm/test/Analysis/UniformityAnalysis/AMDGPU/always_uniform.ll
index 6e75d4ead8cdd..8f64e6967d91e 100644
--- a/llvm/test/Analysis/UniformityAnalysis/AMDGPU/always_uniform.ll
+++ b/llvm/test/Analysis/UniformityAnalysis/AMDGPU/always_uniform.ll
@@ -256,8 +256,36 @@ define void @s_memrealtime(ptr addrspace(1) inreg %out) {
ret void
}
+; CHECK-LABEL: for function 'is_debugging_enabled':
+; CHECK: ALL VALUES UNIFORM
+define i1 @is_debugging_enabled() {
+ %enabled = call i1 @llvm.is.debugging.enabled()
+ ret i1 %enabled
+}
+
+; CHECK-LABEL: for function 'is_debugging_enabled_in_divergent_control_flow':
+; CHECK: DIVERGENT: %workitem = call i32 @llvm.amdgcn.workitem.id.x()
+; CHECK: DIVERGENT: br i1 %lane.zero, label %query, label %exit
+; CHECK-NOT: DIVERGENT
+define amdgpu_kernel void @is_debugging_enabled_in_divergent_control_flow(
+ ptr addrspace(1) inreg %out) {
+entry:
+ %workitem = call i32 @llvm.amdgcn.workitem.id.x()
+ %lane.zero = icmp eq i32 %workitem, 0
+ br i1 %lane.zero, label %query, label %exit
+
+query:
+ %enabled = call i1 @llvm.is.debugging.enabled()
+ store i1 %enabled, ptr addrspace(1) %out
+ br label %exit
+
+exit:
+ ret void
+}
+
declare i32 @llvm.amdgcn.workitem.id.x() #0
+declare i1 @llvm.is.debugging.enabled()
declare i32 @llvm.amdgcn.readfirstlane(i32) #0
declare i64 @llvm.amdgcn.icmp.i32(i32, i32, i32) #1
declare i64 @llvm.amdgcn.fcmp.i32(float, float, i32) #1
diff --git a/llvm/test/CodeGen/AMDGPU/is-debugging-enabled-divergent-exec-guard.ll b/llvm/test/CodeGen/AMDGPU/is-debugging-enabled-divergent-exec-guard.ll
new file mode 100644
index 0000000000000..6b02e4d93414c
--- /dev/null
+++ b/llvm/test/CodeGen/AMDGPU/is-debugging-enabled-divergent-exec-guard.ll
@@ -0,0 +1,236 @@
+; s_cbranch_cdbgsys_or_user and s_trap are scalar and ignore EXEC, so divergent
+; control flow around them needs an EXEC guard.
+;
+; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1150 -O2 < %s | FileCheck %s --check-prefixes=GCN,DBGSTATUS
+; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1150 -O2 -global-isel -global-isel-abort=1 < %s | FileCheck %s --check-prefixes=GCN,DBGSTATUS
+; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1200 -O2 < %s | FileCheck %s --check-prefixes=GCN,DBGPRIV
+; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1200 -O2 -global-isel -global-isel-abort=1 < %s | FileCheck %s --check-prefixes=GCN,DBGPRIV
+; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1310 -O2 < %s | FileCheck %s --check-prefixes=GCN,DBGSTATUS
+; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1310 -O2 -global-isel -global-isel-abort=1 < %s | FileCheck %s --check-prefixes=GCN,DBGSTATUS
+; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1310 -O0 < %s | FileCheck %s --check-prefix=FUSED
+; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1310 -O0 -global-isel -global-isel-abort=1 < %s | FileCheck %s --check-prefix=FUSED
+; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1310 -O1 < %s | FileCheck %s --check-prefix=FUSED
+; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1310 -O1 -global-isel -global-isel-abort=1 < %s | FileCheck %s --check-prefix=FUSED
+
+declare noundef i1 @llvm.is.debugging.enabled()
+declare i1 @llvm.expect.i1(i1, i1 immarg)
+declare void @llvm.debugtrap()
+declare i32 @llvm.amdgcn.workitem.id.x()
+
+; Nested divergent regions: each contributes a guard above the fused branch.
+define amdgpu_kernel void @nested_divergent_debugtrap(ptr addrspace(1) %out) {
+entry:
+ %tid = call i32 @llvm.amdgcn.workitem.id.x()
+ %outer = icmp ult i32 %tid, 8
+ br i1 %outer, label %outer.region, label %exit
+
+outer.region:
+ store volatile i32 1, ptr addrspace(1) %out
+ %inner = icmp eq i32 %tid, 3
+ br i1 %inner, label %inner.region, label %outer.after
+
+inner.region:
+ %e = call i1 @llvm.is.debugging.enabled()
+ br i1 %e, label %dbg, label %inner.after
+
+dbg:
+ call void @llvm.debugtrap()
+ br label %inner.after
+
+inner.after:
+ store volatile i32 2, ptr addrspace(1) %out
+ br label %outer.after
+
+outer.after:
+ store volatile i32 3, ptr addrspace(1) %out
+ br label %exit
+
+exit:
+ ret void
+}
+
+; GCN-LABEL: nested_divergent_debugtrap:
+; GCN: s_cbranch_execz [[OUTER_SKIP:.LBB[0-9_]+]]
+; GCN: s_cbranch_execz [[INNER_SKIP:.LBB[0-9_]+]]
+; GCN: s_cbranch_cdbgsys_or_user [[NESTED_DEBUG:.LBB[0-9_]+]]
+; GCN: [[NESTED_DEBUG]]:
+; GCN-NEXT: s_trap 3
+; GCN: [[INNER_SKIP]]:
+; GCN: [[OUTER_SKIP]]:
+
+; O0/O1 fuse the single-use loop query; O2 materializes it after CFG restructuring.
+define amdgpu_kernel void @loop_debugtrap_exec_guard(ptr addrspace(1) %out) {
+entry:
+ %tid = call i32 @llvm.amdgcn.workitem.id.x()
+ %d = icmp ult i32 %tid, 8
+ br i1 %d, label %L, label %E
+
+L:
+ %e = call i1 @llvm.is.debugging.enabled()
+ br i1 %e, label %dbg, label %M
+
+dbg:
+ call void @llvm.debugtrap()
+ br label %M
+
+M:
+ store i32 2, ptr addrspace(1) %out
+ br i1 %d, label %L, label %E
+
+E:
+ ret void
+}
+
+; GCN-LABEL: loop_debugtrap_exec_guard:
+; GCN: s_cbranch_execz [[LOOP_SKIP:.LBB[0-9_]+]]
+; GCN: [[LOOP_HDR:.LBB[0-9_]+]]:
+; DBGSTATUS: s_getreg_b32 s{{[0-9]+}}, hwreg(HW_REG_{{(WAVE_)?}}STATUS, 20, 2)
+; DBGPRIV: s_getreg_b32 s{{[0-9]+}}, hwreg(HW_REG_WAVE_STATE_PRIV, 16, 2)
+; GCN-NOT: s_cbranch_cdbgsys_or_user
+; GCN: s_trap 3
+; GCN: [[LOOP_SKIP]]:
+
+; FUSED-LABEL: loop_debugtrap_exec_guard:
+; FUSED: s_cbranch_execz
+; FUSED-NOT: s_getreg_b32
+; FUSED: s_cbranch_cdbgsys_or_user
+; FUSED-NOT: s_getreg_b32
+
+define amdgpu_kernel void @divergent_debug_break() {
+entry:
+ %id = call i32 @llvm.amdgcn.workitem.id.x()
+ %divergent = icmp eq i32 %id, 0
+ br i1 %divergent, label %query, label %normal
+
+query:
+ %raw = call i1 @llvm.is.debugging.enabled()
+ %enabled = call i1 @llvm.expect.i1(i1 %raw, i1 false)
+ br i1 %enabled, label %debug, label %normal
+
+debug:
+ call void @llvm.debugtrap()
+ br label %normal
+
+normal:
+ ret void
+}
+
+; GCN-LABEL: divergent_debug_break:
+; GCN: s_cbranch_execz [[DIVERGENT_NORMAL:.LBB[0-9_]+]]
+; GCN-COUNT-1: s_cbranch_cdbgsys_or_user [[DIVERGENT_DEBUG:.LBB[0-9_]+]]
+; GCN-NOT: s_cbranch_cdbgsys_or_user
+; GCN: [[DIVERGENT_DEBUG]]:
+; GCN-NEXT: s_trap 3
+; GCN: [[DIVERGENT_NORMAL]]:{{.*}}%normal
+
+define amdgpu_kernel void @two_query_sites(ptr addrspace(1) %out) {
+entry:
+ %id = call i32 @llvm.amdgcn.workitem.id.x()
+ %choose.left = icmp eq i32 %id, 0
+ br i1 %choose.left, label %left, label %right
+
+left:
+ %left.enabled = call i1 @llvm.is.debugging.enabled()
+ br i1 %left.enabled, label %left.debug, label %join
+
+left.debug:
+ store volatile i32 1, ptr addrspace(1) %out
+ br label %join
+
+right:
+ %right.enabled = call i1 @llvm.is.debugging.enabled()
+ br i1 %right.enabled, label %right.debug, label %join
+
+right.debug:
+ store volatile i32 2, ptr addrspace(1) %out
+ br label %join
+
+join:
+ ret void
+}
+
+; GCN-LABEL: two_query_sites:
+; GCN-COUNT-2: s_cbranch_cdbgsys_or_user
+
+; Uniform query outside a divergent trap: only the inner branch needs a guard.
+define amdgpu_kernel void @debug_enabled_then_divergent_trap(ptr addrspace(1) %out) {
+entry:
+ %e = call i1 @llvm.is.debugging.enabled()
+ br i1 %e, label %dbg.region, label %exit
+
+dbg.region:
+ %tid = call i32 @llvm.amdgcn.workitem.id.x()
+ %gt8 = icmp ugt i32 %tid, 8
+ br i1 %gt8, label %trap, label %after
+
+trap:
+ call void @llvm.debugtrap()
+ br label %after
+
+after:
+ store volatile i32 1, ptr addrspace(1) %out
+ br label %exit
+
+exit:
+ ret void
+}
+
+; GCN-LABEL: debug_enabled_then_divergent_trap:
+; GCN: s_cbranch_cdbgsys_or_user [[DBG_REGION:.LBB[0-9_]+]]
+; GCN: [[DBG_REGION]]:
+; GCN: s_cbranch_execz [[TRAP_SKIP:.LBB[0-9_]+]]
+; GCN: s_trap 3
+; GCN: [[TRAP_SKIP]]:
+
+; Test an out-of-line DebugBreak helper from uniform and divergent call sites.
+define void @debugbreak_helper() noinline {
+entry:
+ %raw = call i1 @llvm.is.debugging.enabled()
+ %enabled = call i1 @llvm.expect.i1(i1 %raw, i1 false)
+ br i1 %enabled, label %debug, label %normal
+
+debug:
+ call void @llvm.debugtrap()
+ br label %normal
+
+normal:
+ ret void
+}
+
+; GCN-LABEL: debugbreak_helper:
+; GCN-NOT: s_cbranch_execz
+; GCN: s_cbranch_cdbgsys_or_user [[HELPER_DEBUG:.LBB[0-9_]+]]
+; GCN-NOT: s_cbranch_cdbgsys_or_user
+; GCN: [[HELPER_DEBUG]]:
+; GCN-NEXT: s_trap 3
+
+define amdgpu_kernel void @uniform_helper_call() {
+entry:
+ call void @debugbreak_helper()
+ ret void
+}
+
+; GCN-LABEL: uniform_helper_call:
+; GCN-NOT: s_cbranch_execz
+; GCN: debugbreak_helper
+; GCN: s_{{swappc_b64|swap_pc_i64}}
+
+define amdgpu_kernel void @divergent_helper_call() {
+entry:
+ %id = call i32 @llvm.amdgcn.workitem.id.x()
+ %divergent = icmp eq i32 %id, 0
+ br i1 %divergent, label %call, label %normal
+
+call:
+ call void @debugbreak_helper()
+ br label %normal
+
+normal:
+ ret void
+}
+
+; GCN-LABEL: divergent_helper_call:
+; GCN: s_cbranch_execz [[CALL_SKIP:.LBB[0-9_]+]]
+; GCN: debugbreak_helper
+; GCN: s_{{swappc_b64|swap_pc_i64}}
+; GCN: [[CALL_SKIP]]:{{.*}}%normal
diff --git a/llvm/test/CodeGen/AMDGPU/is-debugging-enabled-shapes.ll b/llvm/test/CodeGen/AMDGPU/is-debugging-enabled-shapes.ll
new file mode 100644
index 0000000000000..1fd946b65f3f5
--- /dev/null
+++ b/llvm/test/CodeGen/AMDGPU/is-debugging-enabled-shapes.ll
@@ -0,0 +1,449 @@
+; RUN: split-file %s %t
+; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1310 -O2 %t/unused.ll -o - | FileCheck --check-prefixes=UNUSED %s
+; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1310 -O2 %t/store.ll -o - | FileCheck --check-prefixes=STORE %s
+; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1310 -O2 %t/return.ll -o - | FileCheck --check-prefixes=RETURN %s
+; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1310 -O2 %t/select.ll -o - | FileCheck --check-prefixes=SELECT %s
+; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1310 -O2 %t/arithmetic.ll -o - | FileCheck --check-prefixes=ARITH %s
+; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1310 -O2 %t/multiple-branches.ll -o - | FileCheck --check-prefixes=MULTIBR %s
+; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1310 -O2 %t/branch-and-store.ll -o - | FileCheck --check-prefixes=BRSTORE %s
+; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1310 -O2 %t/ordered-placement.ll -o - | FileCheck --check-prefixes=ORDERED %s
+; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1310 -O2 %t/expect-ordered-placement.ll -o - | FileCheck --check-prefixes=EXPECTORDERED %s
+; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1310 -O2 %t/expect-multiple-uses.ll -o - | FileCheck --check-prefixes=EXPECTMULTI %s
+; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1310 -O2 %t/store-between.ll -o - | FileCheck --check-prefixes=STOREBETWEEN %s
+; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1310 -O2 %t/call-between.ll -o - | FileCheck --check-prefixes=CALLBETWEEN %s
+; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1310 -O2 %t/two-observations.ll -o - | FileCheck --check-prefixes=TWOOBS %s
+; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1310 -O2 %t/negated-store-between.ll -o - | FileCheck --check-prefixes=NEGSTORE %s
+; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1310 -O2 %t/load-between.ll -o - | FileCheck --check-prefix=LOADBETWEEN %s
+; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1310 -O2 %t/volatile-load-between.ll -o - | FileCheck --check-prefix=VOLLOADBETWEEN %s
+; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1310 -O2 %t/barrier-between.ll -o - | FileCheck --check-prefix=BARRIERBETWEEN %s
+; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1310 -O2 %t/fence-between.ll -o - | FileCheck --check-prefix=FENCEBETWEEN %s
+
+; The same shapes through GlobalISel.
+; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1310 -O2 -global-isel -global-isel-abort=1 %t/unused.ll -o - | FileCheck --check-prefixes=UNUSED %s
+; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1310 -O2 -global-isel -global-isel-abort=1 %t/store.ll -o - | FileCheck --check-prefixes=STORE %s
+; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1310 -O2 -global-isel -global-isel-abort=1 %t/return.ll -o - | FileCheck --check-prefixes=RETURN %s
+; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1310 -O2 -global-isel -global-isel-abort=1 %t/select.ll -o - | FileCheck --check-prefixes=SELECT %s
+; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1310 -O2 -global-isel -global-isel-abort=1 %t/arithmetic.ll -o - | FileCheck --check-prefixes=ARITH %s
+; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1310 -O2 -global-isel -global-isel-abort=1 %t/multiple-branches.ll -o - | FileCheck --check-prefixes=MULTIBR %s
+; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1310 -O2 -global-isel -global-isel-abort=1 %t/branch-and-store.ll -o - | FileCheck --check-prefixes=BRSTORE %s
+; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1310 -O2 -global-isel -global-isel-abort=1 %t/ordered-placement.ll -o - | FileCheck --check-prefixes=ORDERED %s
+; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1310 -O2 -global-isel -global-isel-abort=1 %t/expect-ordered-placement.ll -o - | FileCheck --check-prefixes=EXPECTORDERED %s
+; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1310 -O2 -global-isel -global-isel-abort=1 %t/expect-multiple-uses.ll -o - | FileCheck --check-prefixes=EXPECTMULTI %s
+; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1310 -O2 -global-isel -global-isel-abort=1 %t/store-between.ll -o - | FileCheck --check-prefixes=STOREBETWEEN %s
+; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1310 -O2 -global-isel -global-isel-abort=1 %t/call-between.ll -o - | FileCheck --check-prefixes=CALLBETWEEN %s
+; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1310 -O2 -global-isel -global-isel-abort=1 %t/two-observations.ll -o - | FileCheck --check-prefixes=TWOOBS %s
+; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1310 -O2 -global-isel -global-isel-abort=1 %t/negated-store-between.ll -o - | FileCheck --check-prefixes=NEGSTORE %s
+; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1310 -O2 -global-isel -global-isel-abort=1 %t/load-between.ll -o - | FileCheck --check-prefix=LOADBETWEEN %s
+; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1310 -O2 -global-isel -global-isel-abort=1 %t/volatile-load-between.ll -o - | FileCheck --check-prefix=VOLLOADBETWEEN %s
+; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1310 -O2 -global-isel -global-isel-abort=1 %t/barrier-between.ll -o - | FileCheck --check-prefix=BARRIERBETWEEN %s
+; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1310 -O2 -global-isel -global-isel-abort=1 %t/fence-between.ll -o - | FileCheck --check-prefix=FENCEBETWEEN %s
+
+; On gfx11.5, where the register prints under its pre-GFX12 name.
+; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1150 -O2 %t/unused.ll -o - | FileCheck --check-prefixes=UNUSED %s
+; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1150 -O2 %t/store.ll -o - | FileCheck --check-prefixes=STORE %s
+; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1150 -O2 %t/return.ll -o - | FileCheck --check-prefixes=RETURN %s
+; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1150 -O2 %t/select.ll -o - | FileCheck --check-prefixes=SELECT %s
+; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1150 -O2 %t/arithmetic.ll -o - | FileCheck --check-prefixes=ARITH %s
+; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1150 -O2 %t/multiple-branches.ll -o - | FileCheck --check-prefixes=MULTIBR %s
+; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1150 -O2 %t/branch-and-store.ll -o - | FileCheck --check-prefixes=BRSTORE %s
+; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1150 -O2 %t/ordered-placement.ll -o - | FileCheck --check-prefixes=ORDERED %s
+; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1150 -O2 %t/expect-ordered-placement.ll -o - | FileCheck --check-prefixes=EXPECTORDERED %s
+; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1150 -O2 %t/expect-multiple-uses.ll -o - | FileCheck --check-prefixes=EXPECTMULTI %s
+; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1150 -O2 %t/store-between.ll -o - | FileCheck --check-prefixes=STOREBETWEEN %s
+; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1150 -O2 %t/call-between.ll -o - | FileCheck --check-prefixes=CALLBETWEEN %s
+; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1150 -O2 %t/two-observations.ll -o - | FileCheck --check-prefixes=TWOOBS %s
+; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1150 -O2 %t/negated-store-between.ll -o - | FileCheck --check-prefixes=NEGSTORE %s
+
+; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1310 -O0 %t/store.ll -o - | FileCheck --check-prefixes=STORE %s
+; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1310 -O0 -global-isel -global-isel-abort=1 %t/store.ll -o - | FileCheck --check-prefixes=STORE %s
+; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1310 -O0 %t/branch-and-store.ll -o - | FileCheck --check-prefixes=BRSTORE %s
+; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1310 -O0 -global-isel -global-isel-abort=1 %t/branch-and-store.ll -o - | FileCheck --check-prefixes=BRSTORE %s
+; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1310 -O0 %t/store-between.ll -o - | FileCheck --check-prefixes=STOREBETWEEN %s
+; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1310 -O0 -global-isel -global-isel-abort=1 %t/store-between.ll -o - | FileCheck --check-prefixes=STOREBETWEEN %s
+; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1310 -O0 %t/call-between.ll -o - | FileCheck --check-prefixes=CALLBETWEEN %s
+; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1310 -O0 -global-isel -global-isel-abort=1 %t/call-between.ll -o - | FileCheck --check-prefixes=CALLBETWEEN %s
+; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1310 -O0 %t/two-observations.ll -o - | FileCheck --check-prefixes=TWOOBS %s
+; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1310 -O0 -global-isel -global-isel-abort=1 %t/two-observations.ll -o - | FileCheck --check-prefixes=TWOOBS %s
+; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1310 -O0 %t/negated-store-between.ll -o - | FileCheck --check-prefixes=NEGSTORE %s
+; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1310 -O0 -global-isel -global-isel-abort=1 %t/negated-store-between.ll -o - | FileCheck --check-prefixes=NEGSTORE %s
+
+; UNUSED-LABEL: unused:
+; STORE-LABEL: store_result:
+; RETURN-LABEL: return_result:
+; SELECT-LABEL: select_result:
+; MULTIBR-LABEL: multiple_branches:
+; BRSTORE-LABEL: branch_and_store:
+; EXPECTMULTI-LABEL: expect_multiple_uses:
+; ORDERED-LABEL: ordered_placement:
+; EXPECTORDERED-LABEL: expect_ordered_placement:
+; STOREBETWEEN-LABEL: store_between:
+; CALLBETWEEN-LABEL: call_between:
+; TWOOBS-LABEL: two_observations:
+; NEGSTORE-LABEL: negated_store_between:
+
+; These either lack a sole branch use or cross an intervening side effect, so
+; the query materializes instead of fusing. Note that llvm.sideeffect cannot
+; serve as the intervening operation, because both instruction selectors drop
+; it without leaving a node behind.
+; UNUSED-NOT: s_cbranch_cdbgsys_or_user
+; UNUSED: s_getreg_b32 s{{[0-9]+}}, hwreg(HW_REG_{{(WAVE_)?}}STATUS, 20, 2)
+; UNUSED-NOT: s_cbranch_cdbgsys_or_user
+; STORE-NOT: s_cbranch_cdbgsys_or_user
+; STORE: s_getreg_b32 s{{[0-9]+}}, hwreg(HW_REG_{{(WAVE_)?}}STATUS, 20, 2)
+; STORE-NOT: s_cbranch_cdbgsys_or_user
+; RETURN-NOT: s_cbranch_cdbgsys_or_user
+; RETURN: s_getreg_b32 s{{[0-9]+}}, hwreg(HW_REG_{{(WAVE_)?}}STATUS, 20, 2)
+; RETURN-NOT: s_cbranch_cdbgsys_or_user
+; SELECT-NOT: s_cbranch_cdbgsys_or_user
+; SELECT: s_getreg_b32 s{{[0-9]+}}, hwreg(HW_REG_{{(WAVE_)?}}STATUS, 20, 2)
+; SELECT-NOT: s_cbranch_cdbgsys_or_user
+; MULTIBR-NOT: s_cbranch_cdbgsys_or_user
+; MULTIBR: s_getreg_b32 s{{[0-9]+}}, hwreg(HW_REG_{{(WAVE_)?}}STATUS, 20, 2)
+; MULTIBR-NOT: s_cbranch_cdbgsys_or_user
+; BRSTORE-NOT: s_cbranch_cdbgsys_or_user
+; BRSTORE: s_getreg_b32 s{{[0-9]+}}, hwreg(HW_REG_{{(WAVE_)?}}STATUS, 20, 2)
+; BRSTORE-NOT: s_cbranch_cdbgsys_or_user
+; EXPECTMULTI-NOT: s_cbranch_cdbgsys_or_user
+; EXPECTMULTI: s_getreg_b32 s{{[0-9]+}}, hwreg(HW_REG_{{(WAVE_)?}}STATUS, 20, 2)
+; EXPECTMULTI-NOT: s_cbranch_cdbgsys_or_user
+; ORDERED-NOT: s_cbranch_cdbgsys_or_user
+; ORDERED: s_getreg_b32 s{{[0-9]+}}, hwreg(HW_REG_{{(WAVE_)?}}STATUS, 20, 2)
+; ORDERED-NOT: s_cbranch_cdbgsys_or_user
+; EXPECTORDERED-NOT: s_cbranch_cdbgsys_or_user
+; EXPECTORDERED: s_getreg_b32 s{{[0-9]+}}, hwreg(HW_REG_{{(WAVE_)?}}STATUS, 20, 2)
+; EXPECTORDERED-NOT: s_cbranch_cdbgsys_or_user
+; STOREBETWEEN-NOT: s_cbranch_cdbgsys_or_user
+; STOREBETWEEN: s_getreg_b32 s{{[0-9]+}}, hwreg(HW_REG_{{(WAVE_)?}}STATUS, 20, 2)
+; STOREBETWEEN-NOT: s_cbranch_cdbgsys_or_user
+; CALLBETWEEN-NOT: s_cbranch_cdbgsys_or_user
+; CALLBETWEEN: s_getreg_b32 s{{[0-9]+}}, hwreg(HW_REG_{{(WAVE_)?}}STATUS, 20, 2)
+; CALLBETWEEN-NOT: s_cbranch_cdbgsys_or_user
+
+; Fusing would swap two distinct observations, so both must
+; materialize.
+; TWOOBS-NOT: s_cbranch_cdbgsys_or_user
+; TWOOBS-COUNT-2: s_getreg_b32 s{{[0-9]+}}, hwreg(HW_REG_{{(WAVE_)?}}STATUS, 20, 2)
+; TWOOBS-NOT: s_cbranch_cdbgsys_or_user
+
+; A negated query rejected by the scan must not lose its negation.
+; NEGSTORE-NOT: s_cbranch_cdbgsys_or_user
+; NEGSTORE: s_getreg_b32 s{{[0-9]+}}, hwreg(HW_REG_{{(WAVE_)?}}STATUS, 20, 2)
+; NEGSTORE-NOT: s_cbranch_cdbgsys_or_user
+
+; LOADBETWEEN-LABEL: load_between:
+; LOADBETWEEN-NOT: s_cbranch_cdbgsys_or_user
+; LOADBETWEEN: s_getreg_b32 s{{[0-9]+}}, hwreg(HW_REG_{{(WAVE_)?}}STATUS, 20, 2)
+; LOADBETWEEN-NOT: s_cbranch_cdbgsys_or_user
+; VOLLOADBETWEEN-LABEL: volatile_load_between:
+; VOLLOADBETWEEN-NOT: s_cbranch_cdbgsys_or_user
+; VOLLOADBETWEEN: s_getreg_b32 s{{[0-9]+}}, hwreg(HW_REG_{{(WAVE_)?}}STATUS, 20, 2)
+; VOLLOADBETWEEN-NOT: s_cbranch_cdbgsys_or_user
+; BARRIERBETWEEN-LABEL: barrier_between:
+; BARRIERBETWEEN-NOT: s_cbranch_cdbgsys_or_user
+; BARRIERBETWEEN: s_getreg_b32 s{{[0-9]+}}, hwreg(HW_REG_{{(WAVE_)?}}STATUS, 20, 2)
+; BARRIERBETWEEN-NOT: s_cbranch_cdbgsys_or_user
+; FENCEBETWEEN-LABEL: fence_between:
+; FENCEBETWEEN-NOT: s_cbranch_cdbgsys_or_user
+; FENCEBETWEEN: s_getreg_b32 s{{[0-9]+}}, hwreg(HW_REG_{{(WAVE_)?}}STATUS, 20, 2)
+; FENCEBETWEEN-NOT: s_cbranch_cdbgsys_or_user
+
+; ARITH-LABEL: arithmetic_use:
+
+; A canonical negation feeding the sole conditional branch is fusable.
+; ARITH-NOT: s_getreg_b32
+; ARITH: s_cbranch_cdbgsys_or_user
+
+;--- unused.ll
+declare noundef i1 @llvm.is.debugging.enabled()
+
+define amdgpu_kernel void @unused() {
+ %enabled = call i1 @llvm.is.debugging.enabled()
+ ret void
+}
+
+;--- store.ll
+declare noundef i1 @llvm.is.debugging.enabled()
+
+define amdgpu_kernel void @store_result(ptr addrspace(1) %out) {
+ %enabled = call i1 @llvm.is.debugging.enabled()
+ store i1 %enabled, ptr addrspace(1) %out
+ ret void
+}
+
+;--- return.ll
+declare noundef i1 @llvm.is.debugging.enabled()
+
+define i1 @return_result() {
+ %enabled = call i1 @llvm.is.debugging.enabled()
+ ret i1 %enabled
+}
+
+;--- select.ll
+declare noundef i1 @llvm.is.debugging.enabled()
+
+define i32 @select_result() {
+ %enabled = call i1 @llvm.is.debugging.enabled()
+ %value = select i1 %enabled, i32 1, i32 0
+ ret i32 %value
+}
+
+;--- arithmetic.ll
+declare noundef i1 @llvm.is.debugging.enabled()
+declare void @llvm.sideeffect()
+
+define amdgpu_kernel void @arithmetic_use() {
+entry:
+ %enabled = call i1 @llvm.is.debugging.enabled()
+ %disabled = xor i1 %enabled, true
+ br i1 %disabled, label %normal, label %debug
+
+debug:
+ call void @llvm.sideeffect()
+ br label %normal
+
+normal:
+ ret void
+}
+
+;--- load-between.ll
+declare noundef i1 @llvm.is.debugging.enabled()
+
+define amdgpu_kernel void @load_between(ptr addrspace(1) %in,
+ ptr addrspace(1) %out) {
+entry:
+ %enabled = call i1 @llvm.is.debugging.enabled()
+ %value = load i32, ptr addrspace(1) %in
+ br i1 %enabled, label %debug, label %normal
+
+debug:
+ store volatile i32 %value, ptr addrspace(1) %out
+ br label %exit
+
+normal:
+ %next = add i32 %value, 1
+ store volatile i32 %next, ptr addrspace(1) %out
+ br label %exit
+
+exit:
+ ret void
+}
+
+;--- volatile-load-between.ll
+declare noundef i1 @llvm.is.debugging.enabled()
+
+define amdgpu_kernel void @volatile_load_between(ptr addrspace(1) %in,
+ ptr addrspace(1) %out) {
+entry:
+ %enabled = call i1 @llvm.is.debugging.enabled()
+ %value = load volatile i32, ptr addrspace(1) %in
+ br i1 %enabled, label %debug, label %exit
+
+debug:
+ store volatile i32 %value, ptr addrspace(1) %out
+ br label %exit
+
+exit:
+ ret void
+}
+
+;--- barrier-between.ll
+declare noundef i1 @llvm.is.debugging.enabled()
+declare void @llvm.amdgcn.s.barrier()
+
+define amdgpu_kernel void @barrier_between(ptr addrspace(1) %out) {
+entry:
+ %enabled = call i1 @llvm.is.debugging.enabled()
+ call void @llvm.amdgcn.s.barrier()
+ br i1 %enabled, label %debug, label %exit
+
+debug:
+ store volatile i32 1, ptr addrspace(1) %out
+ br label %exit
+
+exit:
+ ret void
+}
+
+;--- fence-between.ll
+declare noundef i1 @llvm.is.debugging.enabled()
+
+define amdgpu_kernel void @fence_between(ptr addrspace(1) %out) {
+entry:
+ %enabled = call i1 @llvm.is.debugging.enabled()
+ fence syncscope("agent") seq_cst
+ br i1 %enabled, label %debug, label %exit
+
+debug:
+ store volatile i32 1, ptr addrspace(1) %out
+ br label %exit
+
+exit:
+ ret void
+}
+
+;--- multiple-branches.ll
+declare noundef i1 @llvm.is.debugging.enabled()
+declare void @llvm.sideeffect()
+
+define amdgpu_kernel void @multiple_branches() {
+entry:
+ %enabled = call i1 @llvm.is.debugging.enabled()
+ br i1 %enabled, label %again, label %normal
+
+again:
+ br i1 %enabled, label %debug, label %normal
+
+debug:
+ call void @llvm.sideeffect()
+ br label %normal
+
+normal:
+ ret void
+}
+
+;--- branch-and-store.ll
+declare noundef i1 @llvm.is.debugging.enabled()
+declare void @llvm.sideeffect()
+
+define amdgpu_kernel void @branch_and_store(ptr addrspace(1) %out) {
+entry:
+ %enabled = call i1 @llvm.is.debugging.enabled()
+ store i1 %enabled, ptr addrspace(1) %out
+ br i1 %enabled, label %debug, label %normal
+
+debug:
+ call void @llvm.sideeffect()
+ br label %normal
+
+normal:
+ ret void
+}
+
+;--- ordered-placement.ll
+declare noundef i1 @llvm.is.debugging.enabled()
+declare void @llvm.sideeffect()
+
+define amdgpu_kernel void @ordered_placement(ptr addrspace(1) %out) {
+entry:
+ %enabled = call i1 @llvm.is.debugging.enabled()
+ store volatile i32 7, ptr addrspace(1) %out
+ br i1 %enabled, label %debug, label %normal
+
+debug:
+ call void @llvm.sideeffect()
+ br label %normal
+
+normal:
+ ret void
+}
+
+;--- expect-ordered-placement.ll
+declare noundef i1 @llvm.is.debugging.enabled()
+declare i1 @llvm.expect.i1(i1, i1 immarg)
+declare void @llvm.sideeffect()
+
+define amdgpu_kernel void @expect_ordered_placement(ptr addrspace(1) %out) {
+entry:
+ %raw = call i1 @llvm.is.debugging.enabled()
+ %enabled = call i1 @llvm.expect.i1(i1 %raw, i1 false)
+ store volatile i32 7, ptr addrspace(1) %out
+ br i1 %enabled, label %debug, label %normal
+
+debug:
+ call void @llvm.sideeffect()
+ br label %normal
+
+normal:
+ ret void
+}
+
+;--- expect-multiple-uses.ll
+declare noundef i1 @llvm.is.debugging.enabled()
+declare i1 @llvm.expect.i1(i1, i1 immarg)
+declare void @llvm.sideeffect()
+
+define amdgpu_kernel void @expect_multiple_uses(ptr addrspace(1) %out) {
+entry:
+ %raw = call i1 @llvm.is.debugging.enabled()
+ %enabled = call i1 @llvm.expect.i1(i1 %raw, i1 false)
+ store i1 %enabled, ptr addrspace(1) %out
+ br i1 %enabled, label %debug, label %normal
+
+debug:
+ call void @llvm.sideeffect()
+ br label %normal
+
+normal:
+ ret void
+}
+
+;--- store-between.ll
+declare noundef i1 @llvm.is.debugging.enabled()
+
+define amdgpu_kernel void @store_between(ptr addrspace(1) %out) {
+entry:
+ %enabled = call i1 @llvm.is.debugging.enabled()
+ store volatile i32 7, ptr addrspace(1) %out
+ br i1 %enabled, label %debug, label %normal
+
+debug:
+ store volatile i32 1, ptr addrspace(1) %out
+ br label %normal
+
+normal:
+ ret void
+}
+
+;--- call-between.ll
+declare noundef i1 @llvm.is.debugging.enabled()
+declare void @ext()
+
+define void @call_between(ptr addrspace(1) %out) {
+entry:
+ %enabled = call i1 @llvm.is.debugging.enabled()
+ call void @ext()
+ br i1 %enabled, label %debug, label %normal
+
+debug:
+ store volatile i32 1, ptr addrspace(1) %out
+ br label %normal
+
+normal:
+ ret void
+}
+
+;--- two-observations.ll
+declare noundef i1 @llvm.is.debugging.enabled()
+
+define amdgpu_kernel void @two_observations(ptr addrspace(1) %out) {
+entry:
+ %first = call i1 @llvm.is.debugging.enabled()
+ %second = call i1 @llvm.is.debugging.enabled()
+ store volatile i1 %second, ptr addrspace(1) %out
+ br i1 %first, label %debug, label %normal
+
+debug:
+ store volatile i32 1, ptr addrspace(1) %out
+ br label %normal
+
+normal:
+ ret void
+}
+
+;--- negated-store-between.ll
+declare noundef i1 @llvm.is.debugging.enabled()
+
+define amdgpu_kernel void @negated_store_between(ptr addrspace(1) %out) {
+entry:
+ %enabled = call i1 @llvm.is.debugging.enabled()
+ store volatile i32 7, ptr addrspace(1) %out
+ %disabled = xor i1 %enabled, true
+ br i1 %disabled, label %normal, label %debug
+
+debug:
+ store volatile i32 1, ptr addrspace(1) %out
+ br label %normal
+
+normal:
+ ret void
+}
diff --git a/llvm/test/CodeGen/AMDGPU/is-debugging-enabled-unsupported.ll b/llvm/test/CodeGen/AMDGPU/is-debugging-enabled-unsupported.ll
new file mode 100644
index 0000000000000..4acbb583da834
--- /dev/null
+++ b/llvm/test/CodeGen/AMDGPU/is-debugging-enabled-unsupported.ll
@@ -0,0 +1,32 @@
+; RUN: opt -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1030 -passes=amdgpu-lower-intrinsics -S %s | FileCheck %s --check-prefix=IR
+; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1030 -O2 %s -o - | FileCheck %s --check-prefix=GCN
+; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1100 -O2 %s -o - | FileCheck %s --check-prefix=GCN
+; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx11-generic -O2 %s -o - | FileCheck %s --check-prefix=GCN
+; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1170 -O2 %s -o - | FileCheck %s --check-prefix=GCN
+; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1150 -mattr=-cdbg-sys-or-user-branch -O2 %s -o - | FileCheck %s --check-prefix=GCN
+; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1100 -O2 -global-isel -global-isel-abort=1 %s -o - | FileCheck %s --check-prefix=GCN
+
+declare noundef i1 @llvm.is.debugging.enabled()
+
+define amdgpu_kernel void @unsupported_subtarget() {
+entry:
+ %enabled = call i1 @llvm.is.debugging.enabled()
+ br i1 %enabled, label %debug, label %normal
+
+debug:
+ br label %normal
+
+normal:
+ ret void
+}
+
+; IR-LABEL: define amdgpu_kernel void @unsupported_subtarget()
+; IR: entry:
+; IR-NEXT: br i1 false, label %debug, label %normal
+; IR-NOT: call i1 @llvm.is.debugging.enabled()
+; IR: ret void
+
+; GCN-LABEL: unsupported_subtarget:
+; GCN-NOT: s_cbranch_cdbg
+; GCN-NOT: s_getreg
+; GCN: s_endpgm
diff --git a/llvm/test/CodeGen/AMDGPU/is-debugging-enabled.ll b/llvm/test/CodeGen/AMDGPU/is-debugging-enabled.ll
new file mode 100644
index 0000000000000..a87b1e3e15e0f
--- /dev/null
+++ b/llvm/test/CodeGen/AMDGPU/is-debugging-enabled.ll
@@ -0,0 +1,257 @@
+; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1150 -O2 < %s | FileCheck %s --check-prefix=GCN
+; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1150 -O2 -global-isel -global-isel-abort=1 < %s | FileCheck %s --check-prefix=GCN
+; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1200 -O2 < %s | FileCheck %s --check-prefix=GCN
+; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1200 -O2 -global-isel -global-isel-abort=1 < %s | FileCheck %s --check-prefix=GCN
+; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1310 -O2 < %s | FileCheck %s --check-prefix=GCN
+; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1310 -O2 -global-isel -global-isel-abort=1 < %s | FileCheck %s --check-prefix=GCN
+; At -O0 the DAG combiner never rewrites the inverted condition into a setcc, so
+; these runs cover the bare-xor form of the query reaching branch lowering.
+; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1310 -O0 < %s | FileCheck %s --check-prefix=GCN
+; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1310 -O0 -global-isel -global-isel-abort=1 < %s | FileCheck %s --check-prefix=GCN
+; RUN: opt -passes=lower-expect -S %s | llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1150 -O2 | FileCheck %s --check-prefix=EXPECT
+; RUN: opt -passes=lower-expect -S %s | llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1150 -O2 -global-isel -global-isel-abort=1 | FileCheck %s --check-prefix=EXPECT
+; RUN: opt -passes=lower-expect -S %s | llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1150 -O2 -stop-after=finalize-isel | FileCheck %s --check-prefix=MIR
+; RUN: opt -passes=lower-expect -S %s | llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1150 -O2 -global-isel -global-isel-abort=1 -stop-after=finalize-isel | FileCheck %s --check-prefix=MIR
+; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1150 -O2 < %s | FileCheck %s --check-prefix=LOC
+; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1150 -O2 -global-isel -global-isel-abort=1 < %s | FileCheck %s --check-prefix=LOC
+
+declare noundef i1 @llvm.is.debugging.enabled()
+declare i1 @llvm.expect.i1(i1, i1 immarg)
+declare void @llvm.debugtrap()
+declare i32 @llvm.amdgcn.ballot.i32(i1)
+declare i32 @llvm.amdgcn.workgroup.id.x()
+declare i32 @llvm.amdgcn.workitem.id.x()
+
+; GCN-NOT: s_getreg_b32
+; EXPECT-NOT: s_getreg_b32
+; LOC-NOT: s_getreg_b32
+
+define amdgpu_kernel void @direct_true_first(ptr addrspace(1) %out) {
+entry:
+ %enabled = call i1 @llvm.is.debugging.enabled()
+ br i1 %enabled, label %debug, label %normal
+
+debug:
+ store volatile i32 1, ptr addrspace(1) %out
+ br label %normal
+
+normal:
+ ret void
+}
+
+; GCN-LABEL: direct_true_first:
+; GCN: s_cbranch_cdbgsys_or_user [[TRUE_FIRST_DEBUG:.LBB[0-9_]+]]
+; GCN-NOT: s_cbranch_cdbgsys_or_user
+; GCN-NOT: s_cbranch_scc
+; GCN-NOT: s_cbranch_vcc
+; GCN-NOT: s_cmp
+; GCN: [[TRUE_FIRST_DEBUG]]:
+; GCN: global_store_b32
+
+define amdgpu_kernel void @direct_false_first(ptr addrspace(1) %out) {
+entry:
+ %enabled = call i1 @llvm.is.debugging.enabled()
+ br i1 %enabled, label %debug, label %normal
+
+normal:
+ ret void
+
+debug:
+ store volatile i32 1, ptr addrspace(1) %out
+ br label %normal
+}
+
+; GCN-LABEL: direct_false_first:
+; GCN: s_cbranch_cdbgsys_or_user [[FALSE_FIRST_DEBUG:.LBB[0-9_]+]]
+; GCN-NOT: s_cbranch_cdbgsys_or_user
+; GCN-NOT: s_cbranch_scc
+; GCN-NOT: s_cbranch_vcc
+; GCN-NOT: s_cmp
+; GCN: [[FALSE_FIRST_DEBUG]]:
+; GCN: global_store_b32
+
+define amdgpu_kernel void @chain_ordering(ptr addrspace(1) %out) {
+entry:
+ store volatile i32 0, ptr addrspace(1) %out
+ %enabled = call i1 @llvm.is.debugging.enabled()
+ br i1 %enabled, label %debug, label %normal
+
+debug:
+ store volatile i32 1, ptr addrspace(1) %out
+ br label %normal
+
+normal:
+ ret void
+}
+
+; GCN-LABEL: chain_ordering:
+; GCN: global_store_b32
+; GCN: s_cbranch_cdbgsys_or_user [[CHAIN_DEBUG:.LBB[0-9_]+]]
+; GCN-NOT: s_cbranch_scc
+; GCN-NOT: s_cbranch_vcc
+; GCN: [[CHAIN_DEBUG]]:
+; GCN: global_store_b32
+
+define amdgpu_kernel void @expected_debug_break() {
+entry:
+ %raw = call i1 @llvm.is.debugging.enabled()
+ %enabled = call i1 @llvm.expect.i1(i1 %raw, i1 false)
+ br i1 %enabled, label %debug, label %normal
+
+debug:
+ call void @llvm.debugtrap()
+ br label %normal
+
+normal:
+ ret void
+}
+
+; GCN-LABEL: expected_debug_break:
+; GCN-NOT: s_cbranch_execz
+; GCN: s_cbranch_cdbgsys_or_user [[EXPECTED_DEBUG:.LBB[0-9_]+]]
+; GCN-NOT: s_cbranch_cdbgsys_or_user
+; GCN-NOT: s_cbranch_scc
+; GCN-NOT: s_cbranch_vcc
+; GCN-NOT: s_cmp
+; GCN: [[EXPECTED_DEBUG]]:
+; GCN-NEXT: s_trap 3
+
+; EXPECT-LABEL: expected_debug_break:
+; EXPECT: s_cbranch_cdbgsys_or_user [[COLD_DEBUG:.LBB[0-9_]+]]
+; EXPECT-NOT: s_branch
+; EXPECT: ; %bb.1:{{.*}}%normal
+; EXPECT: s_endpgm
+; EXPECT: [[COLD_DEBUG]]:
+; EXPECT-COUNT-1: s_trap 3
+
+; MIR-LABEL: name: expected_debug_break
+; MIR: bb.{{[0-9]+}}.entry:
+; MIR: successors: %bb.[[MIR_DEBUG:[0-9]+]](0x00106035), %bb.[[MIR_NORMAL:[0-9]+]](0x7fef9fcb)
+; MIR: nomerge S_CBRANCH_CDBGSYS_OR_USER %bb.[[MIR_DEBUG]]
+; MIR-NEXT: S_BRANCH %bb.[[MIR_NORMAL]]
+
+define amdgpu_kernel void @arithmetic_between(i32 %x,
+ ptr addrspace(1) %out) {
+entry:
+ %enabled = call i1 @llvm.is.debugging.enabled()
+ %value = add i32 %x, 1
+ br i1 %enabled, label %debug, label %exit
+
+debug:
+ store volatile i32 1, ptr addrspace(1) %out
+ br label %exit
+
+exit:
+ store volatile i32 %value, ptr addrspace(1) %out
+ ret void
+}
+
+; GCN-LABEL: arithmetic_between:
+; GCN-NOT: s_getreg_b32
+; GCN: s_cbranch_cdbgsys_or_user
+
+; Divergence alone does not block fusion.
+define amdgpu_kernel void @divergent_value_between(ptr addrspace(1) %out) {
+entry:
+ %enabled = call i1 @llvm.is.debugging.enabled()
+ %value = call i32 @llvm.amdgcn.workitem.id.x()
+ br i1 %enabled, label %debug, label %exit
+
+debug:
+ store volatile i32 1, ptr addrspace(1) %out
+ br label %exit
+
+exit:
+ store volatile i32 %value, ptr addrspace(1) %out
+ ret void
+}
+
+; GCN-LABEL: divergent_value_between:
+; GCN-NOT: s_getreg_b32
+; GCN: s_cbranch_cdbgsys_or_user
+
+; Convergent without memory or side effects does not block fusion.
+define amdgpu_kernel void @ballot_between(ptr addrspace(1) %out) {
+entry:
+ %enabled = call i1 @llvm.is.debugging.enabled()
+ %id = call i32 @llvm.amdgcn.workitem.id.x()
+ %predicate = icmp eq i32 %id, 0
+ %value = call i32 @llvm.amdgcn.ballot.i32(i1 %predicate)
+ br i1 %enabled, label %debug, label %exit
+
+debug:
+ store volatile i32 1, ptr addrspace(1) %out
+ br label %exit
+
+exit:
+ store volatile i32 %value, ptr addrspace(1) %out
+ ret void
+}
+
+; GCN-LABEL: ballot_between:
+; GCN-NOT: s_getreg_b32
+; GCN: s_cbranch_cdbgsys_or_user
+
+define amdgpu_kernel void @debug_location(ptr addrspace(1) %out) !dbg !4 {
+entry:
+ %enabled = call i1 @llvm.is.debugging.enabled(), !dbg !7
+ br i1 %enabled, label %debug, label %normal, !dbg !8
+
+debug:
+ store volatile i32 1, ptr addrspace(1) %out, !dbg !9
+ br label %normal, !dbg !10
+
+normal:
+ ret void, !dbg !11
+}
+
+; LOC-LABEL: debug_location:
+; LOC: .loc {{[0-9]+}} 3 3
+; LOC-NEXT: s_cbranch_cdbgsys_or_user
+
+define amdgpu_kernel void @uniform_guard_debugtrap(ptr addrspace(1) %out) {
+entry:
+ %wg = call i32 @llvm.amdgcn.workgroup.id.x()
+ %is3 = icmp eq i32 %wg, 3
+ br i1 %is3, label %region, label %exit
+
+region:
+ %e = call i1 @llvm.is.debugging.enabled()
+ br i1 %e, label %dbg, label %after
+
+dbg:
+ call void @llvm.debugtrap()
+ br label %after
+
+after:
+ store volatile i32 7, ptr addrspace(1) %out
+ br label %exit
+
+exit:
+ ret void
+}
+
+; GCN-LABEL: uniform_guard_debugtrap:
+; GCN-NOT: execz
+; GCN-NOT: v_cmpx
+; GCN: s_cbranch_{{scc1|vccnz}} [[UNIFORM_SKIP:.LBB[0-9_]+]]
+; GCN: s_cbranch_cdbgsys_or_user [[UNIFORM_DEBUG:.LBB[0-9_]+]]
+; GCN: [[UNIFORM_DEBUG]]:
+; GCN-NEXT: s_trap 3
+; GCN: [[UNIFORM_SKIP]]:
+
+!llvm.dbg.cu = !{!0}
+!llvm.module.flags = !{!2, !3}
+
+!0 = distinct !DICompileUnit(language: DW_LANG_C99, file: !1, producer: "llvm", isOptimized: true, runtimeVersion: 0, emissionKind: FullDebug)
+!1 = !DIFile(filename: "is-debugging-enabled.ll", directory: "/")
+!2 = !{i32 2, !"Dwarf Version", i32 5}
+!3 = !{i32 2, !"Debug Info Version", i32 3}
+!4 = distinct !DISubprogram(name: "debug_location", scope: !1, file: !1, line: 1, type: !5, scopeLine: 1, spFlags: DISPFlagDefinition, unit: !0)
+!5 = !DISubroutineType(types: !6)
+!6 = !{}
+!7 = !DILocation(line: 2, column: 3, scope: !4)
+!8 = !DILocation(line: 3, column: 3, scope: !4)
+!9 = !DILocation(line: 4, column: 3, scope: !4)
+!10 = !DILocation(line: 5, column: 3, scope: !4)
+!11 = !DILocation(line: 6, column: 3, scope: !4)
diff --git a/llvm/test/Transforms/PreISelIntrinsicLowering/is-debugging-enabled.ll b/llvm/test/Transforms/PreISelIntrinsicLowering/is-debugging-enabled.ll
index 2ce321558677b..43fe4e4008c12 100644
--- a/llvm/test/Transforms/PreISelIntrinsicLowering/is-debugging-enabled.ll
+++ b/llvm/test/Transforms/PreISelIntrinsicLowering/is-debugging-enabled.ll
@@ -1,11 +1,15 @@
; RUN: %if x86-registered-target %{ opt -mtriple=x86_64 -passes=pre-isel-intrinsic-lowering -S < %s | FileCheck %s --check-prefix=UNSUPPORTED %}
; RUN: %if nvptx-registered-target %{ opt -mtriple=nvptx64-nvidia-cuda -passes=pre-isel-intrinsic-lowering -S < %s | FileCheck %s --check-prefix=UNSUPPORTED %}
; RUN: %if amdgpu-registered-target %{ opt -mtriple=r600 -passes=pre-isel-intrinsic-lowering -S < %s | FileCheck %s --check-prefix=UNSUPPORTED %}
-; RUN: %if amdgpu-registered-target %{ opt -mtriple=amdgcn-amd-amdhsa -passes=pre-isel-intrinsic-lowering -S < %s | FileCheck %s --check-prefix=UNSUPPORTED %}
+; RUN: %if amdgpu-registered-target %{ opt -mtriple=amdgcn-amd-amdhsa -passes=pre-isel-intrinsic-lowering -S < %s | FileCheck %s --check-prefix=GCN %}
define i1 @query() {
; UNSUPPORTED-LABEL: define i1 @query() {
; UNSUPPORTED-NEXT: ret i1 false
+;
+; GCN-LABEL: define i1 @query() {
+; GCN-NEXT: [[ENABLED:%.*]] = call i1 @llvm.is.debugging.enabled()
+; GCN-NEXT: ret i1 [[ENABLED]]
%enabled = call i1 @llvm.is.debugging.enabled()
ret i1 %enabled
}
More information about the llvm-commits
mailing list