[llvm] [AMDGPU] using divergent/uniform information in ISel of zext (PR #174539)
Zeng Wu via llvm-commits
llvm-commits at lists.llvm.org
Tue Mar 31 00:01:39 PDT 2026
https://github.com/zwu-2025 updated https://github.com/llvm/llvm-project/pull/174539
>From 0bb759453caafcefbfd39aee3dfc6ee8bc83ce7d Mon Sep 17 00:00:00 2001
From: zwu-2025 <Zeng.Wu2 at amd.com>
Date: Mon, 23 Mar 2026 17:49:18 +0000
Subject: [PATCH 1/2] changes decision order of UseRC and use
common-sub-regclass
---
llvm/lib/CodeGen/SelectionDAG/InstrEmitter.cpp | 17 +++++++++++++++--
1 file changed, 15 insertions(+), 2 deletions(-)
diff --git a/llvm/lib/CodeGen/SelectionDAG/InstrEmitter.cpp b/llvm/lib/CodeGen/SelectionDAG/InstrEmitter.cpp
index 4ad721bf21959..3096ab5ccb698 100644
--- a/llvm/lib/CodeGen/SelectionDAG/InstrEmitter.cpp
+++ b/llvm/lib/CodeGen/SelectionDAG/InstrEmitter.cpp
@@ -24,9 +24,11 @@
#include "llvm/CodeGen/StackMaps.h"
#include "llvm/CodeGen/TargetInstrInfo.h"
#include "llvm/CodeGen/TargetLowering.h"
+#include "llvm/CodeGen/TargetRegisterInfo.h"
#include "llvm/CodeGen/TargetSubtargetInfo.h"
#include "llvm/IR/DebugInfoMetadata.h"
#include "llvm/IR/PseudoProbe.h"
+#include "llvm/MC/MCInstrDesc.h"
#include "llvm/Support/ErrorHandling.h"
#include "llvm/Target/TargetMachine.h"
using namespace llvm;
@@ -105,8 +107,6 @@ void InstrEmitter::EmitCopyFromReg(SDValue Op, bool IsClone, Register SrcReg,
MVT VT = Op.getSimpleValueType();
// Stick to the preferred register classes for legal types.
- if (TLI->isTypeLegal(VT))
- UseRC = TLI->getRegClassFor(VT, Op->isDivergent());
for (SDNode *User : Op->users()) {
bool Match = true;
@@ -131,6 +131,7 @@ void InstrEmitter::EmitCopyFromReg(SDValue Op, bool IsClone, Register SrcReg,
RC = TRI->getAllocatableClass(
TII->getRegClass(II, i + II.getNumDefs()));
}
+
if (!UseRC)
UseRC = RC;
else if (RC) {
@@ -149,6 +150,18 @@ void InstrEmitter::EmitCopyFromReg(SDValue Op, bool IsClone, Register SrcReg,
break;
}
+ const TargetRegisterClass *RegClassForVT = nullptr;
+ // The check is to be removed in other pending PR, it is kept to make System Z happy.
+ if (TLI->isTypeLegal(VT)) {
+ RegClassForVT = TLI->getRegClassFor(VT, Op->isDivergent());
+ }
+
+ if (UseRC == nullptr || !UseRC->isAllocatable()) {
+ UseRC = RegClassForVT;
+ } else if (auto CommonSubClass = TRI->getCommonSubClass(UseRC, RegClassForVT)) {
+ UseRC = CommonSubClass;
+ }
+
const TargetRegisterClass *SrcRC = nullptr, *DstRC = nullptr;
SrcRC = TRI->getMinimalPhysRegClass(SrcReg, VT);
>From d6c0f2b89afc1f98d34f0cc459d0c286acce82ee Mon Sep 17 00:00:00 2001
From: zwu-2025 <Zeng.Wu2 at amd.com>
Date: Mon, 23 Mar 2026 21:58:21 +0000
Subject: [PATCH 2/2] Select S_AND for ZEXT Only if directly from ICMP and
UNIFORM
---
llvm/lib/Target/AMDGPU/SIInstructions.td | 25 ++++++++++++++++++++++++
1 file changed, 25 insertions(+)
diff --git a/llvm/lib/Target/AMDGPU/SIInstructions.td b/llvm/lib/Target/AMDGPU/SIInstructions.td
index ca5a4d7301bda..eda92e2b8101f 100644
--- a/llvm/lib/Target/AMDGPU/SIInstructions.td
+++ b/llvm/lib/Target/AMDGPU/SIInstructions.td
@@ -2534,6 +2534,31 @@ def : GCNPat <
/*src1mod*/(i32 0), /*src1*/(i32 -1), i1:$src0)
>;
+def UniformEXT: PatFrag<
+ (ops node:$src),
+ (zext $src),
+ [{
+ if (N->isDivergent()) return false;
+
+ if (N->getOpcode() != ISD::ZERO_EXTEND) return false;
+
+ auto CondNode = N->getOperand(0);
+ if (CondNode->getOpcode() == ISD::FREEZE) {
+ CondNode = CondNode->getOperand(0);
+ }
+
+ if (CondNode->getOpcode() != ISD::SETCC)
+ return false;
+
+ return CondNode->getOperand(0).getValueType() == MVT::i32 && CondNode->getOperand(1).getValueType() == MVT::i32;
+ }]
+>;
+
+def : GCNPat <
+ (i32 (UniformEXT i1:$src)),
+ (S_AND_B32 $src, (i32 1))
+>;
+
class Ext32Pat <SDNode ext> : GCNPat <
(i32 (ext i1:$src0)),
(V_CNDMASK_B32_e64 /*src0mod*/(i32 0), /*src0*/(i32 0),
More information about the llvm-commits
mailing list