[llvm] 85305ba - [AMDGPU] Fix llvm.amdgcn.ballot with return width != wavefront size (#211493)
via llvm-commits
llvm-commits at lists.llvm.org
Tue Jul 28 11:50:00 PDT 2026
Author: Arseniy Obolenskiy
Date: 2026-07-28T20:49:55+02:00
New Revision: 85305ba2b67ea806e1e04b61dcc8743b19f51fc8
URL: https://github.com/llvm/llvm-project/commit/85305ba2b67ea806e1e04b61dcc8743b19f51fc8
DIFF: https://github.com/llvm/llvm-project/commit/85305ba2b67ea806e1e04b61dcc8743b19f51fc8.diff
LOG: [AMDGPU] Fix llvm.amdgcn.ballot with return width != wavefront size (#211493)
Before wave mask was emitted directly in the requested return type,
which failed to select for i32 ballots on wave64 (and vice versa)
Compute the mask at the wavefront width and then zext or trunc it to the
result type
Added:
llvm/test/CodeGen/AMDGPU/llvm.amdgcn.ballot.i32.wave64.err.ll
Modified:
llvm/lib/Target/AMDGPU/AMDGPUInstructionSelector.cpp
llvm/lib/Target/AMDGPU/SIISelLowering.cpp
Removed:
################################################################################
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUInstructionSelector.cpp b/llvm/lib/Target/AMDGPU/AMDGPUInstructionSelector.cpp
index 22b8b10554928..0ddb7f771b6c9 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUInstructionSelector.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUInstructionSelector.cpp
@@ -1729,8 +1729,15 @@ bool AMDGPUInstructionSelector::selectBallot(MachineInstr &I) const {
const unsigned BallotSize = MRI->getType(DstReg).getSizeInBits();
const unsigned WaveSize = STI.getWavefrontSize();
- // In the common case, the return type matches the wave size.
- // However we also support emitting i64 ballots in wave32 mode.
+ if (BallotSize < WaveSize) {
+ const Function &Fn = MF->getFunction();
+ Fn.getContext().diagnose(DiagnosticInfoUnsupported(
+ Fn, "ballot return type is narrower than the wavefront size", DL));
+ BuildMI(*BB, &I, DL, TII.get(AMDGPU::IMPLICIT_DEF), DstReg);
+ I.eraseFromParent();
+ return true;
+ }
+
if (BallotSize != WaveSize && (BallotSize != 64 || WaveSize != 32))
return false;
diff --git a/llvm/lib/Target/AMDGPU/SIISelLowering.cpp b/llvm/lib/Target/AMDGPU/SIISelLowering.cpp
index 78bbb7f2d6146..951e70f7f5ee0 100644
--- a/llvm/lib/Target/AMDGPU/SIISelLowering.cpp
+++ b/llvm/lib/Target/AMDGPU/SIISelLowering.cpp
@@ -7942,6 +7942,15 @@ static SDValue lowerBALLOTIntrinsic(const SITargetLowering &TLI, SDNode *N,
SDValue Src = N->getOperand(1);
SDLoc SL(N);
+ unsigned WavefrontSize = TLI.getSubtarget()->getWavefrontSize();
+ if (VT.getScalarSizeInBits() < WavefrontSize) {
+ DAG.getContext()->diagnose(DiagnosticInfoUnsupported(
+ DAG.getMachineFunction().getFunction(),
+ "ballot return type is narrower than the wavefront size",
+ SL.getDebugLoc()));
+ return DAG.getPOISON(VT);
+ }
+
if (Src.getOpcode() == ISD::SETCC) {
SDValue Op0 = Src.getOperand(0);
SDValue Op1 = Src.getOperand(1);
diff --git a/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.ballot.i32.wave64.err.ll b/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.ballot.i32.wave64.err.ll
new file mode 100644
index 0000000000000..54400baf4affe
--- /dev/null
+++ b/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.ballot.i32.wave64.err.ll
@@ -0,0 +1,15 @@
+; RUN: not llc -global-isel=0 -mtriple=amdgpu9.00 -filetype=null %s 2>&1 | FileCheck %s
+; RUN: not llc -global-isel=1 -mtriple=amdgpu9.00 -filetype=null %s 2>&1 | FileCheck %s
+
+; An i32 ballot on a wave64 target cannot represent one bit per lane, so
+; both SelectionDAG and GlobalISel must refuse to lower it instead of
+; dropping the mask bits of the high lanes.
+
+declare i32 @llvm.amdgcn.ballot.i32(i1)
+
+; CHECK: error: {{.*}}ballot return type is narrower than the wavefront size
+define amdgpu_cs i32 @ballot_i32_wave64(i32 %x, i32 %y) {
+ %cmp = icmp eq i32 %x, %y
+ %ballot = call i32 @llvm.amdgcn.ballot.i32(i1 %cmp)
+ ret i32 %ballot
+}
More information about the llvm-commits
mailing list