[llvm] [llvm-calc-occupancy] Add an AMDGPU occupancy calculator tool (PR #208727)
via llvm-commits
llvm-commits at lists.llvm.org
Fri Jul 10 08:38:35 PDT 2026
================
@@ -0,0 +1,410 @@
+//===-- llvm-calc-occupancy.cpp - AMDGPU occupancy calculator -------------===//
+//
+// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
+// See https://llvm.org/LICENSE.txt for license information.
+// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
+//
+//===----------------------------------------------------------------------===//
+//
+// A small standalone utility that answers "what occupancy do I get?" for an
+// AMDGPU kernel, given some subset of its resource usage: workgroup size,
+// VGPRs, SGPRs and LDS. Fields that are left unspecified are treated as
+// unconstrained, and the result is reported as a range (waves per EU).
+//
+// It reuses the compiler's own occupancy math (GCNSubtarget) so the numbers
+// match what the backend would compute for the same inputs.
+//
+// Example:
+// llvm-calc-occupancy -mcpu=gfx90a --wg-size=512 --vgprs=50 --sgprs=30 \
+// --lds=103kb
+//
+//===----------------------------------------------------------------------===//
+
+#include "AMDGPUTargetMachine.h"
+#include "GCNSubtarget.h"
+#include "Utils/AMDGPUBaseInfo.h"
+#include "llvm/ADT/SmallVector.h"
+#include "llvm/ADT/StringExtras.h"
+#include "llvm/ADT/StringRef.h"
+#include "llvm/MC/TargetRegistry.h"
+#include "llvm/Support/CommandLine.h"
+#include "llvm/Support/Format.h"
+#include "llvm/Support/InitLLVM.h"
+#include "llvm/Support/TargetSelect.h"
+#include "llvm/Support/WithColor.h"
+#include "llvm/Support/raw_ostream.h"
+#include "llvm/Target/TargetMachine.h"
+#include "llvm/TargetParser/Triple.h"
+#include <optional>
+
+using namespace llvm;
+
+namespace {
+cl::OptionCategory OccCategory("llvm-calc-occupancy options");
+
+cl::opt<std::string> TripleName("mtriple", cl::desc("Target triple"),
+ cl::init("amdgcn-amd-amdhsa"),
+ cl::cat(OccCategory));
+
+cl::opt<std::string> MCPU("mcpu", cl::desc("Target GPU (e.g. gfx90a)"),
+ cl::init(""), cl::cat(OccCategory));
+
+cl::opt<std::string> MAttr("mattr",
+ cl::desc("Comma-separated subtarget features "
+ "(e.g. +wavefrontsize32)"),
+ cl::init(""), cl::cat(OccCategory));
+
+cl::opt<std::string>
+ WGSizeStr("wg-size",
+ cl::desc("Flat workgroup size: single value 'N' or range "
+ "'MIN:MAX' (default: 1:1024)"),
+ cl::init(""), cl::cat(OccCategory));
+cl::alias WGSizeAlias("flat-workgroup-size", cl::aliasopt(WGSizeStr));
+
+cl::opt<int> NumVGPRs("vgprs", cl::desc("VGPRs used per lane (default: none)"),
+ cl::init(-1), cl::cat(OccCategory));
+
+cl::opt<int> NumSGPRs("sgprs", cl::desc("SGPRs used per wave (default: none)"),
+ cl::init(-1), cl::cat(OccCategory));
+
+cl::opt<std::string> LDSStr("lds",
+ cl::desc("LDS bytes per workgroup, accepts k/kb/m "
+ "suffixes (default: 0)"),
+ cl::init(""), cl::cat(OccCategory));
+
+cl::opt<unsigned>
+ DynVGPRBlockSize("dynamic-vgpr-block-size",
+ cl::desc("Dynamic VGPR block size (0 = disabled)"),
+ cl::init(0), cl::cat(OccCategory));
+
+cl::opt<bool> ShowLimits("limits",
+ cl::desc("Print the per-occupancy VGPR/SGPR limit "
+ "table for this GPU"),
+ cl::init(false), cl::cat(OccCategory));
+} // namespace
+
+// Parse a byte size with an optional binary suffix (k/kb/kib/m/mb/mib), all
+// base 1024. A bare number is interpreted as bytes.
+static bool parseSize(StringRef S, uint64_t &Out) {
+ S = S.trim();
+ if (S.empty())
+ return false;
+ uint64_t Mult = 1;
+ StringRef Lower = S.lower();
+ static const std::pair<StringRef, uint64_t> Suffixes[] = {
+ {"kib", 1024}, {"kb", 1024}, {"k", 1024},
+ {"mib", 1024 * 1024}, {"mb", 1024 * 1024}, {"m", 1024 * 1024}};
+ for (const auto &[Suf, M] : Suffixes) {
+ if (Lower.ends_with(Suf)) {
+ Mult = M;
+ S = S.drop_back(Suf.size()).rtrim();
+ break;
+ }
+ }
+ uint64_t Value;
+ if (S.getAsInteger(10, Value))
+ return false;
+ Out = Value * Mult;
+ return true;
+}
+
+// Parse "N" or "MIN:MAX" (also accepts "MIN-MAX") into a flat workgroup range.
+static bool parseWGRange(StringRef S, unsigned &Min, unsigned &Max) {
+ S = S.trim();
+ StringRef LHS, RHS;
+ if (S.contains(':'))
+ std::tie(LHS, RHS) = S.split(':');
+ else if (S.contains('-'))
+ std::tie(LHS, RHS) = S.split('-');
+ else
+ LHS = RHS = S;
+
+ unsigned Lo, Hi;
+ if (LHS.trim().getAsInteger(10, Lo) || RHS.trim().getAsInteger(10, Hi))
+ return false;
+ if (Lo == 0 || Hi == 0 || Lo > Hi)
+ return false;
+ Min = Lo;
+ Max = Hi;
+ return true;
+}
+
+static std::string formatBytes(uint64_t Bytes) {
+ if (Bytes && Bytes % 1024 == 0)
+ return (Twine(Bytes) + " bytes (" + Twine(Bytes / 1024) + " KiB)").str();
+ return (Twine(Bytes) + " bytes").str();
+}
+
+int main(int argc, char **argv) {
+ InitLLVM X(argc, argv);
+
+ cl::HideUnrelatedOptions(OccCategory);
+ cl::ParseCommandLineOptions(
+ argc, argv,
+ "AMDGPU occupancy calculator\n\n"
+ " Prints the occupancy (waves per EU) implied by a given workgroup "
+ "size,\n"
+ " VGPR/SGPR usage and LDS allocation. Unspecified fields are reported "
+ "as\n"
+ " a range.\n");
+
+ LLVMInitializeAMDGPUTargetInfo();
+ LLVMInitializeAMDGPUTarget();
+ LLVMInitializeAMDGPUTargetMC();
+
+ if (MCPU.empty()) {
+ WithColor::error(errs(), "llvm-calc-occupancy")
+ << "no GPU specified; pass -mcpu=<gfxNNN> (e.g. -mcpu=gfx90a)\n";
+ return 1;
+ }
+
+ Triple TT(Triple::normalize(TripleName));
+ if (!TT.isAMDGCN()) {
+ WithColor::error(errs(), "llvm-calc-occupancy")
+ << "this tool only supports the amdgcn target; got triple '" << TT.str()
+ << "'\n";
+ return 1;
+ }
+
+ std::string Error;
+ const Target *T = TargetRegistry::lookupTarget(TT, Error);
+ if (!T) {
+ WithColor::error(errs(), "llvm-calc-occupancy") << Error << "\n";
+ return 1;
+ }
+
+ TargetOptions Options;
+ std::unique_ptr<TargetMachine> TM(T->createTargetMachine(
+ TT, MCPU, MAttr, Options, std::nullopt, std::nullopt));
+ if (!TM) {
+ WithColor::error(errs(), "llvm-calc-occupancy")
----------------
michaelselehov wrote:
Done
https://github.com/llvm/llvm-project/pull/208727
More information about the llvm-commits
mailing list