[llvm] [AMDGPU][NewPass] Attempt to promote uniform ptr arguments to inreg (PR #210410)
Shilei Tian via llvm-commits
llvm-commits at lists.llvm.org
Fri Sep 25 06:37:42 PDT 2026
================
@@ -0,0 +1,160 @@
+//===-- AMDGPUPromoteUniformArgs.cpp --------------------------------------===//
+//
+// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
+// See https://llvm.org/LICENSE.txt for license information.
+// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
+//
+//===----------------------------------------------------------------------===//
+//
+// Promote scalar and pointer arguments of internal callees to \c inreg when
+// every visible call-site operand is trivially uniform: a constant, an
+// argument passed in an SGPR, or an always-uniform intrinsic in the same
+// block as the call. Vectors are not promoted.
+//
+//===----------------------------------------------------------------------===//
+
+#include "AMDGPU.h"
+#include "Utils/AMDGPUBaseInfo.h"
+#include "llvm/ADT/SmallVector.h"
+#include "llvm/ADT/Statistic.h"
+#include "llvm/IR/Analysis.h"
+#include "llvm/IR/Attributes.h"
+#include "llvm/IR/Instructions.h"
+#include "llvm/IR/IntrinsicInst.h"
+#include "llvm/IR/Module.h"
+#include "llvm/IR/Type.h"
+#include "llvm/Support/CommandLine.h"
+#include "llvm/TargetParser/Triple.h"
+
+using namespace llvm;
+
+#define DEBUG_TYPE "amdgpu-promote-uniform-args"
+
+STATISTIC(NumPromotedInRegArgs,
+ "Number of uniform arguments promoted to inreg");
+STATISTIC(NumPromotedInRegFuncs,
+ "Number of functions with a promoted uniform argument");
+
+static cl::opt<bool> EnablePromoteUniformArgs(
+ "amdgpu-enable-promote-uniform-args", cl::Hidden, cl::init(true),
+ cl::desc("Promote provably uniform internal scalar and pointer arguments "
+ "to inreg"));
+
+namespace {
+
+static bool canPromoteArgToInReg(const Argument &A) {
+ Type *Ty = A.getType();
+ if (!(Ty->isIntOrPtrTy() || Ty->isFloatingPointTy()) || A.hasInRegAttr())
+ return false;
+ // inreg is mutually exclusive with byval, inalloca, preallocated, byref,
+ // sret, and nest. The first five are covered by hasPointeeInMemoryValueAttr.
+ if (A.hasPointeeInMemoryValueAttr() || A.hasNestAttr())
+ return false;
+ return !A.hasAttribute("amdgpu-hidden-argument");
+}
+
+static bool isEligibleInRegUniformCallee(const Function &F) {
+ if (F.isDeclaration() || F.isVarArg() || !F.canChangeSignature())
+ return false;
+ if (!F.hasLocalLinkage())
+ return false;
+ switch (F.getCallingConv()) {
+ case CallingConv::C:
+ case CallingConv::Fast:
+ break;
+ default:
+ return false;
+ }
+
+ // A musttail call requires the enclosing function's parameter ABI attributes
+ // to match the callee's positionally, so adding inreg to any parameter of F
+ // breaks the contract, not just to one forwarded to the tail call.
+ for (const BasicBlock &BB : F)
+ for (const Instruction &I : BB)
+ if (const auto *CB = dyn_cast<CallBase>(&I))
+ if (CB->isMustTailCall())
+ return false;
+
+ // Every use must be a direct call to F. This subsumes hasAddressTaken(),
+ // which by default ignores some uses we care about (e.g. assume-like calls),
+ // and it is what lets the transform treat each user as a call site to update.
+ for (const User *U : F.users()) {
+ const auto *CB = dyn_cast<CallBase>(U);
+ if (!CB || CB->getCalledFunction() != &F)
+ return false;
+ if (CB->isMustTailCall() || isa<InvokeInst>(CB))
+ return false;
+ }
+ return !F.user_empty();
+}
+
+static bool isTriviallyUniform(const Use &U) {
----------------
shiltian wrote:
Actually I was referring to `TargetTransformInfo::isAlwaysUniform` for the starting point.
https://github.com/llvm/llvm-project/pull/210410
More information about the llvm-commits
mailing list