[llvm] cd94d18 - [ABI][AMDGPU] Add AMDGPU target ABI classifier to the LLVM ABI library (#220177)
via llvm-commits
llvm-commits at lists.llvm.org
Wed Sep 30 02:15:59 PDT 2026
Author: Chaitanya
Date: 2026-09-30T14:45:53+05:30
New Revision: cd94d181b2e39bfb0720fd383f17a2f0eefed3d0
URL: https://github.com/llvm/llvm-project/commit/cd94d181b2e39bfb0720fd383f17a2f0eefed3d0
DIFF: https://github.com/llvm/llvm-project/commit/cd94d181b2e39bfb0720fd383f17a2f0eefed3d0.diff
LOG: [ABI][AMDGPU] Add AMDGPU target ABI classifier to the LLVM ABI library (#220177)
**Summary:**
- Port the classic CodeGen AMDGPUABIInfo call-convention classifier to
the new language-agnostic LLVM ABI library.
**Changes:**
- LLVM ABI library. AMDGPUTargetInfo implements return, kernel-argument,
and regular-argument classification.
- Unit tests covering the classifier's argument/return branches.
Relates to issue: https://github.com/llvm/llvm-project/issues/220471
Pre-requisite PRs: #224994, #225005, #225009
Assisted by: Claude Opus 4.8
Added:
llvm/lib/ABI/Targets/AMDGPU.cpp
llvm/unittests/ABI/AMDGPUTargetInfoTest.cpp
Modified:
llvm/include/llvm/ABI/TargetInfo.h
llvm/lib/ABI/CMakeLists.txt
llvm/unittests/ABI/CMakeLists.txt
Removed:
################################################################################
diff --git a/llvm/include/llvm/ABI/TargetInfo.h b/llvm/include/llvm/ABI/TargetInfo.h
index b8da6d6afcc07..f7820caff567e 100644
--- a/llvm/include/llvm/ABI/TargetInfo.h
+++ b/llvm/include/llvm/ABI/TargetInfo.h
@@ -145,6 +145,10 @@ class TargetInfo {
LLVM_ABI std::unique_ptr<TargetInfo> createBPFTargetInfo(TypeBuilder &TB);
+LLVM_ABI std::unique_ptr<TargetInfo>
+createAMDGPUTargetInfo(TypeBuilder &TB,
+ bool CoerceGenericPtrArgToGlobal = false);
+
/// The AVX ABI level for X86 targets.
enum class X86AVXABILevel {
None,
diff --git a/llvm/lib/ABI/CMakeLists.txt b/llvm/lib/ABI/CMakeLists.txt
index 831ea0715f7c0..bd097a8e87139 100644
--- a/llvm/lib/ABI/CMakeLists.txt
+++ b/llvm/lib/ABI/CMakeLists.txt
@@ -5,6 +5,7 @@ add_llvm_component_library(LLVMABI
DefaultTargetInfo.cpp
IRTypeMapper.cpp
Targets/AArch64.cpp
+ Targets/AMDGPU.cpp
Targets/BPF.cpp
Targets/X86.cpp
diff --git a/llvm/lib/ABI/Targets/AMDGPU.cpp b/llvm/lib/ABI/Targets/AMDGPU.cpp
new file mode 100644
index 0000000000000..c6488a1c539fd
--- /dev/null
+++ b/llvm/lib/ABI/Targets/AMDGPU.cpp
@@ -0,0 +1,261 @@
+//===- AMDGPU.cpp - AMDGPU ABI Implementation ----------------------------===//
+//
+// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
+// See https://llvm.org/LICENSE.txt for license information.
+// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
+//
+//===----------------------------------------------------------------------===//
+
+#include "llvm/ABI/DefaultTargetInfo.h"
+#include "llvm/ABI/FunctionInfo.h"
+#include "llvm/ABI/TargetInfo.h"
+#include "llvm/ABI/Types.h"
+#include "llvm/Support/AMDGPUAddrSpace.h"
+#include "llvm/Support/Alignment.h"
+#include "llvm/Support/Casting.h"
+#include "llvm/Support/TypeSize.h"
+#include <algorithm>
+#include <cassert>
+#include <cstdint>
+
+namespace llvm {
+namespace abi {
+
+class AMDGPUTargetInfo final : public DefaultTargetInfo {
+private:
+ static const unsigned MaxNumRegsForArgsRet = 16;
+
+ ABICompatInfo CompatInfo;
+
+ /// HIP coerces a generic scalar-pointer kernel argument to the global
+ /// address space. Gated by the front end, which alone can see LangOpts.HIP.
+ bool CoerceGenericPtrArgToGlobal;
+
+ ArgInfo classifyReturnType(const Type *RetTy) const;
+ ArgInfo classifyKernelArgumentType(const Type *Ty) const;
+ ArgInfo classifyArgumentType(const Type *Ty, bool Variadic,
+ unsigned &NumRegsLeft) const;
+
+ /// Estimate number of registers the type will use when passed in registers.
+ uint64_t getNumRegsForType(const Type *Ty) const;
+
+public:
+ AMDGPUTargetInfo(TypeBuilder &TypeBuilder, const ABICompatInfo &Compat,
+ bool CoerceGenericPtrArgToGlobal)
+ : DefaultTargetInfo(TypeBuilder), CompatInfo(Compat),
+ CoerceGenericPtrArgToGlobal(CoerceGenericPtrArgToGlobal) {}
+
+ const ABICompatInfo &getABICompatInfo() const override { return CompatInfo; }
+
+ /// Indirect arguments live in the private (alloca) address space on AMDGPU.
+ unsigned getAllocaAddrSpace() const override {
+ return AMDGPUAS::PRIVATE_ADDRESS;
+ }
+
+ void computeInfo(FunctionInfo &FI) const override;
+};
+
+uint64_t AMDGPUTargetInfo::getNumRegsForType(const Type *Ty) const {
+ uint64_t NumRegs = 0;
+
+ if (const auto *VT = dyn_cast<VectorType>(Ty)) {
+ // Compute from the number of elements. The reported size is based on the
+ // in-memory size, which includes the padding 4th element for 3-vectors.
+ const Type *EltTy = VT->getElementType();
+ uint64_t EltSize = EltTy->getSizeInBits().getFixedValue();
+ unsigned NumElts = VT->getNumElements().getFixedValue();
+
+ // 16-bit element vectors should be passed as packed.
+ if (EltSize == 16)
+ return (NumElts + 1) / 2;
+
+ uint64_t EltNumRegs = (EltSize + 31) / 32;
+ return EltNumRegs * NumElts;
+ }
+
+ if (const auto *RT = dyn_cast<RecordType>(Ty)) {
+ for (const FieldInfo &Field : RT->getFields())
+ NumRegs += getNumRegsForType(Field.FieldType);
+ return NumRegs;
+ }
+
+ return (Ty->getSizeInBits().getFixedValue() + 31) / 32;
+}
+
+ArgInfo AMDGPUTargetInfo::classifyReturnType(const Type *RetTy) const {
+ if (RetTy->isVoid())
+ return ArgInfo::getIgnore();
+
+ if (isAggregateTypeForABI(RetTy)) {
+ // Records with non-trivial destructors/copy-constructors should not be
+ // returned by value.
+ if (getRecordArgABI(RetTy) == RAA_Default) {
+ const auto *RT = dyn_cast<RecordType>(RetTy);
+
+ // Ignore empty structs/unions.
+ if (RT && RT->isEmpty())
+ return ArgInfo::getIgnore();
+
+ // Lower single-element structs to just return a regular value.
+ if (const Type *SeltTy = isSingleElementStruct(RetTy))
+ return ArgInfo::getDirect(SeltTy);
+
+ if (RT && RT->hasFlexibleArrayMember())
+ return DefaultTargetInfo::classifyReturnType(RetTy);
+
+ // Pack aggregates <= 4 bytes into single VGPR or pair.
+ uint64_t Size = RetTy->getSizeInBits().getFixedValue();
+ if (Size <= 16)
+ return ArgInfo::getDirect(TB.getIntegerType(16, Align(2), false));
+
+ if (Size <= 32)
+ return ArgInfo::getDirect(TB.getIntegerType(32, Align(4), false));
+
+ if (Size <= 64) {
+ const Type *I32Ty = TB.getIntegerType(32, Align(4), false);
+ return ArgInfo::getDirect(TB.getArrayType(I32Ty, 2, /*SizeInBits=*/64));
+ }
+
+ if (getNumRegsForType(RetTy) <= MaxNumRegsForArgsRet)
+ return ArgInfo::getDirect();
+ }
+ }
+
+ // Otherwise just do the default thing.
+ return DefaultTargetInfo::classifyReturnType(RetTy);
+}
+
+/// For kernels all parameters are really passed in a special buffer. It doesn't
+/// make sense to pass anything byval, so everything must be direct.
+ArgInfo AMDGPUTargetInfo::classifyKernelArgumentType(const Type *Ty) const {
+ Ty = useFirstFieldIfTransparentUnion(Ty);
+
+ if (const Type *SeltTy = isSingleElementStruct(Ty))
+ Ty = SeltTy;
+
+ // HIP passes a generic scalar pointer as a global pointer; a pointer is not
+ // an aggregate, so this stays on the direct path.
+ if (CoerceGenericPtrArgToGlobal) {
+ if (const auto *PtrTy = dyn_cast<PointerType>(Ty);
+ PtrTy && PtrTy->getAddrSpace() == AMDGPUAS::FLAT_ADDRESS) {
+ const Type *Coerced =
+ TB.getPointerType(PtrTy->getSizeInBits().getFixedValue(),
+ PtrTy->getAlignment(), AMDGPUAS::GLOBAL_ADDRESS);
+ return ArgInfo::getDirect(Coerced, /*Offset=*/0, /*Align=*/std::nullopt,
+ /*CanBeFlattened=*/false);
+ }
+ }
+
+ // FIXME: This doesn't apply the optimization of coercing pointers in structs
+ // to global address space when using byref. This would require implementing a
+ // new kind of coercion of the in-memory type when for indirect arguments.
+ if (isAggregateTypeForABI(Ty))
+ return ArgInfo::getIndirectAliased(
+ Ty->getAlignment(),
+ /*AddrSpace=*/AMDGPUAS::CONSTANT_ADDRESS);
+
+ // CanBeFlattened=false keeps the struct intact.
+ return ArgInfo::getDirect(Ty, /*Offset=*/0, /*Align=*/std::nullopt,
+ /*CanBeFlattened=*/false);
+}
+
+ArgInfo AMDGPUTargetInfo::classifyArgumentType(const Type *Ty, bool Variadic,
+ unsigned &NumRegsLeft) const {
+ assert(NumRegsLeft <= MaxNumRegsForArgsRet && "register estimate underflow");
+
+ Ty = useFirstFieldIfTransparentUnion(Ty);
+
+ // Variadic aggregates are kept intact rather than flattened into fields.
+ if (Variadic)
+ return ArgInfo::getDirect(/*T=*/nullptr, /*Offset=*/0,
+ /*Align=*/std::nullopt, /*CanBeFlattened=*/false);
+
+ if (isAggregateTypeForABI(Ty)) {
+ // Records with non-trivial destructors/copy-constructors should not be
+ // passed by value.
+ if (RecordArgABI RAA = getRecordArgABI(Ty); RAA != RAA_Default)
+ return ArgInfo::getIndirect(Ty->getAlignment(),
+ /*ByVal=*/RAA == RAA_DirectInMemory,
+ /*AddrSpace=*/AMDGPUAS::PRIVATE_ADDRESS);
+
+ // Ignore empty structs/unions.
+ if (Ty->isEmptyRecord())
+ return ArgInfo::getIgnore();
+
+ // Lower single-element structs to just pass a regular value.
+ if (const Type *SeltTy = isSingleElementStruct(Ty))
+ return ArgInfo::getDirect(SeltTy);
+
+ if (const auto *RT = dyn_cast<RecordType>(Ty);
+ RT && RT->hasFlexibleArrayMember())
+ return DefaultTargetInfo::classifyArgumentType(Ty);
+
+ // Pack aggregates <= 8 bytes into single VGPR or pair.
+ uint64_t Size = Ty->getSizeInBits().getFixedValue();
+ if (Size <= 64) {
+ unsigned NumRegs = (Size + 31) / 32;
+ NumRegsLeft -= std::min(NumRegsLeft, NumRegs);
+
+ if (Size <= 16)
+ return ArgInfo::getDirect(TB.getIntegerType(16, Align(2), false));
+
+ if (Size <= 32)
+ return ArgInfo::getDirect(TB.getIntegerType(32, Align(4), false));
+
+ const Type *I32Ty = TB.getIntegerType(32, Align(4), false);
+ return ArgInfo::getDirect(TB.getArrayType(I32Ty, 2, /*SizeInBits=*/64));
+ }
+
+ if (NumRegsLeft > 0) {
+ uint64_t NumRegs = getNumRegsForType(Ty);
+ if (NumRegsLeft >= NumRegs) {
+ NumRegsLeft -= NumRegs;
+ return ArgInfo::getDirect();
+ }
+ }
+
+ // Pass a struct argument by reference rather than by value.
+ return ArgInfo::getIndirectAliased(Ty->getAlignment(),
+ /*AddrSpace=*/AMDGPUAS::PRIVATE_ADDRESS);
+ }
+
+ // Otherwise just do the default thing.
+ ArgInfo AI = DefaultTargetInfo::classifyArgumentType(Ty);
+ if (!AI.isIndirect()) {
+ uint64_t NumRegs = getNumRegsForType(Ty);
+ NumRegsLeft -= std::min(NumRegs, uint64_t{NumRegsLeft});
+ }
+
+ return AI;
+}
+
+void AMDGPUTargetInfo::computeInfo(FunctionInfo &FI) const {
+ CallingConv::ID CC = FI.getCallingConvention();
+
+ // Non-trivial C++ records are returned indirectly
+ // in the flat address space.
+ if (!maybeCommonClassifyReturnType(FI))
+ FI.getReturnInfo() = classifyReturnType(FI.getReturnType());
+
+ unsigned ArgumentIndex = 0;
+ const unsigned NumFixedArguments = FI.getNumRequiredArgs();
+
+ unsigned NumRegsLeft = MaxNumRegsForArgsRet;
+ for (ArgEntry &Arg : FI.arguments()) {
+ if (CC == CallingConv::AMDGPU_KERNEL) {
+ Arg.Info = classifyKernelArgumentType(Arg.ABIType);
+ } else {
+ bool FixedArgument = ArgumentIndex++ < NumFixedArguments;
+ Arg.Info = classifyArgumentType(Arg.ABIType, !FixedArgument, NumRegsLeft);
+ }
+ }
+}
+
+std::unique_ptr<TargetInfo>
+createAMDGPUTargetInfo(TypeBuilder &TB, bool CoerceGenericPtrArgToGlobal) {
+ return std::make_unique<AMDGPUTargetInfo>(TB, ABICompatInfo(),
+ CoerceGenericPtrArgToGlobal);
+}
+
+} // namespace abi
+} // namespace llvm
diff --git a/llvm/unittests/ABI/AMDGPUTargetInfoTest.cpp b/llvm/unittests/ABI/AMDGPUTargetInfoTest.cpp
new file mode 100644
index 0000000000000..ecb298b69215e
--- /dev/null
+++ b/llvm/unittests/ABI/AMDGPUTargetInfoTest.cpp
@@ -0,0 +1,499 @@
+//===- AMDGPUTargetInfoTest.cpp - AMDGPU ABI unit tests -------------------===//
+//
+// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
+// See https://llvm.org/LICENSE.txt for license information.
+// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
+//
+//===----------------------------------------------------------------------===//
+
+#include "llvm/ABI/FunctionInfo.h"
+#include "llvm/ABI/TargetInfo.h"
+#include "llvm/ABI/Types.h"
+#include "llvm/ADT/APFloat.h"
+#include "llvm/IR/CallingConv.h"
+#include "llvm/Support/AMDGPUAddrSpace.h"
+#include "llvm/Support/Alignment.h"
+#include "llvm/Support/Allocator.h"
+#include "gtest/gtest.h"
+
+namespace {
+
+// RecordFlags' bitmask operators are declared in namespace llvm, so combining
+// two of them needs that namespace visible.
+using namespace llvm;
+
+using ABIType = llvm::abi::Type;
+using llvm::abi::ABICompatInfo;
+using llvm::abi::ArgInfo;
+using llvm::abi::createAMDGPUTargetInfo;
+using llvm::abi::FieldInfo;
+using llvm::abi::FunctionInfo;
+using llvm::abi::RecordFlags;
+using llvm::abi::RequiredArgs;
+using llvm::abi::StructPacking;
+using llvm::abi::TargetInfo;
+using llvm::abi::TypeBuilder;
+
+class AMDGPUTargetInfoTest : public ::testing::Test {
+protected:
+ llvm::BumpPtrAllocator Alloc;
+ TypeBuilder TB;
+ const ABIType *I8;
+ const ABIType *I16;
+ const ABIType *I32;
+ const ABIType *F32;
+ const ABIType *Void;
+ /// An empty class: a record with no fields, one byte wide, register-passable.
+ const ABIType *Empty;
+
+ AMDGPUTargetInfoTest()
+ : TB(Alloc), I8(TB.getIntegerType(8, llvm::Align(1), /*Signed=*/true)),
+ I16(TB.getIntegerType(16, llvm::Align(2), /*Signed=*/true)),
+ I32(TB.getIntegerType(32, llvm::Align(4), /*Signed=*/true)),
+ F32(TB.getFloatType(llvm::APFloat::IEEEsingle(), llvm::Align(4))),
+ Void(TB.getVoidType()),
+ Empty(TB.getRecordType({}, llvm::TypeSize::getFixed(8), llvm::Align(1),
+ llvm::Align(1), StructPacking::Default, {}, {},
+ RecordFlags::CanPassInRegisters)) {}
+
+ std::unique_ptr<TargetInfo> target() const {
+ return createAMDGPUTargetInfo(const_cast<TypeBuilder &>(TB));
+ }
+
+ /// A target configured like a HIP compilation: generic scalar-pointer kernel
+ /// arguments are coerced to the global address space.
+ std::unique_ptr<TargetInfo> hipTarget() const {
+ return createAMDGPUTargetInfo(const_cast<TypeBuilder &>(TB),
+ /*CoerceGenericPtrArgToGlobal=*/true);
+ }
+
+ /// The classification a kernel argument gets on target \p TI.
+ const ArgInfo &classifyKernelArg(const ABIType *ArgTy,
+ std::unique_ptr<FunctionInfo> &FI,
+ std::unique_ptr<TargetInfo> &TI) {
+ FI = FunctionInfo::create(CallingConv::AMDGPU_KERNEL, Void, {ArgTy});
+ TI->computeInfo(*FI);
+ return FI->getArgInfo(0).Info;
+ }
+
+ /// A register-passable record with the given fields, size and alignment.
+ const ABIType *recordOf(llvm::ArrayRef<FieldInfo> Fields, uint64_t SizeInBits,
+ llvm::Align Alignment) {
+ return TB.getRecordType(Fields, llvm::TypeSize::getFixed(SizeInBits),
+ Alignment, Alignment, StructPacking::Default, {},
+ {}, RecordFlags::CanPassInRegisters);
+ }
+
+ /// The argument classification the target computes for a single parameter
+ /// under calling convention \p CC.
+ const ArgInfo &classifyArg(const ABIType *ArgTy,
+ std::unique_ptr<FunctionInfo> &FI,
+ std::unique_ptr<TargetInfo> &TI,
+ CallingConv::ID CC = CallingConv::C) {
+ TI = target();
+ FI = FunctionInfo::create(CC, Void, {ArgTy});
+ TI->computeInfo(*FI);
+ return FI->getArgInfo(0).Info;
+ }
+
+ /// The return classification the target computes for \p RetTy.
+ const ArgInfo &classifyRet(const ABIType *RetTy,
+ std::unique_ptr<FunctionInfo> &FI,
+ std::unique_ptr<TargetInfo> &TI) {
+ TI = target();
+ FI = FunctionInfo::create(CallingConv::C, RetTy, {});
+ TI->computeInfo(*FI);
+ return FI->getReturnInfo();
+ }
+};
+
+// A direct arg coerced to a pointer in address space \p AS.
+static void expectDirectPointerAS(const ArgInfo &Info, unsigned AS) {
+ ASSERT_TRUE(Info.isDirect());
+ const auto *PT =
+ llvm::dyn_cast_or_null<llvm::abi::PointerType>(Info.getCoerceToType());
+ ASSERT_NE(PT, nullptr);
+ EXPECT_EQ(PT->getAddrSpace(), AS);
+}
+
+static void expectUncoercedDirect(const ArgInfo &Info) {
+ ASSERT_TRUE(Info.isDirect());
+ EXPECT_EQ(Info.getCoerceToType(), nullptr);
+}
+
+static void expectDirectInteger(const ArgInfo &Info, unsigned Bits) {
+ ASSERT_TRUE(Info.isDirect());
+ const ABIType *Coerce = Info.getCoerceToType();
+ ASSERT_NE(Coerce, nullptr);
+ const auto *IT = llvm::dyn_cast<llvm::abi::IntegerType>(Coerce);
+ ASSERT_NE(IT, nullptr);
+ EXPECT_EQ(IT->getSizeInBits().getFixedValue(), Bits);
+}
+
+static void expectDirectFloat(const ArgInfo &Info,
+ const llvm::fltSemantics &Sem) {
+ ASSERT_TRUE(Info.isDirect());
+ const ABIType *Coerce = Info.getCoerceToType();
+ ASSERT_NE(Coerce, nullptr);
+ const auto *FT = llvm::dyn_cast<llvm::abi::FloatType>(Coerce);
+ ASSERT_NE(FT, nullptr);
+ EXPECT_EQ(FT->getSemantics(), &Sem);
+}
+
+// A <= 8-byte aggregate coerces to [2 x i32].
+static void expectDirectI32Pair(const ArgInfo &Info) {
+ ASSERT_TRUE(Info.isDirect());
+ const auto *AT =
+ llvm::dyn_cast_or_null<llvm::abi::ArrayType>(Info.getCoerceToType());
+ ASSERT_NE(AT, nullptr);
+ EXPECT_EQ(AT->getNumElements(), 2u);
+ const auto *IT = llvm::dyn_cast<llvm::abi::IntegerType>(AT->getElementType());
+ ASSERT_NE(IT, nullptr);
+ EXPECT_EQ(IT->getSizeInBits().getFixedValue(), 32u);
+}
+
+static void expectIndirect(const ArgInfo &Info, llvm::Align ExpectedAlign,
+ bool ByVal, unsigned AddrSpace) {
+ ASSERT_TRUE(Info.isIndirect());
+ EXPECT_EQ(Info.getIndirectAlign(), ExpectedAlign);
+ EXPECT_EQ(Info.getIndirectByVal(), ByVal);
+ EXPECT_EQ(Info.getIndirectAddrSpace(), AddrSpace);
+}
+
+// Aliased indirect carries an address space but no byval, since the pointer
+// refers to an object the caller owns.
+static void expectIndirectAliased(const ArgInfo &Info,
+ llvm::Align ExpectedAlign,
+ unsigned AddrSpace) {
+ ASSERT_TRUE(Info.isIndirectAliased());
+ EXPECT_EQ(Info.getIndirectAlign(), ExpectedAlign);
+ EXPECT_EQ(Info.getIndirectAddrSpace(), AddrSpace);
+}
+
+//===----------------------------------------------------------------------===//
+// Argument classification
+//===----------------------------------------------------------------------===//
+
+// A 32-bit integer and a pointer-sized scalar pass directly in their own type.
+TEST_F(AMDGPUTargetInfoTest, ScalarPassesDirect) {
+ std::unique_ptr<FunctionInfo> FI;
+ std::unique_ptr<TargetInfo> TI;
+ const ArgInfo &Info = classifyArg(I32, FI, TI);
+ expectUncoercedDirect(Info);
+ // Ordinary direct args stay flattenable (the getDirect default).
+ EXPECT_TRUE(Info.getCanBeFlattened());
+ expectUncoercedDirect(classifyArg(F32, FI, TI));
+}
+
+// A sub-word integer is sign/zero extended to fill its register.
+TEST_F(AMDGPUTargetInfoTest, PromotableIntegerExtends) {
+ std::unique_ptr<FunctionInfo> FI;
+ std::unique_ptr<TargetInfo> TI;
+ const ArgInfo &Info = classifyArg(I8, FI, TI);
+ ASSERT_TRUE(Info.isExtend());
+ EXPECT_TRUE(Info.isSignExt());
+}
+
+// An oversized _BitInt (> 128 bits) is passed indirectly, byval.
+// The 128-bit boundary width stays direct.
+TEST_F(AMDGPUTargetInfoTest, OversizedBitIntArgumentIsIndirect) {
+ std::unique_ptr<FunctionInfo> FI;
+ std::unique_ptr<TargetInfo> TI;
+ const ABIType *BitInt256 = TB.getIntegerType(256, llvm::Align(16),
+ /*Signed=*/true,
+ /*IsBitInt=*/true);
+ expectIndirect(classifyArg(BitInt256, FI, TI), llvm::Align(16),
+ /*ByVal=*/true, llvm::AMDGPUAS::PRIVATE_ADDRESS);
+
+ const ABIType *BitInt128 = TB.getIntegerType(128, llvm::Align(16),
+ /*Signed=*/true,
+ /*IsBitInt=*/true);
+ EXPECT_TRUE(classifyArg(BitInt128, FI, TI).isDirect());
+}
+
+// Aggregates <= 16/32/64 bits pack into i16 / i32 / [2 x i32].
+TEST_F(AMDGPUTargetInfoTest, SmallAggregatesPackIntoRegisters) {
+ std::unique_ptr<FunctionInfo> FI;
+ std::unique_ptr<TargetInfo> TI;
+
+ const ABIType *S16 =
+ recordOf({FieldInfo(I8, 0), FieldInfo(I8, 8)}, 16, llvm::Align(1));
+ expectDirectInteger(classifyArg(S16, FI, TI), 16);
+
+ const ABIType *S32 =
+ recordOf({FieldInfo(I16, 0), FieldInfo(I16, 16)}, 32, llvm::Align(2));
+ expectDirectInteger(classifyArg(S32, FI, TI), 32);
+
+ const ABIType *S64 =
+ recordOf({FieldInfo(I32, 0), FieldInfo(I32, 32)}, 64, llvm::Align(4));
+ expectDirectI32Pair(classifyArg(S64, FI, TI));
+}
+
+// A mid-sized aggregate (> 8 bytes) that still fits the 16-register budget is
+// passed and returned directly, uncoerced.
+TEST_F(AMDGPUTargetInfoTest, MidSizeAggregateFitsInRegistersIsDirect) {
+ std::unique_ptr<FunctionInfo> FI;
+ std::unique_ptr<TargetInfo> TI;
+ // Two i32[4] fields => 8 registers, within MaxNumRegsForArgsRet (16).
+ const ABIType *Arr =
+ TB.getArrayType(I32, /*NumElements=*/4, /*SizeInBits=*/128);
+ const ABIType *Mid =
+ recordOf({FieldInfo(Arr, 0), FieldInfo(Arr, 128)}, 256, llvm::Align(4));
+ expectUncoercedDirect(classifyArg(Mid, FI, TI));
+ expectUncoercedDirect(classifyRet(Mid, FI, TI));
+}
+
+// An empty struct is dropped from the argument list.
+TEST_F(AMDGPUTargetInfoTest, EmptyAggregateIsIgnored) {
+ std::unique_ptr<FunctionInfo> FI;
+ std::unique_ptr<TargetInfo> TI;
+ EXPECT_TRUE(classifyArg(Empty, FI, TI).isIgnore());
+}
+
+// A single-element struct is passed as its inner scalar.
+TEST_F(AMDGPUTargetInfoTest, SingleElementStructUnwraps) {
+ std::unique_ptr<FunctionInfo> FI;
+ std::unique_ptr<TargetInfo> TI;
+ const ABIType *Wrapper = recordOf({FieldInfo(F32, 0)}, 32, llvm::Align(4));
+ expectDirectFloat(classifyArg(Wrapper, FI, TI), llvm::APFloat::IEEEsingle());
+}
+
+// A large aggregate that does not fit the 16-register budget is passed by
+// reference (aliased) in the private address space.
+TEST_F(AMDGPUTargetInfoTest, OversizedAggregateIsIndirectPrivate) {
+ std::unique_ptr<FunctionInfo> FI;
+ std::unique_ptr<TargetInfo> TI;
+ // Two i32[10] fields => 20 registers, above MaxNumRegsForArgsRet (16).
+ const ABIType *ArrTy = TB.getArrayType(I32, /*NumElements=*/10,
+ /*SizeInBits=*/320);
+ const ABIType *Big = recordOf({FieldInfo(ArrTy, 0), FieldInfo(ArrTy, 320)},
+ 640, llvm::Align(4));
+ expectIndirectAliased(classifyArg(Big, FI, TI), llvm::Align(4),
+ llvm::AMDGPUAS::PRIVATE_ADDRESS);
+}
+
+// Once the 16-register budget is spent by earlier arguments, a later aggregate
+// that would otherwise fit is passed by reference (aliased) in private memory.
+TEST_F(AMDGPUTargetInfoTest, ArgumentRegisterBudgetExhaustionForcesIndirect) {
+ std::unique_ptr<TargetInfo> TI = target();
+ // Each argument uses 8 of the 16 available registers.
+ const ABIType *Arr =
+ TB.getArrayType(I32, /*NumElements=*/4, /*SizeInBits=*/128);
+ const ABIType *Mid =
+ recordOf({FieldInfo(Arr, 0), FieldInfo(Arr, 128)}, 256, llvm::Align(4));
+ std::unique_ptr<FunctionInfo> FI =
+ FunctionInfo::create(CallingConv::C, Void, {Mid, Mid, Mid});
+ TI->computeInfo(*FI);
+ // The first two fit and pass directly.
+ expectUncoercedDirect(FI->getArgInfo(0).Info);
+ expectUncoercedDirect(FI->getArgInfo(1).Info);
+ // The third exhausts the budget and is passed by reference.
+ expectIndirectAliased(FI->getArgInfo(2).Info, llvm::Align(4),
+ llvm::AMDGPUAS::PRIVATE_ADDRESS);
+}
+
+// A record that cannot pass in registers (non-trivial C++ type) is passed
+// indirectly in the private address space.
+TEST_F(AMDGPUTargetInfoTest, NonTrivialRecordIsIndirectPrivate) {
+ std::unique_ptr<FunctionInfo> FI;
+ std::unique_ptr<TargetInfo> TI;
+ const ABIType *CannotPass = TB.getRecordType(
+ {FieldInfo(I32, 0)}, llvm::TypeSize::getFixed(32), llvm::Align(4),
+ llvm::Align(4), StructPacking::Default, {}, {}, RecordFlags::IsCXXRecord);
+ expectIndirect(classifyArg(CannotPass, FI, TI), llvm::Align(4),
+ /*ByVal=*/false, llvm::AMDGPUAS::PRIVATE_ADDRESS);
+}
+
+// A record with a flexible array member falls back to the default classifier,
+// so it is passed indirectly (byval) in private memory.
+TEST_F(AMDGPUTargetInfoTest, FlexibleArrayMemberFallsBackToDefault) {
+ std::unique_ptr<FunctionInfo> FI;
+ std::unique_ptr<TargetInfo> TI;
+ const ABIType *Flex = TB.getRecordType(
+ {FieldInfo(I32, 0), FieldInfo(I32, 32)}, llvm::TypeSize::getFixed(64),
+ llvm::Align(4), llvm::Align(4), StructPacking::Default, {}, {},
+ RecordFlags::CanPassInRegisters | RecordFlags::HasFlexibleArrayMember);
+ expectIndirect(classifyArg(Flex, FI, TI), llvm::Align(4), /*ByVal=*/true,
+ llvm::AMDGPUAS::PRIVATE_ADDRESS);
+}
+
+// A variadic argument bypasses register packing and passes through unchanged,
+// kept intact rather than flattened into per-field wire arguments.
+TEST_F(AMDGPUTargetInfoTest, VariadicArgumentPassesDirect) {
+ std::unique_ptr<TargetInfo> TI = target();
+ const ABIType *S16 =
+ recordOf({FieldInfo(I8, 0), FieldInfo(I8, 8)}, 16, llvm::Align(1));
+ // Zero declared parameters, so the sole argument is variadic.
+ std::unique_ptr<FunctionInfo> FI =
+ FunctionInfo::create(CallingConv::C, Void, {S16}, RequiredArgs(0));
+ TI->computeInfo(*FI);
+ expectUncoercedDirect(FI->getArgInfo(0).Info);
+ EXPECT_FALSE(FI->getArgInfo(0).Info.getCanBeFlattened());
+}
+
+//===----------------------------------------------------------------------===//
+// Return classification
+//===----------------------------------------------------------------------===//
+
+// A void return is ignored.
+TEST_F(AMDGPUTargetInfoTest, VoidReturnIsIgnored) {
+ std::unique_ptr<FunctionInfo> FI;
+ std::unique_ptr<TargetInfo> TI;
+ EXPECT_TRUE(classifyRet(Void, FI, TI).isIgnore());
+}
+
+// An empty struct return is ignored, like an empty argument.
+TEST_F(AMDGPUTargetInfoTest, EmptyAggregateReturnIsIgnored) {
+ std::unique_ptr<FunctionInfo> FI;
+ std::unique_ptr<TargetInfo> TI;
+ EXPECT_TRUE(classifyRet(Empty, FI, TI).isIgnore());
+}
+
+// A single-element struct return is unwrapped to its inner scalar.
+TEST_F(AMDGPUTargetInfoTest, SingleElementStructReturnUnwraps) {
+ std::unique_ptr<FunctionInfo> FI;
+ std::unique_ptr<TargetInfo> TI;
+ const ABIType *Wrapper = recordOf({FieldInfo(F32, 0)}, 32, llvm::Align(4));
+ expectDirectFloat(classifyRet(Wrapper, FI, TI), llvm::APFloat::IEEEsingle());
+}
+
+// Aggregate returns <= 16/32/64 bits pack into i16 / i32 / [2 x i32].
+TEST_F(AMDGPUTargetInfoTest, SmallAggregateReturnPacks) {
+ std::unique_ptr<FunctionInfo> FI;
+ std::unique_ptr<TargetInfo> TI;
+
+ const ABIType *S16 =
+ recordOf({FieldInfo(I8, 0), FieldInfo(I8, 8)}, 16, llvm::Align(1));
+ expectDirectInteger(classifyRet(S16, FI, TI), 16);
+
+ const ABIType *S32 =
+ recordOf({FieldInfo(I16, 0), FieldInfo(I16, 16)}, 32, llvm::Align(2));
+ expectDirectInteger(classifyRet(S32, FI, TI), 32);
+
+ const ABIType *S64 =
+ recordOf({FieldInfo(I32, 0), FieldInfo(I32, 32)}, 64, llvm::Align(4));
+ expectDirectI32Pair(classifyRet(S64, FI, TI));
+}
+
+// An oversized _BitInt return is indirect too. Classic DefaultABIInfo's
+// getNaturalAlignIndirect defaults ByVal to true, so returns carry it too.
+TEST_F(AMDGPUTargetInfoTest, OversizedBitIntReturnIsIndirect) {
+ std::unique_ptr<FunctionInfo> FI;
+ std::unique_ptr<TargetInfo> TI;
+ const ABIType *BitInt256 = TB.getIntegerType(256, llvm::Align(16),
+ /*Signed=*/true,
+ /*IsBitInt=*/true);
+ expectIndirect(classifyRet(BitInt256, FI, TI), llvm::Align(16),
+ /*ByVal=*/true, llvm::AMDGPUAS::PRIVATE_ADDRESS);
+}
+
+// A large trivial aggregate return that overflows the register budget is
+// returned indirectly (byval) in the private address space. This is distinct
+// from a non-trivial return, which lands in the flat address space.
+TEST_F(AMDGPUTargetInfoTest, OversizedAggregateReturnIsIndirectPrivate) {
+ std::unique_ptr<FunctionInfo> FI;
+ std::unique_ptr<TargetInfo> TI;
+ // Two i32[10] fields => 20 registers, above MaxNumRegsForArgsRet (16).
+ const ABIType *ArrTy =
+ TB.getArrayType(I32, /*NumElements=*/10, /*SizeInBits=*/320);
+ const ABIType *Big = recordOf({FieldInfo(ArrTy, 0), FieldInfo(ArrTy, 320)},
+ 640, llvm::Align(4));
+ expectIndirect(classifyRet(Big, FI, TI), llvm::Align(4), /*ByVal=*/true,
+ llvm::AMDGPUAS::PRIVATE_ADDRESS);
+}
+
+// A non-trivial C++ record return is indirect in the generic (flat) address
+// space with ByVal=false, matching classic's getSRetAddrSpace(LangAS::Default).
+TEST_F(AMDGPUTargetInfoTest, NonTrivialRecordReturnIsIndirectFlat) {
+ std::unique_ptr<FunctionInfo> FI;
+ std::unique_ptr<TargetInfo> TI;
+ const ABIType *CannotPass = TB.getRecordType(
+ {FieldInfo(I32, 0)}, llvm::TypeSize::getFixed(32), llvm::Align(4),
+ llvm::Align(4), StructPacking::Default, {}, {}, RecordFlags::IsCXXRecord);
+ expectIndirect(classifyRet(CannotPass, FI, TI), llvm::Align(4),
+ /*ByVal=*/false, llvm::AMDGPUAS::FLAT_ADDRESS);
+}
+
+// A record that cannot pass in registers is returned indirectly (sret-style,
+// ByVal=false) via the target-independent return rule.
+TEST_F(AMDGPUTargetInfoTest, NonTrivialRecordReturnIsIndirect) {
+ std::unique_ptr<FunctionInfo> FI;
+ std::unique_ptr<TargetInfo> TI;
+ const ABIType *CannotPass = TB.getRecordType(
+ {FieldInfo(I32, 0)}, llvm::TypeSize::getFixed(32), llvm::Align(4),
+ llvm::Align(4), StructPacking::Default, {}, {}, RecordFlags::IsCXXRecord);
+ const ArgInfo &Info = classifyRet(CannotPass, FI, TI);
+ ASSERT_TRUE(Info.isIndirect());
+ EXPECT_FALSE(Info.getIndirectByVal());
+}
+
+//===----------------------------------------------------------------------===//
+// Kernel argument classification
+//===----------------------------------------------------------------------===//
+
+// Kernel direct arguments are passed directly and kept intact.
+TEST_F(AMDGPUTargetInfoTest, KernelScalarIsDirect) {
+ std::unique_ptr<FunctionInfo> FI;
+ std::unique_ptr<TargetInfo> TI;
+ const ArgInfo &Info = classifyArg(I32, FI, TI, CallingConv::AMDGPU_KERNEL);
+ expectDirectInteger(Info, 32);
+ EXPECT_FALSE(Info.getCanBeFlattened());
+}
+
+// A single-element struct kernel argument is unwrapped and passed directly,
+// kept intact.
+TEST_F(AMDGPUTargetInfoTest, KernelSingleElementStructUnwraps) {
+ std::unique_ptr<FunctionInfo> FI;
+ std::unique_ptr<TargetInfo> TI;
+ const ABIType *Wrapper = recordOf({FieldInfo(F32, 0)}, 32, llvm::Align(4));
+ const ArgInfo &Info =
+ classifyArg(Wrapper, FI, TI, CallingConv::AMDGPU_KERNEL);
+ expectDirectFloat(Info, llvm::APFloat::IEEEsingle());
+ EXPECT_FALSE(Info.getCanBeFlattened());
+}
+
+// Kernel aggregate arguments are passed by reference (aliased) in the constant
+// address space, never byval.
+TEST_F(AMDGPUTargetInfoTest, KernelAggregateIsIndirectConstant) {
+ std::unique_ptr<FunctionInfo> FI;
+ std::unique_ptr<TargetInfo> TI;
+ const ABIType *S =
+ recordOf({FieldInfo(I32, 0), FieldInfo(I32, 32)}, 64, llvm::Align(4));
+ expectIndirectAliased(classifyArg(S, FI, TI, CallingConv::AMDGPU_KERNEL),
+ llvm::Align(4), llvm::AMDGPUAS::CONSTANT_ADDRESS);
+}
+
+// Under HIP a generic scalar-pointer kernel argument is coerced to a global
+// pointer and passed directly, kept intact.
+TEST_F(AMDGPUTargetInfoTest, KernelGenericPointerCoercesToGlobalUnderHIP) {
+ std::unique_ptr<FunctionInfo> FI;
+ std::unique_ptr<TargetInfo> TI = hipTarget();
+ const ABIType *GenericPtr =
+ TB.getPointerType(64, llvm::Align(8), llvm::AMDGPUAS::FLAT_ADDRESS);
+ const ArgInfo &Info = classifyKernelArg(GenericPtr, FI, TI);
+ expectDirectPointerAS(Info, llvm::AMDGPUAS::GLOBAL_ADDRESS);
+ EXPECT_FALSE(Info.getCanBeFlattened());
+}
+
+// A pointer already in the global address space is left untouched under HIP.
+TEST_F(AMDGPUTargetInfoTest, KernelGlobalPointerUnchangedUnderHIP) {
+ std::unique_ptr<FunctionInfo> FI;
+ std::unique_ptr<TargetInfo> TI = hipTarget();
+ const ABIType *GlobalPtr =
+ TB.getPointerType(64, llvm::Align(8), llvm::AMDGPUAS::GLOBAL_ADDRESS);
+ expectDirectPointerAS(classifyKernelArg(GlobalPtr, FI, TI),
+ llvm::AMDGPUAS::GLOBAL_ADDRESS);
+}
+
+// Without the HIP rule a generic pointer kernel argument stays generic.
+TEST_F(AMDGPUTargetInfoTest, KernelGenericPointerUnchangedByDefault) {
+ std::unique_ptr<FunctionInfo> FI;
+ std::unique_ptr<TargetInfo> TI;
+ const ABIType *GenericPtr =
+ TB.getPointerType(64, llvm::Align(8), llvm::AMDGPUAS::FLAT_ADDRESS);
+ expectDirectPointerAS(
+ classifyArg(GenericPtr, FI, TI, CallingConv::AMDGPU_KERNEL),
+ llvm::AMDGPUAS::FLAT_ADDRESS);
+}
+
+} // namespace
diff --git a/llvm/unittests/ABI/CMakeLists.txt b/llvm/unittests/ABI/CMakeLists.txt
index 07c1cdbb9d80e..eb50b88b87d2d 100644
--- a/llvm/unittests/ABI/CMakeLists.txt
+++ b/llvm/unittests/ABI/CMakeLists.txt
@@ -9,6 +9,7 @@ add_llvm_unittest(ABITests
IRTypeMapperTest.cpp
FunctionInfoTest.cpp
TargetInfoTest.cpp
+ AMDGPUTargetInfoTest.cpp
X86TargetInfoTest.cpp
TypesTest.cpp
)
More information about the llvm-commits
mailing list