[llvm] [ABI][AMDGPU] Add AMDGPU target ABI classifier to the LLVM ABI library (PR #220177)
via llvm-commits
llvm-commits at lists.llvm.org
Mon Sep 21 00:21:38 PDT 2026
https://github.com/skc7 updated https://github.com/llvm/llvm-project/pull/220177
>From 1975f4834262630e31d7970cf7de82d9a40e7c5a Mon Sep 17 00:00:00 2001
From: skc7 <Krishna.Sankisa at amd.com>
Date: Tue, 1 Sep 2026 12:35:15 +0530
Subject: [PATCH 1/4] [ABI] Add AMDGPU target ABI classifier to the LLVM ABI
library
---
llvm/include/llvm/ABI/TargetInfo.h | 6 +
llvm/lib/ABI/CMakeLists.txt | 1 +
llvm/lib/ABI/TargetInfo.cpp | 54 ++++
llvm/lib/ABI/Targets/AMDGPU.cpp | 248 ++++++++++++++++++
llvm/lib/ABI/Targets/X86.cpp | 55 ----
llvm/unittests/ABI/AMDGPUTargetInfoTest.cpp | 274 ++++++++++++++++++++
llvm/unittests/ABI/CMakeLists.txt | 1 +
7 files changed, 584 insertions(+), 55 deletions(-)
create mode 100644 llvm/lib/ABI/Targets/AMDGPU.cpp
create mode 100644 llvm/unittests/ABI/AMDGPUTargetInfoTest.cpp
diff --git a/llvm/include/llvm/ABI/TargetInfo.h b/llvm/include/llvm/ABI/TargetInfo.h
index 6f7f29e70bb3da..a1fe1aa01ba998 100644
--- a/llvm/include/llvm/ABI/TargetInfo.h
+++ b/llvm/include/llvm/ABI/TargetInfo.h
@@ -92,6 +92,10 @@ class TargetInfo {
/// return Ty unchanged.
LLVM_ABI const Type *useFirstFieldIfTransparentUnion(const Type *Ty) const;
+ /// If the record reduces to a single scalar element,
+ /// return that element type; otherwise null.
+ LLVM_ABI const Type *isSingleElementStruct(const Type *Ty) const;
+
/// Apply rules for classifying return types that are common to all targets.
LLVM_ABI bool maybeCommonClassifyReturnType(FunctionInfo &FI) const;
@@ -127,6 +131,8 @@ class TargetInfo {
LLVM_ABI std::unique_ptr<TargetInfo> createBPFTargetInfo(TypeBuilder &TB);
+LLVM_ABI std::unique_ptr<TargetInfo> createAMDGPUTargetInfo(TypeBuilder &TB);
+
/// The AVX ABI level for X86 targets.
enum class X86AVXABILevel {
None,
diff --git a/llvm/lib/ABI/CMakeLists.txt b/llvm/lib/ABI/CMakeLists.txt
index 39e725ca9fd3b7..54307bae8e07df 100644
--- a/llvm/lib/ABI/CMakeLists.txt
+++ b/llvm/lib/ABI/CMakeLists.txt
@@ -4,6 +4,7 @@ add_llvm_component_library(LLVMABI
TargetInfo.cpp
IRTypeMapper.cpp
Targets/AArch64.cpp
+ Targets/AMDGPU.cpp
Targets/BPF.cpp
Targets/X86.cpp
diff --git a/llvm/lib/ABI/TargetInfo.cpp b/llvm/lib/ABI/TargetInfo.cpp
index 9ad8cb8f5829a4..3342e642ad1f55 100644
--- a/llvm/lib/ABI/TargetInfo.cpp
+++ b/llvm/lib/ABI/TargetInfo.cpp
@@ -71,6 +71,60 @@ const Type *TargetInfo::useFirstFieldIfTransparentUnion(const Type *Ty) const {
return Ty;
}
+const Type *TargetInfo::isSingleElementStruct(const Type *Ty) const {
+ const auto *RT = dyn_cast<RecordType>(Ty);
+ if (!RT)
+ return nullptr;
+
+ if (RT->hasFlexibleArrayMember())
+ return nullptr;
+
+ const Type *Found = nullptr;
+
+ for (const auto &Base : RT->getBaseClasses()) {
+ const Type *BaseTy = Base.FieldType;
+ const auto *BaseRT = dyn_cast<RecordType>(BaseTy);
+ if (!BaseRT || BaseRT->isEmpty())
+ continue;
+
+ const Type *Elem = isSingleElementStruct(BaseTy);
+ if (!Elem || Found)
+ return nullptr;
+ Found = Elem;
+ }
+
+ for (const auto &FI : RT->getFields()) {
+ if (FI.isEmpty())
+ continue;
+
+ const Type *FTy = FI.FieldType;
+
+ // A single-element array is transparent for this reduction.
+ while (const auto *AT = dyn_cast<ArrayType>(FTy)) {
+ if (AT->getNumElements() != 1)
+ break;
+ FTy = AT->getElementType();
+ }
+
+ const Type *Elem;
+ if (const auto *InnerRT = dyn_cast<RecordType>(FTy))
+ Elem = isSingleElementStruct(InnerRT);
+ else
+ Elem = FTy;
+ if (!Elem || Found)
+ return nullptr;
+ Found = Elem;
+ }
+
+ if (!Found)
+ return nullptr;
+ // The reduced element must cover the whole record (no tail padding).
+ if (Found->getSizeInBits() != Ty->getSizeInBits())
+ return nullptr;
+
+ return Found;
+}
+
bool TargetInfo::maybeCommonClassifyReturnType(FunctionInfo &FI) const {
const abi::Type *Ty = FI.getReturnType();
diff --git a/llvm/lib/ABI/Targets/AMDGPU.cpp b/llvm/lib/ABI/Targets/AMDGPU.cpp
new file mode 100644
index 00000000000000..92130664f8b62d
--- /dev/null
+++ b/llvm/lib/ABI/Targets/AMDGPU.cpp
@@ -0,0 +1,248 @@
+//===- AMDGPU.cpp - AMDGPU ABI Implementation ----------------------------===//
+//
+// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
+// See https://llvm.org/LICENSE.txt for license information.
+// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
+//
+//===----------------------------------------------------------------------===//
+
+#include "llvm/ABI/FunctionInfo.h"
+#include "llvm/ABI/TargetInfo.h"
+#include "llvm/ABI/Types.h"
+#include "llvm/Support/AMDGPUAddrSpace.h"
+#include "llvm/Support/Alignment.h"
+#include "llvm/Support/Casting.h"
+#include "llvm/Support/TypeSize.h"
+#include <algorithm>
+#include <cassert>
+#include <cstdint>
+
+namespace llvm {
+namespace abi {
+
+class AMDGPUTargetInfo : public TargetInfo {
+private:
+ TypeBuilder &TB;
+ static const unsigned MaxNumRegsForArgsRet = 16;
+
+ ArgInfo classifyReturnType(const Type *RetTy) const;
+ ArgInfo classifyKernelArgumentType(const Type *Ty) const;
+ ArgInfo classifyArgumentType(const Type *Ty, bool Variadic,
+ unsigned &NumRegsLeft) const;
+
+ /// Target-independent fallback, mirroring classic CodeGen's DefaultABIInfo.
+ ArgInfo classifyDefaultType(const Type *Ty, bool IsReturn) const;
+
+ /// Estimate number of registers the type will use when passed in registers.
+ uint64_t numRegsForType(const Type *Ty) const;
+
+public:
+ AMDGPUTargetInfo(TypeBuilder &TypeBuilder, const ABICompatInfo &Compat)
+ : TargetInfo(Compat), TB(TypeBuilder) {}
+
+ void computeInfo(FunctionInfo &FI) const override;
+};
+
+uint64_t AMDGPUTargetInfo::numRegsForType(const Type *Ty) const {
+ uint64_t NumRegs = 0;
+
+ if (const auto *VT = dyn_cast<VectorType>(Ty)) {
+ // Compute from the number of elements. The reported size is based on the
+ // in-memory size, which includes the padding 4th element for 3-vectors.
+ const Type *EltTy = VT->getElementType();
+ uint64_t EltSize = EltTy->getSizeInBits().getFixedValue();
+ unsigned NumElts = VT->getNumElements().getFixedValue();
+
+ // 16-bit element vectors should be passed as packed.
+ if (EltSize == 16)
+ return (NumElts + 1) / 2;
+
+ uint64_t EltNumRegs = (EltSize + 31) / 32;
+ return EltNumRegs * NumElts;
+ }
+
+ if (const auto *RT = dyn_cast<RecordType>(Ty)) {
+ for (const FieldInfo &Field : RT->getFields())
+ NumRegs += numRegsForType(Field.FieldType);
+ return NumRegs;
+ }
+
+ return (Ty->getSizeInBits().getFixedValue() + 31) / 32;
+}
+
+ArgInfo AMDGPUTargetInfo::classifyDefaultType(const Type *Ty,
+ bool IsReturn) const {
+ if (IsReturn && Ty->isVoid())
+ return ArgInfo::getIgnore();
+
+ if (isAggregateTypeForABI(Ty)) {
+ if (RecordArgABI RAA = getRecordArgABI(Ty); RAA != RAA_Default)
+ return getNaturalAlignIndirect(Ty, /*ByVal=*/RAA == RAA_DirectInMemory);
+ return getNaturalAlignIndirect(Ty, /*ByVal=*/!IsReturn);
+ }
+
+ if (const auto *IT = dyn_cast<IntegerType>(Ty))
+ if (isPromotableInteger(IT))
+ return ArgInfo::getExtend(Ty);
+
+ return ArgInfo::getDirect();
+}
+
+ArgInfo AMDGPUTargetInfo::classifyReturnType(const Type *RetTy) const {
+ if (RetTy->isVoid())
+ return ArgInfo::getIgnore();
+
+ if (isAggregateTypeForABI(RetTy)) {
+ // Records with non-trivial destructors/copy-constructors should not be
+ // returned by value.
+ if (getRecordArgABI(RetTy) == RAA_Default) {
+ const auto *RT = dyn_cast<RecordType>(RetTy);
+
+ // Ignore empty structs/unions.
+ if (RT && RT->isEmpty())
+ return ArgInfo::getIgnore();
+
+ // Lower single-element structs to just return a regular value.
+ if (const Type *SeltTy = isSingleElementStruct(RetTy))
+ return ArgInfo::getDirect(SeltTy);
+
+ if (RT && RT->hasFlexibleArrayMember())
+ return classifyDefaultType(RetTy, /*IsReturn=*/true);
+
+ // Pack aggregates <= 4 bytes into single VGPR or pair.
+ uint64_t Size = RetTy->getSizeInBits().getFixedValue();
+ if (Size <= 16)
+ return ArgInfo::getDirect(TB.getIntegerType(16, Align(2), false));
+
+ if (Size <= 32)
+ return ArgInfo::getDirect(TB.getIntegerType(32, Align(4), false));
+
+ if (Size <= 64) {
+ const Type *I32Ty = TB.getIntegerType(32, Align(4), false);
+ return ArgInfo::getDirect(TB.getArrayType(I32Ty, 2, /*SizeInBits=*/64));
+ }
+
+ if (numRegsForType(RetTy) <= MaxNumRegsForArgsRet)
+ return ArgInfo::getDirect();
+ }
+ }
+
+ // Otherwise just do the default thing.
+ return classifyDefaultType(RetTy, /*IsReturn=*/true);
+}
+
+/// For kernels all parameters are really passed in a special buffer. It doesn't
+/// make sense to pass anything byval, so everything must be direct.
+ArgInfo AMDGPUTargetInfo::classifyKernelArgumentType(const Type *Ty) const {
+ Ty = useFirstFieldIfTransparentUnion(Ty);
+
+ if (const Type *SeltTy = isSingleElementStruct(Ty))
+ Ty = SeltTy;
+
+ // TODO: Classic coerces HIP scalar pointers from generic to global here; that
+ // depends on LangOptions the ABI library cannot see, so it is skipped.
+ if (isAggregateTypeForABI(Ty))
+ return ArgInfo::getIndirect(Ty->getAlignment(), /*ByVal=*/false,
+ /*AddrSpace=*/AMDGPUAS::CONSTANT_ADDRESS);
+
+ // TODO: Classic passes CanBeFlattened=false here to keep a struct intact;
+ // ArgInfo cannot model that yet, so a multi-field record coerce may be
+ // flattened.
+ return ArgInfo::getDirect(Ty);
+}
+
+ArgInfo AMDGPUTargetInfo::classifyArgumentType(const Type *Ty, bool Variadic,
+ unsigned &NumRegsLeft) const {
+ assert(NumRegsLeft <= MaxNumRegsForArgsRet && "register estimate underflow");
+
+ Ty = useFirstFieldIfTransparentUnion(Ty);
+
+ // TODO: Classic sets CanBeFlattened=false for variadics; not modeled here.
+ if (Variadic)
+ return ArgInfo::getDirect();
+
+ if (isAggregateTypeForABI(Ty)) {
+ // Records with non-trivial destructors/copy-constructors should not be
+ // passed by value.
+ if (RecordArgABI RAA = getRecordArgABI(Ty); RAA != RAA_Default)
+ return ArgInfo::getIndirect(Ty->getAlignment(),
+ /*ByVal=*/RAA == RAA_DirectInMemory,
+ /*AddrSpace=*/AMDGPUAS::PRIVATE_ADDRESS);
+
+ // Ignore empty structs/unions.
+ if (const auto *RT = dyn_cast<RecordType>(Ty); RT && RT->isEmpty())
+ return ArgInfo::getIgnore();
+
+ // Lower single-element structs to just pass a regular value.
+ if (const Type *SeltTy = isSingleElementStruct(Ty))
+ return ArgInfo::getDirect(SeltTy);
+
+ if (const auto *RT = dyn_cast<RecordType>(Ty);
+ RT && RT->hasFlexibleArrayMember())
+ return classifyDefaultType(Ty, /*IsReturn=*/false);
+
+ // Pack aggregates <= 8 bytes into single VGPR or pair.
+ uint64_t Size = Ty->getSizeInBits().getFixedValue();
+ if (Size <= 64) {
+ unsigned NumRegs = (Size + 31) / 32;
+ NumRegsLeft -= std::min(NumRegsLeft, NumRegs);
+
+ if (Size <= 16)
+ return ArgInfo::getDirect(TB.getIntegerType(16, Align(2), false));
+
+ if (Size <= 32)
+ return ArgInfo::getDirect(TB.getIntegerType(32, Align(4), false));
+
+ const Type *I32Ty = TB.getIntegerType(32, Align(4), false);
+ return ArgInfo::getDirect(TB.getArrayType(I32Ty, 2, /*SizeInBits=*/64));
+ }
+
+ if (NumRegsLeft > 0) {
+ uint64_t NumRegs = numRegsForType(Ty);
+ if (NumRegsLeft >= NumRegs) {
+ NumRegsLeft -= NumRegs;
+ return ArgInfo::getDirect();
+ }
+ }
+
+ // Pass a struct argument by reference rather than by value.
+ return ArgInfo::getIndirect(Ty->getAlignment(), /*ByVal=*/false,
+ /*AddrSpace=*/AMDGPUAS::PRIVATE_ADDRESS);
+ }
+
+ // Otherwise just do the default thing.
+ ArgInfo AI = classifyDefaultType(Ty, /*IsReturn=*/false);
+ if (!AI.isIndirect()) {
+ uint64_t NumRegs = numRegsForType(Ty);
+ NumRegsLeft -= std::min(NumRegs, uint64_t{NumRegsLeft});
+ }
+
+ return AI;
+}
+
+void AMDGPUTargetInfo::computeInfo(FunctionInfo &FI) const {
+ CallingConv::ID CC = FI.getCallingConvention();
+
+ if (!maybeCommonClassifyReturnType(FI))
+ FI.getReturnInfo() = classifyReturnType(FI.getReturnType());
+
+ unsigned ArgumentIndex = 0;
+ const unsigned NumFixedArguments = FI.getNumRequiredArgs();
+
+ unsigned NumRegsLeft = MaxNumRegsForArgsRet;
+ for (ArgEntry &Arg : FI.arguments()) {
+ if (CC == CallingConv::AMDGPU_KERNEL) {
+ Arg.Info = classifyKernelArgumentType(Arg.ABIType);
+ } else {
+ bool FixedArgument = ArgumentIndex++ < NumFixedArguments;
+ Arg.Info = classifyArgumentType(Arg.ABIType, !FixedArgument, NumRegsLeft);
+ }
+ }
+}
+
+std::unique_ptr<TargetInfo> createAMDGPUTargetInfo(TypeBuilder &TB) {
+ return std::make_unique<AMDGPUTargetInfo>(TB, ABICompatInfo());
+}
+
+} // namespace abi
+} // namespace llvm
diff --git a/llvm/lib/ABI/Targets/X86.cpp b/llvm/lib/ABI/Targets/X86.cpp
index 78aa9cd1a25190..20f174db5d698b 100644
--- a/llvm/lib/ABI/Targets/X86.cpp
+++ b/llvm/lib/ABI/Targets/X86.cpp
@@ -98,7 +98,6 @@ class X86_64TargetInfo : public TargetInfo {
ArgInfo getIndirectReturnResult(const Type *Ty) const;
const Type *getFPTypeAtOffset(const Type *Ty, unsigned Offset) const;
- const Type *isSingleElementStruct(const Type *Ty) const;
const Type *getByteVectorType(const Type *Ty) const;
const Type *createPairType(const Type *Lo, const Type *Hi) const;
@@ -1261,60 +1260,6 @@ const Type *X86_64TargetInfo::getByteVectorType(const Type *Ty) const {
ElementCount::getFixed(Size / 64), Align(Size / 8));
}
-// Returns the single element if this is a single-element struct wrapper
-const Type *X86_64TargetInfo::isSingleElementStruct(const Type *Ty) const {
- const auto *RT = dyn_cast<RecordType>(Ty);
- if (!RT)
- return nullptr;
-
- if (RT->hasFlexibleArrayMember())
- return nullptr;
-
- const Type *Found = nullptr;
-
- for (const auto &Base : RT->getBaseClasses()) {
- const Type *BaseTy = Base.FieldType;
- auto *BaseRT = dyn_cast<RecordType>(BaseTy);
-
- if (!BaseRT || BaseRT->isEmpty())
- continue;
-
- const Type *Elem = isSingleElementStruct(BaseTy);
- if (!Elem || Found)
- return nullptr;
- Found = Elem;
- }
-
- for (const auto &FI : RT->getFields()) {
- if (FI.isEmpty())
- continue;
-
- const Type *FTy = FI.FieldType;
-
- while (auto *AT = dyn_cast<ArrayType>(FTy)) {
- if (AT->getNumElements() != 1)
- break;
- FTy = AT->getElementType();
- }
-
- const Type *Elem;
- if (auto *InnerRT = dyn_cast<RecordType>(FTy))
- Elem = isSingleElementStruct(InnerRT);
- else
- Elem = FTy;
- if (!Elem || Found)
- return nullptr;
- Found = Elem;
- }
-
- if (!Found)
- return nullptr;
- if (Found->getSizeInBits() != Ty->getSizeInBits())
- return nullptr;
-
- return Found;
-}
-
bool X86_64TargetInfo::isIllegalVectorType(const Type *Ty) const {
if (const auto *VecTy = dyn_cast<VectorType>(Ty)) {
uint64_t Size = VecTy->getSizeInBits().getFixedValue();
diff --git a/llvm/unittests/ABI/AMDGPUTargetInfoTest.cpp b/llvm/unittests/ABI/AMDGPUTargetInfoTest.cpp
new file mode 100644
index 00000000000000..38d2bbfc37017a
--- /dev/null
+++ b/llvm/unittests/ABI/AMDGPUTargetInfoTest.cpp
@@ -0,0 +1,274 @@
+//===- AMDGPUTargetInfoTest.cpp - AMDGPU ABI unit tests -------------------===//
+//
+// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
+// See https://llvm.org/LICENSE.txt for license information.
+// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
+//
+//===----------------------------------------------------------------------===//
+
+#include "llvm/ABI/FunctionInfo.h"
+#include "llvm/ABI/TargetInfo.h"
+#include "llvm/ABI/Types.h"
+#include "llvm/ADT/APFloat.h"
+#include "llvm/IR/CallingConv.h"
+#include "llvm/Support/AMDGPUAddrSpace.h"
+#include "llvm/Support/Alignment.h"
+#include "llvm/Support/Allocator.h"
+#include "gtest/gtest.h"
+
+namespace {
+
+// RecordFlags' bitmask operators are declared in namespace llvm, so combining
+// two of them needs that namespace visible.
+using namespace llvm;
+
+using ABIType = llvm::abi::Type;
+using llvm::abi::ABICompatInfo;
+using llvm::abi::ArgInfo;
+using llvm::abi::createAMDGPUTargetInfo;
+using llvm::abi::FieldInfo;
+using llvm::abi::FunctionInfo;
+using llvm::abi::RecordFlags;
+using llvm::abi::RequiredArgs;
+using llvm::abi::StructPacking;
+using llvm::abi::TargetInfo;
+using llvm::abi::TypeBuilder;
+
+class AMDGPUTargetInfoTest : public ::testing::Test {
+protected:
+ llvm::BumpPtrAllocator Alloc;
+ TypeBuilder TB;
+ const ABIType *I8;
+ const ABIType *I16;
+ const ABIType *I32;
+ const ABIType *F32;
+ const ABIType *Void;
+ /// An empty class: a record with no fields, one byte wide, register-passable.
+ const ABIType *Empty;
+
+ AMDGPUTargetInfoTest()
+ : TB(Alloc), I8(TB.getIntegerType(8, llvm::Align(1), /*Signed=*/true)),
+ I16(TB.getIntegerType(16, llvm::Align(2), /*Signed=*/true)),
+ I32(TB.getIntegerType(32, llvm::Align(4), /*Signed=*/true)),
+ F32(TB.getFloatType(llvm::APFloat::IEEEsingle(), llvm::Align(4))),
+ Void(TB.getVoidType()),
+ Empty(TB.getRecordType({}, llvm::TypeSize::getFixed(8), llvm::Align(1),
+ StructPacking::Default, {}, {},
+ RecordFlags::CanPassInRegisters)) {}
+
+ std::unique_ptr<TargetInfo> target() const {
+ return createAMDGPUTargetInfo(const_cast<TypeBuilder &>(TB));
+ }
+
+ /// A register-passable record with the given fields, size and alignment.
+ const ABIType *recordOf(llvm::ArrayRef<FieldInfo> Fields, uint64_t SizeInBits,
+ llvm::Align Alignment) {
+ return TB.getRecordType(Fields, llvm::TypeSize::getFixed(SizeInBits),
+ Alignment, StructPacking::Default, {}, {},
+ RecordFlags::CanPassInRegisters);
+ }
+
+ /// The argument classification the target computes for a single parameter
+ /// under calling convention \p CC.
+ const ArgInfo &classifyArg(const ABIType *ArgTy,
+ std::unique_ptr<FunctionInfo> &FI,
+ std::unique_ptr<TargetInfo> &TI,
+ CallingConv::ID CC = CallingConv::C) {
+ TI = target();
+ FI = FunctionInfo::create(CC, Void, {ArgTy});
+ TI->computeInfo(*FI);
+ return FI->getArgInfo(0).Info;
+ }
+
+ /// The return classification the target computes for \p RetTy.
+ const ArgInfo &classifyRet(const ABIType *RetTy,
+ std::unique_ptr<FunctionInfo> &FI,
+ std::unique_ptr<TargetInfo> &TI) {
+ TI = target();
+ FI = FunctionInfo::create(CallingConv::C, RetTy, {});
+ TI->computeInfo(*FI);
+ return FI->getReturnInfo();
+ }
+};
+
+static void expectUncoercedDirect(const ArgInfo &Info) {
+ ASSERT_TRUE(Info.isDirect());
+ EXPECT_EQ(Info.getCoerceToType(), nullptr);
+}
+
+static void expectDirectInteger(const ArgInfo &Info, unsigned Bits) {
+ ASSERT_TRUE(Info.isDirect());
+ const ABIType *Coerce = Info.getCoerceToType();
+ ASSERT_NE(Coerce, nullptr);
+ const auto *IT = llvm::dyn_cast<llvm::abi::IntegerType>(Coerce);
+ ASSERT_NE(IT, nullptr);
+ EXPECT_EQ(IT->getSizeInBits().getFixedValue(), Bits);
+}
+
+static void expectDirectFloat(const ArgInfo &Info,
+ const llvm::fltSemantics &Sem) {
+ ASSERT_TRUE(Info.isDirect());
+ const ABIType *Coerce = Info.getCoerceToType();
+ ASSERT_NE(Coerce, nullptr);
+ const auto *FT = llvm::dyn_cast<llvm::abi::FloatType>(Coerce);
+ ASSERT_NE(FT, nullptr);
+ EXPECT_EQ(FT->getSemantics(), &Sem);
+}
+
+// A <= 8-byte aggregate coerces to [2 x i32].
+static void expectDirectI32Pair(const ArgInfo &Info) {
+ ASSERT_TRUE(Info.isDirect());
+ const auto *AT =
+ llvm::dyn_cast_or_null<llvm::abi::ArrayType>(Info.getCoerceToType());
+ ASSERT_NE(AT, nullptr);
+ EXPECT_EQ(AT->getNumElements(), 2u);
+ const auto *IT = llvm::dyn_cast<llvm::abi::IntegerType>(AT->getElementType());
+ ASSERT_NE(IT, nullptr);
+ EXPECT_EQ(IT->getSizeInBits().getFixedValue(), 32u);
+}
+
+static void expectIndirect(const ArgInfo &Info, llvm::Align ExpectedAlign,
+ bool ByVal, unsigned AddrSpace) {
+ ASSERT_TRUE(Info.isIndirect());
+ EXPECT_EQ(Info.getIndirectAlign(), ExpectedAlign);
+ EXPECT_EQ(Info.getIndirectByVal(), ByVal);
+ EXPECT_EQ(Info.getIndirectAddrSpace(), AddrSpace);
+}
+
+// A 32-bit integer and a pointer-sized scalar pass directly in their own type.
+TEST_F(AMDGPUTargetInfoTest, ScalarPassesDirect) {
+ std::unique_ptr<FunctionInfo> FI;
+ std::unique_ptr<TargetInfo> TI;
+ expectUncoercedDirect(classifyArg(I32, FI, TI));
+ expectUncoercedDirect(classifyArg(F32, FI, TI));
+}
+
+// A sub-word integer is sign/zero extended to fill its register.
+TEST_F(AMDGPUTargetInfoTest, PromotableIntegerExtends) {
+ std::unique_ptr<FunctionInfo> FI;
+ std::unique_ptr<TargetInfo> TI;
+ const ArgInfo &Info = classifyArg(I8, FI, TI);
+ ASSERT_TRUE(Info.isExtend());
+ EXPECT_TRUE(Info.isSignExt());
+}
+
+// Aggregates <= 16/32/64 bits pack into i16 / i32 / [2 x i32].
+TEST_F(AMDGPUTargetInfoTest, SmallAggregatesPackIntoRegisters) {
+ std::unique_ptr<FunctionInfo> FI;
+ std::unique_ptr<TargetInfo> TI;
+
+ const ABIType *S16 =
+ recordOf({FieldInfo(I8, 0), FieldInfo(I8, 8)}, 16, llvm::Align(1));
+ expectDirectInteger(classifyArg(S16, FI, TI), 16);
+
+ const ABIType *S32 =
+ recordOf({FieldInfo(I16, 0), FieldInfo(I16, 16)}, 32, llvm::Align(2));
+ expectDirectInteger(classifyArg(S32, FI, TI), 32);
+
+ const ABIType *S64 =
+ recordOf({FieldInfo(I32, 0), FieldInfo(I32, 32)}, 64, llvm::Align(4));
+ expectDirectI32Pair(classifyArg(S64, FI, TI));
+}
+
+// An empty struct is dropped from the argument list.
+TEST_F(AMDGPUTargetInfoTest, EmptyAggregateIsIgnored) {
+ std::unique_ptr<FunctionInfo> FI;
+ std::unique_ptr<TargetInfo> TI;
+ EXPECT_TRUE(classifyArg(Empty, FI, TI).isIgnore());
+}
+
+// A single-element struct is passed as its inner scalar.
+TEST_F(AMDGPUTargetInfoTest, SingleElementStructUnwraps) {
+ std::unique_ptr<FunctionInfo> FI;
+ std::unique_ptr<TargetInfo> TI;
+ const ABIType *Wrapper = recordOf({FieldInfo(F32, 0)}, 32, llvm::Align(4));
+ expectDirectFloat(classifyArg(Wrapper, FI, TI), llvm::APFloat::IEEEsingle());
+}
+
+// A large aggregate that does not fit the 16-register budget is passed by
+// reference in the private address space.
+TEST_F(AMDGPUTargetInfoTest, OversizedAggregateIsIndirectPrivate) {
+ std::unique_ptr<FunctionInfo> FI;
+ std::unique_ptr<TargetInfo> TI;
+ // Two i32[10] fields => 20 registers, above MaxNumRegsForArgsRet (16).
+ const ABIType *ArrTy = TB.getArrayType(I32, /*NumElements=*/10,
+ /*SizeInBits=*/320);
+ const ABIType *Big = recordOf({FieldInfo(ArrTy, 0), FieldInfo(ArrTy, 320)},
+ 640, llvm::Align(4));
+ expectIndirect(classifyArg(Big, FI, TI), llvm::Align(4), /*ByVal=*/false,
+ llvm::AMDGPUAS::PRIVATE_ADDRESS);
+}
+
+// A record that cannot pass in registers (non-trivial C++ type) is passed
+// indirectly in the private address space.
+TEST_F(AMDGPUTargetInfoTest, NonTrivialRecordIsIndirectPrivate) {
+ std::unique_ptr<FunctionInfo> FI;
+ std::unique_ptr<TargetInfo> TI;
+ const ABIType *CannotPass = TB.getRecordType(
+ {FieldInfo(I32, 0)}, llvm::TypeSize::getFixed(32), llvm::Align(4),
+ StructPacking::Default, {}, {}, RecordFlags::IsCXXRecord);
+ expectIndirect(classifyArg(CannotPass, FI, TI), llvm::Align(4),
+ /*ByVal=*/false, llvm::AMDGPUAS::PRIVATE_ADDRESS);
+}
+
+// A variadic argument bypasses register packing and passes through unchanged.
+TEST_F(AMDGPUTargetInfoTest, VariadicArgumentPassesDirect) {
+ std::unique_ptr<TargetInfo> TI = target();
+ const ABIType *S16 =
+ recordOf({FieldInfo(I8, 0), FieldInfo(I8, 8)}, 16, llvm::Align(1));
+ // Zero declared parameters, so the sole argument is variadic.
+ std::unique_ptr<FunctionInfo> FI =
+ FunctionInfo::create(CallingConv::C, Void, {S16}, RequiredArgs(0));
+ TI->computeInfo(*FI);
+ expectUncoercedDirect(FI->getArgInfo(0).Info);
+}
+
+// Kernel aggregate arguments are passed by reference in the constant address
+// space (the kernarg segment), never byval.
+TEST_F(AMDGPUTargetInfoTest, KernelAggregateIsIndirectConstant) {
+ std::unique_ptr<FunctionInfo> FI;
+ std::unique_ptr<TargetInfo> TI;
+ const ABIType *S =
+ recordOf({FieldInfo(I32, 0), FieldInfo(I32, 32)}, 64, llvm::Align(4));
+ expectIndirect(classifyArg(S, FI, TI, CallingConv::AMDGPU_KERNEL),
+ llvm::Align(4), /*ByVal=*/false,
+ llvm::AMDGPUAS::CONSTANT_ADDRESS);
+}
+
+// Kernel scalar arguments are passed directly.
+TEST_F(AMDGPUTargetInfoTest, KernelScalarIsDirect) {
+ std::unique_ptr<FunctionInfo> FI;
+ std::unique_ptr<TargetInfo> TI;
+ expectDirectInteger(classifyArg(I32, FI, TI, CallingConv::AMDGPU_KERNEL), 32);
+}
+
+// A void return is ignored.
+TEST_F(AMDGPUTargetInfoTest, VoidReturnIsIgnored) {
+ std::unique_ptr<FunctionInfo> FI;
+ std::unique_ptr<TargetInfo> TI;
+ EXPECT_TRUE(classifyRet(Void, FI, TI).isIgnore());
+}
+
+// A <= 8-byte aggregate return packs into [2 x i32].
+TEST_F(AMDGPUTargetInfoTest, SmallAggregateReturnPacks) {
+ std::unique_ptr<FunctionInfo> FI;
+ std::unique_ptr<TargetInfo> TI;
+ const ABIType *S64 =
+ recordOf({FieldInfo(I32, 0), FieldInfo(I32, 32)}, 64, llvm::Align(4));
+ expectDirectI32Pair(classifyRet(S64, FI, TI));
+}
+
+// A record that cannot pass in registers is returned indirectly (sret-style,
+// ByVal=false) via the target-independent return rule.
+TEST_F(AMDGPUTargetInfoTest, NonTrivialRecordReturnIsIndirect) {
+ std::unique_ptr<FunctionInfo> FI;
+ std::unique_ptr<TargetInfo> TI;
+ const ABIType *CannotPass = TB.getRecordType(
+ {FieldInfo(I32, 0)}, llvm::TypeSize::getFixed(32), llvm::Align(4),
+ StructPacking::Default, {}, {}, RecordFlags::IsCXXRecord);
+ const ArgInfo &Info = classifyRet(CannotPass, FI, TI);
+ ASSERT_TRUE(Info.isIndirect());
+ EXPECT_FALSE(Info.getIndirectByVal());
+}
+
+} // namespace
diff --git a/llvm/unittests/ABI/CMakeLists.txt b/llvm/unittests/ABI/CMakeLists.txt
index 4a1910ab8809b4..ffd5e82ef082cb 100644
--- a/llvm/unittests/ABI/CMakeLists.txt
+++ b/llvm/unittests/ABI/CMakeLists.txt
@@ -8,6 +8,7 @@ add_llvm_unittest(ABITests
AArch64TargetInfoTest.cpp
IRTypeMapperTest.cpp
FunctionInfoTest.cpp
+ AMDGPUTargetInfoTest.cpp
X86TargetInfoTest.cpp
TypesTest.cpp
)
>From ef132d0700f0209d4ad545f24e6078ff12c80380 Mon Sep 17 00:00:00 2001
From: skc7 <Krishna.Sankisa at amd.com>
Date: Thu, 17 Sep 2026 16:24:53 +0530
Subject: [PATCH 2/4] support CoerceGenericPtrArgToGlobal
---
llvm/include/llvm/ABI/TargetInfo.h | 4 +-
llvm/lib/ABI/Targets/AMDGPU.cpp | 110 ++++++++++++++------
llvm/unittests/ABI/AMDGPUTargetInfoTest.cpp | 102 +++++++++++++++++-
3 files changed, 182 insertions(+), 34 deletions(-)
diff --git a/llvm/include/llvm/ABI/TargetInfo.h b/llvm/include/llvm/ABI/TargetInfo.h
index a1fe1aa01ba998..31157b4a9c9ac8 100644
--- a/llvm/include/llvm/ABI/TargetInfo.h
+++ b/llvm/include/llvm/ABI/TargetInfo.h
@@ -131,7 +131,9 @@ class TargetInfo {
LLVM_ABI std::unique_ptr<TargetInfo> createBPFTargetInfo(TypeBuilder &TB);
-LLVM_ABI std::unique_ptr<TargetInfo> createAMDGPUTargetInfo(TypeBuilder &TB);
+LLVM_ABI std::unique_ptr<TargetInfo>
+createAMDGPUTargetInfo(TypeBuilder &TB,
+ bool CoerceGenericPtrArgToGlobal = false);
/// The AVX ABI level for X86 targets.
enum class X86AVXABILevel {
diff --git a/llvm/lib/ABI/Targets/AMDGPU.cpp b/llvm/lib/ABI/Targets/AMDGPU.cpp
index 92130664f8b62d..1c27afd598f327 100644
--- a/llvm/lib/ABI/Targets/AMDGPU.cpp
+++ b/llvm/lib/ABI/Targets/AMDGPU.cpp
@@ -22,23 +22,29 @@ namespace abi {
class AMDGPUTargetInfo : public TargetInfo {
private:
- TypeBuilder &TB;
static const unsigned MaxNumRegsForArgsRet = 16;
+ /// HIP coerces a generic scalar-pointer kernel argument to the global
+ /// address space. Gated by the front end, which alone can see LangOpts.HIP.
+ bool CoerceGenericPtrArgToGlobal;
+
ArgInfo classifyReturnType(const Type *RetTy) const;
ArgInfo classifyKernelArgumentType(const Type *Ty) const;
ArgInfo classifyArgumentType(const Type *Ty, bool Variadic,
unsigned &NumRegsLeft) const;
- /// Target-independent fallback, mirroring classic CodeGen's DefaultABIInfo.
- ArgInfo classifyDefaultType(const Type *Ty, bool IsReturn) const;
+ /// Target-independent fallback classification for arguments and returns.
+ ArgInfo classifyDefaultArgumentType(const Type *Ty) const;
+ ArgInfo classifyDefaultReturnType(const Type *Ty) const;
/// Estimate number of registers the type will use when passed in registers.
uint64_t numRegsForType(const Type *Ty) const;
public:
- AMDGPUTargetInfo(TypeBuilder &TypeBuilder, const ABICompatInfo &Compat)
- : TargetInfo(Compat), TB(TypeBuilder) {}
+ AMDGPUTargetInfo(TypeBuilder &TypeBuilder, const ABICompatInfo &Compat,
+ bool CoerceGenericPtrArgToGlobal)
+ : TargetInfo(TypeBuilder, Compat),
+ CoerceGenericPtrArgToGlobal(CoerceGenericPtrArgToGlobal) {}
void computeInfo(FunctionInfo &FI) const override;
};
@@ -70,22 +76,55 @@ uint64_t AMDGPUTargetInfo::numRegsForType(const Type *Ty) const {
return (Ty->getSizeInBits().getFixedValue() + 31) / 32;
}
-ArgInfo AMDGPUTargetInfo::classifyDefaultType(const Type *Ty,
- bool IsReturn) const {
- if (IsReturn && Ty->isVoid())
- return ArgInfo::getIgnore();
+// Natural-alignment indirect in the alloca address space (PRIVATE on AMDGPU),
+// ByVal by default.
+static ArgInfo getNaturalAlignIndirectPrivate(const Type *Ty,
+ bool ByVal = true) {
+ return ArgInfo::getIndirect(Ty->getAlignment(), ByVal,
+ /*AddrSpace=*/AMDGPUAS::PRIVATE_ADDRESS);
+}
+
+// A _BitInt wider than int128 is passed indirectly. AMDGPU has int128, so the
+// threshold is 128 bits.
+static bool isOversizedBitInt(const Type *Ty) {
+ const auto *IT = dyn_cast<IntegerType>(Ty);
+ return IT && IT->isBitInt() && IT->getSizeInBits().getFixedValue() > 128;
+}
+
+// Default argument classification for types not handled by the AMDGPU rules.
+ArgInfo AMDGPUTargetInfo::classifyDefaultArgumentType(const Type *Ty) const {
+ Ty = useFirstFieldIfTransparentUnion(Ty);
if (isAggregateTypeForABI(Ty)) {
if (RecordArgABI RAA = getRecordArgABI(Ty); RAA != RAA_Default)
- return getNaturalAlignIndirect(Ty, /*ByVal=*/RAA == RAA_DirectInMemory);
- return getNaturalAlignIndirect(Ty, /*ByVal=*/!IsReturn);
+ return getNaturalAlignIndirectPrivate(Ty,
+ /*ByVal=*/RAA ==
+ RAA_DirectInMemory);
+ return getNaturalAlignIndirectPrivate(Ty);
}
- if (const auto *IT = dyn_cast<IntegerType>(Ty))
- if (isPromotableInteger(IT))
- return ArgInfo::getExtend(Ty);
+ if (isOversizedBitInt(Ty))
+ return getNaturalAlignIndirectPrivate(Ty);
- return ArgInfo::getDirect();
+ const auto *IT = dyn_cast<IntegerType>(Ty);
+ return (IT && isPromotableInteger(IT)) ? ArgInfo::getExtend(Ty)
+ : ArgInfo::getDirect();
+}
+
+// Default return classification for types not handled by the AMDGPU rules.
+ArgInfo AMDGPUTargetInfo::classifyDefaultReturnType(const Type *Ty) const {
+ if (Ty->isVoid())
+ return ArgInfo::getIgnore();
+
+ if (isAggregateTypeForABI(Ty))
+ return getNaturalAlignIndirectPrivate(Ty);
+
+ if (isOversizedBitInt(Ty))
+ return getNaturalAlignIndirectPrivate(Ty);
+
+ const auto *IT = dyn_cast<IntegerType>(Ty);
+ return (IT && isPromotableInteger(IT)) ? ArgInfo::getExtend(Ty)
+ : ArgInfo::getDirect();
}
ArgInfo AMDGPUTargetInfo::classifyReturnType(const Type *RetTy) const {
@@ -107,7 +146,7 @@ ArgInfo AMDGPUTargetInfo::classifyReturnType(const Type *RetTy) const {
return ArgInfo::getDirect(SeltTy);
if (RT && RT->hasFlexibleArrayMember())
- return classifyDefaultType(RetTy, /*IsReturn=*/true);
+ return classifyDefaultReturnType(RetTy);
// Pack aggregates <= 4 bytes into single VGPR or pair.
uint64_t Size = RetTy->getSizeInBits().getFixedValue();
@@ -128,7 +167,7 @@ ArgInfo AMDGPUTargetInfo::classifyReturnType(const Type *RetTy) const {
}
// Otherwise just do the default thing.
- return classifyDefaultType(RetTy, /*IsReturn=*/true);
+ return classifyDefaultReturnType(RetTy);
}
/// For kernels all parameters are really passed in a special buffer. It doesn't
@@ -139,16 +178,26 @@ ArgInfo AMDGPUTargetInfo::classifyKernelArgumentType(const Type *Ty) const {
if (const Type *SeltTy = isSingleElementStruct(Ty))
Ty = SeltTy;
- // TODO: Classic coerces HIP scalar pointers from generic to global here; that
- // depends on LangOptions the ABI library cannot see, so it is skipped.
+ // HIP passes a generic scalar pointer as a global pointer; a pointer is not
+ // an aggregate, so this stays on the direct path.
+ if (CoerceGenericPtrArgToGlobal) {
+ if (const auto *PtrTy = dyn_cast<PointerType>(Ty);
+ PtrTy && PtrTy->getAddrSpace() == AMDGPUAS::FLAT_ADDRESS) {
+ const Type *Coerced =
+ TB.getPointerType(PtrTy->getSizeInBits().getFixedValue(),
+ PtrTy->getAlignment(), AMDGPUAS::GLOBAL_ADDRESS);
+ return ArgInfo::getDirect(Coerced, /*Offset=*/0, /*Align=*/std::nullopt,
+ /*CanBeFlattened=*/false);
+ }
+ }
+
if (isAggregateTypeForABI(Ty))
return ArgInfo::getIndirect(Ty->getAlignment(), /*ByVal=*/false,
/*AddrSpace=*/AMDGPUAS::CONSTANT_ADDRESS);
- // TODO: Classic passes CanBeFlattened=false here to keep a struct intact;
- // ArgInfo cannot model that yet, so a multi-field record coerce may be
- // flattened.
- return ArgInfo::getDirect(Ty);
+ // CanBeFlattened=false keeps the struct intact.
+ return ArgInfo::getDirect(Ty, /*Offset=*/0, /*Align=*/std::nullopt,
+ /*CanBeFlattened=*/false);
}
ArgInfo AMDGPUTargetInfo::classifyArgumentType(const Type *Ty, bool Variadic,
@@ -157,9 +206,10 @@ ArgInfo AMDGPUTargetInfo::classifyArgumentType(const Type *Ty, bool Variadic,
Ty = useFirstFieldIfTransparentUnion(Ty);
- // TODO: Classic sets CanBeFlattened=false for variadics; not modeled here.
+ // Variadic aggregates are kept intact rather than flattened into fields.
if (Variadic)
- return ArgInfo::getDirect();
+ return ArgInfo::getDirect(/*T=*/nullptr, /*Offset=*/0,
+ /*Align=*/std::nullopt, /*CanBeFlattened=*/false);
if (isAggregateTypeForABI(Ty)) {
// Records with non-trivial destructors/copy-constructors should not be
@@ -179,7 +229,7 @@ ArgInfo AMDGPUTargetInfo::classifyArgumentType(const Type *Ty, bool Variadic,
if (const auto *RT = dyn_cast<RecordType>(Ty);
RT && RT->hasFlexibleArrayMember())
- return classifyDefaultType(Ty, /*IsReturn=*/false);
+ return classifyDefaultArgumentType(Ty);
// Pack aggregates <= 8 bytes into single VGPR or pair.
uint64_t Size = Ty->getSizeInBits().getFixedValue();
@@ -211,7 +261,7 @@ ArgInfo AMDGPUTargetInfo::classifyArgumentType(const Type *Ty, bool Variadic,
}
// Otherwise just do the default thing.
- ArgInfo AI = classifyDefaultType(Ty, /*IsReturn=*/false);
+ ArgInfo AI = classifyDefaultArgumentType(Ty);
if (!AI.isIndirect()) {
uint64_t NumRegs = numRegsForType(Ty);
NumRegsLeft -= std::min(NumRegs, uint64_t{NumRegsLeft});
@@ -240,8 +290,10 @@ void AMDGPUTargetInfo::computeInfo(FunctionInfo &FI) const {
}
}
-std::unique_ptr<TargetInfo> createAMDGPUTargetInfo(TypeBuilder &TB) {
- return std::make_unique<AMDGPUTargetInfo>(TB, ABICompatInfo());
+std::unique_ptr<TargetInfo>
+createAMDGPUTargetInfo(TypeBuilder &TB, bool CoerceGenericPtrArgToGlobal) {
+ return std::make_unique<AMDGPUTargetInfo>(TB, ABICompatInfo(),
+ CoerceGenericPtrArgToGlobal);
}
} // namespace abi
diff --git a/llvm/unittests/ABI/AMDGPUTargetInfoTest.cpp b/llvm/unittests/ABI/AMDGPUTargetInfoTest.cpp
index 38d2bbfc37017a..b85d0f9e4f058f 100644
--- a/llvm/unittests/ABI/AMDGPUTargetInfoTest.cpp
+++ b/llvm/unittests/ABI/AMDGPUTargetInfoTest.cpp
@@ -60,6 +60,22 @@ class AMDGPUTargetInfoTest : public ::testing::Test {
return createAMDGPUTargetInfo(const_cast<TypeBuilder &>(TB));
}
+ /// A target configured like a HIP compilation: generic scalar-pointer kernel
+ /// arguments are coerced to the global address space.
+ std::unique_ptr<TargetInfo> hipTarget() const {
+ return createAMDGPUTargetInfo(const_cast<TypeBuilder &>(TB),
+ /*CoerceGenericPtrArgToGlobal=*/true);
+ }
+
+ /// The classification a kernel argument gets on target \p TI.
+ const ArgInfo &classifyKernelArg(const ABIType *ArgTy,
+ std::unique_ptr<FunctionInfo> &FI,
+ std::unique_ptr<TargetInfo> &TI) {
+ FI = FunctionInfo::create(CallingConv::AMDGPU_KERNEL, Void, {ArgTy});
+ TI->computeInfo(*FI);
+ return FI->getArgInfo(0).Info;
+ }
+
/// A register-passable record with the given fields, size and alignment.
const ABIType *recordOf(llvm::ArrayRef<FieldInfo> Fields, uint64_t SizeInBits,
llvm::Align Alignment) {
@@ -91,6 +107,15 @@ class AMDGPUTargetInfoTest : public ::testing::Test {
}
};
+// A direct arg coerced to a pointer in address space \p AS.
+static void expectDirectPointerAS(const ArgInfo &Info, unsigned AS) {
+ ASSERT_TRUE(Info.isDirect());
+ const auto *PT =
+ llvm::dyn_cast_or_null<llvm::abi::PointerType>(Info.getCoerceToType());
+ ASSERT_NE(PT, nullptr);
+ EXPECT_EQ(PT->getAddrSpace(), AS);
+}
+
static void expectUncoercedDirect(const ArgInfo &Info) {
ASSERT_TRUE(Info.isDirect());
EXPECT_EQ(Info.getCoerceToType(), nullptr);
@@ -139,7 +164,10 @@ static void expectIndirect(const ArgInfo &Info, llvm::Align ExpectedAlign,
TEST_F(AMDGPUTargetInfoTest, ScalarPassesDirect) {
std::unique_ptr<FunctionInfo> FI;
std::unique_ptr<TargetInfo> TI;
- expectUncoercedDirect(classifyArg(I32, FI, TI));
+ const ArgInfo &Info = classifyArg(I32, FI, TI);
+ expectUncoercedDirect(Info);
+ // Ordinary direct args stay flattenable (the getDirect default).
+ EXPECT_TRUE(Info.getCanBeFlattened());
expectUncoercedDirect(classifyArg(F32, FI, TI));
}
@@ -152,6 +180,35 @@ TEST_F(AMDGPUTargetInfoTest, PromotableIntegerExtends) {
EXPECT_TRUE(Info.isSignExt());
}
+// An oversized _BitInt (> 128 bits) is passed indirectly, byval, like classic
+// DefaultABIInfo. The 128-bit boundary width stays direct.
+TEST_F(AMDGPUTargetInfoTest, OversizedBitIntArgumentIsIndirect) {
+ std::unique_ptr<FunctionInfo> FI;
+ std::unique_ptr<TargetInfo> TI;
+ const ABIType *BitInt256 = TB.getIntegerType(256, llvm::Align(16),
+ /*Signed=*/true,
+ /*IsBitInt=*/true);
+ expectIndirect(classifyArg(BitInt256, FI, TI), llvm::Align(16),
+ /*ByVal=*/true, llvm::AMDGPUAS::PRIVATE_ADDRESS);
+
+ const ABIType *BitInt128 = TB.getIntegerType(128, llvm::Align(16),
+ /*Signed=*/true,
+ /*IsBitInt=*/true);
+ EXPECT_TRUE(classifyArg(BitInt128, FI, TI).isDirect());
+}
+
+// An oversized _BitInt return is indirect too. Classic DefaultABIInfo's
+// getNaturalAlignIndirect defaults ByVal to true, so returns carry it too.
+TEST_F(AMDGPUTargetInfoTest, OversizedBitIntReturnIsIndirect) {
+ std::unique_ptr<FunctionInfo> FI;
+ std::unique_ptr<TargetInfo> TI;
+ const ABIType *BitInt256 = TB.getIntegerType(256, llvm::Align(16),
+ /*Signed=*/true,
+ /*IsBitInt=*/true);
+ expectIndirect(classifyRet(BitInt256, FI, TI), llvm::Align(16),
+ /*ByVal=*/true, llvm::AMDGPUAS::PRIVATE_ADDRESS);
+}
+
// Aggregates <= 16/32/64 bits pack into i16 / i32 / [2 x i32].
TEST_F(AMDGPUTargetInfoTest, SmallAggregatesPackIntoRegisters) {
std::unique_ptr<FunctionInfo> FI;
@@ -211,7 +268,8 @@ TEST_F(AMDGPUTargetInfoTest, NonTrivialRecordIsIndirectPrivate) {
/*ByVal=*/false, llvm::AMDGPUAS::PRIVATE_ADDRESS);
}
-// A variadic argument bypasses register packing and passes through unchanged.
+// A variadic argument bypasses register packing and passes through unchanged,
+// kept intact rather than flattened into per-field wire arguments.
TEST_F(AMDGPUTargetInfoTest, VariadicArgumentPassesDirect) {
std::unique_ptr<TargetInfo> TI = target();
const ABIType *S16 =
@@ -221,6 +279,7 @@ TEST_F(AMDGPUTargetInfoTest, VariadicArgumentPassesDirect) {
FunctionInfo::create(CallingConv::C, Void, {S16}, RequiredArgs(0));
TI->computeInfo(*FI);
expectUncoercedDirect(FI->getArgInfo(0).Info);
+ EXPECT_FALSE(FI->getArgInfo(0).Info.getCanBeFlattened());
}
// Kernel aggregate arguments are passed by reference in the constant address
@@ -235,11 +294,46 @@ TEST_F(AMDGPUTargetInfoTest, KernelAggregateIsIndirectConstant) {
llvm::AMDGPUAS::CONSTANT_ADDRESS);
}
-// Kernel scalar arguments are passed directly.
+// Kernel direct arguments are passed directly and kept intact.
TEST_F(AMDGPUTargetInfoTest, KernelScalarIsDirect) {
std::unique_ptr<FunctionInfo> FI;
std::unique_ptr<TargetInfo> TI;
- expectDirectInteger(classifyArg(I32, FI, TI, CallingConv::AMDGPU_KERNEL), 32);
+ const ArgInfo &Info = classifyArg(I32, FI, TI, CallingConv::AMDGPU_KERNEL);
+ expectDirectInteger(Info, 32);
+ EXPECT_FALSE(Info.getCanBeFlattened());
+}
+
+// Under HIP a generic scalar-pointer kernel argument is coerced to a global
+// pointer and passed directly, kept intact.
+TEST_F(AMDGPUTargetInfoTest, KernelGenericPointerCoercesToGlobalUnderHIP) {
+ std::unique_ptr<FunctionInfo> FI;
+ std::unique_ptr<TargetInfo> TI = hipTarget();
+ const ABIType *GenericPtr =
+ TB.getPointerType(64, llvm::Align(8), llvm::AMDGPUAS::FLAT_ADDRESS);
+ const ArgInfo &Info = classifyKernelArg(GenericPtr, FI, TI);
+ expectDirectPointerAS(Info, llvm::AMDGPUAS::GLOBAL_ADDRESS);
+ EXPECT_FALSE(Info.getCanBeFlattened());
+}
+
+// A pointer already in the global address space is left untouched under HIP.
+TEST_F(AMDGPUTargetInfoTest, KernelGlobalPointerUnchangedUnderHIP) {
+ std::unique_ptr<FunctionInfo> FI;
+ std::unique_ptr<TargetInfo> TI = hipTarget();
+ const ABIType *GlobalPtr =
+ TB.getPointerType(64, llvm::Align(8), llvm::AMDGPUAS::GLOBAL_ADDRESS);
+ expectDirectPointerAS(classifyKernelArg(GlobalPtr, FI, TI),
+ llvm::AMDGPUAS::GLOBAL_ADDRESS);
+}
+
+// Without the HIP rule a generic pointer kernel argument stays generic.
+TEST_F(AMDGPUTargetInfoTest, KernelGenericPointerUnchangedByDefault) {
+ std::unique_ptr<FunctionInfo> FI;
+ std::unique_ptr<TargetInfo> TI;
+ const ABIType *GenericPtr =
+ TB.getPointerType(64, llvm::Align(8), llvm::AMDGPUAS::FLAT_ADDRESS);
+ expectDirectPointerAS(
+ classifyArg(GenericPtr, FI, TI, CallingConv::AMDGPU_KERNEL),
+ llvm::AMDGPUAS::FLAT_ADDRESS);
}
// A void return is ignored.
>From 723bc931b56db28973658ce7d585d7b49faf02fc Mon Sep 17 00:00:00 2001
From: skc7 <Krishna.Sankisa at amd.com>
Date: Fri, 18 Sep 2026 13:21:52 +0530
Subject: [PATCH 3/4] fix unittest
---
llvm/unittests/ABI/AMDGPUTargetInfoTest.cpp | 10 +++++-----
1 file changed, 5 insertions(+), 5 deletions(-)
diff --git a/llvm/unittests/ABI/AMDGPUTargetInfoTest.cpp b/llvm/unittests/ABI/AMDGPUTargetInfoTest.cpp
index b85d0f9e4f058f..eb7cd65de867f6 100644
--- a/llvm/unittests/ABI/AMDGPUTargetInfoTest.cpp
+++ b/llvm/unittests/ABI/AMDGPUTargetInfoTest.cpp
@@ -53,7 +53,7 @@ class AMDGPUTargetInfoTest : public ::testing::Test {
F32(TB.getFloatType(llvm::APFloat::IEEEsingle(), llvm::Align(4))),
Void(TB.getVoidType()),
Empty(TB.getRecordType({}, llvm::TypeSize::getFixed(8), llvm::Align(1),
- StructPacking::Default, {}, {},
+ llvm::Align(1), StructPacking::Default, {}, {},
RecordFlags::CanPassInRegisters)) {}
std::unique_ptr<TargetInfo> target() const {
@@ -80,8 +80,8 @@ class AMDGPUTargetInfoTest : public ::testing::Test {
const ABIType *recordOf(llvm::ArrayRef<FieldInfo> Fields, uint64_t SizeInBits,
llvm::Align Alignment) {
return TB.getRecordType(Fields, llvm::TypeSize::getFixed(SizeInBits),
- Alignment, StructPacking::Default, {}, {},
- RecordFlags::CanPassInRegisters);
+ Alignment, Alignment, StructPacking::Default, {},
+ {}, RecordFlags::CanPassInRegisters);
}
/// The argument classification the target computes for a single parameter
@@ -263,7 +263,7 @@ TEST_F(AMDGPUTargetInfoTest, NonTrivialRecordIsIndirectPrivate) {
std::unique_ptr<TargetInfo> TI;
const ABIType *CannotPass = TB.getRecordType(
{FieldInfo(I32, 0)}, llvm::TypeSize::getFixed(32), llvm::Align(4),
- StructPacking::Default, {}, {}, RecordFlags::IsCXXRecord);
+ llvm::Align(4), StructPacking::Default, {}, {}, RecordFlags::IsCXXRecord);
expectIndirect(classifyArg(CannotPass, FI, TI), llvm::Align(4),
/*ByVal=*/false, llvm::AMDGPUAS::PRIVATE_ADDRESS);
}
@@ -359,7 +359,7 @@ TEST_F(AMDGPUTargetInfoTest, NonTrivialRecordReturnIsIndirect) {
std::unique_ptr<TargetInfo> TI;
const ABIType *CannotPass = TB.getRecordType(
{FieldInfo(I32, 0)}, llvm::TypeSize::getFixed(32), llvm::Align(4),
- StructPacking::Default, {}, {}, RecordFlags::IsCXXRecord);
+ llvm::Align(4), StructPacking::Default, {}, {}, RecordFlags::IsCXXRecord);
const ArgInfo &Info = classifyRet(CannotPass, FI, TI);
ASSERT_TRUE(Info.isIndirect());
EXPECT_FALSE(Info.getIndirectByVal());
>From 39c1ce606294999af1383359d5397def1aa656a3 Mon Sep 17 00:00:00 2001
From: skc7 <Krishna.Sankisa at amd.com>
Date: Mon, 21 Sep 2026 12:24:58 +0530
Subject: [PATCH 4/4] update
---
llvm/lib/ABI/Targets/AMDGPU.cpp | 22 +++++++++++++--------
llvm/unittests/ABI/AMDGPUTargetInfoTest.cpp | 12 +++++++++++
2 files changed, 26 insertions(+), 8 deletions(-)
diff --git a/llvm/lib/ABI/Targets/AMDGPU.cpp b/llvm/lib/ABI/Targets/AMDGPU.cpp
index 1c27afd598f327..49785390ffeece 100644
--- a/llvm/lib/ABI/Targets/AMDGPU.cpp
+++ b/llvm/lib/ABI/Targets/AMDGPU.cpp
@@ -20,10 +20,12 @@
namespace llvm {
namespace abi {
-class AMDGPUTargetInfo : public TargetInfo {
+class AMDGPUTargetInfo final : public TargetInfo {
private:
static const unsigned MaxNumRegsForArgsRet = 16;
+ ABICompatInfo CompatInfo;
+
/// HIP coerces a generic scalar-pointer kernel argument to the global
/// address space. Gated by the front end, which alone can see LangOpts.HIP.
bool CoerceGenericPtrArgToGlobal;
@@ -38,18 +40,20 @@ class AMDGPUTargetInfo : public TargetInfo {
ArgInfo classifyDefaultReturnType(const Type *Ty) const;
/// Estimate number of registers the type will use when passed in registers.
- uint64_t numRegsForType(const Type *Ty) const;
+ uint64_t getNumRegsForType(const Type *Ty) const;
public:
AMDGPUTargetInfo(TypeBuilder &TypeBuilder, const ABICompatInfo &Compat,
bool CoerceGenericPtrArgToGlobal)
- : TargetInfo(TypeBuilder, Compat),
+ : TargetInfo(TypeBuilder), CompatInfo(Compat),
CoerceGenericPtrArgToGlobal(CoerceGenericPtrArgToGlobal) {}
+ const ABICompatInfo &getABICompatInfo() const override { return CompatInfo; }
+
void computeInfo(FunctionInfo &FI) const override;
};
-uint64_t AMDGPUTargetInfo::numRegsForType(const Type *Ty) const {
+uint64_t AMDGPUTargetInfo::getNumRegsForType(const Type *Ty) const {
uint64_t NumRegs = 0;
if (const auto *VT = dyn_cast<VectorType>(Ty)) {
@@ -69,7 +73,7 @@ uint64_t AMDGPUTargetInfo::numRegsForType(const Type *Ty) const {
if (const auto *RT = dyn_cast<RecordType>(Ty)) {
for (const FieldInfo &Field : RT->getFields())
- NumRegs += numRegsForType(Field.FieldType);
+ NumRegs += getNumRegsForType(Field.FieldType);
return NumRegs;
}
@@ -161,7 +165,7 @@ ArgInfo AMDGPUTargetInfo::classifyReturnType(const Type *RetTy) const {
return ArgInfo::getDirect(TB.getArrayType(I32Ty, 2, /*SizeInBits=*/64));
}
- if (numRegsForType(RetTy) <= MaxNumRegsForArgsRet)
+ if (getNumRegsForType(RetTy) <= MaxNumRegsForArgsRet)
return ArgInfo::getDirect();
}
}
@@ -248,7 +252,7 @@ ArgInfo AMDGPUTargetInfo::classifyArgumentType(const Type *Ty, bool Variadic,
}
if (NumRegsLeft > 0) {
- uint64_t NumRegs = numRegsForType(Ty);
+ uint64_t NumRegs = getNumRegsForType(Ty);
if (NumRegsLeft >= NumRegs) {
NumRegsLeft -= NumRegs;
return ArgInfo::getDirect();
@@ -263,7 +267,7 @@ ArgInfo AMDGPUTargetInfo::classifyArgumentType(const Type *Ty, bool Variadic,
// Otherwise just do the default thing.
ArgInfo AI = classifyDefaultArgumentType(Ty);
if (!AI.isIndirect()) {
- uint64_t NumRegs = numRegsForType(Ty);
+ uint64_t NumRegs = getNumRegsForType(Ty);
NumRegsLeft -= std::min(NumRegs, uint64_t{NumRegsLeft});
}
@@ -273,6 +277,8 @@ ArgInfo AMDGPUTargetInfo::classifyArgumentType(const Type *Ty, bool Variadic,
void AMDGPUTargetInfo::computeInfo(FunctionInfo &FI) const {
CallingConv::ID CC = FI.getCallingConvention();
+ // Non-trivial C++ records are returned indirectly
+ // in the flat address space.
if (!maybeCommonClassifyReturnType(FI))
FI.getReturnInfo() = classifyReturnType(FI.getReturnType());
diff --git a/llvm/unittests/ABI/AMDGPUTargetInfoTest.cpp b/llvm/unittests/ABI/AMDGPUTargetInfoTest.cpp
index eb7cd65de867f6..d43f7ddfccdaba 100644
--- a/llvm/unittests/ABI/AMDGPUTargetInfoTest.cpp
+++ b/llvm/unittests/ABI/AMDGPUTargetInfoTest.cpp
@@ -268,6 +268,18 @@ TEST_F(AMDGPUTargetInfoTest, NonTrivialRecordIsIndirectPrivate) {
/*ByVal=*/false, llvm::AMDGPUAS::PRIVATE_ADDRESS);
}
+// A non-trivial C++ record return is indirect in the generic (flat) address
+// space with ByVal=false, matching classic's getSRetAddrSpace(LangAS::Default).
+TEST_F(AMDGPUTargetInfoTest, NonTrivialRecordReturnIsIndirectFlat) {
+ std::unique_ptr<FunctionInfo> FI;
+ std::unique_ptr<TargetInfo> TI;
+ const ABIType *CannotPass = TB.getRecordType(
+ {FieldInfo(I32, 0)}, llvm::TypeSize::getFixed(32), llvm::Align(4),
+ llvm::Align(4), StructPacking::Default, {}, {}, RecordFlags::IsCXXRecord);
+ expectIndirect(classifyRet(CannotPass, FI, TI), llvm::Align(4),
+ /*ByVal=*/false, llvm::AMDGPUAS::FLAT_ADDRESS);
+}
+
// A variadic argument bypasses register packing and passes through unchanged,
// kept intact rather than flattened into per-field wire arguments.
TEST_F(AMDGPUTargetInfoTest, VariadicArgumentPassesDirect) {
More information about the llvm-commits
mailing list