[llvm] [ABI][AMDGPU] Add AMDGPU target ABI classifier to the LLVM ABI library (PR #220177)

via llvm-commits llvm-commits at lists.llvm.org
Mon Sep 21 00:21:38 PDT 2026


https://github.com/skc7 updated https://github.com/llvm/llvm-project/pull/220177

>From 1975f4834262630e31d7970cf7de82d9a40e7c5a Mon Sep 17 00:00:00 2001
From: skc7 <Krishna.Sankisa at amd.com>
Date: Tue, 1 Sep 2026 12:35:15 +0530
Subject: [PATCH 1/4] [ABI] Add AMDGPU target ABI classifier to the LLVM ABI
 library

---
 llvm/include/llvm/ABI/TargetInfo.h          |   6 +
 llvm/lib/ABI/CMakeLists.txt                 |   1 +
 llvm/lib/ABI/TargetInfo.cpp                 |  54 ++++
 llvm/lib/ABI/Targets/AMDGPU.cpp             | 248 ++++++++++++++++++
 llvm/lib/ABI/Targets/X86.cpp                |  55 ----
 llvm/unittests/ABI/AMDGPUTargetInfoTest.cpp | 274 ++++++++++++++++++++
 llvm/unittests/ABI/CMakeLists.txt           |   1 +
 7 files changed, 584 insertions(+), 55 deletions(-)
 create mode 100644 llvm/lib/ABI/Targets/AMDGPU.cpp
 create mode 100644 llvm/unittests/ABI/AMDGPUTargetInfoTest.cpp

diff --git a/llvm/include/llvm/ABI/TargetInfo.h b/llvm/include/llvm/ABI/TargetInfo.h
index 6f7f29e70bb3da..a1fe1aa01ba998 100644
--- a/llvm/include/llvm/ABI/TargetInfo.h
+++ b/llvm/include/llvm/ABI/TargetInfo.h
@@ -92,6 +92,10 @@ class TargetInfo {
   /// return Ty unchanged.
   LLVM_ABI const Type *useFirstFieldIfTransparentUnion(const Type *Ty) const;
 
+  /// If the record reduces to a single scalar element,
+  /// return that element type; otherwise null.
+  LLVM_ABI const Type *isSingleElementStruct(const Type *Ty) const;
+
   /// Apply rules for classifying return types that are common to all targets.
   LLVM_ABI bool maybeCommonClassifyReturnType(FunctionInfo &FI) const;
 
@@ -127,6 +131,8 @@ class TargetInfo {
 
 LLVM_ABI std::unique_ptr<TargetInfo> createBPFTargetInfo(TypeBuilder &TB);
 
+LLVM_ABI std::unique_ptr<TargetInfo> createAMDGPUTargetInfo(TypeBuilder &TB);
+
 /// The AVX ABI level for X86 targets.
 enum class X86AVXABILevel {
   None,
diff --git a/llvm/lib/ABI/CMakeLists.txt b/llvm/lib/ABI/CMakeLists.txt
index 39e725ca9fd3b7..54307bae8e07df 100644
--- a/llvm/lib/ABI/CMakeLists.txt
+++ b/llvm/lib/ABI/CMakeLists.txt
@@ -4,6 +4,7 @@ add_llvm_component_library(LLVMABI
   TargetInfo.cpp
   IRTypeMapper.cpp
   Targets/AArch64.cpp
+  Targets/AMDGPU.cpp
   Targets/BPF.cpp
   Targets/X86.cpp
 
diff --git a/llvm/lib/ABI/TargetInfo.cpp b/llvm/lib/ABI/TargetInfo.cpp
index 9ad8cb8f5829a4..3342e642ad1f55 100644
--- a/llvm/lib/ABI/TargetInfo.cpp
+++ b/llvm/lib/ABI/TargetInfo.cpp
@@ -71,6 +71,60 @@ const Type *TargetInfo::useFirstFieldIfTransparentUnion(const Type *Ty) const {
   return Ty;
 }
 
+const Type *TargetInfo::isSingleElementStruct(const Type *Ty) const {
+  const auto *RT = dyn_cast<RecordType>(Ty);
+  if (!RT)
+    return nullptr;
+
+  if (RT->hasFlexibleArrayMember())
+    return nullptr;
+
+  const Type *Found = nullptr;
+
+  for (const auto &Base : RT->getBaseClasses()) {
+    const Type *BaseTy = Base.FieldType;
+    const auto *BaseRT = dyn_cast<RecordType>(BaseTy);
+    if (!BaseRT || BaseRT->isEmpty())
+      continue;
+
+    const Type *Elem = isSingleElementStruct(BaseTy);
+    if (!Elem || Found)
+      return nullptr;
+    Found = Elem;
+  }
+
+  for (const auto &FI : RT->getFields()) {
+    if (FI.isEmpty())
+      continue;
+
+    const Type *FTy = FI.FieldType;
+
+    // A single-element array is transparent for this reduction.
+    while (const auto *AT = dyn_cast<ArrayType>(FTy)) {
+      if (AT->getNumElements() != 1)
+        break;
+      FTy = AT->getElementType();
+    }
+
+    const Type *Elem;
+    if (const auto *InnerRT = dyn_cast<RecordType>(FTy))
+      Elem = isSingleElementStruct(InnerRT);
+    else
+      Elem = FTy;
+    if (!Elem || Found)
+      return nullptr;
+    Found = Elem;
+  }
+
+  if (!Found)
+    return nullptr;
+  // The reduced element must cover the whole record (no tail padding).
+  if (Found->getSizeInBits() != Ty->getSizeInBits())
+    return nullptr;
+
+  return Found;
+}
+
 bool TargetInfo::maybeCommonClassifyReturnType(FunctionInfo &FI) const {
   const abi::Type *Ty = FI.getReturnType();
 
diff --git a/llvm/lib/ABI/Targets/AMDGPU.cpp b/llvm/lib/ABI/Targets/AMDGPU.cpp
new file mode 100644
index 00000000000000..92130664f8b62d
--- /dev/null
+++ b/llvm/lib/ABI/Targets/AMDGPU.cpp
@@ -0,0 +1,248 @@
+//===- AMDGPU.cpp - AMDGPU ABI Implementation ----------------------------===//
+//
+// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
+// See https://llvm.org/LICENSE.txt for license information.
+// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
+//
+//===----------------------------------------------------------------------===//
+
+#include "llvm/ABI/FunctionInfo.h"
+#include "llvm/ABI/TargetInfo.h"
+#include "llvm/ABI/Types.h"
+#include "llvm/Support/AMDGPUAddrSpace.h"
+#include "llvm/Support/Alignment.h"
+#include "llvm/Support/Casting.h"
+#include "llvm/Support/TypeSize.h"
+#include <algorithm>
+#include <cassert>
+#include <cstdint>
+
+namespace llvm {
+namespace abi {
+
+class AMDGPUTargetInfo : public TargetInfo {
+private:
+  TypeBuilder &TB;
+  static const unsigned MaxNumRegsForArgsRet = 16;
+
+  ArgInfo classifyReturnType(const Type *RetTy) const;
+  ArgInfo classifyKernelArgumentType(const Type *Ty) const;
+  ArgInfo classifyArgumentType(const Type *Ty, bool Variadic,
+                               unsigned &NumRegsLeft) const;
+
+  /// Target-independent fallback, mirroring classic CodeGen's DefaultABIInfo.
+  ArgInfo classifyDefaultType(const Type *Ty, bool IsReturn) const;
+
+  /// Estimate number of registers the type will use when passed in registers.
+  uint64_t numRegsForType(const Type *Ty) const;
+
+public:
+  AMDGPUTargetInfo(TypeBuilder &TypeBuilder, const ABICompatInfo &Compat)
+      : TargetInfo(Compat), TB(TypeBuilder) {}
+
+  void computeInfo(FunctionInfo &FI) const override;
+};
+
+uint64_t AMDGPUTargetInfo::numRegsForType(const Type *Ty) const {
+  uint64_t NumRegs = 0;
+
+  if (const auto *VT = dyn_cast<VectorType>(Ty)) {
+    // Compute from the number of elements. The reported size is based on the
+    // in-memory size, which includes the padding 4th element for 3-vectors.
+    const Type *EltTy = VT->getElementType();
+    uint64_t EltSize = EltTy->getSizeInBits().getFixedValue();
+    unsigned NumElts = VT->getNumElements().getFixedValue();
+
+    // 16-bit element vectors should be passed as packed.
+    if (EltSize == 16)
+      return (NumElts + 1) / 2;
+
+    uint64_t EltNumRegs = (EltSize + 31) / 32;
+    return EltNumRegs * NumElts;
+  }
+
+  if (const auto *RT = dyn_cast<RecordType>(Ty)) {
+    for (const FieldInfo &Field : RT->getFields())
+      NumRegs += numRegsForType(Field.FieldType);
+    return NumRegs;
+  }
+
+  return (Ty->getSizeInBits().getFixedValue() + 31) / 32;
+}
+
+ArgInfo AMDGPUTargetInfo::classifyDefaultType(const Type *Ty,
+                                              bool IsReturn) const {
+  if (IsReturn && Ty->isVoid())
+    return ArgInfo::getIgnore();
+
+  if (isAggregateTypeForABI(Ty)) {
+    if (RecordArgABI RAA = getRecordArgABI(Ty); RAA != RAA_Default)
+      return getNaturalAlignIndirect(Ty, /*ByVal=*/RAA == RAA_DirectInMemory);
+    return getNaturalAlignIndirect(Ty, /*ByVal=*/!IsReturn);
+  }
+
+  if (const auto *IT = dyn_cast<IntegerType>(Ty))
+    if (isPromotableInteger(IT))
+      return ArgInfo::getExtend(Ty);
+
+  return ArgInfo::getDirect();
+}
+
+ArgInfo AMDGPUTargetInfo::classifyReturnType(const Type *RetTy) const {
+  if (RetTy->isVoid())
+    return ArgInfo::getIgnore();
+
+  if (isAggregateTypeForABI(RetTy)) {
+    // Records with non-trivial destructors/copy-constructors should not be
+    // returned by value.
+    if (getRecordArgABI(RetTy) == RAA_Default) {
+      const auto *RT = dyn_cast<RecordType>(RetTy);
+
+      // Ignore empty structs/unions.
+      if (RT && RT->isEmpty())
+        return ArgInfo::getIgnore();
+
+      // Lower single-element structs to just return a regular value.
+      if (const Type *SeltTy = isSingleElementStruct(RetTy))
+        return ArgInfo::getDirect(SeltTy);
+
+      if (RT && RT->hasFlexibleArrayMember())
+        return classifyDefaultType(RetTy, /*IsReturn=*/true);
+
+      // Pack aggregates <= 4 bytes into single VGPR or pair.
+      uint64_t Size = RetTy->getSizeInBits().getFixedValue();
+      if (Size <= 16)
+        return ArgInfo::getDirect(TB.getIntegerType(16, Align(2), false));
+
+      if (Size <= 32)
+        return ArgInfo::getDirect(TB.getIntegerType(32, Align(4), false));
+
+      if (Size <= 64) {
+        const Type *I32Ty = TB.getIntegerType(32, Align(4), false);
+        return ArgInfo::getDirect(TB.getArrayType(I32Ty, 2, /*SizeInBits=*/64));
+      }
+
+      if (numRegsForType(RetTy) <= MaxNumRegsForArgsRet)
+        return ArgInfo::getDirect();
+    }
+  }
+
+  // Otherwise just do the default thing.
+  return classifyDefaultType(RetTy, /*IsReturn=*/true);
+}
+
+/// For kernels all parameters are really passed in a special buffer. It doesn't
+/// make sense to pass anything byval, so everything must be direct.
+ArgInfo AMDGPUTargetInfo::classifyKernelArgumentType(const Type *Ty) const {
+  Ty = useFirstFieldIfTransparentUnion(Ty);
+
+  if (const Type *SeltTy = isSingleElementStruct(Ty))
+    Ty = SeltTy;
+
+  // TODO: Classic coerces HIP scalar pointers from generic to global here; that
+  // depends on LangOptions the ABI library cannot see, so it is skipped.
+  if (isAggregateTypeForABI(Ty))
+    return ArgInfo::getIndirect(Ty->getAlignment(), /*ByVal=*/false,
+                                /*AddrSpace=*/AMDGPUAS::CONSTANT_ADDRESS);
+
+  // TODO: Classic passes CanBeFlattened=false here to keep a struct intact;
+  // ArgInfo cannot model that yet, so a multi-field record coerce may be
+  // flattened.
+  return ArgInfo::getDirect(Ty);
+}
+
+ArgInfo AMDGPUTargetInfo::classifyArgumentType(const Type *Ty, bool Variadic,
+                                               unsigned &NumRegsLeft) const {
+  assert(NumRegsLeft <= MaxNumRegsForArgsRet && "register estimate underflow");
+
+  Ty = useFirstFieldIfTransparentUnion(Ty);
+
+  // TODO: Classic sets CanBeFlattened=false for variadics; not modeled here.
+  if (Variadic)
+    return ArgInfo::getDirect();
+
+  if (isAggregateTypeForABI(Ty)) {
+    // Records with non-trivial destructors/copy-constructors should not be
+    // passed by value.
+    if (RecordArgABI RAA = getRecordArgABI(Ty); RAA != RAA_Default)
+      return ArgInfo::getIndirect(Ty->getAlignment(),
+                                  /*ByVal=*/RAA == RAA_DirectInMemory,
+                                  /*AddrSpace=*/AMDGPUAS::PRIVATE_ADDRESS);
+
+    // Ignore empty structs/unions.
+    if (const auto *RT = dyn_cast<RecordType>(Ty); RT && RT->isEmpty())
+      return ArgInfo::getIgnore();
+
+    // Lower single-element structs to just pass a regular value.
+    if (const Type *SeltTy = isSingleElementStruct(Ty))
+      return ArgInfo::getDirect(SeltTy);
+
+    if (const auto *RT = dyn_cast<RecordType>(Ty);
+        RT && RT->hasFlexibleArrayMember())
+      return classifyDefaultType(Ty, /*IsReturn=*/false);
+
+    // Pack aggregates <= 8 bytes into single VGPR or pair.
+    uint64_t Size = Ty->getSizeInBits().getFixedValue();
+    if (Size <= 64) {
+      unsigned NumRegs = (Size + 31) / 32;
+      NumRegsLeft -= std::min(NumRegsLeft, NumRegs);
+
+      if (Size <= 16)
+        return ArgInfo::getDirect(TB.getIntegerType(16, Align(2), false));
+
+      if (Size <= 32)
+        return ArgInfo::getDirect(TB.getIntegerType(32, Align(4), false));
+
+      const Type *I32Ty = TB.getIntegerType(32, Align(4), false);
+      return ArgInfo::getDirect(TB.getArrayType(I32Ty, 2, /*SizeInBits=*/64));
+    }
+
+    if (NumRegsLeft > 0) {
+      uint64_t NumRegs = numRegsForType(Ty);
+      if (NumRegsLeft >= NumRegs) {
+        NumRegsLeft -= NumRegs;
+        return ArgInfo::getDirect();
+      }
+    }
+
+    // Pass a struct argument by reference rather than by value.
+    return ArgInfo::getIndirect(Ty->getAlignment(), /*ByVal=*/false,
+                                /*AddrSpace=*/AMDGPUAS::PRIVATE_ADDRESS);
+  }
+
+  // Otherwise just do the default thing.
+  ArgInfo AI = classifyDefaultType(Ty, /*IsReturn=*/false);
+  if (!AI.isIndirect()) {
+    uint64_t NumRegs = numRegsForType(Ty);
+    NumRegsLeft -= std::min(NumRegs, uint64_t{NumRegsLeft});
+  }
+
+  return AI;
+}
+
+void AMDGPUTargetInfo::computeInfo(FunctionInfo &FI) const {
+  CallingConv::ID CC = FI.getCallingConvention();
+
+  if (!maybeCommonClassifyReturnType(FI))
+    FI.getReturnInfo() = classifyReturnType(FI.getReturnType());
+
+  unsigned ArgumentIndex = 0;
+  const unsigned NumFixedArguments = FI.getNumRequiredArgs();
+
+  unsigned NumRegsLeft = MaxNumRegsForArgsRet;
+  for (ArgEntry &Arg : FI.arguments()) {
+    if (CC == CallingConv::AMDGPU_KERNEL) {
+      Arg.Info = classifyKernelArgumentType(Arg.ABIType);
+    } else {
+      bool FixedArgument = ArgumentIndex++ < NumFixedArguments;
+      Arg.Info = classifyArgumentType(Arg.ABIType, !FixedArgument, NumRegsLeft);
+    }
+  }
+}
+
+std::unique_ptr<TargetInfo> createAMDGPUTargetInfo(TypeBuilder &TB) {
+  return std::make_unique<AMDGPUTargetInfo>(TB, ABICompatInfo());
+}
+
+} // namespace abi
+} // namespace llvm
diff --git a/llvm/lib/ABI/Targets/X86.cpp b/llvm/lib/ABI/Targets/X86.cpp
index 78aa9cd1a25190..20f174db5d698b 100644
--- a/llvm/lib/ABI/Targets/X86.cpp
+++ b/llvm/lib/ABI/Targets/X86.cpp
@@ -98,7 +98,6 @@ class X86_64TargetInfo : public TargetInfo {
   ArgInfo getIndirectReturnResult(const Type *Ty) const;
   const Type *getFPTypeAtOffset(const Type *Ty, unsigned Offset) const;
 
-  const Type *isSingleElementStruct(const Type *Ty) const;
   const Type *getByteVectorType(const Type *Ty) const;
 
   const Type *createPairType(const Type *Lo, const Type *Hi) const;
@@ -1261,60 +1260,6 @@ const Type *X86_64TargetInfo::getByteVectorType(const Type *Ty) const {
                           ElementCount::getFixed(Size / 64), Align(Size / 8));
 }
 
-// Returns the single element if this is a single-element struct wrapper
-const Type *X86_64TargetInfo::isSingleElementStruct(const Type *Ty) const {
-  const auto *RT = dyn_cast<RecordType>(Ty);
-  if (!RT)
-    return nullptr;
-
-  if (RT->hasFlexibleArrayMember())
-    return nullptr;
-
-  const Type *Found = nullptr;
-
-  for (const auto &Base : RT->getBaseClasses()) {
-    const Type *BaseTy = Base.FieldType;
-    auto *BaseRT = dyn_cast<RecordType>(BaseTy);
-
-    if (!BaseRT || BaseRT->isEmpty())
-      continue;
-
-    const Type *Elem = isSingleElementStruct(BaseTy);
-    if (!Elem || Found)
-      return nullptr;
-    Found = Elem;
-  }
-
-  for (const auto &FI : RT->getFields()) {
-    if (FI.isEmpty())
-      continue;
-
-    const Type *FTy = FI.FieldType;
-
-    while (auto *AT = dyn_cast<ArrayType>(FTy)) {
-      if (AT->getNumElements() != 1)
-        break;
-      FTy = AT->getElementType();
-    }
-
-    const Type *Elem;
-    if (auto *InnerRT = dyn_cast<RecordType>(FTy))
-      Elem = isSingleElementStruct(InnerRT);
-    else
-      Elem = FTy;
-    if (!Elem || Found)
-      return nullptr;
-    Found = Elem;
-  }
-
-  if (!Found)
-    return nullptr;
-  if (Found->getSizeInBits() != Ty->getSizeInBits())
-    return nullptr;
-
-  return Found;
-}
-
 bool X86_64TargetInfo::isIllegalVectorType(const Type *Ty) const {
   if (const auto *VecTy = dyn_cast<VectorType>(Ty)) {
     uint64_t Size = VecTy->getSizeInBits().getFixedValue();
diff --git a/llvm/unittests/ABI/AMDGPUTargetInfoTest.cpp b/llvm/unittests/ABI/AMDGPUTargetInfoTest.cpp
new file mode 100644
index 00000000000000..38d2bbfc37017a
--- /dev/null
+++ b/llvm/unittests/ABI/AMDGPUTargetInfoTest.cpp
@@ -0,0 +1,274 @@
+//===- AMDGPUTargetInfoTest.cpp - AMDGPU ABI unit tests -------------------===//
+//
+// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
+// See https://llvm.org/LICENSE.txt for license information.
+// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
+//
+//===----------------------------------------------------------------------===//
+
+#include "llvm/ABI/FunctionInfo.h"
+#include "llvm/ABI/TargetInfo.h"
+#include "llvm/ABI/Types.h"
+#include "llvm/ADT/APFloat.h"
+#include "llvm/IR/CallingConv.h"
+#include "llvm/Support/AMDGPUAddrSpace.h"
+#include "llvm/Support/Alignment.h"
+#include "llvm/Support/Allocator.h"
+#include "gtest/gtest.h"
+
+namespace {
+
+// RecordFlags' bitmask operators are declared in namespace llvm, so combining
+// two of them needs that namespace visible.
+using namespace llvm;
+
+using ABIType = llvm::abi::Type;
+using llvm::abi::ABICompatInfo;
+using llvm::abi::ArgInfo;
+using llvm::abi::createAMDGPUTargetInfo;
+using llvm::abi::FieldInfo;
+using llvm::abi::FunctionInfo;
+using llvm::abi::RecordFlags;
+using llvm::abi::RequiredArgs;
+using llvm::abi::StructPacking;
+using llvm::abi::TargetInfo;
+using llvm::abi::TypeBuilder;
+
+class AMDGPUTargetInfoTest : public ::testing::Test {
+protected:
+  llvm::BumpPtrAllocator Alloc;
+  TypeBuilder TB;
+  const ABIType *I8;
+  const ABIType *I16;
+  const ABIType *I32;
+  const ABIType *F32;
+  const ABIType *Void;
+  /// An empty class: a record with no fields, one byte wide, register-passable.
+  const ABIType *Empty;
+
+  AMDGPUTargetInfoTest()
+      : TB(Alloc), I8(TB.getIntegerType(8, llvm::Align(1), /*Signed=*/true)),
+        I16(TB.getIntegerType(16, llvm::Align(2), /*Signed=*/true)),
+        I32(TB.getIntegerType(32, llvm::Align(4), /*Signed=*/true)),
+        F32(TB.getFloatType(llvm::APFloat::IEEEsingle(), llvm::Align(4))),
+        Void(TB.getVoidType()),
+        Empty(TB.getRecordType({}, llvm::TypeSize::getFixed(8), llvm::Align(1),
+                               StructPacking::Default, {}, {},
+                               RecordFlags::CanPassInRegisters)) {}
+
+  std::unique_ptr<TargetInfo> target() const {
+    return createAMDGPUTargetInfo(const_cast<TypeBuilder &>(TB));
+  }
+
+  /// A register-passable record with the given fields, size and alignment.
+  const ABIType *recordOf(llvm::ArrayRef<FieldInfo> Fields, uint64_t SizeInBits,
+                          llvm::Align Alignment) {
+    return TB.getRecordType(Fields, llvm::TypeSize::getFixed(SizeInBits),
+                            Alignment, StructPacking::Default, {}, {},
+                            RecordFlags::CanPassInRegisters);
+  }
+
+  /// The argument classification the target computes for a single parameter
+  /// under calling convention \p CC.
+  const ArgInfo &classifyArg(const ABIType *ArgTy,
+                             std::unique_ptr<FunctionInfo> &FI,
+                             std::unique_ptr<TargetInfo> &TI,
+                             CallingConv::ID CC = CallingConv::C) {
+    TI = target();
+    FI = FunctionInfo::create(CC, Void, {ArgTy});
+    TI->computeInfo(*FI);
+    return FI->getArgInfo(0).Info;
+  }
+
+  /// The return classification the target computes for \p RetTy.
+  const ArgInfo &classifyRet(const ABIType *RetTy,
+                             std::unique_ptr<FunctionInfo> &FI,
+                             std::unique_ptr<TargetInfo> &TI) {
+    TI = target();
+    FI = FunctionInfo::create(CallingConv::C, RetTy, {});
+    TI->computeInfo(*FI);
+    return FI->getReturnInfo();
+  }
+};
+
+static void expectUncoercedDirect(const ArgInfo &Info) {
+  ASSERT_TRUE(Info.isDirect());
+  EXPECT_EQ(Info.getCoerceToType(), nullptr);
+}
+
+static void expectDirectInteger(const ArgInfo &Info, unsigned Bits) {
+  ASSERT_TRUE(Info.isDirect());
+  const ABIType *Coerce = Info.getCoerceToType();
+  ASSERT_NE(Coerce, nullptr);
+  const auto *IT = llvm::dyn_cast<llvm::abi::IntegerType>(Coerce);
+  ASSERT_NE(IT, nullptr);
+  EXPECT_EQ(IT->getSizeInBits().getFixedValue(), Bits);
+}
+
+static void expectDirectFloat(const ArgInfo &Info,
+                              const llvm::fltSemantics &Sem) {
+  ASSERT_TRUE(Info.isDirect());
+  const ABIType *Coerce = Info.getCoerceToType();
+  ASSERT_NE(Coerce, nullptr);
+  const auto *FT = llvm::dyn_cast<llvm::abi::FloatType>(Coerce);
+  ASSERT_NE(FT, nullptr);
+  EXPECT_EQ(FT->getSemantics(), &Sem);
+}
+
+// A <= 8-byte aggregate coerces to [2 x i32].
+static void expectDirectI32Pair(const ArgInfo &Info) {
+  ASSERT_TRUE(Info.isDirect());
+  const auto *AT =
+      llvm::dyn_cast_or_null<llvm::abi::ArrayType>(Info.getCoerceToType());
+  ASSERT_NE(AT, nullptr);
+  EXPECT_EQ(AT->getNumElements(), 2u);
+  const auto *IT = llvm::dyn_cast<llvm::abi::IntegerType>(AT->getElementType());
+  ASSERT_NE(IT, nullptr);
+  EXPECT_EQ(IT->getSizeInBits().getFixedValue(), 32u);
+}
+
+static void expectIndirect(const ArgInfo &Info, llvm::Align ExpectedAlign,
+                           bool ByVal, unsigned AddrSpace) {
+  ASSERT_TRUE(Info.isIndirect());
+  EXPECT_EQ(Info.getIndirectAlign(), ExpectedAlign);
+  EXPECT_EQ(Info.getIndirectByVal(), ByVal);
+  EXPECT_EQ(Info.getIndirectAddrSpace(), AddrSpace);
+}
+
+// A 32-bit integer and a pointer-sized scalar pass directly in their own type.
+TEST_F(AMDGPUTargetInfoTest, ScalarPassesDirect) {
+  std::unique_ptr<FunctionInfo> FI;
+  std::unique_ptr<TargetInfo> TI;
+  expectUncoercedDirect(classifyArg(I32, FI, TI));
+  expectUncoercedDirect(classifyArg(F32, FI, TI));
+}
+
+// A sub-word integer is sign/zero extended to fill its register.
+TEST_F(AMDGPUTargetInfoTest, PromotableIntegerExtends) {
+  std::unique_ptr<FunctionInfo> FI;
+  std::unique_ptr<TargetInfo> TI;
+  const ArgInfo &Info = classifyArg(I8, FI, TI);
+  ASSERT_TRUE(Info.isExtend());
+  EXPECT_TRUE(Info.isSignExt());
+}
+
+// Aggregates <= 16/32/64 bits pack into i16 / i32 / [2 x i32].
+TEST_F(AMDGPUTargetInfoTest, SmallAggregatesPackIntoRegisters) {
+  std::unique_ptr<FunctionInfo> FI;
+  std::unique_ptr<TargetInfo> TI;
+
+  const ABIType *S16 =
+      recordOf({FieldInfo(I8, 0), FieldInfo(I8, 8)}, 16, llvm::Align(1));
+  expectDirectInteger(classifyArg(S16, FI, TI), 16);
+
+  const ABIType *S32 =
+      recordOf({FieldInfo(I16, 0), FieldInfo(I16, 16)}, 32, llvm::Align(2));
+  expectDirectInteger(classifyArg(S32, FI, TI), 32);
+
+  const ABIType *S64 =
+      recordOf({FieldInfo(I32, 0), FieldInfo(I32, 32)}, 64, llvm::Align(4));
+  expectDirectI32Pair(classifyArg(S64, FI, TI));
+}
+
+// An empty struct is dropped from the argument list.
+TEST_F(AMDGPUTargetInfoTest, EmptyAggregateIsIgnored) {
+  std::unique_ptr<FunctionInfo> FI;
+  std::unique_ptr<TargetInfo> TI;
+  EXPECT_TRUE(classifyArg(Empty, FI, TI).isIgnore());
+}
+
+// A single-element struct is passed as its inner scalar.
+TEST_F(AMDGPUTargetInfoTest, SingleElementStructUnwraps) {
+  std::unique_ptr<FunctionInfo> FI;
+  std::unique_ptr<TargetInfo> TI;
+  const ABIType *Wrapper = recordOf({FieldInfo(F32, 0)}, 32, llvm::Align(4));
+  expectDirectFloat(classifyArg(Wrapper, FI, TI), llvm::APFloat::IEEEsingle());
+}
+
+// A large aggregate that does not fit the 16-register budget is passed by
+// reference in the private address space.
+TEST_F(AMDGPUTargetInfoTest, OversizedAggregateIsIndirectPrivate) {
+  std::unique_ptr<FunctionInfo> FI;
+  std::unique_ptr<TargetInfo> TI;
+  // Two i32[10] fields => 20 registers, above MaxNumRegsForArgsRet (16).
+  const ABIType *ArrTy = TB.getArrayType(I32, /*NumElements=*/10,
+                                         /*SizeInBits=*/320);
+  const ABIType *Big = recordOf({FieldInfo(ArrTy, 0), FieldInfo(ArrTy, 320)},
+                                640, llvm::Align(4));
+  expectIndirect(classifyArg(Big, FI, TI), llvm::Align(4), /*ByVal=*/false,
+                 llvm::AMDGPUAS::PRIVATE_ADDRESS);
+}
+
+// A record that cannot pass in registers (non-trivial C++ type) is passed
+// indirectly in the private address space.
+TEST_F(AMDGPUTargetInfoTest, NonTrivialRecordIsIndirectPrivate) {
+  std::unique_ptr<FunctionInfo> FI;
+  std::unique_ptr<TargetInfo> TI;
+  const ABIType *CannotPass = TB.getRecordType(
+      {FieldInfo(I32, 0)}, llvm::TypeSize::getFixed(32), llvm::Align(4),
+      StructPacking::Default, {}, {}, RecordFlags::IsCXXRecord);
+  expectIndirect(classifyArg(CannotPass, FI, TI), llvm::Align(4),
+                 /*ByVal=*/false, llvm::AMDGPUAS::PRIVATE_ADDRESS);
+}
+
+// A variadic argument bypasses register packing and passes through unchanged.
+TEST_F(AMDGPUTargetInfoTest, VariadicArgumentPassesDirect) {
+  std::unique_ptr<TargetInfo> TI = target();
+  const ABIType *S16 =
+      recordOf({FieldInfo(I8, 0), FieldInfo(I8, 8)}, 16, llvm::Align(1));
+  // Zero declared parameters, so the sole argument is variadic.
+  std::unique_ptr<FunctionInfo> FI =
+      FunctionInfo::create(CallingConv::C, Void, {S16}, RequiredArgs(0));
+  TI->computeInfo(*FI);
+  expectUncoercedDirect(FI->getArgInfo(0).Info);
+}
+
+// Kernel aggregate arguments are passed by reference in the constant address
+// space (the kernarg segment), never byval.
+TEST_F(AMDGPUTargetInfoTest, KernelAggregateIsIndirectConstant) {
+  std::unique_ptr<FunctionInfo> FI;
+  std::unique_ptr<TargetInfo> TI;
+  const ABIType *S =
+      recordOf({FieldInfo(I32, 0), FieldInfo(I32, 32)}, 64, llvm::Align(4));
+  expectIndirect(classifyArg(S, FI, TI, CallingConv::AMDGPU_KERNEL),
+                 llvm::Align(4), /*ByVal=*/false,
+                 llvm::AMDGPUAS::CONSTANT_ADDRESS);
+}
+
+// Kernel scalar arguments are passed directly.
+TEST_F(AMDGPUTargetInfoTest, KernelScalarIsDirect) {
+  std::unique_ptr<FunctionInfo> FI;
+  std::unique_ptr<TargetInfo> TI;
+  expectDirectInteger(classifyArg(I32, FI, TI, CallingConv::AMDGPU_KERNEL), 32);
+}
+
+// A void return is ignored.
+TEST_F(AMDGPUTargetInfoTest, VoidReturnIsIgnored) {
+  std::unique_ptr<FunctionInfo> FI;
+  std::unique_ptr<TargetInfo> TI;
+  EXPECT_TRUE(classifyRet(Void, FI, TI).isIgnore());
+}
+
+// A <= 8-byte aggregate return packs into [2 x i32].
+TEST_F(AMDGPUTargetInfoTest, SmallAggregateReturnPacks) {
+  std::unique_ptr<FunctionInfo> FI;
+  std::unique_ptr<TargetInfo> TI;
+  const ABIType *S64 =
+      recordOf({FieldInfo(I32, 0), FieldInfo(I32, 32)}, 64, llvm::Align(4));
+  expectDirectI32Pair(classifyRet(S64, FI, TI));
+}
+
+// A record that cannot pass in registers is returned indirectly (sret-style,
+// ByVal=false) via the target-independent return rule.
+TEST_F(AMDGPUTargetInfoTest, NonTrivialRecordReturnIsIndirect) {
+  std::unique_ptr<FunctionInfo> FI;
+  std::unique_ptr<TargetInfo> TI;
+  const ABIType *CannotPass = TB.getRecordType(
+      {FieldInfo(I32, 0)}, llvm::TypeSize::getFixed(32), llvm::Align(4),
+      StructPacking::Default, {}, {}, RecordFlags::IsCXXRecord);
+  const ArgInfo &Info = classifyRet(CannotPass, FI, TI);
+  ASSERT_TRUE(Info.isIndirect());
+  EXPECT_FALSE(Info.getIndirectByVal());
+}
+
+} // namespace
diff --git a/llvm/unittests/ABI/CMakeLists.txt b/llvm/unittests/ABI/CMakeLists.txt
index 4a1910ab8809b4..ffd5e82ef082cb 100644
--- a/llvm/unittests/ABI/CMakeLists.txt
+++ b/llvm/unittests/ABI/CMakeLists.txt
@@ -8,6 +8,7 @@ add_llvm_unittest(ABITests
   AArch64TargetInfoTest.cpp
   IRTypeMapperTest.cpp
   FunctionInfoTest.cpp
+  AMDGPUTargetInfoTest.cpp
   X86TargetInfoTest.cpp
   TypesTest.cpp
   )

>From ef132d0700f0209d4ad545f24e6078ff12c80380 Mon Sep 17 00:00:00 2001
From: skc7 <Krishna.Sankisa at amd.com>
Date: Thu, 17 Sep 2026 16:24:53 +0530
Subject: [PATCH 2/4] support CoerceGenericPtrArgToGlobal

---
 llvm/include/llvm/ABI/TargetInfo.h          |   4 +-
 llvm/lib/ABI/Targets/AMDGPU.cpp             | 110 ++++++++++++++------
 llvm/unittests/ABI/AMDGPUTargetInfoTest.cpp | 102 +++++++++++++++++-
 3 files changed, 182 insertions(+), 34 deletions(-)

diff --git a/llvm/include/llvm/ABI/TargetInfo.h b/llvm/include/llvm/ABI/TargetInfo.h
index a1fe1aa01ba998..31157b4a9c9ac8 100644
--- a/llvm/include/llvm/ABI/TargetInfo.h
+++ b/llvm/include/llvm/ABI/TargetInfo.h
@@ -131,7 +131,9 @@ class TargetInfo {
 
 LLVM_ABI std::unique_ptr<TargetInfo> createBPFTargetInfo(TypeBuilder &TB);
 
-LLVM_ABI std::unique_ptr<TargetInfo> createAMDGPUTargetInfo(TypeBuilder &TB);
+LLVM_ABI std::unique_ptr<TargetInfo>
+createAMDGPUTargetInfo(TypeBuilder &TB,
+                       bool CoerceGenericPtrArgToGlobal = false);
 
 /// The AVX ABI level for X86 targets.
 enum class X86AVXABILevel {
diff --git a/llvm/lib/ABI/Targets/AMDGPU.cpp b/llvm/lib/ABI/Targets/AMDGPU.cpp
index 92130664f8b62d..1c27afd598f327 100644
--- a/llvm/lib/ABI/Targets/AMDGPU.cpp
+++ b/llvm/lib/ABI/Targets/AMDGPU.cpp
@@ -22,23 +22,29 @@ namespace abi {
 
 class AMDGPUTargetInfo : public TargetInfo {
 private:
-  TypeBuilder &TB;
   static const unsigned MaxNumRegsForArgsRet = 16;
 
+  /// HIP coerces a generic scalar-pointer kernel argument to the global
+  /// address space. Gated by the front end, which alone can see LangOpts.HIP.
+  bool CoerceGenericPtrArgToGlobal;
+
   ArgInfo classifyReturnType(const Type *RetTy) const;
   ArgInfo classifyKernelArgumentType(const Type *Ty) const;
   ArgInfo classifyArgumentType(const Type *Ty, bool Variadic,
                                unsigned &NumRegsLeft) const;
 
-  /// Target-independent fallback, mirroring classic CodeGen's DefaultABIInfo.
-  ArgInfo classifyDefaultType(const Type *Ty, bool IsReturn) const;
+  /// Target-independent fallback classification for arguments and returns.
+  ArgInfo classifyDefaultArgumentType(const Type *Ty) const;
+  ArgInfo classifyDefaultReturnType(const Type *Ty) const;
 
   /// Estimate number of registers the type will use when passed in registers.
   uint64_t numRegsForType(const Type *Ty) const;
 
 public:
-  AMDGPUTargetInfo(TypeBuilder &TypeBuilder, const ABICompatInfo &Compat)
-      : TargetInfo(Compat), TB(TypeBuilder) {}
+  AMDGPUTargetInfo(TypeBuilder &TypeBuilder, const ABICompatInfo &Compat,
+                   bool CoerceGenericPtrArgToGlobal)
+      : TargetInfo(TypeBuilder, Compat),
+        CoerceGenericPtrArgToGlobal(CoerceGenericPtrArgToGlobal) {}
 
   void computeInfo(FunctionInfo &FI) const override;
 };
@@ -70,22 +76,55 @@ uint64_t AMDGPUTargetInfo::numRegsForType(const Type *Ty) const {
   return (Ty->getSizeInBits().getFixedValue() + 31) / 32;
 }
 
-ArgInfo AMDGPUTargetInfo::classifyDefaultType(const Type *Ty,
-                                              bool IsReturn) const {
-  if (IsReturn && Ty->isVoid())
-    return ArgInfo::getIgnore();
+// Natural-alignment indirect in the alloca address space (PRIVATE on AMDGPU),
+// ByVal by default.
+static ArgInfo getNaturalAlignIndirectPrivate(const Type *Ty,
+                                              bool ByVal = true) {
+  return ArgInfo::getIndirect(Ty->getAlignment(), ByVal,
+                              /*AddrSpace=*/AMDGPUAS::PRIVATE_ADDRESS);
+}
+
+// A _BitInt wider than int128 is passed indirectly. AMDGPU has int128, so the
+// threshold is 128 bits.
+static bool isOversizedBitInt(const Type *Ty) {
+  const auto *IT = dyn_cast<IntegerType>(Ty);
+  return IT && IT->isBitInt() && IT->getSizeInBits().getFixedValue() > 128;
+}
+
+// Default argument classification for types not handled by the AMDGPU rules.
+ArgInfo AMDGPUTargetInfo::classifyDefaultArgumentType(const Type *Ty) const {
+  Ty = useFirstFieldIfTransparentUnion(Ty);
 
   if (isAggregateTypeForABI(Ty)) {
     if (RecordArgABI RAA = getRecordArgABI(Ty); RAA != RAA_Default)
-      return getNaturalAlignIndirect(Ty, /*ByVal=*/RAA == RAA_DirectInMemory);
-    return getNaturalAlignIndirect(Ty, /*ByVal=*/!IsReturn);
+      return getNaturalAlignIndirectPrivate(Ty,
+                                            /*ByVal=*/RAA ==
+                                                RAA_DirectInMemory);
+    return getNaturalAlignIndirectPrivate(Ty);
   }
 
-  if (const auto *IT = dyn_cast<IntegerType>(Ty))
-    if (isPromotableInteger(IT))
-      return ArgInfo::getExtend(Ty);
+  if (isOversizedBitInt(Ty))
+    return getNaturalAlignIndirectPrivate(Ty);
 
-  return ArgInfo::getDirect();
+  const auto *IT = dyn_cast<IntegerType>(Ty);
+  return (IT && isPromotableInteger(IT)) ? ArgInfo::getExtend(Ty)
+                                         : ArgInfo::getDirect();
+}
+
+// Default return classification for types not handled by the AMDGPU rules.
+ArgInfo AMDGPUTargetInfo::classifyDefaultReturnType(const Type *Ty) const {
+  if (Ty->isVoid())
+    return ArgInfo::getIgnore();
+
+  if (isAggregateTypeForABI(Ty))
+    return getNaturalAlignIndirectPrivate(Ty);
+
+  if (isOversizedBitInt(Ty))
+    return getNaturalAlignIndirectPrivate(Ty);
+
+  const auto *IT = dyn_cast<IntegerType>(Ty);
+  return (IT && isPromotableInteger(IT)) ? ArgInfo::getExtend(Ty)
+                                         : ArgInfo::getDirect();
 }
 
 ArgInfo AMDGPUTargetInfo::classifyReturnType(const Type *RetTy) const {
@@ -107,7 +146,7 @@ ArgInfo AMDGPUTargetInfo::classifyReturnType(const Type *RetTy) const {
         return ArgInfo::getDirect(SeltTy);
 
       if (RT && RT->hasFlexibleArrayMember())
-        return classifyDefaultType(RetTy, /*IsReturn=*/true);
+        return classifyDefaultReturnType(RetTy);
 
       // Pack aggregates <= 4 bytes into single VGPR or pair.
       uint64_t Size = RetTy->getSizeInBits().getFixedValue();
@@ -128,7 +167,7 @@ ArgInfo AMDGPUTargetInfo::classifyReturnType(const Type *RetTy) const {
   }
 
   // Otherwise just do the default thing.
-  return classifyDefaultType(RetTy, /*IsReturn=*/true);
+  return classifyDefaultReturnType(RetTy);
 }
 
 /// For kernels all parameters are really passed in a special buffer. It doesn't
@@ -139,16 +178,26 @@ ArgInfo AMDGPUTargetInfo::classifyKernelArgumentType(const Type *Ty) const {
   if (const Type *SeltTy = isSingleElementStruct(Ty))
     Ty = SeltTy;
 
-  // TODO: Classic coerces HIP scalar pointers from generic to global here; that
-  // depends on LangOptions the ABI library cannot see, so it is skipped.
+  // HIP passes a generic scalar pointer as a global pointer; a pointer is not
+  // an aggregate, so this stays on the direct path.
+  if (CoerceGenericPtrArgToGlobal) {
+    if (const auto *PtrTy = dyn_cast<PointerType>(Ty);
+        PtrTy && PtrTy->getAddrSpace() == AMDGPUAS::FLAT_ADDRESS) {
+      const Type *Coerced =
+          TB.getPointerType(PtrTy->getSizeInBits().getFixedValue(),
+                            PtrTy->getAlignment(), AMDGPUAS::GLOBAL_ADDRESS);
+      return ArgInfo::getDirect(Coerced, /*Offset=*/0, /*Align=*/std::nullopt,
+                                /*CanBeFlattened=*/false);
+    }
+  }
+
   if (isAggregateTypeForABI(Ty))
     return ArgInfo::getIndirect(Ty->getAlignment(), /*ByVal=*/false,
                                 /*AddrSpace=*/AMDGPUAS::CONSTANT_ADDRESS);
 
-  // TODO: Classic passes CanBeFlattened=false here to keep a struct intact;
-  // ArgInfo cannot model that yet, so a multi-field record coerce may be
-  // flattened.
-  return ArgInfo::getDirect(Ty);
+  // CanBeFlattened=false keeps the struct intact.
+  return ArgInfo::getDirect(Ty, /*Offset=*/0, /*Align=*/std::nullopt,
+                            /*CanBeFlattened=*/false);
 }
 
 ArgInfo AMDGPUTargetInfo::classifyArgumentType(const Type *Ty, bool Variadic,
@@ -157,9 +206,10 @@ ArgInfo AMDGPUTargetInfo::classifyArgumentType(const Type *Ty, bool Variadic,
 
   Ty = useFirstFieldIfTransparentUnion(Ty);
 
-  // TODO: Classic sets CanBeFlattened=false for variadics; not modeled here.
+  // Variadic aggregates are kept intact rather than flattened into fields.
   if (Variadic)
-    return ArgInfo::getDirect();
+    return ArgInfo::getDirect(/*T=*/nullptr, /*Offset=*/0,
+                              /*Align=*/std::nullopt, /*CanBeFlattened=*/false);
 
   if (isAggregateTypeForABI(Ty)) {
     // Records with non-trivial destructors/copy-constructors should not be
@@ -179,7 +229,7 @@ ArgInfo AMDGPUTargetInfo::classifyArgumentType(const Type *Ty, bool Variadic,
 
     if (const auto *RT = dyn_cast<RecordType>(Ty);
         RT && RT->hasFlexibleArrayMember())
-      return classifyDefaultType(Ty, /*IsReturn=*/false);
+      return classifyDefaultArgumentType(Ty);
 
     // Pack aggregates <= 8 bytes into single VGPR or pair.
     uint64_t Size = Ty->getSizeInBits().getFixedValue();
@@ -211,7 +261,7 @@ ArgInfo AMDGPUTargetInfo::classifyArgumentType(const Type *Ty, bool Variadic,
   }
 
   // Otherwise just do the default thing.
-  ArgInfo AI = classifyDefaultType(Ty, /*IsReturn=*/false);
+  ArgInfo AI = classifyDefaultArgumentType(Ty);
   if (!AI.isIndirect()) {
     uint64_t NumRegs = numRegsForType(Ty);
     NumRegsLeft -= std::min(NumRegs, uint64_t{NumRegsLeft});
@@ -240,8 +290,10 @@ void AMDGPUTargetInfo::computeInfo(FunctionInfo &FI) const {
   }
 }
 
-std::unique_ptr<TargetInfo> createAMDGPUTargetInfo(TypeBuilder &TB) {
-  return std::make_unique<AMDGPUTargetInfo>(TB, ABICompatInfo());
+std::unique_ptr<TargetInfo>
+createAMDGPUTargetInfo(TypeBuilder &TB, bool CoerceGenericPtrArgToGlobal) {
+  return std::make_unique<AMDGPUTargetInfo>(TB, ABICompatInfo(),
+                                            CoerceGenericPtrArgToGlobal);
 }
 
 } // namespace abi
diff --git a/llvm/unittests/ABI/AMDGPUTargetInfoTest.cpp b/llvm/unittests/ABI/AMDGPUTargetInfoTest.cpp
index 38d2bbfc37017a..b85d0f9e4f058f 100644
--- a/llvm/unittests/ABI/AMDGPUTargetInfoTest.cpp
+++ b/llvm/unittests/ABI/AMDGPUTargetInfoTest.cpp
@@ -60,6 +60,22 @@ class AMDGPUTargetInfoTest : public ::testing::Test {
     return createAMDGPUTargetInfo(const_cast<TypeBuilder &>(TB));
   }
 
+  /// A target configured like a HIP compilation: generic scalar-pointer kernel
+  /// arguments are coerced to the global address space.
+  std::unique_ptr<TargetInfo> hipTarget() const {
+    return createAMDGPUTargetInfo(const_cast<TypeBuilder &>(TB),
+                                  /*CoerceGenericPtrArgToGlobal=*/true);
+  }
+
+  /// The classification a kernel argument gets on target \p TI.
+  const ArgInfo &classifyKernelArg(const ABIType *ArgTy,
+                                   std::unique_ptr<FunctionInfo> &FI,
+                                   std::unique_ptr<TargetInfo> &TI) {
+    FI = FunctionInfo::create(CallingConv::AMDGPU_KERNEL, Void, {ArgTy});
+    TI->computeInfo(*FI);
+    return FI->getArgInfo(0).Info;
+  }
+
   /// A register-passable record with the given fields, size and alignment.
   const ABIType *recordOf(llvm::ArrayRef<FieldInfo> Fields, uint64_t SizeInBits,
                           llvm::Align Alignment) {
@@ -91,6 +107,15 @@ class AMDGPUTargetInfoTest : public ::testing::Test {
   }
 };
 
+// A direct arg coerced to a pointer in address space \p AS.
+static void expectDirectPointerAS(const ArgInfo &Info, unsigned AS) {
+  ASSERT_TRUE(Info.isDirect());
+  const auto *PT =
+      llvm::dyn_cast_or_null<llvm::abi::PointerType>(Info.getCoerceToType());
+  ASSERT_NE(PT, nullptr);
+  EXPECT_EQ(PT->getAddrSpace(), AS);
+}
+
 static void expectUncoercedDirect(const ArgInfo &Info) {
   ASSERT_TRUE(Info.isDirect());
   EXPECT_EQ(Info.getCoerceToType(), nullptr);
@@ -139,7 +164,10 @@ static void expectIndirect(const ArgInfo &Info, llvm::Align ExpectedAlign,
 TEST_F(AMDGPUTargetInfoTest, ScalarPassesDirect) {
   std::unique_ptr<FunctionInfo> FI;
   std::unique_ptr<TargetInfo> TI;
-  expectUncoercedDirect(classifyArg(I32, FI, TI));
+  const ArgInfo &Info = classifyArg(I32, FI, TI);
+  expectUncoercedDirect(Info);
+  // Ordinary direct args stay flattenable (the getDirect default).
+  EXPECT_TRUE(Info.getCanBeFlattened());
   expectUncoercedDirect(classifyArg(F32, FI, TI));
 }
 
@@ -152,6 +180,35 @@ TEST_F(AMDGPUTargetInfoTest, PromotableIntegerExtends) {
   EXPECT_TRUE(Info.isSignExt());
 }
 
+// An oversized _BitInt (> 128 bits) is passed indirectly, byval, like classic
+// DefaultABIInfo. The 128-bit boundary width stays direct.
+TEST_F(AMDGPUTargetInfoTest, OversizedBitIntArgumentIsIndirect) {
+  std::unique_ptr<FunctionInfo> FI;
+  std::unique_ptr<TargetInfo> TI;
+  const ABIType *BitInt256 = TB.getIntegerType(256, llvm::Align(16),
+                                               /*Signed=*/true,
+                                               /*IsBitInt=*/true);
+  expectIndirect(classifyArg(BitInt256, FI, TI), llvm::Align(16),
+                 /*ByVal=*/true, llvm::AMDGPUAS::PRIVATE_ADDRESS);
+
+  const ABIType *BitInt128 = TB.getIntegerType(128, llvm::Align(16),
+                                               /*Signed=*/true,
+                                               /*IsBitInt=*/true);
+  EXPECT_TRUE(classifyArg(BitInt128, FI, TI).isDirect());
+}
+
+// An oversized _BitInt return is indirect too. Classic DefaultABIInfo's
+// getNaturalAlignIndirect defaults ByVal to true, so returns carry it too.
+TEST_F(AMDGPUTargetInfoTest, OversizedBitIntReturnIsIndirect) {
+  std::unique_ptr<FunctionInfo> FI;
+  std::unique_ptr<TargetInfo> TI;
+  const ABIType *BitInt256 = TB.getIntegerType(256, llvm::Align(16),
+                                               /*Signed=*/true,
+                                               /*IsBitInt=*/true);
+  expectIndirect(classifyRet(BitInt256, FI, TI), llvm::Align(16),
+                 /*ByVal=*/true, llvm::AMDGPUAS::PRIVATE_ADDRESS);
+}
+
 // Aggregates <= 16/32/64 bits pack into i16 / i32 / [2 x i32].
 TEST_F(AMDGPUTargetInfoTest, SmallAggregatesPackIntoRegisters) {
   std::unique_ptr<FunctionInfo> FI;
@@ -211,7 +268,8 @@ TEST_F(AMDGPUTargetInfoTest, NonTrivialRecordIsIndirectPrivate) {
                  /*ByVal=*/false, llvm::AMDGPUAS::PRIVATE_ADDRESS);
 }
 
-// A variadic argument bypasses register packing and passes through unchanged.
+// A variadic argument bypasses register packing and passes through unchanged,
+// kept intact rather than flattened into per-field wire arguments.
 TEST_F(AMDGPUTargetInfoTest, VariadicArgumentPassesDirect) {
   std::unique_ptr<TargetInfo> TI = target();
   const ABIType *S16 =
@@ -221,6 +279,7 @@ TEST_F(AMDGPUTargetInfoTest, VariadicArgumentPassesDirect) {
       FunctionInfo::create(CallingConv::C, Void, {S16}, RequiredArgs(0));
   TI->computeInfo(*FI);
   expectUncoercedDirect(FI->getArgInfo(0).Info);
+  EXPECT_FALSE(FI->getArgInfo(0).Info.getCanBeFlattened());
 }
 
 // Kernel aggregate arguments are passed by reference in the constant address
@@ -235,11 +294,46 @@ TEST_F(AMDGPUTargetInfoTest, KernelAggregateIsIndirectConstant) {
                  llvm::AMDGPUAS::CONSTANT_ADDRESS);
 }
 
-// Kernel scalar arguments are passed directly.
+// Kernel direct arguments are passed directly and kept intact.
 TEST_F(AMDGPUTargetInfoTest, KernelScalarIsDirect) {
   std::unique_ptr<FunctionInfo> FI;
   std::unique_ptr<TargetInfo> TI;
-  expectDirectInteger(classifyArg(I32, FI, TI, CallingConv::AMDGPU_KERNEL), 32);
+  const ArgInfo &Info = classifyArg(I32, FI, TI, CallingConv::AMDGPU_KERNEL);
+  expectDirectInteger(Info, 32);
+  EXPECT_FALSE(Info.getCanBeFlattened());
+}
+
+// Under HIP a generic scalar-pointer kernel argument is coerced to a global
+// pointer and passed directly, kept intact.
+TEST_F(AMDGPUTargetInfoTest, KernelGenericPointerCoercesToGlobalUnderHIP) {
+  std::unique_ptr<FunctionInfo> FI;
+  std::unique_ptr<TargetInfo> TI = hipTarget();
+  const ABIType *GenericPtr =
+      TB.getPointerType(64, llvm::Align(8), llvm::AMDGPUAS::FLAT_ADDRESS);
+  const ArgInfo &Info = classifyKernelArg(GenericPtr, FI, TI);
+  expectDirectPointerAS(Info, llvm::AMDGPUAS::GLOBAL_ADDRESS);
+  EXPECT_FALSE(Info.getCanBeFlattened());
+}
+
+// A pointer already in the global address space is left untouched under HIP.
+TEST_F(AMDGPUTargetInfoTest, KernelGlobalPointerUnchangedUnderHIP) {
+  std::unique_ptr<FunctionInfo> FI;
+  std::unique_ptr<TargetInfo> TI = hipTarget();
+  const ABIType *GlobalPtr =
+      TB.getPointerType(64, llvm::Align(8), llvm::AMDGPUAS::GLOBAL_ADDRESS);
+  expectDirectPointerAS(classifyKernelArg(GlobalPtr, FI, TI),
+                        llvm::AMDGPUAS::GLOBAL_ADDRESS);
+}
+
+// Without the HIP rule a generic pointer kernel argument stays generic.
+TEST_F(AMDGPUTargetInfoTest, KernelGenericPointerUnchangedByDefault) {
+  std::unique_ptr<FunctionInfo> FI;
+  std::unique_ptr<TargetInfo> TI;
+  const ABIType *GenericPtr =
+      TB.getPointerType(64, llvm::Align(8), llvm::AMDGPUAS::FLAT_ADDRESS);
+  expectDirectPointerAS(
+      classifyArg(GenericPtr, FI, TI, CallingConv::AMDGPU_KERNEL),
+      llvm::AMDGPUAS::FLAT_ADDRESS);
 }
 
 // A void return is ignored.

>From 723bc931b56db28973658ce7d585d7b49faf02fc Mon Sep 17 00:00:00 2001
From: skc7 <Krishna.Sankisa at amd.com>
Date: Fri, 18 Sep 2026 13:21:52 +0530
Subject: [PATCH 3/4] fix unittest

---
 llvm/unittests/ABI/AMDGPUTargetInfoTest.cpp | 10 +++++-----
 1 file changed, 5 insertions(+), 5 deletions(-)

diff --git a/llvm/unittests/ABI/AMDGPUTargetInfoTest.cpp b/llvm/unittests/ABI/AMDGPUTargetInfoTest.cpp
index b85d0f9e4f058f..eb7cd65de867f6 100644
--- a/llvm/unittests/ABI/AMDGPUTargetInfoTest.cpp
+++ b/llvm/unittests/ABI/AMDGPUTargetInfoTest.cpp
@@ -53,7 +53,7 @@ class AMDGPUTargetInfoTest : public ::testing::Test {
         F32(TB.getFloatType(llvm::APFloat::IEEEsingle(), llvm::Align(4))),
         Void(TB.getVoidType()),
         Empty(TB.getRecordType({}, llvm::TypeSize::getFixed(8), llvm::Align(1),
-                               StructPacking::Default, {}, {},
+                               llvm::Align(1), StructPacking::Default, {}, {},
                                RecordFlags::CanPassInRegisters)) {}
 
   std::unique_ptr<TargetInfo> target() const {
@@ -80,8 +80,8 @@ class AMDGPUTargetInfoTest : public ::testing::Test {
   const ABIType *recordOf(llvm::ArrayRef<FieldInfo> Fields, uint64_t SizeInBits,
                           llvm::Align Alignment) {
     return TB.getRecordType(Fields, llvm::TypeSize::getFixed(SizeInBits),
-                            Alignment, StructPacking::Default, {}, {},
-                            RecordFlags::CanPassInRegisters);
+                            Alignment, Alignment, StructPacking::Default, {},
+                            {}, RecordFlags::CanPassInRegisters);
   }
 
   /// The argument classification the target computes for a single parameter
@@ -263,7 +263,7 @@ TEST_F(AMDGPUTargetInfoTest, NonTrivialRecordIsIndirectPrivate) {
   std::unique_ptr<TargetInfo> TI;
   const ABIType *CannotPass = TB.getRecordType(
       {FieldInfo(I32, 0)}, llvm::TypeSize::getFixed(32), llvm::Align(4),
-      StructPacking::Default, {}, {}, RecordFlags::IsCXXRecord);
+      llvm::Align(4), StructPacking::Default, {}, {}, RecordFlags::IsCXXRecord);
   expectIndirect(classifyArg(CannotPass, FI, TI), llvm::Align(4),
                  /*ByVal=*/false, llvm::AMDGPUAS::PRIVATE_ADDRESS);
 }
@@ -359,7 +359,7 @@ TEST_F(AMDGPUTargetInfoTest, NonTrivialRecordReturnIsIndirect) {
   std::unique_ptr<TargetInfo> TI;
   const ABIType *CannotPass = TB.getRecordType(
       {FieldInfo(I32, 0)}, llvm::TypeSize::getFixed(32), llvm::Align(4),
-      StructPacking::Default, {}, {}, RecordFlags::IsCXXRecord);
+      llvm::Align(4), StructPacking::Default, {}, {}, RecordFlags::IsCXXRecord);
   const ArgInfo &Info = classifyRet(CannotPass, FI, TI);
   ASSERT_TRUE(Info.isIndirect());
   EXPECT_FALSE(Info.getIndirectByVal());

>From 39c1ce606294999af1383359d5397def1aa656a3 Mon Sep 17 00:00:00 2001
From: skc7 <Krishna.Sankisa at amd.com>
Date: Mon, 21 Sep 2026 12:24:58 +0530
Subject: [PATCH 4/4] update

---
 llvm/lib/ABI/Targets/AMDGPU.cpp             | 22 +++++++++++++--------
 llvm/unittests/ABI/AMDGPUTargetInfoTest.cpp | 12 +++++++++++
 2 files changed, 26 insertions(+), 8 deletions(-)

diff --git a/llvm/lib/ABI/Targets/AMDGPU.cpp b/llvm/lib/ABI/Targets/AMDGPU.cpp
index 1c27afd598f327..49785390ffeece 100644
--- a/llvm/lib/ABI/Targets/AMDGPU.cpp
+++ b/llvm/lib/ABI/Targets/AMDGPU.cpp
@@ -20,10 +20,12 @@
 namespace llvm {
 namespace abi {
 
-class AMDGPUTargetInfo : public TargetInfo {
+class AMDGPUTargetInfo final : public TargetInfo {
 private:
   static const unsigned MaxNumRegsForArgsRet = 16;
 
+  ABICompatInfo CompatInfo;
+
   /// HIP coerces a generic scalar-pointer kernel argument to the global
   /// address space. Gated by the front end, which alone can see LangOpts.HIP.
   bool CoerceGenericPtrArgToGlobal;
@@ -38,18 +40,20 @@ class AMDGPUTargetInfo : public TargetInfo {
   ArgInfo classifyDefaultReturnType(const Type *Ty) const;
 
   /// Estimate number of registers the type will use when passed in registers.
-  uint64_t numRegsForType(const Type *Ty) const;
+  uint64_t getNumRegsForType(const Type *Ty) const;
 
 public:
   AMDGPUTargetInfo(TypeBuilder &TypeBuilder, const ABICompatInfo &Compat,
                    bool CoerceGenericPtrArgToGlobal)
-      : TargetInfo(TypeBuilder, Compat),
+      : TargetInfo(TypeBuilder), CompatInfo(Compat),
         CoerceGenericPtrArgToGlobal(CoerceGenericPtrArgToGlobal) {}
 
+  const ABICompatInfo &getABICompatInfo() const override { return CompatInfo; }
+
   void computeInfo(FunctionInfo &FI) const override;
 };
 
-uint64_t AMDGPUTargetInfo::numRegsForType(const Type *Ty) const {
+uint64_t AMDGPUTargetInfo::getNumRegsForType(const Type *Ty) const {
   uint64_t NumRegs = 0;
 
   if (const auto *VT = dyn_cast<VectorType>(Ty)) {
@@ -69,7 +73,7 @@ uint64_t AMDGPUTargetInfo::numRegsForType(const Type *Ty) const {
 
   if (const auto *RT = dyn_cast<RecordType>(Ty)) {
     for (const FieldInfo &Field : RT->getFields())
-      NumRegs += numRegsForType(Field.FieldType);
+      NumRegs += getNumRegsForType(Field.FieldType);
     return NumRegs;
   }
 
@@ -161,7 +165,7 @@ ArgInfo AMDGPUTargetInfo::classifyReturnType(const Type *RetTy) const {
         return ArgInfo::getDirect(TB.getArrayType(I32Ty, 2, /*SizeInBits=*/64));
       }
 
-      if (numRegsForType(RetTy) <= MaxNumRegsForArgsRet)
+      if (getNumRegsForType(RetTy) <= MaxNumRegsForArgsRet)
         return ArgInfo::getDirect();
     }
   }
@@ -248,7 +252,7 @@ ArgInfo AMDGPUTargetInfo::classifyArgumentType(const Type *Ty, bool Variadic,
     }
 
     if (NumRegsLeft > 0) {
-      uint64_t NumRegs = numRegsForType(Ty);
+      uint64_t NumRegs = getNumRegsForType(Ty);
       if (NumRegsLeft >= NumRegs) {
         NumRegsLeft -= NumRegs;
         return ArgInfo::getDirect();
@@ -263,7 +267,7 @@ ArgInfo AMDGPUTargetInfo::classifyArgumentType(const Type *Ty, bool Variadic,
   // Otherwise just do the default thing.
   ArgInfo AI = classifyDefaultArgumentType(Ty);
   if (!AI.isIndirect()) {
-    uint64_t NumRegs = numRegsForType(Ty);
+    uint64_t NumRegs = getNumRegsForType(Ty);
     NumRegsLeft -= std::min(NumRegs, uint64_t{NumRegsLeft});
   }
 
@@ -273,6 +277,8 @@ ArgInfo AMDGPUTargetInfo::classifyArgumentType(const Type *Ty, bool Variadic,
 void AMDGPUTargetInfo::computeInfo(FunctionInfo &FI) const {
   CallingConv::ID CC = FI.getCallingConvention();
 
+  // Non-trivial C++ records are returned indirectly
+  // in the flat address space.
   if (!maybeCommonClassifyReturnType(FI))
     FI.getReturnInfo() = classifyReturnType(FI.getReturnType());
 
diff --git a/llvm/unittests/ABI/AMDGPUTargetInfoTest.cpp b/llvm/unittests/ABI/AMDGPUTargetInfoTest.cpp
index eb7cd65de867f6..d43f7ddfccdaba 100644
--- a/llvm/unittests/ABI/AMDGPUTargetInfoTest.cpp
+++ b/llvm/unittests/ABI/AMDGPUTargetInfoTest.cpp
@@ -268,6 +268,18 @@ TEST_F(AMDGPUTargetInfoTest, NonTrivialRecordIsIndirectPrivate) {
                  /*ByVal=*/false, llvm::AMDGPUAS::PRIVATE_ADDRESS);
 }
 
+// A non-trivial C++ record return is indirect in the generic (flat) address
+// space with ByVal=false, matching classic's getSRetAddrSpace(LangAS::Default).
+TEST_F(AMDGPUTargetInfoTest, NonTrivialRecordReturnIsIndirectFlat) {
+  std::unique_ptr<FunctionInfo> FI;
+  std::unique_ptr<TargetInfo> TI;
+  const ABIType *CannotPass = TB.getRecordType(
+      {FieldInfo(I32, 0)}, llvm::TypeSize::getFixed(32), llvm::Align(4),
+      llvm::Align(4), StructPacking::Default, {}, {}, RecordFlags::IsCXXRecord);
+  expectIndirect(classifyRet(CannotPass, FI, TI), llvm::Align(4),
+                 /*ByVal=*/false, llvm::AMDGPUAS::FLAT_ADDRESS);
+}
+
 // A variadic argument bypasses register packing and passes through unchanged,
 // kept intact rather than flattened into per-field wire arguments.
 TEST_F(AMDGPUTargetInfoTest, VariadicArgumentPassesDirect) {



More information about the llvm-commits mailing list