[llvm-branch-commits] [llvm] [AMDGPU][SwLowerLDS] Lower non-kernels with LDS instructions and assign kernel IDs to their callers (PR #228043)

via llvm-branch-commits llvm-branch-commits at lists.llvm.org
Thu Oct 1 04:24:17 PDT 2026


https://github.com/skc7 created https://github.com/llvm/llvm-project/pull/228043

Summary:
- Lower every non-kernel that a kernel reaches and that has `LDS memory instructions` or `flat to LDS` casts. 
- Previously only non-kernels using LDS globals or taking LDS pointer arguments were lowered, so a device function casting a flat argument back to LDS accessed hardware LDS directly.
- Find these functions with a call graph walk from each kernel. Indirect calls are treated as reaching any address taken non-kernel, matching `getTransitiveUsesOfLDS`.
- Give every kernel that reaches such a function a kernel ID. Kernels without LDS get a null base table entry and a poison offset table row.
- Remove `amdgpu-no-lds-kernel-id` after the kernel set is final, so the kernel ID is passed to every function the kernel can reach.

Assisted by: claude opus 5.5

>From 9594aa490f1e68affb8672f049ba7025910d651a Mon Sep 17 00:00:00 2001
From: skc7 <Krishna.Sankisa at amd.com>
Date: Thu, 1 Oct 2026 16:49:27 +0530
Subject: [PATCH] [AMDGPU][SwLowerLDS] Lower non-kernels with LDS instructions
 and assign kernel IDs to their callers

---
 llvm/lib/Target/AMDGPU/AMDGPUSwLowerLDS.cpp   | 202 +++++++++++-------
 .../amdgpu-sw-lower-lds-flat-arg-kernel-id.ll | 131 ++++++++++++
 2 files changed, 251 insertions(+), 82 deletions(-)
 create mode 100644 llvm/test/CodeGen/AMDGPU/amdgpu-sw-lower-lds-flat-arg-kernel-id.ll

diff --git a/llvm/lib/Target/AMDGPU/AMDGPUSwLowerLDS.cpp b/llvm/lib/Target/AMDGPU/AMDGPUSwLowerLDS.cpp
index e581e26057790..460b41dea9bf4 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUSwLowerLDS.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUSwLowerLDS.cpp
@@ -60,9 +60,10 @@
 //
 // Replacement of non-kernel LDS accesses:
 //    Multiple kernels can access the same non-kernel function.
-//    All the kernels accessing LDS through non-kernels are sorted and
-//    assigned a kernel-id. All the LDS globals accessed by non-kernels
-//    are sorted. This information is used to build two tables:
+//    All the kernels accessing LDS through non-kernels, or reaching
+//    non-kernels with LDS instructions, are sorted and assigned a kernel-id.
+//    All the LDS globals accessed by non-kernels are sorted. This
+//    information is used to build two tables:
 //    - Base table:
 //        Base table will have single row, with elements of the row
 //        placed as per kernel ID. Each element in the row corresponds
@@ -95,6 +96,7 @@
 #include "llvm/IR/DebugInfo.h"
 #include "llvm/IR/DebugInfoMetadata.h"
 #include "llvm/IR/IRBuilder.h"
+#include "llvm/IR/InstIterator.h"
 #include "llvm/IR/Instructions.h"
 #include "llvm/IR/MDBuilder.h"
 #include "llvm/IR/ReplaceConstant.h"
@@ -160,7 +162,7 @@ struct AsanInstrumentInfo {
 struct FunctionsAndLDSAccess {
   MapVector<Function *, KernelLDSParameters> KernelToLDSParametersMap;
   SetVector<Function *> KernelsWithIndirectLDSAccess;
-  SetVector<Function *> NonKernelsWithLDSArgument;
+  SetVector<Function *> NonKernelsWithLDSInstructions;
   SetVector<GlobalVariable *> AllNonKernelLDSAccess;
   FunctionVariableMap NonKernelToLDSAccessMap;
 };
@@ -171,7 +173,7 @@ class AMDGPUSwLowerLDS {
       : M(Mod), IRB(M.getContext()), DTCallback(Callback) {}
   bool run();
   void getUsesOfLDSByNonKernels();
-  void getNonKernelsWithLDSArguments(const CallGraph &CG);
+  void getNonKernelsWithLDSInstructions(const CallGraph &CG);
   SetVector<Function *>
   getOrderedIndirectLDSAccessingKernels(SetVector<Function *> &Kernels);
   SetVector<GlobalVariable *>
@@ -251,35 +253,92 @@ SetVector<Function *> AMDGPUSwLowerLDS::getOrderedIndirectLDSAccessingKernels(
   return OrderedKernels;
 }
 
-void AMDGPUSwLowerLDS::getNonKernelsWithLDSArguments(const CallGraph &CG) {
-  // Among the kernels accessing LDS, get list of
-  // Non-kernels to which a call is made and a ptr
-  // to addrspace(3) is passed as argument.
-  for (auto &K : FuncLDSAccessInfo.KernelToLDSParametersMap) {
-    Function *Func = K.first;
-    const CallGraphNode *CGN = CG[Func];
-    if (!CGN)
+// LDS memory operations and casts between flat and LDS.
+static bool isLDSMemoryInstruction(const Instruction &Inst) {
+  if (auto *LI = dyn_cast<LoadInst>(&Inst))
+    return LI->getPointerAddressSpace() == AMDGPUAS::LOCAL_ADDRESS;
+  if (auto *SI = dyn_cast<StoreInst>(&Inst))
+    return SI->getPointerAddressSpace() == AMDGPUAS::LOCAL_ADDRESS;
+  if (auto *RMW = dyn_cast<AtomicRMWInst>(&Inst))
+    return RMW->getPointerAddressSpace() == AMDGPUAS::LOCAL_ADDRESS;
+  if (auto *XCHG = dyn_cast<AtomicCmpXchgInst>(&Inst))
+    return XCHG->getPointerAddressSpace() == AMDGPUAS::LOCAL_ADDRESS;
+  if (auto *ASC = dyn_cast<AddrSpaceCastInst>(&Inst)) {
+    unsigned SrcAS = ASC->getSrcAddressSpace();
+    unsigned DstAS = ASC->getDestAddressSpace();
+    return (SrcAS == AMDGPUAS::LOCAL_ADDRESS &&
+            DstAS == AMDGPUAS::FLAT_ADDRESS) ||
+           (SrcAS == AMDGPUAS::FLAT_ADDRESS &&
+            DstAS == AMDGPUAS::LOCAL_ADDRESS);
+  }
+  if (auto *MI = dyn_cast<AnyMemIntrinsic>(&Inst)) {
+    if (MI->getDestAddressSpace() == AMDGPUAS::LOCAL_ADDRESS)
+      return true;
+    if (auto *MTI = dyn_cast<AnyMemTransferInst>(MI))
+      return MTI->getSourceAddressSpace() == AMDGPUAS::LOCAL_ADDRESS;
+  }
+  return false;
+}
+
+void AMDGPUSwLowerLDS::getNonKernelsWithLDSInstructions(const CallGraph &CG) {
+  // Non-kernels with LDS instructions find the malloc buffer through the base
+  // table, so every kernel reaching them needs a kernel ID, even one without
+  // LDS.
+  DenseMap<Function *, bool> HasLDSInstructionsCache;
+  auto HasLDSInstructions = [&](Function *F) {
+    auto [It, Inserted] = HasLDSInstructionsCache.try_emplace(F, false);
+    if (Inserted) {
+      It->second = any_of(instructions(*F), isLDSMemoryInstruction);
+      if (It->second)
+        FuncLDSAccessInfo.NonKernelsWithLDSInstructions.insert(F);
+    }
+    return It->second;
+  };
+
+  // Indirect calls may reach any address taken non-kernel.
+  SmallVector<Function *> AddressTakenFuncs;
+  for (Function &F : M)
+    if (!F.isDeclaration() && !isKernel(F) &&
+        F.hasAddressTaken(nullptr, /*IgnoreCallbackUses=*/false,
+                          /*IgnoreAssumeLikeCalls=*/false,
+                          /*IgnoreLLVMUsed=*/true))
+      AddressTakenFuncs.push_back(&F);
+
+  for (Function &Kernel : M) {
+    if (Kernel.isDeclaration() || !isKernel(Kernel))
       continue;
-    for (auto &I : *CGN) {
-      CallGraphNode *CallerCGN = I.second;
-      Function *CalledFunc = CallerCGN->getFunction();
-      if (!CalledFunc || CalledFunc->isDeclaration())
-        continue;
-      if (AMDGPU::isKernel(*CalledFunc))
-        continue;
-      for (auto AI = CalledFunc->arg_begin(), E = CalledFunc->arg_end();
-           AI != E; ++AI) {
-        Type *ArgTy = (*AI).getType();
-        if (!ArgTy->isPointerTy())
+
+    bool ReachesLDSInstructions = false;
+    bool SeenUnknownCall = false;
+    SmallVector<Function *> WorkList = {&Kernel};
+    SmallPtrSet<Function *, 8> Visited;
+    auto Visit = [&](Function *Callee) {
+      if (Callee->isDeclaration() || isKernel(*Callee) ||
+          !Visited.insert(Callee).second)
+        return;
+      ReachesLDSInstructions |= HasLDSInstructions(Callee);
+      WorkList.push_back(Callee);
+    };
+
+    while (!WorkList.empty()) {
+      Function *F = WorkList.pop_back_val();
+      for (const CallGraphNode::CallRecord &CallRecord : *CG[F]) {
+        if (!CallRecord.second)
           continue;
-        if (ArgTy->getPointerAddressSpace() != AMDGPUAS::LOCAL_ADDRESS)
+        if (Function *Callee = CallRecord.second->getFunction()) {
+          Visit(Callee);
+          continue;
+        }
+        if (SeenUnknownCall)
           continue;
-        FuncLDSAccessInfo.NonKernelsWithLDSArgument.insert(CalledFunc);
-        // Also add the Calling function to KernelsWithIndirectLDSAccess list
-        // so that base table of LDS is generated.
-        FuncLDSAccessInfo.KernelsWithIndirectLDSAccess.insert(Func);
+        SeenUnknownCall = true;
+        for (Function *PotentialCallee : AddressTakenFuncs)
+          Visit(PotentialCallee);
       }
     }
+
+    if (ReachesLDSInstructions)
+      FuncLDSAccessInfo.KernelsWithIndirectLDSAccess.insert(&Kernel);
   }
 }
 
@@ -633,39 +692,9 @@ static DebugLoc getOrCreateDebugLoc(const Instruction *InsertBefore,
 
 void AMDGPUSwLowerLDS::getLDSMemoryInstructions(
     Function *Func, SetVector<Instruction *> &LDSInstructions) {
-  for (BasicBlock &BB : *Func) {
-    for (Instruction &Inst : BB) {
-      if (LoadInst *LI = dyn_cast<LoadInst>(&Inst)) {
-        if (LI->getPointerAddressSpace() == AMDGPUAS::LOCAL_ADDRESS)
-          LDSInstructions.insert(&Inst);
-      } else if (StoreInst *SI = dyn_cast<StoreInst>(&Inst)) {
-        if (SI->getPointerAddressSpace() == AMDGPUAS::LOCAL_ADDRESS)
-          LDSInstructions.insert(&Inst);
-      } else if (AtomicRMWInst *RMW = dyn_cast<AtomicRMWInst>(&Inst)) {
-        if (RMW->getPointerAddressSpace() == AMDGPUAS::LOCAL_ADDRESS)
-          LDSInstructions.insert(&Inst);
-      } else if (AtomicCmpXchgInst *XCHG = dyn_cast<AtomicCmpXchgInst>(&Inst)) {
-        if (XCHG->getPointerAddressSpace() == AMDGPUAS::LOCAL_ADDRESS)
-          LDSInstructions.insert(&Inst);
-      } else if (AddrSpaceCastInst *ASC = dyn_cast<AddrSpaceCastInst>(&Inst)) {
-        unsigned SrcAS = ASC->getSrcAddressSpace();
-        unsigned DstAS = ASC->getDestAddressSpace();
-        if ((SrcAS == AMDGPUAS::LOCAL_ADDRESS &&
-             DstAS == AMDGPUAS::FLAT_ADDRESS) ||
-            (SrcAS == AMDGPUAS::FLAT_ADDRESS &&
-             DstAS == AMDGPUAS::LOCAL_ADDRESS))
-          LDSInstructions.insert(&Inst);
-      } else if (AnyMemIntrinsic *MI = dyn_cast<AnyMemIntrinsic>(&Inst)) {
-        if (MI->getDestAddressSpace() == AMDGPUAS::LOCAL_ADDRESS) {
-          LDSInstructions.insert(&Inst);
-        } else if (auto *MTI = dyn_cast<AnyMemTransferInst>(MI)) {
-          if (MTI->getSourceAddressSpace() == AMDGPUAS::LOCAL_ADDRESS)
-            LDSInstructions.insert(&Inst);
-        }
-      } else
-        continue;
-    }
-  }
+  for (Instruction &Inst : instructions(Func))
+    if (isLDSMemoryInstruction(Inst))
+      LDSInstructions.insert(&Inst);
 }
 
 Value *AMDGPUSwLowerLDS::getTranslatedGlobalMemoryPtrOfLDS(Value *LoadMallocPtr,
@@ -1035,14 +1064,19 @@ void AMDGPUSwLowerLDS::lowerKernelLDSAccesses(Function *Func,
 Constant *AMDGPUSwLowerLDS::getAddressesOfVariablesInKernel(
     Function *Func, SetVector<GlobalVariable *> &Variables) {
   Type *Int32Ty = IRB.getInt32Ty();
-  auto &LDSParams = FuncLDSAccessInfo.KernelToLDSParametersMap[Func];
+  ArrayType *KernelOffsetsType =
+      ArrayType::get(IRB.getPtrTy(AMDGPUAS::GLOBAL_ADDRESS), Variables.size());
+
+  // Kernels with no direct or indirect LDS uses never read their offset row.
+  auto KernelIt = FuncLDSAccessInfo.KernelToLDSParametersMap.find(Func);
+  if (KernelIt == FuncLDSAccessInfo.KernelToLDSParametersMap.end() ||
+      !KernelIt->second.SwLDSMetadata)
+    return PoisonValue::get(KernelOffsetsType);
 
+  auto &LDSParams = KernelIt->second;
   GlobalVariable *SwLDSMetadata = LDSParams.SwLDSMetadata;
-  assert(SwLDSMetadata);
   auto *SwLDSMetadataStructType =
       cast<StructType>(SwLDSMetadata->getValueType());
-  ArrayType *KernelOffsetsType =
-      ArrayType::get(IRB.getPtrTy(AMDGPUAS::GLOBAL_ADDRESS), Variables.size());
 
   SmallVector<Constant *> Elements;
   for (auto *GV : Variables) {
@@ -1073,13 +1107,18 @@ void AMDGPUSwLowerLDS::buildNonKernelLDSBaseTable(
   if (Kernels.empty())
     return;
   const size_t NumberKernels = Kernels.size();
-  ArrayType *AllKernelsOffsetsType =
-      ArrayType::get(IRB.getPtrTy(AMDGPUAS::LOCAL_ADDRESS), NumberKernels);
+  PointerType *LDSPtrTy = IRB.getPtrTy(AMDGPUAS::LOCAL_ADDRESS);
+  ArrayType *AllKernelsOffsetsType = ArrayType::get(LDSPtrTy, NumberKernels);
   std::vector<Constant *> OverallConstantExprElts(NumberKernels);
   for (size_t i = 0; i < NumberKernels; i++) {
-    Function *Func = Kernels[i];
-    auto &LDSParams = FuncLDSAccessInfo.KernelToLDSParametersMap[Func];
-    OverallConstantExprElts[i] = LDSParams.SwLDS;
+    // Null for kernels without LDS, they reach LDS instructions only on
+    // invalid paths.
+    Constant *SwLDS = ConstantPointerNull::get(LDSPtrTy);
+    auto It = FuncLDSAccessInfo.KernelToLDSParametersMap.find(Kernels[i]);
+    if (It != FuncLDSAccessInfo.KernelToLDSParametersMap.end() &&
+        It->second.SwLDS)
+      SwLDS = It->second.SwLDS;
+    OverallConstantExprElts[i] = SwLDS;
   }
   Constant *init =
       ConstantArray::get(AllKernelsOffsetsType, OverallConstantExprElts);
@@ -1291,9 +1330,6 @@ bool AMDGPUSwLowerLDS::run() {
         CG, Func,
         {"amdgpu-no-workitem-id-x", "amdgpu-no-workitem-id-y",
          "amdgpu-no-workitem-id-z", "amdgpu-no-heap-ptr"});
-    if (!LDSParams.IndirectAccess.StaticLDSGlobals.empty() ||
-        !LDSParams.IndirectAccess.DynamicLDSGlobals.empty())
-      removeFnAttrFromReachable(CG, Func, {"amdgpu-no-lds-kernel-id"});
     reorderStaticDynamicIndirectLDSSet(LDSParams);
     buildSwLDSGlobal(Func);
     buildSwDynLDSGlobal(Func);
@@ -1308,12 +1344,15 @@ bool AMDGPUSwLowerLDS::run() {
   // Get the Uses of LDS from non-kernels.
   getUsesOfLDSByNonKernels();
 
-  // Get non-kernels with LDS ptr as argument and called by kernels.
-  getNonKernelsWithLDSArguments(CG);
+  // Get non-kernels with LDS instructions and the kernels reaching them.
+  getNonKernelsWithLDSInstructions(CG);
+
+  for (Function *Func : FuncLDSAccessInfo.KernelsWithIndirectLDSAccess)
+    removeFnAttrFromReachable(CG, Func, {"amdgpu-no-lds-kernel-id"});
 
   // Lower LDS accesses in non-kernels.
   if (!FuncLDSAccessInfo.NonKernelToLDSAccessMap.empty() ||
-      !FuncLDSAccessInfo.NonKernelsWithLDSArgument.empty()) {
+      !FuncLDSAccessInfo.NonKernelsWithLDSInstructions.empty()) {
     NonKernelLDSParameters NKLDSParams;
     NKLDSParams.OrderedKernels = getOrderedIndirectLDSAccessingKernels(
         FuncLDSAccessInfo.KernelsWithIndirectLDSAccess);
@@ -1328,12 +1367,11 @@ bool AMDGPUSwLowerLDS::run() {
           std::vector<GlobalVariable *>(LDSGlobals.begin(), LDSGlobals.end()));
       lowerNonKernelLDSAccesses(Func, OrderedLDSGlobals, NKLDSParams);
     }
-    for (Function *Func : FuncLDSAccessInfo.NonKernelsWithLDSArgument) {
-      auto &K = FuncLDSAccessInfo.NonKernelToLDSAccessMap;
-      if (K.contains(Func))
+    for (Function *Func : FuncLDSAccessInfo.NonKernelsWithLDSInstructions) {
+      if (FuncLDSAccessInfo.NonKernelToLDSAccessMap.contains(Func))
         continue;
-      SetVector<llvm::GlobalVariable *> Vec;
-      lowerNonKernelLDSAccesses(Func, Vec, NKLDSParams);
+      SetVector<GlobalVariable *> NoLDSGlobals;
+      lowerNonKernelLDSAccesses(Func, NoLDSGlobals, NKLDSParams);
     }
     Changed = true;
   }
diff --git a/llvm/test/CodeGen/AMDGPU/amdgpu-sw-lower-lds-flat-arg-kernel-id.ll b/llvm/test/CodeGen/AMDGPU/amdgpu-sw-lower-lds-flat-arg-kernel-id.ll
new file mode 100644
index 0000000000000..b73c24834958f
--- /dev/null
+++ b/llvm/test/CodeGen/AMDGPU/amdgpu-sw-lower-lds-flat-arg-kernel-id.ll
@@ -0,0 +1,131 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --check-globals all --version 6
+; RUN: opt < %s -passes=amdgpu-sw-lower-lds -amdgpu-asan-instrument-lds=false -S -mtriple=amdgpu-amd-amdhsa | FileCheck %s
+
+; Test that a non-kernel with only a flat to LDS cast is lowered through the
+; base table, and that every kernel reaching it gets a kernel ID.
+; @k1 reaches @writer through @mid. @k2 has no LDS and gets a null base entry.
+; amdgpu-no-lds-kernel-id must be removed from @mid and @writer.
+
+ at lds = internal addrspace(3) global [4 x i32] poison, align 4
+
+;.
+; CHECK: @llvm.amdgcn.sw.lds.k1 = internal addrspace(3) global ptr poison, no_sanitize_address, align 4, !absolute_symbol [[META0:![0-9]+]]
+; CHECK: @llvm.amdgcn.sw.lds.k1.md = internal addrspace(1) global %llvm.amdgcn.sw.lds.k1.md.type { %llvm.amdgcn.sw.lds.k1.md.item { i32 0, i32 8, i32 32 }, %llvm.amdgcn.sw.lds.k1.md.item { i32 32, i32 16, i32 32 } }, no_sanitize_address
+; CHECK: @llvm.amdgcn.sw.lds.base.table = internal addrspace(1) constant [2 x ptr addrspace(3)] [ptr addrspace(3) @llvm.amdgcn.sw.lds.k1, ptr addrspace(3) null], no_sanitize_address
+;.
+define internal void @writer(ptr %p) #0 {
+; CHECK-LABEL: define internal void @writer(
+; CHECK-SAME: ptr [[P:%.*]]) #[[ATTR0:[0-9]+]] {
+; CHECK-NEXT:    [[TMP1:%.*]] = call i32 @llvm.amdgcn.lds.kernel.id()
+; CHECK-NEXT:    [[TMP2:%.*]] = getelementptr inbounds [2 x ptr addrspace(3)], ptr addrspace(1) @llvm.amdgcn.sw.lds.base.table, i32 0, i32 [[TMP1]]
+; CHECK-NEXT:    [[TMP3:%.*]] = load ptr addrspace(3), ptr addrspace(1) [[TMP2]], align 4
+; CHECK-NEXT:    [[TMP4:%.*]] = load ptr addrspace(1), ptr addrspace(3) [[TMP3]], align 8
+; CHECK-NEXT:    [[TMP5:%.*]] = ptrtoint ptr addrspace(1) [[TMP4]] to i64
+; CHECK-NEXT:    [[TMP6:%.*]] = ptrtoint ptr [[P]] to i64
+; CHECK-NEXT:    [[TMP7:%.*]] = sub i64 [[TMP6]], [[TMP5]]
+; CHECK-NEXT:    [[TMP8:%.*]] = trunc i64 [[TMP7]] to i32
+; CHECK-NEXT:    [[TMP9:%.*]] = inttoptr i32 [[TMP8]] to ptr addrspace(3)
+; CHECK-NEXT:    [[TMP10:%.*]] = ptrtoint ptr addrspace(3) [[TMP9]] to i32
+; CHECK-NEXT:    [[TMP11:%.*]] = getelementptr inbounds i8, ptr addrspace(1) [[TMP4]], i32 [[TMP10]]
+; CHECK-NEXT:    store i32 1, ptr addrspace(1) [[TMP11]], align 4
+; CHECK-NEXT:    ret void
+;
+  %new_lds = addrspacecast ptr %p to ptr addrspace(3)
+  store i32 1, ptr addrspace(3) %new_lds, align 4
+  ret void
+}
+
+define internal void @mid(ptr %p) #0 {
+; CHECK-LABEL: define internal void @mid(
+; CHECK-SAME: ptr [[P:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT:    call void @writer(ptr [[P]])
+; CHECK-NEXT:    ret void
+;
+  call void @writer(ptr %p)
+  ret void
+}
+
+define amdgpu_kernel void @k1() sanitize_address {
+; CHECK-LABEL: define amdgpu_kernel void @k1(
+; CHECK-SAME: ) #[[ATTR1:[0-9]+]] !llvm.amdgcn.lds.kernel.id [[META2:![0-9]+]] {
+; CHECK-NEXT:  [[WID:.*]]:
+; CHECK-NEXT:    [[TMP0:%.*]] = call i32 @llvm.amdgcn.workitem.id.x()
+; CHECK-NEXT:    [[TMP1:%.*]] = call i32 @llvm.amdgcn.workitem.id.y()
+; CHECK-NEXT:    [[TMP2:%.*]] = call i32 @llvm.amdgcn.workitem.id.z()
+; CHECK-NEXT:    [[TMP3:%.*]] = or i32 [[TMP0]], [[TMP1]]
+; CHECK-NEXT:    [[TMP4:%.*]] = or i32 [[TMP3]], [[TMP2]]
+; CHECK-NEXT:    [[TMP5:%.*]] = icmp eq i32 [[TMP4]], 0
+; CHECK-NEXT:    br i1 [[TMP5]], label %[[MALLOC:.*]], label %[[BB18:.*]]
+; CHECK:       [[MALLOC]]:
+; CHECK-NEXT:    [[TMP6:%.*]] = load i32, ptr addrspace(1) getelementptr inbounds (i8, ptr addrspace(1) @llvm.amdgcn.sw.lds.k1.md, i64 12), align 4
+; CHECK-NEXT:    [[TMP7:%.*]] = load i32, ptr addrspace(1) getelementptr inbounds (i8, ptr addrspace(1) @llvm.amdgcn.sw.lds.k1.md, i64 20), align 4
+; CHECK-NEXT:    [[TMP8:%.*]] = add i32 [[TMP6]], [[TMP7]]
+; CHECK-NEXT:    [[TMP9:%.*]] = zext i32 [[TMP8]] to i64
+; CHECK-NEXT:    [[TMP10:%.*]] = call ptr @llvm.returnaddress.p0(i32 0)
+; CHECK-NEXT:    [[TMP11:%.*]] = ptrtoint ptr [[TMP10]] to i64
+; CHECK-NEXT:    [[TMP12:%.*]] = call i64 @__asan_malloc_impl(i64 [[TMP9]], i64 [[TMP11]])
+; CHECK-NEXT:    [[TMP13:%.*]] = inttoptr i64 [[TMP12]] to ptr addrspace(1)
+; CHECK-NEXT:    store ptr addrspace(1) [[TMP13]], ptr addrspace(3) @llvm.amdgcn.sw.lds.k1, align 8
+; CHECK-NEXT:    [[TMP14:%.*]] = getelementptr inbounds i8, ptr addrspace(1) [[TMP13]], i64 8
+; CHECK-NEXT:    [[TMP15:%.*]] = ptrtoint ptr addrspace(1) [[TMP14]] to i64
+; CHECK-NEXT:    call void @__asan_poison_region(i64 [[TMP15]], i64 24)
+; CHECK-NEXT:    [[TMP16:%.*]] = getelementptr inbounds i8, ptr addrspace(1) [[TMP13]], i64 48
+; CHECK-NEXT:    [[TMP17:%.*]] = ptrtoint ptr addrspace(1) [[TMP16]] to i64
+; CHECK-NEXT:    call void @__asan_poison_region(i64 [[TMP17]], i64 16)
+; CHECK-NEXT:    br label %[[BB18]]
+; CHECK:       [[BB18]]:
+; CHECK-NEXT:    [[XYZCOND:%.*]] = phi i1 [ false, %[[WID]] ], [ true, %[[MALLOC]] ]
+; CHECK-NEXT:    call void @llvm.amdgcn.s.barrier()
+; CHECK-NEXT:    [[TMP19:%.*]] = load ptr addrspace(1), ptr addrspace(3) @llvm.amdgcn.sw.lds.k1, align 8
+; CHECK-NEXT:    [[TMP20:%.*]] = load i32, ptr addrspace(1) getelementptr inbounds (i8, ptr addrspace(1) @llvm.amdgcn.sw.lds.k1.md, i64 12), align 4
+; CHECK-NEXT:    [[TMP21:%.*]] = getelementptr inbounds i8, ptr addrspace(3) @llvm.amdgcn.sw.lds.k1, i32 [[TMP20]]
+; CHECK-NEXT:    [[GEP:%.*]] = getelementptr inbounds [4 x i32], ptr addrspace(3) [[TMP21]], i32 0, i32 2
+; CHECK-NEXT:    [[TMP22:%.*]] = ptrtoint ptr addrspace(3) [[GEP]] to i32
+; CHECK-NEXT:    [[TMP23:%.*]] = getelementptr inbounds i8, ptr addrspace(1) [[TMP19]], i32 [[TMP22]]
+; CHECK-NEXT:    [[TMP24:%.*]] = addrspacecast ptr addrspace(1) [[TMP23]] to ptr
+; CHECK-NEXT:    call void @mid(ptr [[TMP24]])
+; CHECK-NEXT:    br label %[[CONDFREE:.*]]
+; CHECK:       [[CONDFREE]]:
+; CHECK-NEXT:    call void @llvm.amdgcn.s.barrier()
+; CHECK-NEXT:    br i1 [[XYZCOND]], label %[[FREE:.*]], label %[[END:.*]]
+; CHECK:       [[FREE]]:
+; CHECK-NEXT:    [[TMP25:%.*]] = call ptr @llvm.returnaddress.p0(i32 0)
+; CHECK-NEXT:    [[TMP26:%.*]] = ptrtoint ptr [[TMP25]] to i64
+; CHECK-NEXT:    [[TMP27:%.*]] = ptrtoint ptr addrspace(1) [[TMP19]] to i64
+; CHECK-NEXT:    call void @__asan_free_impl(i64 [[TMP27]], i64 [[TMP26]])
+; CHECK-NEXT:    br label %[[END]]
+; CHECK:       [[END]]:
+; CHECK-NEXT:    ret void
+;
+  %gep = getelementptr inbounds [4 x i32], ptr addrspace(3) @lds, i32 0, i32 2
+  %flat = addrspacecast ptr addrspace(3) %gep to ptr
+  call void @mid(ptr %flat)
+  ret void
+}
+
+define amdgpu_kernel void @k2(ptr %p) sanitize_address {
+; CHECK-LABEL: define amdgpu_kernel void @k2(
+; CHECK-SAME: ptr [[P:%.*]]) #[[ATTR0]] !llvm.amdgcn.lds.kernel.id [[META3:![0-9]+]] {
+; CHECK-NEXT:    call void @writer(ptr [[P]])
+; CHECK-NEXT:    ret void
+;
+  call void @writer(ptr %p)
+  ret void
+}
+
+attributes #0 = { sanitize_address "amdgpu-no-lds-kernel-id" }
+
+!llvm.module.flags = !{!0}
+!0 = !{i32 4, !"nosanitize_address", i32 1}
+;.
+; CHECK: attributes #[[ATTR0]] = { sanitize_address }
+; CHECK: attributes #[[ATTR1]] = { sanitize_address "amdgpu-lds-size"="8" }
+; CHECK: attributes #[[ATTR2:[0-9]+]] = { nocallback nofree nosync nounwind speculatable willreturn memory(none) }
+; CHECK: attributes #[[ATTR3:[0-9]+]] = { nocallback nofree nosync nounwind willreturn memory(none) }
+; CHECK: attributes #[[ATTR4:[0-9]+]] = { convergent nocallback nofree nounwind willreturn }
+;.
+; CHECK: [[META0]] = !{i32 0, i32 1}
+; CHECK: [[META1:![0-9]+]] = !{i32 4, !"nosanitize_address", i32 1}
+; CHECK: [[META2]] = !{i32 0}
+; CHECK: [[META3]] = !{i32 1}
+;.



More information about the llvm-branch-commits mailing list