[llvm-branch-commits] [llvm] [AMDGPU][SwLowerLDS] Lower non-kernels with LDS instructions and assign kernel IDs to their callers (PR #228043)
via llvm-branch-commits
llvm-branch-commits at lists.llvm.org
Thu Oct 1 04:24:17 PDT 2026
https://github.com/skc7 created https://github.com/llvm/llvm-project/pull/228043
Summary:
- Lower every non-kernel that a kernel reaches and that has `LDS memory instructions` or `flat to LDS` casts.
- Previously only non-kernels using LDS globals or taking LDS pointer arguments were lowered, so a device function casting a flat argument back to LDS accessed hardware LDS directly.
- Find these functions with a call graph walk from each kernel. Indirect calls are treated as reaching any address taken non-kernel, matching `getTransitiveUsesOfLDS`.
- Give every kernel that reaches such a function a kernel ID. Kernels without LDS get a null base table entry and a poison offset table row.
- Remove `amdgpu-no-lds-kernel-id` after the kernel set is final, so the kernel ID is passed to every function the kernel can reach.
Assisted by: claude opus 5.5
>From 9594aa490f1e68affb8672f049ba7025910d651a Mon Sep 17 00:00:00 2001
From: skc7 <Krishna.Sankisa at amd.com>
Date: Thu, 1 Oct 2026 16:49:27 +0530
Subject: [PATCH] [AMDGPU][SwLowerLDS] Lower non-kernels with LDS instructions
and assign kernel IDs to their callers
---
llvm/lib/Target/AMDGPU/AMDGPUSwLowerLDS.cpp | 202 +++++++++++-------
.../amdgpu-sw-lower-lds-flat-arg-kernel-id.ll | 131 ++++++++++++
2 files changed, 251 insertions(+), 82 deletions(-)
create mode 100644 llvm/test/CodeGen/AMDGPU/amdgpu-sw-lower-lds-flat-arg-kernel-id.ll
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUSwLowerLDS.cpp b/llvm/lib/Target/AMDGPU/AMDGPUSwLowerLDS.cpp
index e581e26057790..460b41dea9bf4 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUSwLowerLDS.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUSwLowerLDS.cpp
@@ -60,9 +60,10 @@
//
// Replacement of non-kernel LDS accesses:
// Multiple kernels can access the same non-kernel function.
-// All the kernels accessing LDS through non-kernels are sorted and
-// assigned a kernel-id. All the LDS globals accessed by non-kernels
-// are sorted. This information is used to build two tables:
+// All the kernels accessing LDS through non-kernels, or reaching
+// non-kernels with LDS instructions, are sorted and assigned a kernel-id.
+// All the LDS globals accessed by non-kernels are sorted. This
+// information is used to build two tables:
// - Base table:
// Base table will have single row, with elements of the row
// placed as per kernel ID. Each element in the row corresponds
@@ -95,6 +96,7 @@
#include "llvm/IR/DebugInfo.h"
#include "llvm/IR/DebugInfoMetadata.h"
#include "llvm/IR/IRBuilder.h"
+#include "llvm/IR/InstIterator.h"
#include "llvm/IR/Instructions.h"
#include "llvm/IR/MDBuilder.h"
#include "llvm/IR/ReplaceConstant.h"
@@ -160,7 +162,7 @@ struct AsanInstrumentInfo {
struct FunctionsAndLDSAccess {
MapVector<Function *, KernelLDSParameters> KernelToLDSParametersMap;
SetVector<Function *> KernelsWithIndirectLDSAccess;
- SetVector<Function *> NonKernelsWithLDSArgument;
+ SetVector<Function *> NonKernelsWithLDSInstructions;
SetVector<GlobalVariable *> AllNonKernelLDSAccess;
FunctionVariableMap NonKernelToLDSAccessMap;
};
@@ -171,7 +173,7 @@ class AMDGPUSwLowerLDS {
: M(Mod), IRB(M.getContext()), DTCallback(Callback) {}
bool run();
void getUsesOfLDSByNonKernels();
- void getNonKernelsWithLDSArguments(const CallGraph &CG);
+ void getNonKernelsWithLDSInstructions(const CallGraph &CG);
SetVector<Function *>
getOrderedIndirectLDSAccessingKernels(SetVector<Function *> &Kernels);
SetVector<GlobalVariable *>
@@ -251,35 +253,92 @@ SetVector<Function *> AMDGPUSwLowerLDS::getOrderedIndirectLDSAccessingKernels(
return OrderedKernels;
}
-void AMDGPUSwLowerLDS::getNonKernelsWithLDSArguments(const CallGraph &CG) {
- // Among the kernels accessing LDS, get list of
- // Non-kernels to which a call is made and a ptr
- // to addrspace(3) is passed as argument.
- for (auto &K : FuncLDSAccessInfo.KernelToLDSParametersMap) {
- Function *Func = K.first;
- const CallGraphNode *CGN = CG[Func];
- if (!CGN)
+// LDS memory operations and casts between flat and LDS.
+static bool isLDSMemoryInstruction(const Instruction &Inst) {
+ if (auto *LI = dyn_cast<LoadInst>(&Inst))
+ return LI->getPointerAddressSpace() == AMDGPUAS::LOCAL_ADDRESS;
+ if (auto *SI = dyn_cast<StoreInst>(&Inst))
+ return SI->getPointerAddressSpace() == AMDGPUAS::LOCAL_ADDRESS;
+ if (auto *RMW = dyn_cast<AtomicRMWInst>(&Inst))
+ return RMW->getPointerAddressSpace() == AMDGPUAS::LOCAL_ADDRESS;
+ if (auto *XCHG = dyn_cast<AtomicCmpXchgInst>(&Inst))
+ return XCHG->getPointerAddressSpace() == AMDGPUAS::LOCAL_ADDRESS;
+ if (auto *ASC = dyn_cast<AddrSpaceCastInst>(&Inst)) {
+ unsigned SrcAS = ASC->getSrcAddressSpace();
+ unsigned DstAS = ASC->getDestAddressSpace();
+ return (SrcAS == AMDGPUAS::LOCAL_ADDRESS &&
+ DstAS == AMDGPUAS::FLAT_ADDRESS) ||
+ (SrcAS == AMDGPUAS::FLAT_ADDRESS &&
+ DstAS == AMDGPUAS::LOCAL_ADDRESS);
+ }
+ if (auto *MI = dyn_cast<AnyMemIntrinsic>(&Inst)) {
+ if (MI->getDestAddressSpace() == AMDGPUAS::LOCAL_ADDRESS)
+ return true;
+ if (auto *MTI = dyn_cast<AnyMemTransferInst>(MI))
+ return MTI->getSourceAddressSpace() == AMDGPUAS::LOCAL_ADDRESS;
+ }
+ return false;
+}
+
+void AMDGPUSwLowerLDS::getNonKernelsWithLDSInstructions(const CallGraph &CG) {
+ // Non-kernels with LDS instructions find the malloc buffer through the base
+ // table, so every kernel reaching them needs a kernel ID, even one without
+ // LDS.
+ DenseMap<Function *, bool> HasLDSInstructionsCache;
+ auto HasLDSInstructions = [&](Function *F) {
+ auto [It, Inserted] = HasLDSInstructionsCache.try_emplace(F, false);
+ if (Inserted) {
+ It->second = any_of(instructions(*F), isLDSMemoryInstruction);
+ if (It->second)
+ FuncLDSAccessInfo.NonKernelsWithLDSInstructions.insert(F);
+ }
+ return It->second;
+ };
+
+ // Indirect calls may reach any address taken non-kernel.
+ SmallVector<Function *> AddressTakenFuncs;
+ for (Function &F : M)
+ if (!F.isDeclaration() && !isKernel(F) &&
+ F.hasAddressTaken(nullptr, /*IgnoreCallbackUses=*/false,
+ /*IgnoreAssumeLikeCalls=*/false,
+ /*IgnoreLLVMUsed=*/true))
+ AddressTakenFuncs.push_back(&F);
+
+ for (Function &Kernel : M) {
+ if (Kernel.isDeclaration() || !isKernel(Kernel))
continue;
- for (auto &I : *CGN) {
- CallGraphNode *CallerCGN = I.second;
- Function *CalledFunc = CallerCGN->getFunction();
- if (!CalledFunc || CalledFunc->isDeclaration())
- continue;
- if (AMDGPU::isKernel(*CalledFunc))
- continue;
- for (auto AI = CalledFunc->arg_begin(), E = CalledFunc->arg_end();
- AI != E; ++AI) {
- Type *ArgTy = (*AI).getType();
- if (!ArgTy->isPointerTy())
+
+ bool ReachesLDSInstructions = false;
+ bool SeenUnknownCall = false;
+ SmallVector<Function *> WorkList = {&Kernel};
+ SmallPtrSet<Function *, 8> Visited;
+ auto Visit = [&](Function *Callee) {
+ if (Callee->isDeclaration() || isKernel(*Callee) ||
+ !Visited.insert(Callee).second)
+ return;
+ ReachesLDSInstructions |= HasLDSInstructions(Callee);
+ WorkList.push_back(Callee);
+ };
+
+ while (!WorkList.empty()) {
+ Function *F = WorkList.pop_back_val();
+ for (const CallGraphNode::CallRecord &CallRecord : *CG[F]) {
+ if (!CallRecord.second)
continue;
- if (ArgTy->getPointerAddressSpace() != AMDGPUAS::LOCAL_ADDRESS)
+ if (Function *Callee = CallRecord.second->getFunction()) {
+ Visit(Callee);
+ continue;
+ }
+ if (SeenUnknownCall)
continue;
- FuncLDSAccessInfo.NonKernelsWithLDSArgument.insert(CalledFunc);
- // Also add the Calling function to KernelsWithIndirectLDSAccess list
- // so that base table of LDS is generated.
- FuncLDSAccessInfo.KernelsWithIndirectLDSAccess.insert(Func);
+ SeenUnknownCall = true;
+ for (Function *PotentialCallee : AddressTakenFuncs)
+ Visit(PotentialCallee);
}
}
+
+ if (ReachesLDSInstructions)
+ FuncLDSAccessInfo.KernelsWithIndirectLDSAccess.insert(&Kernel);
}
}
@@ -633,39 +692,9 @@ static DebugLoc getOrCreateDebugLoc(const Instruction *InsertBefore,
void AMDGPUSwLowerLDS::getLDSMemoryInstructions(
Function *Func, SetVector<Instruction *> &LDSInstructions) {
- for (BasicBlock &BB : *Func) {
- for (Instruction &Inst : BB) {
- if (LoadInst *LI = dyn_cast<LoadInst>(&Inst)) {
- if (LI->getPointerAddressSpace() == AMDGPUAS::LOCAL_ADDRESS)
- LDSInstructions.insert(&Inst);
- } else if (StoreInst *SI = dyn_cast<StoreInst>(&Inst)) {
- if (SI->getPointerAddressSpace() == AMDGPUAS::LOCAL_ADDRESS)
- LDSInstructions.insert(&Inst);
- } else if (AtomicRMWInst *RMW = dyn_cast<AtomicRMWInst>(&Inst)) {
- if (RMW->getPointerAddressSpace() == AMDGPUAS::LOCAL_ADDRESS)
- LDSInstructions.insert(&Inst);
- } else if (AtomicCmpXchgInst *XCHG = dyn_cast<AtomicCmpXchgInst>(&Inst)) {
- if (XCHG->getPointerAddressSpace() == AMDGPUAS::LOCAL_ADDRESS)
- LDSInstructions.insert(&Inst);
- } else if (AddrSpaceCastInst *ASC = dyn_cast<AddrSpaceCastInst>(&Inst)) {
- unsigned SrcAS = ASC->getSrcAddressSpace();
- unsigned DstAS = ASC->getDestAddressSpace();
- if ((SrcAS == AMDGPUAS::LOCAL_ADDRESS &&
- DstAS == AMDGPUAS::FLAT_ADDRESS) ||
- (SrcAS == AMDGPUAS::FLAT_ADDRESS &&
- DstAS == AMDGPUAS::LOCAL_ADDRESS))
- LDSInstructions.insert(&Inst);
- } else if (AnyMemIntrinsic *MI = dyn_cast<AnyMemIntrinsic>(&Inst)) {
- if (MI->getDestAddressSpace() == AMDGPUAS::LOCAL_ADDRESS) {
- LDSInstructions.insert(&Inst);
- } else if (auto *MTI = dyn_cast<AnyMemTransferInst>(MI)) {
- if (MTI->getSourceAddressSpace() == AMDGPUAS::LOCAL_ADDRESS)
- LDSInstructions.insert(&Inst);
- }
- } else
- continue;
- }
- }
+ for (Instruction &Inst : instructions(Func))
+ if (isLDSMemoryInstruction(Inst))
+ LDSInstructions.insert(&Inst);
}
Value *AMDGPUSwLowerLDS::getTranslatedGlobalMemoryPtrOfLDS(Value *LoadMallocPtr,
@@ -1035,14 +1064,19 @@ void AMDGPUSwLowerLDS::lowerKernelLDSAccesses(Function *Func,
Constant *AMDGPUSwLowerLDS::getAddressesOfVariablesInKernel(
Function *Func, SetVector<GlobalVariable *> &Variables) {
Type *Int32Ty = IRB.getInt32Ty();
- auto &LDSParams = FuncLDSAccessInfo.KernelToLDSParametersMap[Func];
+ ArrayType *KernelOffsetsType =
+ ArrayType::get(IRB.getPtrTy(AMDGPUAS::GLOBAL_ADDRESS), Variables.size());
+
+ // Kernels with no direct or indirect LDS uses never read their offset row.
+ auto KernelIt = FuncLDSAccessInfo.KernelToLDSParametersMap.find(Func);
+ if (KernelIt == FuncLDSAccessInfo.KernelToLDSParametersMap.end() ||
+ !KernelIt->second.SwLDSMetadata)
+ return PoisonValue::get(KernelOffsetsType);
+ auto &LDSParams = KernelIt->second;
GlobalVariable *SwLDSMetadata = LDSParams.SwLDSMetadata;
- assert(SwLDSMetadata);
auto *SwLDSMetadataStructType =
cast<StructType>(SwLDSMetadata->getValueType());
- ArrayType *KernelOffsetsType =
- ArrayType::get(IRB.getPtrTy(AMDGPUAS::GLOBAL_ADDRESS), Variables.size());
SmallVector<Constant *> Elements;
for (auto *GV : Variables) {
@@ -1073,13 +1107,18 @@ void AMDGPUSwLowerLDS::buildNonKernelLDSBaseTable(
if (Kernels.empty())
return;
const size_t NumberKernels = Kernels.size();
- ArrayType *AllKernelsOffsetsType =
- ArrayType::get(IRB.getPtrTy(AMDGPUAS::LOCAL_ADDRESS), NumberKernels);
+ PointerType *LDSPtrTy = IRB.getPtrTy(AMDGPUAS::LOCAL_ADDRESS);
+ ArrayType *AllKernelsOffsetsType = ArrayType::get(LDSPtrTy, NumberKernels);
std::vector<Constant *> OverallConstantExprElts(NumberKernels);
for (size_t i = 0; i < NumberKernels; i++) {
- Function *Func = Kernels[i];
- auto &LDSParams = FuncLDSAccessInfo.KernelToLDSParametersMap[Func];
- OverallConstantExprElts[i] = LDSParams.SwLDS;
+ // Null for kernels without LDS, they reach LDS instructions only on
+ // invalid paths.
+ Constant *SwLDS = ConstantPointerNull::get(LDSPtrTy);
+ auto It = FuncLDSAccessInfo.KernelToLDSParametersMap.find(Kernels[i]);
+ if (It != FuncLDSAccessInfo.KernelToLDSParametersMap.end() &&
+ It->second.SwLDS)
+ SwLDS = It->second.SwLDS;
+ OverallConstantExprElts[i] = SwLDS;
}
Constant *init =
ConstantArray::get(AllKernelsOffsetsType, OverallConstantExprElts);
@@ -1291,9 +1330,6 @@ bool AMDGPUSwLowerLDS::run() {
CG, Func,
{"amdgpu-no-workitem-id-x", "amdgpu-no-workitem-id-y",
"amdgpu-no-workitem-id-z", "amdgpu-no-heap-ptr"});
- if (!LDSParams.IndirectAccess.StaticLDSGlobals.empty() ||
- !LDSParams.IndirectAccess.DynamicLDSGlobals.empty())
- removeFnAttrFromReachable(CG, Func, {"amdgpu-no-lds-kernel-id"});
reorderStaticDynamicIndirectLDSSet(LDSParams);
buildSwLDSGlobal(Func);
buildSwDynLDSGlobal(Func);
@@ -1308,12 +1344,15 @@ bool AMDGPUSwLowerLDS::run() {
// Get the Uses of LDS from non-kernels.
getUsesOfLDSByNonKernels();
- // Get non-kernels with LDS ptr as argument and called by kernels.
- getNonKernelsWithLDSArguments(CG);
+ // Get non-kernels with LDS instructions and the kernels reaching them.
+ getNonKernelsWithLDSInstructions(CG);
+
+ for (Function *Func : FuncLDSAccessInfo.KernelsWithIndirectLDSAccess)
+ removeFnAttrFromReachable(CG, Func, {"amdgpu-no-lds-kernel-id"});
// Lower LDS accesses in non-kernels.
if (!FuncLDSAccessInfo.NonKernelToLDSAccessMap.empty() ||
- !FuncLDSAccessInfo.NonKernelsWithLDSArgument.empty()) {
+ !FuncLDSAccessInfo.NonKernelsWithLDSInstructions.empty()) {
NonKernelLDSParameters NKLDSParams;
NKLDSParams.OrderedKernels = getOrderedIndirectLDSAccessingKernels(
FuncLDSAccessInfo.KernelsWithIndirectLDSAccess);
@@ -1328,12 +1367,11 @@ bool AMDGPUSwLowerLDS::run() {
std::vector<GlobalVariable *>(LDSGlobals.begin(), LDSGlobals.end()));
lowerNonKernelLDSAccesses(Func, OrderedLDSGlobals, NKLDSParams);
}
- for (Function *Func : FuncLDSAccessInfo.NonKernelsWithLDSArgument) {
- auto &K = FuncLDSAccessInfo.NonKernelToLDSAccessMap;
- if (K.contains(Func))
+ for (Function *Func : FuncLDSAccessInfo.NonKernelsWithLDSInstructions) {
+ if (FuncLDSAccessInfo.NonKernelToLDSAccessMap.contains(Func))
continue;
- SetVector<llvm::GlobalVariable *> Vec;
- lowerNonKernelLDSAccesses(Func, Vec, NKLDSParams);
+ SetVector<GlobalVariable *> NoLDSGlobals;
+ lowerNonKernelLDSAccesses(Func, NoLDSGlobals, NKLDSParams);
}
Changed = true;
}
diff --git a/llvm/test/CodeGen/AMDGPU/amdgpu-sw-lower-lds-flat-arg-kernel-id.ll b/llvm/test/CodeGen/AMDGPU/amdgpu-sw-lower-lds-flat-arg-kernel-id.ll
new file mode 100644
index 0000000000000..b73c24834958f
--- /dev/null
+++ b/llvm/test/CodeGen/AMDGPU/amdgpu-sw-lower-lds-flat-arg-kernel-id.ll
@@ -0,0 +1,131 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --check-globals all --version 6
+; RUN: opt < %s -passes=amdgpu-sw-lower-lds -amdgpu-asan-instrument-lds=false -S -mtriple=amdgpu-amd-amdhsa | FileCheck %s
+
+; Test that a non-kernel with only a flat to LDS cast is lowered through the
+; base table, and that every kernel reaching it gets a kernel ID.
+; @k1 reaches @writer through @mid. @k2 has no LDS and gets a null base entry.
+; amdgpu-no-lds-kernel-id must be removed from @mid and @writer.
+
+ at lds = internal addrspace(3) global [4 x i32] poison, align 4
+
+;.
+; CHECK: @llvm.amdgcn.sw.lds.k1 = internal addrspace(3) global ptr poison, no_sanitize_address, align 4, !absolute_symbol [[META0:![0-9]+]]
+; CHECK: @llvm.amdgcn.sw.lds.k1.md = internal addrspace(1) global %llvm.amdgcn.sw.lds.k1.md.type { %llvm.amdgcn.sw.lds.k1.md.item { i32 0, i32 8, i32 32 }, %llvm.amdgcn.sw.lds.k1.md.item { i32 32, i32 16, i32 32 } }, no_sanitize_address
+; CHECK: @llvm.amdgcn.sw.lds.base.table = internal addrspace(1) constant [2 x ptr addrspace(3)] [ptr addrspace(3) @llvm.amdgcn.sw.lds.k1, ptr addrspace(3) null], no_sanitize_address
+;.
+define internal void @writer(ptr %p) #0 {
+; CHECK-LABEL: define internal void @writer(
+; CHECK-SAME: ptr [[P:%.*]]) #[[ATTR0:[0-9]+]] {
+; CHECK-NEXT: [[TMP1:%.*]] = call i32 @llvm.amdgcn.lds.kernel.id()
+; CHECK-NEXT: [[TMP2:%.*]] = getelementptr inbounds [2 x ptr addrspace(3)], ptr addrspace(1) @llvm.amdgcn.sw.lds.base.table, i32 0, i32 [[TMP1]]
+; CHECK-NEXT: [[TMP3:%.*]] = load ptr addrspace(3), ptr addrspace(1) [[TMP2]], align 4
+; CHECK-NEXT: [[TMP4:%.*]] = load ptr addrspace(1), ptr addrspace(3) [[TMP3]], align 8
+; CHECK-NEXT: [[TMP5:%.*]] = ptrtoint ptr addrspace(1) [[TMP4]] to i64
+; CHECK-NEXT: [[TMP6:%.*]] = ptrtoint ptr [[P]] to i64
+; CHECK-NEXT: [[TMP7:%.*]] = sub i64 [[TMP6]], [[TMP5]]
+; CHECK-NEXT: [[TMP8:%.*]] = trunc i64 [[TMP7]] to i32
+; CHECK-NEXT: [[TMP9:%.*]] = inttoptr i32 [[TMP8]] to ptr addrspace(3)
+; CHECK-NEXT: [[TMP10:%.*]] = ptrtoint ptr addrspace(3) [[TMP9]] to i32
+; CHECK-NEXT: [[TMP11:%.*]] = getelementptr inbounds i8, ptr addrspace(1) [[TMP4]], i32 [[TMP10]]
+; CHECK-NEXT: store i32 1, ptr addrspace(1) [[TMP11]], align 4
+; CHECK-NEXT: ret void
+;
+ %new_lds = addrspacecast ptr %p to ptr addrspace(3)
+ store i32 1, ptr addrspace(3) %new_lds, align 4
+ ret void
+}
+
+define internal void @mid(ptr %p) #0 {
+; CHECK-LABEL: define internal void @mid(
+; CHECK-SAME: ptr [[P:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT: call void @writer(ptr [[P]])
+; CHECK-NEXT: ret void
+;
+ call void @writer(ptr %p)
+ ret void
+}
+
+define amdgpu_kernel void @k1() sanitize_address {
+; CHECK-LABEL: define amdgpu_kernel void @k1(
+; CHECK-SAME: ) #[[ATTR1:[0-9]+]] !llvm.amdgcn.lds.kernel.id [[META2:![0-9]+]] {
+; CHECK-NEXT: [[WID:.*]]:
+; CHECK-NEXT: [[TMP0:%.*]] = call i32 @llvm.amdgcn.workitem.id.x()
+; CHECK-NEXT: [[TMP1:%.*]] = call i32 @llvm.amdgcn.workitem.id.y()
+; CHECK-NEXT: [[TMP2:%.*]] = call i32 @llvm.amdgcn.workitem.id.z()
+; CHECK-NEXT: [[TMP3:%.*]] = or i32 [[TMP0]], [[TMP1]]
+; CHECK-NEXT: [[TMP4:%.*]] = or i32 [[TMP3]], [[TMP2]]
+; CHECK-NEXT: [[TMP5:%.*]] = icmp eq i32 [[TMP4]], 0
+; CHECK-NEXT: br i1 [[TMP5]], label %[[MALLOC:.*]], label %[[BB18:.*]]
+; CHECK: [[MALLOC]]:
+; CHECK-NEXT: [[TMP6:%.*]] = load i32, ptr addrspace(1) getelementptr inbounds (i8, ptr addrspace(1) @llvm.amdgcn.sw.lds.k1.md, i64 12), align 4
+; CHECK-NEXT: [[TMP7:%.*]] = load i32, ptr addrspace(1) getelementptr inbounds (i8, ptr addrspace(1) @llvm.amdgcn.sw.lds.k1.md, i64 20), align 4
+; CHECK-NEXT: [[TMP8:%.*]] = add i32 [[TMP6]], [[TMP7]]
+; CHECK-NEXT: [[TMP9:%.*]] = zext i32 [[TMP8]] to i64
+; CHECK-NEXT: [[TMP10:%.*]] = call ptr @llvm.returnaddress.p0(i32 0)
+; CHECK-NEXT: [[TMP11:%.*]] = ptrtoint ptr [[TMP10]] to i64
+; CHECK-NEXT: [[TMP12:%.*]] = call i64 @__asan_malloc_impl(i64 [[TMP9]], i64 [[TMP11]])
+; CHECK-NEXT: [[TMP13:%.*]] = inttoptr i64 [[TMP12]] to ptr addrspace(1)
+; CHECK-NEXT: store ptr addrspace(1) [[TMP13]], ptr addrspace(3) @llvm.amdgcn.sw.lds.k1, align 8
+; CHECK-NEXT: [[TMP14:%.*]] = getelementptr inbounds i8, ptr addrspace(1) [[TMP13]], i64 8
+; CHECK-NEXT: [[TMP15:%.*]] = ptrtoint ptr addrspace(1) [[TMP14]] to i64
+; CHECK-NEXT: call void @__asan_poison_region(i64 [[TMP15]], i64 24)
+; CHECK-NEXT: [[TMP16:%.*]] = getelementptr inbounds i8, ptr addrspace(1) [[TMP13]], i64 48
+; CHECK-NEXT: [[TMP17:%.*]] = ptrtoint ptr addrspace(1) [[TMP16]] to i64
+; CHECK-NEXT: call void @__asan_poison_region(i64 [[TMP17]], i64 16)
+; CHECK-NEXT: br label %[[BB18]]
+; CHECK: [[BB18]]:
+; CHECK-NEXT: [[XYZCOND:%.*]] = phi i1 [ false, %[[WID]] ], [ true, %[[MALLOC]] ]
+; CHECK-NEXT: call void @llvm.amdgcn.s.barrier()
+; CHECK-NEXT: [[TMP19:%.*]] = load ptr addrspace(1), ptr addrspace(3) @llvm.amdgcn.sw.lds.k1, align 8
+; CHECK-NEXT: [[TMP20:%.*]] = load i32, ptr addrspace(1) getelementptr inbounds (i8, ptr addrspace(1) @llvm.amdgcn.sw.lds.k1.md, i64 12), align 4
+; CHECK-NEXT: [[TMP21:%.*]] = getelementptr inbounds i8, ptr addrspace(3) @llvm.amdgcn.sw.lds.k1, i32 [[TMP20]]
+; CHECK-NEXT: [[GEP:%.*]] = getelementptr inbounds [4 x i32], ptr addrspace(3) [[TMP21]], i32 0, i32 2
+; CHECK-NEXT: [[TMP22:%.*]] = ptrtoint ptr addrspace(3) [[GEP]] to i32
+; CHECK-NEXT: [[TMP23:%.*]] = getelementptr inbounds i8, ptr addrspace(1) [[TMP19]], i32 [[TMP22]]
+; CHECK-NEXT: [[TMP24:%.*]] = addrspacecast ptr addrspace(1) [[TMP23]] to ptr
+; CHECK-NEXT: call void @mid(ptr [[TMP24]])
+; CHECK-NEXT: br label %[[CONDFREE:.*]]
+; CHECK: [[CONDFREE]]:
+; CHECK-NEXT: call void @llvm.amdgcn.s.barrier()
+; CHECK-NEXT: br i1 [[XYZCOND]], label %[[FREE:.*]], label %[[END:.*]]
+; CHECK: [[FREE]]:
+; CHECK-NEXT: [[TMP25:%.*]] = call ptr @llvm.returnaddress.p0(i32 0)
+; CHECK-NEXT: [[TMP26:%.*]] = ptrtoint ptr [[TMP25]] to i64
+; CHECK-NEXT: [[TMP27:%.*]] = ptrtoint ptr addrspace(1) [[TMP19]] to i64
+; CHECK-NEXT: call void @__asan_free_impl(i64 [[TMP27]], i64 [[TMP26]])
+; CHECK-NEXT: br label %[[END]]
+; CHECK: [[END]]:
+; CHECK-NEXT: ret void
+;
+ %gep = getelementptr inbounds [4 x i32], ptr addrspace(3) @lds, i32 0, i32 2
+ %flat = addrspacecast ptr addrspace(3) %gep to ptr
+ call void @mid(ptr %flat)
+ ret void
+}
+
+define amdgpu_kernel void @k2(ptr %p) sanitize_address {
+; CHECK-LABEL: define amdgpu_kernel void @k2(
+; CHECK-SAME: ptr [[P:%.*]]) #[[ATTR0]] !llvm.amdgcn.lds.kernel.id [[META3:![0-9]+]] {
+; CHECK-NEXT: call void @writer(ptr [[P]])
+; CHECK-NEXT: ret void
+;
+ call void @writer(ptr %p)
+ ret void
+}
+
+attributes #0 = { sanitize_address "amdgpu-no-lds-kernel-id" }
+
+!llvm.module.flags = !{!0}
+!0 = !{i32 4, !"nosanitize_address", i32 1}
+;.
+; CHECK: attributes #[[ATTR0]] = { sanitize_address }
+; CHECK: attributes #[[ATTR1]] = { sanitize_address "amdgpu-lds-size"="8" }
+; CHECK: attributes #[[ATTR2:[0-9]+]] = { nocallback nofree nosync nounwind speculatable willreturn memory(none) }
+; CHECK: attributes #[[ATTR3:[0-9]+]] = { nocallback nofree nosync nounwind willreturn memory(none) }
+; CHECK: attributes #[[ATTR4:[0-9]+]] = { convergent nocallback nofree nounwind willreturn }
+;.
+; CHECK: [[META0]] = !{i32 0, i32 1}
+; CHECK: [[META1:![0-9]+]] = !{i32 4, !"nosanitize_address", i32 1}
+; CHECK: [[META2]] = !{i32 0}
+; CHECK: [[META3]] = !{i32 1}
+;.
More information about the llvm-branch-commits
mailing list