[clang] fa45fec - [CIR][SYCL] Enable relocatable device code for SYCL (#226596)
via cfe-commits
cfe-commits at lists.llvm.org
Wed Sep 30 16:47:06 PDT 2026
Author: Konstantinos Parasyris
Date: 2026-09-30T16:46:57-07:00
New Revision: fa45fecae6bb18a24a9ff0aaa674fb80b83ceccc
URL: https://github.com/llvm/llvm-project/commit/fa45fecae6bb18a24a9ff0aaa674fb80b83ceccc
DIFF: https://github.com/llvm/llvm-project/commit/fa45fecae6bb18a24a9ff0aaa674fb80b83ceccc.diff
LOG: [CIR][SYCL] Enable relocatable device code for SYCL (#226596)
Emit `sycl_external` functions with sycl-module-id, allow -fgpu-rdc
mangling, and embed offload objects in the host.
Co-authored-by: Claude Opus 5.5 <noreply at anthropic.com>
Added:
clang/test/CIR/CodeGenSYCL/embed-offload-object.cpp
clang/test/CIR/CodeGenSYCL/gpu-rdc-mangled-name.cpp
clang/test/CIR/CodeGenSYCL/sycl-external.cpp
Modified:
clang/lib/CIR/CodeGen/CIRGenModule.cpp
clang/lib/CIR/FrontendAction/CIRGenAction.cpp
Removed:
################################################################################
diff --git a/clang/lib/CIR/CodeGen/CIRGenModule.cpp b/clang/lib/CIR/CodeGen/CIRGenModule.cpp
index 65bab0394d59a..c45d6d6ca48fa 100644
--- a/clang/lib/CIR/CodeGen/CIRGenModule.cpp
+++ b/clang/lib/CIR/CodeGen/CIRGenModule.cpp
@@ -2829,7 +2829,10 @@ static std::string getMangledNameImpl(CIRGenModule &cgm, GlobalDecl gd,
"getMangledName: multi-version functions");
}
}
- if (cgm.getLangOpts().GPURelocatableDeviceCode) {
+ // SYCL does not externalize file-scope statics, so RDC does not change the
+ // mangled name.
+ if (cgm.getLangOpts().GPURelocatableDeviceCode &&
+ !cgm.getLangOpts().isSYCL()) {
cgm.errorNYI(nd->getSourceRange(),
"getMangledName: GPU relocatable device code");
}
@@ -2996,10 +2999,10 @@ bool CIRGenModule::mayBeEmittedEagerly(const ValueDecl *global) {
// Defer until all versions have been semantically checked.
if (fd->hasAttr<TargetVersionAttr>() && !fd->isMultiVersion())
return false;
- if (langOpts.SYCLIsDevice) {
- errorNYI(fd->getSourceRange(), "mayBeEmittedEagerly: SYCL");
+ // Defer emission of SYCL kernel entry point functions during device
+ // compilation.
+ if (langOpts.SYCLIsDevice && fd->hasAttr<SYCLKernelEntryPointAttr>())
return false;
- }
}
const auto *vd = dyn_cast<VarDecl>(global);
if (vd)
@@ -3451,6 +3454,11 @@ void CIRGenModule::setCIRFunctionAttributesForDefinition(
if (isa<CXXMethodDecl>(decl) && f.getAlignment().value_or(1) < 2)
f.setAlignment(2);
}
+
+ // Attach "sycl-module-id" to sycl_external function definitions to mark
+ // them as entry points for per-translation-unit device-code splitting.
+ if (getLangOpts().SYCLIsDevice && decl->hasAttr<SYCLExternalAttr>())
+ addSYCLModuleIdAttr(f);
}
// Maps an AST address space to the OpenCL logical address space kind recorded
diff --git a/clang/lib/CIR/FrontendAction/CIRGenAction.cpp b/clang/lib/CIR/FrontendAction/CIRGenAction.cpp
index ee895760a4057..94aea278b4fa9 100644
--- a/clang/lib/CIR/FrontendAction/CIRGenAction.cpp
+++ b/clang/lib/CIR/FrontendAction/CIRGenAction.cpp
@@ -238,6 +238,24 @@ class CIRGenConsumer : public clang::ASTConsumer {
if (C.getLangOpts().SYCLIsHost && !CGO.OffloadBinaryToEmbedFile.empty())
embedSYCLDeviceBinary(*LLVMModule);
+ // CUDA, HIP and OpenMP offloading rely on host-side offload entries that
+ // are not emitted on the ClangIR path yet, so embedding their device
+ // objects would produce a host object that cannot be registered.
+ const LangOptions &LangOpts = C.getLangOpts();
+ if (!CGO.OffloadObjects.empty() &&
+ (LangOpts.CUDA || !LangOpts.OMPTargetTriples.empty())) {
+ DiagnosticsEngine &Diags = CI.getDiagnostics();
+ Diags.Report(Diags.getCustomDiagID(
+ DiagnosticsEngine::Error,
+ "ClangIR code gen Not Yet Implemented: embedding offload objects "
+ "for CUDA, HIP or OpenMP offloading"));
+ return;
+ }
+
+ // If there is device offloading code embed it in the host now.
+ EmbedObject(LLVMModule.get(), CGO, CI.getVirtualFileSystem(),
+ CI.getDiagnostics());
+
BackendAction BEAction = getBackendActionFromOutputType(Action);
emitBackendOutput(CI, CI.getCodeGenOpts(), LLVMModule.get(), BEAction, FS,
std::move(OutputStream));
diff --git a/clang/test/CIR/CodeGenSYCL/embed-offload-object.cpp b/clang/test/CIR/CodeGenSYCL/embed-offload-object.cpp
new file mode 100644
index 0000000000000..d2b3968e15e39
--- /dev/null
+++ b/clang/test/CIR/CodeGenSYCL/embed-offload-object.cpp
@@ -0,0 +1,74 @@
+// REQUIRES: x86-registered-target
+
+// Verify that on the ClangIR path -fembed-offload-object embeds the packaged
+// device object into the ".llvm.offloading" section of the host module,
+// matching classic CodeGen. This is the default relocatable device code flow
+// for SYCL.
+// RUN: echo -n 'FAKE_OFFLOAD_OBJECT' > %t.out
+// RUN: %clang_cc1 -triple x86_64-unknown-linux-gnu -fsycl-is-host -fgpu-rdc \
+// RUN: -fclangir -fembed-offload-object=%t.out -emit-llvm %s -o - \
+// RUN: | FileCheck %s
+// RUN: %clang_cc1 -triple x86_64-unknown-linux-gnu -fsycl-is-host -fgpu-rdc \
+// RUN: -fembed-offload-object=%t.out -emit-llvm %s -o - \
+// RUN: | FileCheck %s
+
+// Object emission must produce the ".llvm.offloading" section.
+// RUN: %clang_cc1 -triple x86_64-unknown-linux-gnu -fsycl-is-host -fgpu-rdc \
+// RUN: -fclangir -fembed-offload-object=%t.out -emit-obj %s -o %t.o
+// RUN: llvm-readelf -S %t.o | FileCheck %s --check-prefix=OBJ
+
+// Without the flag nothing is embedded.
+// RUN: %clang_cc1 -triple x86_64-unknown-linux-gnu -fsycl-is-host -fgpu-rdc \
+// RUN: -fclangir -emit-llvm %s -o - \
+// RUN: | FileCheck %s --check-prefix=NONE --implicit-check-not='.llvm.offloading'
+
+// Embedding does not depend on an offloading language model, as in classic
+// CodeGen.
+// RUN: %clang_cc1 -triple x86_64-unknown-linux-gnu -x c++ -fclangir \
+// RUN: -fembed-offload-object=%t.out -emit-llvm /dev/null -o - \
+// RUN: | FileCheck %s
+
+// A missing object file must be diagnosed.
+// RUN: not %clang_cc1 -triple x86_64-unknown-linux-gnu -fsycl-is-host \
+// RUN: -fgpu-rdc -fclangir -fembed-offload-object=%t.does-not-exist \
+// RUN: -emit-llvm %s -o - 2>&1 | FileCheck %s --check-prefix=ERROR
+
+// Embedding for CUDA, HIP and OpenMP offloading is not implemented on the
+// ClangIR path yet, including when OpenMP offloading is combined with SYCL.
+// RUN: not %clang_cc1 -triple x86_64-unknown-linux-gnu -x cuda -fclangir \
+// RUN: -fembed-offload-object=%t.out -emit-llvm /dev/null -o - 2>&1 \
+// RUN: | FileCheck %s --check-prefix=NYI
+// RUN: not %clang_cc1 -triple x86_64-unknown-linux-gnu -x hip -fclangir \
+// RUN: -fembed-offload-object=%t.out -emit-llvm /dev/null -o - 2>&1 \
+// RUN: | FileCheck %s --check-prefix=NYI
+// RUN: not %clang_cc1 -triple x86_64-unknown-linux-gnu -x c -fopenmp \
+// RUN: -fopenmp-targets=amdgcn-amd-amdhsa -fclangir \
+// RUN: -fembed-offload-object=%t.out -emit-llvm /dev/null -o - 2>&1 \
+// RUN: | FileCheck %s --check-prefix=NYI
+// RUN: not %clang_cc1 -triple x86_64-unknown-linux-gnu -x c++ -fsycl-is-host \
+// RUN: -fopenmp -fopenmp-targets=amdgcn-amd-amdhsa -fclangir \
+// RUN: -fembed-offload-object=%t.out -emit-llvm /dev/null -o - 2>&1 \
+// RUN: | FileCheck %s --check-prefix=NYI
+
+template <typename KN, typename... Ts>
+void sycl_kernel_launch(const char *, Ts...) {}
+
+template <typename KN, typename KT>
+[[clang::sycl_kernel_entry_point(KN)]] void kernel_entry(KT k) { k(); }
+
+struct KN;
+
+int main() { kernel_entry<KN>([] {}); }
+
+// CHECK: @llvm.embedded.object = private constant [19 x i8] c"FAKE_OFFLOAD_OBJECT", section ".llvm.offloading", align 8, !exclude
+// CHECK: @llvm.compiler.used = appending global [1 x ptr] [ptr @llvm.embedded.object], section "llvm.metadata"
+// CHECK: !llvm.embedded.objects = !{![[EMB:[0-9]+]]}
+// CHECK: ![[EMB]] = !{ptr @llvm.embedded.object, !".llvm.offloading"}
+
+// OBJ: .llvm.offloading
+
+// NONE: define {{.*}}@main
+
+// ERROR: error: could not open '{{.*}}.does-not-exist' for embedding
+
+// NYI: error: ClangIR code gen Not Yet Implemented: embedding offload objects for CUDA, HIP or OpenMP offloading
diff --git a/clang/test/CIR/CodeGenSYCL/gpu-rdc-mangled-name.cpp b/clang/test/CIR/CodeGenSYCL/gpu-rdc-mangled-name.cpp
new file mode 100644
index 0000000000000..7e63eccea0f86
--- /dev/null
+++ b/clang/test/CIR/CodeGenSYCL/gpu-rdc-mangled-name.cpp
@@ -0,0 +1,53 @@
+// SYCL enables relocatable device code by default. SYCL does not externalize
+// file-scope statics, so -fgpu-rdc must not change mangled names on either the
+// host or the device side.
+
+// RUN: %clang_cc1 -triple x86_64-unknown-linux-gnu -fsycl-is-host -fgpu-rdc \
+// RUN: -fclangir -emit-cir %s -o %t.cir
+// RUN: FileCheck --check-prefix=CIR --input-file=%t.cir %s
+// RUN: %clang_cc1 -triple x86_64-unknown-linux-gnu -fsycl-is-host -fgpu-rdc \
+// RUN: -fclangir -emit-llvm %s -o %t-cir.ll
+// RUN: FileCheck --check-prefix=LLVM --input-file=%t-cir.ll %s
+// RUN: %clang_cc1 -triple x86_64-unknown-linux-gnu -fsycl-is-host -fgpu-rdc \
+// RUN: -emit-llvm %s -o %t.ll
+// RUN: FileCheck --check-prefix=LLVM --input-file=%t.ll %s
+
+// RUN: %clang_cc1 -triple spirv64-unknown-unknown \
+// RUN: -aux-triple x86_64-unknown-linux-gnu -fsycl-is-device -fgpu-rdc \
+// RUN: -fclangir -emit-cir %s -o %t-dev.cir
+// RUN: FileCheck --check-prefix=CIR-DEV --input-file=%t-dev.cir %s
+// RUN: %clang_cc1 -triple spirv64-unknown-unknown \
+// RUN: -aux-triple x86_64-unknown-linux-gnu -fsycl-is-device -fgpu-rdc \
+// RUN: -fclangir -emit-llvm %s -o %t-dev-cir.ll
+// RUN: FileCheck --check-prefix=LLVM-DEV --input-file=%t-dev-cir.ll %s
+// RUN: %clang_cc1 -triple spirv64-unknown-unknown \
+// RUN: -aux-triple x86_64-unknown-linux-gnu -fsycl-is-device -fgpu-rdc \
+// RUN: -emit-llvm %s -o %t-dev.ll
+// RUN: FileCheck --check-prefix=LLVM-DEV --input-file=%t-dev.ll %s
+
+template <typename KN, typename... Ts>
+void sycl_kernel_launch(const char *, Ts...) {}
+
+template <typename KN, typename KT>
+[[clang::sycl_kernel_entry_point(KN)]] void kernel_entry(KT k) { k(); }
+
+struct KN;
+
+static int hostStatic;
+
+template <typename T> static T getValue() { return T(1); }
+
+int main() {
+ hostStatic = 1;
+ kernel_entry<KN>([] { (void)getValue<int>(); });
+}
+
+// CIR: cir.global "private" internal dso_local @_ZL10hostStatic
+// CIR: cir.func {{.*}}@main
+// LLVM: @_ZL10hostStatic = internal global i32 0
+// LLVM: define {{.*}}@main
+
+// CIR-DEV: cir.func {{.*}}@_ZTS2KN
+// CIR-DEV: cir.func {{.*}}@_ZL8getValueIiET_v
+// LLVM-DEV: define {{.*}}spir_kernel void @_ZTS2KN
+// LLVM-DEV: define internal spir_func {{.*}}@_ZL8getValueIiET_v
diff --git a/clang/test/CIR/CodeGenSYCL/sycl-external.cpp b/clang/test/CIR/CodeGenSYCL/sycl-external.cpp
new file mode 100644
index 0000000000000..d3676f06c2d23
--- /dev/null
+++ b/clang/test/CIR/CodeGenSYCL/sycl-external.cpp
@@ -0,0 +1,74 @@
+// RUN: %clang_cc1 -fsycl-is-device -triple spirv64-unknown-unknown -fclangir -emit-cir %s -o %t.cir
+// RUN: FileCheck --input-file=%t.cir %s -check-prefix=CIR
+// RUN: %clang_cc1 -fsycl-is-device -triple spirv64-unknown-unknown -fclangir -emit-llvm %s -o %t-cir.ll
+// RUN: FileCheck --input-file=%t-cir.ll %s --check-prefixes=LLVM,LLVM-CIR
+// RUN: %clang_cc1 -fsycl-is-device -triple spirv64-unknown-unknown -emit-llvm %s -o %t.ll
+// RUN: FileCheck --input-file=%t.ll %s --check-prefixes=LLVM,OGCG
+
+// Verify that sycl_external functions are emitted in device code, matching
+// classic CodeGen, and that their definitions carry the "sycl-module-id"
+// attribute that marks them as entry points for device-code splitting.
+
+template <typename KernelName, typename... Ts>
+void sycl_kernel_launch(const char *, Ts...) {}
+
+template <typename KernelName, typename KernelType>
+[[clang::sycl_kernel_entry_point(KernelName)]]
+void kernel_single_task(KernelType kf) { kf(); }
+
+struct KN;
+void use() { kernel_single_task<KN>([] {}); }
+
+// Defined and not used: emitted.
+[[clang::sycl_external]] int square(int x) { return x * x; }
+
+// Declared but not defined or used: not emitted.
+[[clang::sycl_external]] int declOnly();
+
+// Declared and used in device code but not defined: external reference.
+[[clang::sycl_external]] void declUsedInDevice(int y);
+[[clang::sycl_external]] void deviceUse() { declUsedInDevice(3); }
+
+// Declared with the attribute and later defined: definition emitted.
+[[clang::sycl_external]] int func1(int arg);
+int func1(int arg) { return arg; }
+
+// Reachable from a sycl_external function: emitted, but not an entry point.
+int ret1() { return 1; }
+[[clang::sycl_external]] int withAttr() { return ret1(); }
+
+// Explicit specialization defined: emitted.
+template <typename T> [[clang::sycl_external]] void tFunc(T arg) {}
+template <> [[clang::sycl_external]] void tFunc<int>(int arg) {}
+
+// Defined without the attribute and not used in device code: not emitted.
+int squareNoAttr(int x) { return x * x; }
+
+// CIR: cir.func {{.*}}@_Z6squarei({{.*}} attributes {{{.*}}"sycl-module-id" = "{{.*}}sycl-external.cpp"}
+// CIR: cir.func {{.*}}@_Z9deviceUsev() {{.*}} attributes {{{.*}}"sycl-module-id" = "{{.*}}sycl-external.cpp"}
+// CIR: cir.func private @_Z16declUsedInDevicei({{.*}} attributes {convergent{{[^"]*}}} loc
+// CIR: cir.func {{.*}}@_Z5func1i({{.*}} attributes {{{.*}}"sycl-module-id" = "{{.*}}sycl-external.cpp"}
+// CIR: cir.func {{.*}}@_Z8withAttrv() {{.*}} attributes {{{.*}}"sycl-module-id" = "{{.*}}sycl-external.cpp"}
+// CIR: cir.func {{.*}}@_Z4ret1v() {{.*}} attributes {convergent{{[^"]*}}} {
+// CIR: cir.func {{.*}}@_Z5tFuncIiEvT_({{.*}} attributes {{{.*}}"sycl-module-id" = "{{.*}}sycl-external.cpp"}
+// CIR: cir.func {{.*}}@_ZTS2KN({{.*}} attributes {{{.*}}"sycl-module-id" = "{{.*}}sycl-external.cpp"}
+// CIR-NOT: @_Z8declOnlyv
+// CIR-NOT: @_Z12squareNoAttri
+
+// LLVM: define spir_func noundef i32 @_Z6squarei(i32 noundef %{{.*}}) #[[EXT:[0-9]+]]
+// LLVM: define spir_func void @_Z9deviceUsev() #[[EXT]]
+// LLVM: declare spir_func void @_Z16declUsedInDevicei(i32 noundef) #[[DECL:[0-9]+]]
+// LLVM: define spir_func noundef i32 @_Z5func1i(i32 noundef %{{.*}}) #[[EXT]]
+// LLVM: define spir_func noundef i32 @_Z8withAttrv() #[[EXT]]
+// LLVM: define spir_func noundef i32 @_Z4ret1v() #[[NOEXT:[0-9]+]]
+// LLVM: define spir_func void @_Z5tFuncIiEvT_(i32 noundef %{{.*}}) #[[EXT]]
+// ClangIR does not yet pass kernel arguments byval or emit the attributes that
+// would let the kernel share the attribute group of the other entry points.
+// LLVM-CIR: define spir_kernel void @_ZTS2KN({{.*}}) #[[KERNEL:[0-9]+]]
+// OGCG: define spir_kernel void @_ZTS2KN(ptr noundef byval({{.*}}) #[[EXT]]
+// LLVM-NOT: @_Z8declOnlyv
+// LLVM-NOT: @_Z12squareNoAttri
+// LLVM-DAG: attributes #[[EXT]] = { {{.*}}"sycl-module-id"="{{.*}}sycl-external.cpp" }
+// LLVM-DAG: attributes #[[DECL]] = { convergent{{.*}} }
+// LLVM-DAG: attributes #[[NOEXT]] = { convergent {{.*}}noinline{{.*}} }
+// LLVM-CIR-DAG: attributes #[[KERNEL]] = { {{.*}}"sycl-module-id"="{{.*}}sycl-external.cpp" }
More information about the cfe-commits
mailing list