llvmorg-github-actions[bot] wrote:
<!--LLVM PR SUMMARY COMMENT--> @llvm/pr-subscribers-clang Author: Konstantinos Parasyris (koparasy) <details> <summary>Changes</summary> Emit `sycl_external` functions with sycl-module-id, allow -fgpu-rdc mangling, and embed offload objects in the host. --- Full diff: https://github.com/llvm/llvm-project/pull/226596.diff 5 Files Affected: - (modified) clang/lib/CIR/CodeGen/CIRGenModule.cpp (+12-4) - (modified) clang/lib/CIR/FrontendAction/CIRGenAction.cpp (+18) - (added) clang/test/CIR/CodeGenSYCL/embed-offload-object.cpp (+74) - (added) clang/test/CIR/CodeGenSYCL/gpu-rdc-mangled-name.cpp (+53) - (added) clang/test/CIR/CodeGenSYCL/sycl-external.cpp (+74) ``````````diff diff --git a/clang/lib/CIR/CodeGen/CIRGenModule.cpp b/clang/lib/CIR/CodeGen/CIRGenModule.cpp index adffa7dfe2969..b6baf082deb9a 100644 --- a/clang/lib/CIR/CodeGen/CIRGenModule.cpp +++ b/clang/lib/CIR/CodeGen/CIRGenModule.cpp @@ -2836,7 +2836,10 @@ static std::string getMangledNameImpl(CIRGenModule &cgm, GlobalDecl gd, "getMangledName: multi-version functions"); } } - if (cgm.getLangOpts().GPURelocatableDeviceCode) { + // SYCL does not externalize file-scope statics, so RDC does not change the + // mangled name. + if (cgm.getLangOpts().GPURelocatableDeviceCode && + !cgm.getLangOpts().isSYCL()) { cgm.errorNYI(nd->getSourceRange(), "getMangledName: GPU relocatable device code"); } @@ -3003,10 +3006,10 @@ bool CIRGenModule::mayBeEmittedEagerly(const ValueDecl *global) { // Defer until all versions have been semantically checked. if (fd->hasAttr<TargetVersionAttr>() && !fd->isMultiVersion()) return false; - if (langOpts.SYCLIsDevice) { - errorNYI(fd->getSourceRange(), "mayBeEmittedEagerly: SYCL"); + // Defer emission of SYCL kernel entry point functions during device + // compilation. + if (langOpts.SYCLIsDevice && fd->hasAttr<SYCLKernelEntryPointAttr>()) return false; - } } const auto *vd = dyn_cast<VarDecl>(global); if (vd) @@ -3445,6 +3448,11 @@ void CIRGenModule::setCIRFunctionAttributesForDefinition( if (isa<CXXMethodDecl>(decl) && f.getAlignment().value_or(1) < 2) f.setAlignment(2); } + + // Attach "sycl-module-id" to sycl_external function definitions to mark + // them as entry points for per-translation-unit device-code splitting. + if (getLangOpts().SYCLIsDevice && decl->hasAttr<SYCLExternalAttr>()) + addSYCLModuleIdAttr(f); } // Maps an AST address space to the OpenCL logical address space kind recorded diff --git a/clang/lib/CIR/FrontendAction/CIRGenAction.cpp b/clang/lib/CIR/FrontendAction/CIRGenAction.cpp index 240601f9834e5..ad68a6526e4d1 100644 --- a/clang/lib/CIR/FrontendAction/CIRGenAction.cpp +++ b/clang/lib/CIR/FrontendAction/CIRGenAction.cpp @@ -215,6 +215,24 @@ class CIRGenConsumer : public clang::ASTConsumer { if (C.getLangOpts().SYCLIsHost && !CGO.OffloadBinaryToEmbedFile.empty()) embedSYCLDeviceBinary(*LLVMModule); + // CUDA, HIP and OpenMP offloading rely on host-side offload entries that + // are not emitted on the ClangIR path yet, so embedding their device + // objects would produce a host object that cannot be registered. + const LangOptions &LangOpts = C.getLangOpts(); + if (!CGO.OffloadObjects.empty() && + (LangOpts.CUDA || !LangOpts.OMPTargetTriples.empty())) { + DiagnosticsEngine &Diags = CI.getDiagnostics(); + Diags.Report(Diags.getCustomDiagID( + DiagnosticsEngine::Error, + "ClangIR code gen Not Yet Implemented: embedding offload objects " + "for CUDA, HIP or OpenMP offloading")); + return; + } + + // If there is device offloading code embed it in the host now. + EmbedObject(LLVMModule.get(), CGO, CI.getVirtualFileSystem(), + CI.getDiagnostics()); + BackendAction BEAction = getBackendActionFromOutputType(Action); emitBackendOutput(CI, CI.getCodeGenOpts(), LLVMModule.get(), BEAction, FS, std::move(OutputStream)); diff --git a/clang/test/CIR/CodeGenSYCL/embed-offload-object.cpp b/clang/test/CIR/CodeGenSYCL/embed-offload-object.cpp new file mode 100644 index 0000000000000..d2b3968e15e39 --- /dev/null +++ b/clang/test/CIR/CodeGenSYCL/embed-offload-object.cpp @@ -0,0 +1,74 @@ +// REQUIRES: x86-registered-target + +// Verify that on the ClangIR path -fembed-offload-object embeds the packaged +// device object into the ".llvm.offloading" section of the host module, +// matching classic CodeGen. This is the default relocatable device code flow +// for SYCL. +// RUN: echo -n 'FAKE_OFFLOAD_OBJECT' > %t.out +// RUN: %clang_cc1 -triple x86_64-unknown-linux-gnu -fsycl-is-host -fgpu-rdc \ +// RUN: -fclangir -fembed-offload-object=%t.out -emit-llvm %s -o - \ +// RUN: | FileCheck %s +// RUN: %clang_cc1 -triple x86_64-unknown-linux-gnu -fsycl-is-host -fgpu-rdc \ +// RUN: -fembed-offload-object=%t.out -emit-llvm %s -o - \ +// RUN: | FileCheck %s + +// Object emission must produce the ".llvm.offloading" section. +// RUN: %clang_cc1 -triple x86_64-unknown-linux-gnu -fsycl-is-host -fgpu-rdc \ +// RUN: -fclangir -fembed-offload-object=%t.out -emit-obj %s -o %t.o +// RUN: llvm-readelf -S %t.o | FileCheck %s --check-prefix=OBJ + +// Without the flag nothing is embedded. +// RUN: %clang_cc1 -triple x86_64-unknown-linux-gnu -fsycl-is-host -fgpu-rdc \ +// RUN: -fclangir -emit-llvm %s -o - \ +// RUN: | FileCheck %s --check-prefix=NONE --implicit-check-not='.llvm.offloading' + +// Embedding does not depend on an offloading language model, as in classic +// CodeGen. +// RUN: %clang_cc1 -triple x86_64-unknown-linux-gnu -x c++ -fclangir \ +// RUN: -fembed-offload-object=%t.out -emit-llvm /dev/null -o - \ +// RUN: | FileCheck %s + +// A missing object file must be diagnosed. +// RUN: not %clang_cc1 -triple x86_64-unknown-linux-gnu -fsycl-is-host \ +// RUN: -fgpu-rdc -fclangir -fembed-offload-object=%t.does-not-exist \ +// RUN: -emit-llvm %s -o - 2>&1 | FileCheck %s --check-prefix=ERROR + +// Embedding for CUDA, HIP and OpenMP offloading is not implemented on the +// ClangIR path yet, including when OpenMP offloading is combined with SYCL. +// RUN: not %clang_cc1 -triple x86_64-unknown-linux-gnu -x cuda -fclangir \ +// RUN: -fembed-offload-object=%t.out -emit-llvm /dev/null -o - 2>&1 \ +// RUN: | FileCheck %s --check-prefix=NYI +// RUN: not %clang_cc1 -triple x86_64-unknown-linux-gnu -x hip -fclangir \ +// RUN: -fembed-offload-object=%t.out -emit-llvm /dev/null -o - 2>&1 \ +// RUN: | FileCheck %s --check-prefix=NYI +// RUN: not %clang_cc1 -triple x86_64-unknown-linux-gnu -x c -fopenmp \ +// RUN: -fopenmp-targets=amdgcn-amd-amdhsa -fclangir \ +// RUN: -fembed-offload-object=%t.out -emit-llvm /dev/null -o - 2>&1 \ +// RUN: | FileCheck %s --check-prefix=NYI +// RUN: not %clang_cc1 -triple x86_64-unknown-linux-gnu -x c++ -fsycl-is-host \ +// RUN: -fopenmp -fopenmp-targets=amdgcn-amd-amdhsa -fclangir \ +// RUN: -fembed-offload-object=%t.out -emit-llvm /dev/null -o - 2>&1 \ +// RUN: | FileCheck %s --check-prefix=NYI + +template <typename KN, typename... Ts> +void sycl_kernel_launch(const char *, Ts...) {} + +template <typename KN, typename KT> +[[clang::sycl_kernel_entry_point(KN)]] void kernel_entry(KT k) { k(); } + +struct KN; + +int main() { kernel_entry<KN>([] {}); } + +// CHECK: @llvm.embedded.object = private constant [19 x i8] c"FAKE_OFFLOAD_OBJECT", section ".llvm.offloading", align 8, !exclude +// CHECK: @llvm.compiler.used = appending global [1 x ptr] [ptr @llvm.embedded.object], section "llvm.metadata" +// CHECK: !llvm.embedded.objects = !{![[EMB:[0-9]+]]} +// CHECK: ![[EMB]] = !{ptr @llvm.embedded.object, !".llvm.offloading"} + +// OBJ: .llvm.offloading + +// NONE: define {{.*}}@main + +// ERROR: error: could not open '{{.*}}.does-not-exist' for embedding + +// NYI: error: ClangIR code gen Not Yet Implemented: embedding offload objects for CUDA, HIP or OpenMP offloading diff --git a/clang/test/CIR/CodeGenSYCL/gpu-rdc-mangled-name.cpp b/clang/test/CIR/CodeGenSYCL/gpu-rdc-mangled-name.cpp new file mode 100644 index 0000000000000..7e63eccea0f86 --- /dev/null +++ b/clang/test/CIR/CodeGenSYCL/gpu-rdc-mangled-name.cpp @@ -0,0 +1,53 @@ +// SYCL enables relocatable device code by default. SYCL does not externalize +// file-scope statics, so -fgpu-rdc must not change mangled names on either the +// host or the device side. + +// RUN: %clang_cc1 -triple x86_64-unknown-linux-gnu -fsycl-is-host -fgpu-rdc \ +// RUN: -fclangir -emit-cir %s -o %t.cir +// RUN: FileCheck --check-prefix=CIR --input-file=%t.cir %s +// RUN: %clang_cc1 -triple x86_64-unknown-linux-gnu -fsycl-is-host -fgpu-rdc \ +// RUN: -fclangir -emit-llvm %s -o %t-cir.ll +// RUN: FileCheck --check-prefix=LLVM --input-file=%t-cir.ll %s +// RUN: %clang_cc1 -triple x86_64-unknown-linux-gnu -fsycl-is-host -fgpu-rdc \ +// RUN: -emit-llvm %s -o %t.ll +// RUN: FileCheck --check-prefix=LLVM --input-file=%t.ll %s + +// RUN: %clang_cc1 -triple spirv64-unknown-unknown \ +// RUN: -aux-triple x86_64-unknown-linux-gnu -fsycl-is-device -fgpu-rdc \ +// RUN: -fclangir -emit-cir %s -o %t-dev.cir +// RUN: FileCheck --check-prefix=CIR-DEV --input-file=%t-dev.cir %s +// RUN: %clang_cc1 -triple spirv64-unknown-unknown \ +// RUN: -aux-triple x86_64-unknown-linux-gnu -fsycl-is-device -fgpu-rdc \ +// RUN: -fclangir -emit-llvm %s -o %t-dev-cir.ll +// RUN: FileCheck --check-prefix=LLVM-DEV --input-file=%t-dev-cir.ll %s +// RUN: %clang_cc1 -triple spirv64-unknown-unknown \ +// RUN: -aux-triple x86_64-unknown-linux-gnu -fsycl-is-device -fgpu-rdc \ +// RUN: -emit-llvm %s -o %t-dev.ll +// RUN: FileCheck --check-prefix=LLVM-DEV --input-file=%t-dev.ll %s + +template <typename KN, typename... Ts> +void sycl_kernel_launch(const char *, Ts...) {} + +template <typename KN, typename KT> +[[clang::sycl_kernel_entry_point(KN)]] void kernel_entry(KT k) { k(); } + +struct KN; + +static int hostStatic; + +template <typename T> static T getValue() { return T(1); } + +int main() { + hostStatic = 1; + kernel_entry<KN>([] { (void)getValue<int>(); }); +} + +// CIR: cir.global "private" internal dso_local @_ZL10hostStatic +// CIR: cir.func {{.*}}@main +// LLVM: @_ZL10hostStatic = internal global i32 0 +// LLVM: define {{.*}}@main + +// CIR-DEV: cir.func {{.*}}@_ZTS2KN +// CIR-DEV: cir.func {{.*}}@_ZL8getValueIiET_v +// LLVM-DEV: define {{.*}}spir_kernel void @_ZTS2KN +// LLVM-DEV: define internal spir_func {{.*}}@_ZL8getValueIiET_v diff --git a/clang/test/CIR/CodeGenSYCL/sycl-external.cpp b/clang/test/CIR/CodeGenSYCL/sycl-external.cpp new file mode 100644 index 0000000000000..d3676f06c2d23 --- /dev/null +++ b/clang/test/CIR/CodeGenSYCL/sycl-external.cpp @@ -0,0 +1,74 @@ +// RUN: %clang_cc1 -fsycl-is-device -triple spirv64-unknown-unknown -fclangir -emit-cir %s -o %t.cir +// RUN: FileCheck --input-file=%t.cir %s -check-prefix=CIR +// RUN: %clang_cc1 -fsycl-is-device -triple spirv64-unknown-unknown -fclangir -emit-llvm %s -o %t-cir.ll +// RUN: FileCheck --input-file=%t-cir.ll %s --check-prefixes=LLVM,LLVM-CIR +// RUN: %clang_cc1 -fsycl-is-device -triple spirv64-unknown-unknown -emit-llvm %s -o %t.ll +// RUN: FileCheck --input-file=%t.ll %s --check-prefixes=LLVM,OGCG + +// Verify that sycl_external functions are emitted in device code, matching +// classic CodeGen, and that their definitions carry the "sycl-module-id" +// attribute that marks them as entry points for device-code splitting. + +template <typename KernelName, typename... Ts> +void sycl_kernel_launch(const char *, Ts...) {} + +template <typename KernelName, typename KernelType> +[[clang::sycl_kernel_entry_point(KernelName)]] +void kernel_single_task(KernelType kf) { kf(); } + +struct KN; +void use() { kernel_single_task<KN>([] {}); } + +// Defined and not used: emitted. +[[clang::sycl_external]] int square(int x) { return x * x; } + +// Declared but not defined or used: not emitted. +[[clang::sycl_external]] int declOnly(); + +// Declared and used in device code but not defined: external reference. +[[clang::sycl_external]] void declUsedInDevice(int y); +[[clang::sycl_external]] void deviceUse() { declUsedInDevice(3); } + +// Declared with the attribute and later defined: definition emitted. +[[clang::sycl_external]] int func1(int arg); +int func1(int arg) { return arg; } + +// Reachable from a sycl_external function: emitted, but not an entry point. +int ret1() { return 1; } +[[clang::sycl_external]] int withAttr() { return ret1(); } + +// Explicit specialization defined: emitted. +template <typename T> [[clang::sycl_external]] void tFunc(T arg) {} +template <> [[clang::sycl_external]] void tFunc<int>(int arg) {} + +// Defined without the attribute and not used in device code: not emitted. +int squareNoAttr(int x) { return x * x; } + +// CIR: cir.func {{.*}}@_Z6squarei({{.*}} attributes {{{.*}}"sycl-module-id" = "{{.*}}sycl-external.cpp"} +// CIR: cir.func {{.*}}@_Z9deviceUsev() {{.*}} attributes {{{.*}}"sycl-module-id" = "{{.*}}sycl-external.cpp"} +// CIR: cir.func private @_Z16declUsedInDevicei({{.*}} attributes {convergent{{[^"]*}}} loc +// CIR: cir.func {{.*}}@_Z5func1i({{.*}} attributes {{{.*}}"sycl-module-id" = "{{.*}}sycl-external.cpp"} +// CIR: cir.func {{.*}}@_Z8withAttrv() {{.*}} attributes {{{.*}}"sycl-module-id" = "{{.*}}sycl-external.cpp"} +// CIR: cir.func {{.*}}@_Z4ret1v() {{.*}} attributes {convergent{{[^"]*}}} { +// CIR: cir.func {{.*}}@_Z5tFuncIiEvT_({{.*}} attributes {{{.*}}"sycl-module-id" = "{{.*}}sycl-external.cpp"} +// CIR: cir.func {{.*}}@_ZTS2KN({{.*}} attributes {{{.*}}"sycl-module-id" = "{{.*}}sycl-external.cpp"} +// CIR-NOT: @_Z8declOnlyv +// CIR-NOT: @_Z12squareNoAttri + +// LLVM: define spir_func noundef i32 @_Z6squarei(i32 noundef %{{.*}}) #[[EXT:[0-9]+]] +// LLVM: define spir_func void @_Z9deviceUsev() #[[EXT]] +// LLVM: declare spir_func void @_Z16declUsedInDevicei(i32 noundef) #[[DECL:[0-9]+]] +// LLVM: define spir_func noundef i32 @_Z5func1i(i32 noundef %{{.*}}) #[[EXT]] +// LLVM: define spir_func noundef i32 @_Z8withAttrv() #[[EXT]] +// LLVM: define spir_func noundef i32 @_Z4ret1v() #[[NOEXT:[0-9]+]] +// LLVM: define spir_func void @_Z5tFuncIiEvT_(i32 noundef %{{.*}}) #[[EXT]] +// ClangIR does not yet pass kernel arguments byval or emit the attributes that +// would let the kernel share the attribute group of the other entry points. +// LLVM-CIR: define spir_kernel void @_ZTS2KN({{.*}}) #[[KERNEL:[0-9]+]] +// OGCG: define spir_kernel void @_ZTS2KN(ptr noundef byval({{.*}}) #[[EXT]] +// LLVM-NOT: @_Z8declOnlyv +// LLVM-NOT: @_Z12squareNoAttri +// LLVM-DAG: attributes #[[EXT]] = { {{.*}}"sycl-module-id"="{{.*}}sycl-external.cpp" } +// LLVM-DAG: attributes #[[DECL]] = { convergent{{.*}} } +// LLVM-DAG: attributes #[[NOEXT]] = { convergent {{.*}}noinline{{.*}} } +// LLVM-CIR-DAG: attributes #[[KERNEL]] = { {{.*}}"sycl-module-id"="{{.*}}sycl-external.cpp" } `````````` </details> https://github.com/llvm/llvm-project/pull/226596 _______________________________________________ cfe-commits mailing list [email protected] https://lists.llvm.org/cgi-bin/mailman/listinfo/cfe-commits
