llvmorg-github-actions[bot] wrote:

<!--LLVM PR SUMMARY COMMENT-->

@llvm/pr-subscribers-clang

Author: Konstantinos Parasyris (koparasy)

<details>
<summary>Changes</summary>

Emit `sycl_external` functions with sycl-module-id, allow -fgpu-rdc mangling, 
and embed offload objects in the host.

---
Full diff: https://github.com/llvm/llvm-project/pull/226596.diff


5 Files Affected:

- (modified) clang/lib/CIR/CodeGen/CIRGenModule.cpp (+12-4) 
- (modified) clang/lib/CIR/FrontendAction/CIRGenAction.cpp (+18) 
- (added) clang/test/CIR/CodeGenSYCL/embed-offload-object.cpp (+74) 
- (added) clang/test/CIR/CodeGenSYCL/gpu-rdc-mangled-name.cpp (+53) 
- (added) clang/test/CIR/CodeGenSYCL/sycl-external.cpp (+74) 


``````````diff
diff --git a/clang/lib/CIR/CodeGen/CIRGenModule.cpp 
b/clang/lib/CIR/CodeGen/CIRGenModule.cpp
index adffa7dfe2969..b6baf082deb9a 100644
--- a/clang/lib/CIR/CodeGen/CIRGenModule.cpp
+++ b/clang/lib/CIR/CodeGen/CIRGenModule.cpp
@@ -2836,7 +2836,10 @@ static std::string getMangledNameImpl(CIRGenModule &cgm, 
GlobalDecl gd,
                    "getMangledName: multi-version functions");
     }
   }
-  if (cgm.getLangOpts().GPURelocatableDeviceCode) {
+  // SYCL does not externalize file-scope statics, so RDC does not change the
+  // mangled name.
+  if (cgm.getLangOpts().GPURelocatableDeviceCode &&
+      !cgm.getLangOpts().isSYCL()) {
     cgm.errorNYI(nd->getSourceRange(),
                  "getMangledName: GPU relocatable device code");
   }
@@ -3003,10 +3006,10 @@ bool CIRGenModule::mayBeEmittedEagerly(const ValueDecl 
*global) {
     // Defer until all versions have been semantically checked.
     if (fd->hasAttr<TargetVersionAttr>() && !fd->isMultiVersion())
       return false;
-    if (langOpts.SYCLIsDevice) {
-      errorNYI(fd->getSourceRange(), "mayBeEmittedEagerly: SYCL");
+    // Defer emission of SYCL kernel entry point functions during device
+    // compilation.
+    if (langOpts.SYCLIsDevice && fd->hasAttr<SYCLKernelEntryPointAttr>())
       return false;
-    }
   }
   const auto *vd = dyn_cast<VarDecl>(global);
   if (vd)
@@ -3445,6 +3448,11 @@ void CIRGenModule::setCIRFunctionAttributesForDefinition(
     if (isa<CXXMethodDecl>(decl) && f.getAlignment().value_or(1) < 2)
       f.setAlignment(2);
   }
+
+  // Attach "sycl-module-id" to sycl_external function definitions to mark
+  // them as entry points for per-translation-unit device-code splitting.
+  if (getLangOpts().SYCLIsDevice && decl->hasAttr<SYCLExternalAttr>())
+    addSYCLModuleIdAttr(f);
 }
 
 // Maps an AST address space to the OpenCL logical address space kind recorded
diff --git a/clang/lib/CIR/FrontendAction/CIRGenAction.cpp 
b/clang/lib/CIR/FrontendAction/CIRGenAction.cpp
index 240601f9834e5..ad68a6526e4d1 100644
--- a/clang/lib/CIR/FrontendAction/CIRGenAction.cpp
+++ b/clang/lib/CIR/FrontendAction/CIRGenAction.cpp
@@ -215,6 +215,24 @@ class CIRGenConsumer : public clang::ASTConsumer {
       if (C.getLangOpts().SYCLIsHost && !CGO.OffloadBinaryToEmbedFile.empty())
         embedSYCLDeviceBinary(*LLVMModule);
 
+      // CUDA, HIP and OpenMP offloading rely on host-side offload entries that
+      // are not emitted on the ClangIR path yet, so embedding their device
+      // objects would produce a host object that cannot be registered.
+      const LangOptions &LangOpts = C.getLangOpts();
+      if (!CGO.OffloadObjects.empty() &&
+          (LangOpts.CUDA || !LangOpts.OMPTargetTriples.empty())) {
+        DiagnosticsEngine &Diags = CI.getDiagnostics();
+        Diags.Report(Diags.getCustomDiagID(
+            DiagnosticsEngine::Error,
+            "ClangIR code gen Not Yet Implemented: embedding offload objects "
+            "for CUDA, HIP or OpenMP offloading"));
+        return;
+      }
+
+      // If there is device offloading code embed it in the host now.
+      EmbedObject(LLVMModule.get(), CGO, CI.getVirtualFileSystem(),
+                  CI.getDiagnostics());
+
       BackendAction BEAction = getBackendActionFromOutputType(Action);
       emitBackendOutput(CI, CI.getCodeGenOpts(), LLVMModule.get(), BEAction, 
FS,
                         std::move(OutputStream));
diff --git a/clang/test/CIR/CodeGenSYCL/embed-offload-object.cpp 
b/clang/test/CIR/CodeGenSYCL/embed-offload-object.cpp
new file mode 100644
index 0000000000000..d2b3968e15e39
--- /dev/null
+++ b/clang/test/CIR/CodeGenSYCL/embed-offload-object.cpp
@@ -0,0 +1,74 @@
+// REQUIRES: x86-registered-target
+
+// Verify that on the ClangIR path -fembed-offload-object embeds the packaged
+// device object into the ".llvm.offloading" section of the host module,
+// matching classic CodeGen. This is the default relocatable device code flow
+// for SYCL.
+// RUN: echo -n 'FAKE_OFFLOAD_OBJECT' > %t.out
+// RUN: %clang_cc1 -triple x86_64-unknown-linux-gnu -fsycl-is-host -fgpu-rdc \
+// RUN:   -fclangir -fembed-offload-object=%t.out -emit-llvm %s -o - \
+// RUN:   | FileCheck %s
+// RUN: %clang_cc1 -triple x86_64-unknown-linux-gnu -fsycl-is-host -fgpu-rdc \
+// RUN:   -fembed-offload-object=%t.out -emit-llvm %s -o - \
+// RUN:   | FileCheck %s
+
+// Object emission must produce the ".llvm.offloading" section.
+// RUN: %clang_cc1 -triple x86_64-unknown-linux-gnu -fsycl-is-host -fgpu-rdc \
+// RUN:   -fclangir -fembed-offload-object=%t.out -emit-obj %s -o %t.o
+// RUN: llvm-readelf -S %t.o | FileCheck %s --check-prefix=OBJ
+
+// Without the flag nothing is embedded.
+// RUN: %clang_cc1 -triple x86_64-unknown-linux-gnu -fsycl-is-host -fgpu-rdc \
+// RUN:   -fclangir -emit-llvm %s -o - \
+// RUN:   | FileCheck %s --check-prefix=NONE 
--implicit-check-not='.llvm.offloading'
+
+// Embedding does not depend on an offloading language model, as in classic
+// CodeGen.
+// RUN: %clang_cc1 -triple x86_64-unknown-linux-gnu -x c++ -fclangir \
+// RUN:   -fembed-offload-object=%t.out -emit-llvm /dev/null -o - \
+// RUN:   | FileCheck %s
+
+// A missing object file must be diagnosed.
+// RUN: not %clang_cc1 -triple x86_64-unknown-linux-gnu -fsycl-is-host \
+// RUN:   -fgpu-rdc -fclangir -fembed-offload-object=%t.does-not-exist \
+// RUN:   -emit-llvm %s -o - 2>&1 | FileCheck %s --check-prefix=ERROR
+
+// Embedding for CUDA, HIP and OpenMP offloading is not implemented on the
+// ClangIR path yet, including when OpenMP offloading is combined with SYCL.
+// RUN: not %clang_cc1 -triple x86_64-unknown-linux-gnu -x cuda -fclangir \
+// RUN:   -fembed-offload-object=%t.out -emit-llvm /dev/null -o - 2>&1 \
+// RUN:   | FileCheck %s --check-prefix=NYI
+// RUN: not %clang_cc1 -triple x86_64-unknown-linux-gnu -x hip -fclangir \
+// RUN:   -fembed-offload-object=%t.out -emit-llvm /dev/null -o - 2>&1 \
+// RUN:   | FileCheck %s --check-prefix=NYI
+// RUN: not %clang_cc1 -triple x86_64-unknown-linux-gnu -x c -fopenmp \
+// RUN:   -fopenmp-targets=amdgcn-amd-amdhsa -fclangir \
+// RUN:   -fembed-offload-object=%t.out -emit-llvm /dev/null -o - 2>&1 \
+// RUN:   | FileCheck %s --check-prefix=NYI
+// RUN: not %clang_cc1 -triple x86_64-unknown-linux-gnu -x c++ -fsycl-is-host \
+// RUN:   -fopenmp -fopenmp-targets=amdgcn-amd-amdhsa -fclangir \
+// RUN:   -fembed-offload-object=%t.out -emit-llvm /dev/null -o - 2>&1 \
+// RUN:   | FileCheck %s --check-prefix=NYI
+
+template <typename KN, typename... Ts>
+void sycl_kernel_launch(const char *, Ts...) {}
+
+template <typename KN, typename KT>
+[[clang::sycl_kernel_entry_point(KN)]] void kernel_entry(KT k) { k(); }
+
+struct KN;
+
+int main() { kernel_entry<KN>([] {}); }
+
+// CHECK: @llvm.embedded.object = private constant [19 x i8] 
c"FAKE_OFFLOAD_OBJECT", section ".llvm.offloading", align 8, !exclude
+// CHECK: @llvm.compiler.used = appending global [1 x ptr] [ptr 
@llvm.embedded.object], section "llvm.metadata"
+// CHECK: !llvm.embedded.objects = !{![[EMB:[0-9]+]]}
+// CHECK: ![[EMB]] = !{ptr @llvm.embedded.object, !".llvm.offloading"}
+
+// OBJ: .llvm.offloading
+
+// NONE: define {{.*}}@main
+
+// ERROR: error: could not open '{{.*}}.does-not-exist' for embedding
+
+// NYI: error: ClangIR code gen Not Yet Implemented: embedding offload objects 
for CUDA, HIP or OpenMP offloading
diff --git a/clang/test/CIR/CodeGenSYCL/gpu-rdc-mangled-name.cpp 
b/clang/test/CIR/CodeGenSYCL/gpu-rdc-mangled-name.cpp
new file mode 100644
index 0000000000000..7e63eccea0f86
--- /dev/null
+++ b/clang/test/CIR/CodeGenSYCL/gpu-rdc-mangled-name.cpp
@@ -0,0 +1,53 @@
+// SYCL enables relocatable device code by default. SYCL does not externalize
+// file-scope statics, so -fgpu-rdc must not change mangled names on either the
+// host or the device side.
+
+// RUN: %clang_cc1 -triple x86_64-unknown-linux-gnu -fsycl-is-host -fgpu-rdc \
+// RUN:   -fclangir -emit-cir %s -o %t.cir
+// RUN: FileCheck --check-prefix=CIR --input-file=%t.cir %s
+// RUN: %clang_cc1 -triple x86_64-unknown-linux-gnu -fsycl-is-host -fgpu-rdc \
+// RUN:   -fclangir -emit-llvm %s -o %t-cir.ll
+// RUN: FileCheck --check-prefix=LLVM --input-file=%t-cir.ll %s
+// RUN: %clang_cc1 -triple x86_64-unknown-linux-gnu -fsycl-is-host -fgpu-rdc \
+// RUN:   -emit-llvm %s -o %t.ll
+// RUN: FileCheck --check-prefix=LLVM --input-file=%t.ll %s
+
+// RUN: %clang_cc1 -triple spirv64-unknown-unknown \
+// RUN:   -aux-triple x86_64-unknown-linux-gnu -fsycl-is-device -fgpu-rdc \
+// RUN:   -fclangir -emit-cir %s -o %t-dev.cir
+// RUN: FileCheck --check-prefix=CIR-DEV --input-file=%t-dev.cir %s
+// RUN: %clang_cc1 -triple spirv64-unknown-unknown \
+// RUN:   -aux-triple x86_64-unknown-linux-gnu -fsycl-is-device -fgpu-rdc \
+// RUN:   -fclangir -emit-llvm %s -o %t-dev-cir.ll
+// RUN: FileCheck --check-prefix=LLVM-DEV --input-file=%t-dev-cir.ll %s
+// RUN: %clang_cc1 -triple spirv64-unknown-unknown \
+// RUN:   -aux-triple x86_64-unknown-linux-gnu -fsycl-is-device -fgpu-rdc \
+// RUN:   -emit-llvm %s -o %t-dev.ll
+// RUN: FileCheck --check-prefix=LLVM-DEV --input-file=%t-dev.ll %s
+
+template <typename KN, typename... Ts>
+void sycl_kernel_launch(const char *, Ts...) {}
+
+template <typename KN, typename KT>
+[[clang::sycl_kernel_entry_point(KN)]] void kernel_entry(KT k) { k(); }
+
+struct KN;
+
+static int hostStatic;
+
+template <typename T> static T getValue() { return T(1); }
+
+int main() {
+  hostStatic = 1;
+  kernel_entry<KN>([] { (void)getValue<int>(); });
+}
+
+// CIR: cir.global "private" internal dso_local @_ZL10hostStatic
+// CIR: cir.func {{.*}}@main
+// LLVM: @_ZL10hostStatic = internal global i32 0
+// LLVM: define {{.*}}@main
+
+// CIR-DEV: cir.func {{.*}}@_ZTS2KN
+// CIR-DEV: cir.func {{.*}}@_ZL8getValueIiET_v
+// LLVM-DEV: define {{.*}}spir_kernel void @_ZTS2KN
+// LLVM-DEV: define internal spir_func {{.*}}@_ZL8getValueIiET_v
diff --git a/clang/test/CIR/CodeGenSYCL/sycl-external.cpp 
b/clang/test/CIR/CodeGenSYCL/sycl-external.cpp
new file mode 100644
index 0000000000000..d3676f06c2d23
--- /dev/null
+++ b/clang/test/CIR/CodeGenSYCL/sycl-external.cpp
@@ -0,0 +1,74 @@
+// RUN: %clang_cc1 -fsycl-is-device -triple spirv64-unknown-unknown -fclangir 
-emit-cir %s -o %t.cir
+// RUN: FileCheck --input-file=%t.cir %s -check-prefix=CIR
+// RUN: %clang_cc1 -fsycl-is-device -triple spirv64-unknown-unknown -fclangir 
-emit-llvm %s -o %t-cir.ll
+// RUN: FileCheck --input-file=%t-cir.ll %s --check-prefixes=LLVM,LLVM-CIR
+// RUN: %clang_cc1 -fsycl-is-device -triple spirv64-unknown-unknown -emit-llvm 
%s -o %t.ll
+// RUN: FileCheck --input-file=%t.ll %s --check-prefixes=LLVM,OGCG
+
+// Verify that sycl_external functions are emitted in device code, matching
+// classic CodeGen, and that their definitions carry the "sycl-module-id"
+// attribute that marks them as entry points for device-code splitting.
+
+template <typename KernelName, typename... Ts>
+void sycl_kernel_launch(const char *, Ts...) {}
+
+template <typename KernelName, typename KernelType>
+[[clang::sycl_kernel_entry_point(KernelName)]]
+void kernel_single_task(KernelType kf) { kf(); }
+
+struct KN;
+void use() { kernel_single_task<KN>([] {}); }
+
+// Defined and not used: emitted.
+[[clang::sycl_external]] int square(int x) { return x * x; }
+
+// Declared but not defined or used: not emitted.
+[[clang::sycl_external]] int declOnly();
+
+// Declared and used in device code but not defined: external reference.
+[[clang::sycl_external]] void declUsedInDevice(int y);
+[[clang::sycl_external]] void deviceUse() { declUsedInDevice(3); }
+
+// Declared with the attribute and later defined: definition emitted.
+[[clang::sycl_external]] int func1(int arg);
+int func1(int arg) { return arg; }
+
+// Reachable from a sycl_external function: emitted, but not an entry point.
+int ret1() { return 1; }
+[[clang::sycl_external]] int withAttr() { return ret1(); }
+
+// Explicit specialization defined: emitted.
+template <typename T> [[clang::sycl_external]] void tFunc(T arg) {}
+template <> [[clang::sycl_external]] void tFunc<int>(int arg) {}
+
+// Defined without the attribute and not used in device code: not emitted.
+int squareNoAttr(int x) { return x * x; }
+
+// CIR: cir.func {{.*}}@_Z6squarei({{.*}} attributes {{{.*}}"sycl-module-id" = 
"{{.*}}sycl-external.cpp"}
+// CIR: cir.func {{.*}}@_Z9deviceUsev() {{.*}} attributes 
{{{.*}}"sycl-module-id" = "{{.*}}sycl-external.cpp"}
+// CIR: cir.func private @_Z16declUsedInDevicei({{.*}} attributes 
{convergent{{[^"]*}}} loc
+// CIR: cir.func {{.*}}@_Z5func1i({{.*}} attributes {{{.*}}"sycl-module-id" = 
"{{.*}}sycl-external.cpp"}
+// CIR: cir.func {{.*}}@_Z8withAttrv() {{.*}} attributes 
{{{.*}}"sycl-module-id" = "{{.*}}sycl-external.cpp"}
+// CIR: cir.func {{.*}}@_Z4ret1v() {{.*}} attributes {convergent{{[^"]*}}} {
+// CIR: cir.func {{.*}}@_Z5tFuncIiEvT_({{.*}} attributes 
{{{.*}}"sycl-module-id" = "{{.*}}sycl-external.cpp"}
+// CIR: cir.func {{.*}}@_ZTS2KN({{.*}} attributes {{{.*}}"sycl-module-id" = 
"{{.*}}sycl-external.cpp"}
+// CIR-NOT: @_Z8declOnlyv
+// CIR-NOT: @_Z12squareNoAttri
+
+// LLVM: define spir_func noundef i32 @_Z6squarei(i32 noundef %{{.*}}) 
#[[EXT:[0-9]+]]
+// LLVM: define spir_func void @_Z9deviceUsev() #[[EXT]]
+// LLVM: declare spir_func void @_Z16declUsedInDevicei(i32 noundef) 
#[[DECL:[0-9]+]]
+// LLVM: define spir_func noundef i32 @_Z5func1i(i32 noundef %{{.*}}) #[[EXT]]
+// LLVM: define spir_func noundef i32 @_Z8withAttrv() #[[EXT]]
+// LLVM: define spir_func noundef i32 @_Z4ret1v() #[[NOEXT:[0-9]+]]
+// LLVM: define spir_func void @_Z5tFuncIiEvT_(i32 noundef %{{.*}}) #[[EXT]]
+// ClangIR does not yet pass kernel arguments byval or emit the attributes that
+// would let the kernel share the attribute group of the other entry points.
+// LLVM-CIR: define spir_kernel void @_ZTS2KN({{.*}}) #[[KERNEL:[0-9]+]]
+// OGCG: define spir_kernel void @_ZTS2KN(ptr noundef byval({{.*}}) #[[EXT]]
+// LLVM-NOT: @_Z8declOnlyv
+// LLVM-NOT: @_Z12squareNoAttri
+// LLVM-DAG: attributes #[[EXT]] = { 
{{.*}}"sycl-module-id"="{{.*}}sycl-external.cpp" }
+// LLVM-DAG: attributes #[[DECL]] = { convergent{{.*}} }
+// LLVM-DAG: attributes #[[NOEXT]] = { convergent {{.*}}noinline{{.*}} }
+// LLVM-CIR-DAG: attributes #[[KERNEL]] = { 
{{.*}}"sycl-module-id"="{{.*}}sycl-external.cpp" }

``````````

</details>


https://github.com/llvm/llvm-project/pull/226596
_______________________________________________
cfe-commits mailing list
[email protected]
https://lists.llvm.org/cgi-bin/mailman/listinfo/cfe-commits

Reply via email to