diff options
| author | Dimitry Andric <dim@FreeBSD.org> | 2017-01-02 19:18:08 +0000 |
|---|---|---|
| committer | Dimitry Andric <dim@FreeBSD.org> | 2017-01-02 19:18:08 +0000 |
| commit | bab175ec4b075c8076ba14c762900392533f6ee4 (patch) | |
| tree | 01f4f29419a2cb10abe13c1e63cd2a66068b0137 /test/CodeGenCUDA | |
| parent | 8b7a8012d223fac5d17d16a66bb39168a9a1dfc0 (diff) | |
Notes
Diffstat (limited to 'test/CodeGenCUDA')
| -rw-r--r-- | test/CodeGenCUDA/convergent.cu | 4 | ||||
| -rw-r--r-- | test/CodeGenCUDA/cuda-builtin-vars.cu | 2 | ||||
| -rw-r--r-- | test/CodeGenCUDA/device-stub.cu | 6 | ||||
| -rw-r--r-- | test/CodeGenCUDA/device-var-init.cu | 6 | ||||
| -rw-r--r-- | test/CodeGenCUDA/function-overload.cu | 8 | ||||
| -rw-r--r-- | test/CodeGenCUDA/host-device-calls-host.cu | 32 | ||||
| -rw-r--r-- | test/CodeGenCUDA/kernel-args-alignment.cu | 36 | ||||
| -rw-r--r-- | test/CodeGenCUDA/launch-bounds.cu | 6 | ||||
| -rw-r--r-- | test/CodeGenCUDA/nothrow.cu | 39 |
9 files changed, 88 insertions, 51 deletions
diff --git a/test/CodeGenCUDA/convergent.cu b/test/CodeGenCUDA/convergent.cu index 6827c57d29fb..62818f9e5af4 100644 --- a/test/CodeGenCUDA/convergent.cu +++ b/test/CodeGenCUDA/convergent.cu @@ -36,8 +36,8 @@ __host__ __device__ void bar() { // DEVICE: attributes [[BAZ_ATTR]] = { // DEVICE-SAME: convergent // DEVICE-SAME: } -// DEVICE: attributes [[CALL_ATTR]] = { convergent } -// DEVICE: attributes [[ASM_ATTR]] = { convergent +// DEVICE-DAG: attributes [[CALL_ATTR]] = { convergent +// DEVICE-DAG: attributes [[ASM_ATTR]] = { convergent // HOST: declare void @_Z3bazv() [[BAZ_ATTR:#[0-9]+]] // HOST: attributes [[BAZ_ATTR]] = { diff --git a/test/CodeGenCUDA/cuda-builtin-vars.cu b/test/CodeGenCUDA/cuda-builtin-vars.cu index c2159f5af141..c1edff936a0d 100644 --- a/test/CodeGenCUDA/cuda-builtin-vars.cu +++ b/test/CodeGenCUDA/cuda-builtin-vars.cu @@ -1,6 +1,6 @@ // RUN: %clang_cc1 "-triple" "nvptx-nvidia-cuda" -emit-llvm -fcuda-is-device -o - %s | FileCheck %s -#include "cuda_builtin_vars.h" +#include "__clang_cuda_builtin_vars.h" // CHECK: define void @_Z6kernelPi(i32* %out) __attribute__((global)) diff --git a/test/CodeGenCUDA/device-stub.cu b/test/CodeGenCUDA/device-stub.cu index 5979ba3fce60..3376803c5020 100644 --- a/test/CodeGenCUDA/device-stub.cu +++ b/test/CodeGenCUDA/device-stub.cu @@ -45,10 +45,12 @@ void use_pointers() { // * constant unnamed string with the kernel name // CHECK: private unnamed_addr constant{{.*}}kernelfunc{{.*}}\00" // * constant unnamed string with GPU binary -// CHECK: private unnamed_addr constant{{.*}}\00" +// CHECK: private unnamed_addr constant{{.*GPU binary would be here.*}}\00" +// CHECK-SAME: section ".nv_fatbin", align 8 // * constant struct that wraps GPU binary // CHECK: @__cuda_fatbin_wrapper = internal constant { i32, i32, i8*, i8* } -// CHECK: { i32 1180844977, i32 1, {{.*}}, i8* null } +// CHECK-SAME: { i32 1180844977, i32 1, {{.*}}, i8* null } +// CHECK-SAME: section ".nvFatBinSegment" // * variable to save GPU binary handle after initialization // CHECK: @__cuda_gpubin_handle = internal global i8** null // * Make sure our constructor/destructor was added to global ctor/dtor list. diff --git a/test/CodeGenCUDA/device-var-init.cu b/test/CodeGenCUDA/device-var-init.cu index 6f2d9294131f..0f4c64813406 100644 --- a/test/CodeGenCUDA/device-var-init.cu +++ b/test/CodeGenCUDA/device-var-init.cu @@ -182,9 +182,9 @@ __device__ void df() { df(); // CHECK: call void @_Z2dfv() // Verify that we only call non-empty destructors - // CHECK-NEXT: call void @_ZN8T_FA_NEDD1Ev(%struct.T_FA_NED* %t_fa_ned) #6 - // CHECK-NEXT: call void @_ZN7T_F_NEDD1Ev(%struct.T_F_NED* %t_f_ned) #6 - // CHECK-NEXT: call void @_ZN7T_B_NEDD1Ev(%struct.T_B_NED* %t_b_ned) #6 + // CHECK-NEXT: call void @_ZN8T_FA_NEDD1Ev(%struct.T_FA_NED* %t_fa_ned) + // CHECK-NEXT: call void @_ZN7T_F_NEDD1Ev(%struct.T_F_NED* %t_f_ned) + // CHECK-NEXT: call void @_ZN7T_B_NEDD1Ev(%struct.T_B_NED* %t_b_ned) // CHECK-NEXT: call void @_ZN2VDD1Ev(%struct.VD* %vd) // CHECK-NEXT: call void @_ZN3NEDD1Ev(%struct.NED* %ned) // CHECK-NEXT: call void @_ZN2UDD1Ev(%struct.UD* %ud) diff --git a/test/CodeGenCUDA/function-overload.cu b/test/CodeGenCUDA/function-overload.cu index 380304af8222..c82b2e96f6c3 100644 --- a/test/CodeGenCUDA/function-overload.cu +++ b/test/CodeGenCUDA/function-overload.cu @@ -16,8 +16,6 @@ int x; struct s_cd_dh { __host__ s_cd_dh() { x = 11; } __device__ s_cd_dh() { x = 12; } - __host__ ~s_cd_dh() { x = 21; } - __device__ ~s_cd_dh() { x = 22; } }; struct s_cd_hd { @@ -38,7 +36,6 @@ void wrapper() { // CHECK-BOTH: call void @_ZN7s_cd_hdC1Ev // CHECK-BOTH: call void @_ZN7s_cd_hdD1Ev( - // CHECK-BOTH: call void @_ZN7s_cd_dhD1Ev( } // CHECK-BOTH: ret void @@ -56,8 +53,3 @@ void wrapper() { // CHECK-BOTH: define linkonce_odr void @_ZN7s_cd_hdD2Ev( // CHECK-BOTH: store i32 32, // CHECK-BOTH: ret void - -// CHECK-BOTH: define linkonce_odr void @_ZN7s_cd_dhD2Ev( -// CHECK-HOST: store i32 21, -// CHECK-DEVICE: store i32 22, -// CHECK-BOTH: ret void diff --git a/test/CodeGenCUDA/host-device-calls-host.cu b/test/CodeGenCUDA/host-device-calls-host.cu deleted file mode 100644 index 94796a3c233c..000000000000 --- a/test/CodeGenCUDA/host-device-calls-host.cu +++ /dev/null @@ -1,32 +0,0 @@ -// RUN: %clang_cc1 %s -triple nvptx-unknown-unknown -fcuda-is-device -Wno-cuda-compat -emit-llvm -o - | FileCheck %s - -#include "Inputs/cuda.h" - -extern "C" -void host_function() {} - -// CHECK-LABEL: define void @hd_function_a -extern "C" -__host__ __device__ void hd_function_a() { - // CHECK: call void @host_function - host_function(); -} - -// CHECK: declare void @host_function - -// CHECK-LABEL: define void @hd_function_b -extern "C" -__host__ __device__ void hd_function_b(bool b) { if (b) host_function(); } - -// CHECK-LABEL: define void @device_function_b -extern "C" -__device__ void device_function_b() { hd_function_b(false); } - -// CHECK-LABEL: define void @global_function -extern "C" -__global__ void global_function() { - // CHECK: call void @device_function_b - device_function_b(); -} - -// CHECK: !{{[0-9]+}} = !{void ()* @global_function, !"kernel", i32 1} diff --git a/test/CodeGenCUDA/kernel-args-alignment.cu b/test/CodeGenCUDA/kernel-args-alignment.cu new file mode 100644 index 000000000000..4bd5eb1bb1ff --- /dev/null +++ b/test/CodeGenCUDA/kernel-args-alignment.cu @@ -0,0 +1,36 @@ +// RUN: %clang_cc1 --std=c++11 -triple x86_64-unknown-linux-gnu -emit-llvm -o - %s | \ +// RUN: FileCheck -check-prefix HOST -check-prefix CHECK %s + +// RUN: %clang_cc1 --std=c++11 -fcuda-is-device -triple nvptx64-nvidia-cuda \ +// RUN: -emit-llvm -o - %s | FileCheck -check-prefix DEVICE -check-prefix CHECK %s + +#include "Inputs/cuda.h" + +struct U { + short x; +} __attribute__((packed)); + +struct S { + int *ptr; + char a; + U u; +}; + +// Clang should generate a packed LLVM struct for S (denoted by the <>s), +// otherwise this test isn't interesting. +// CHECK: %struct.S = type <{ i32*, i8, %struct.U, [5 x i8] }> + +static_assert(alignof(S) == 8, "Unexpected alignment."); + +// HOST-LABEL: @_Z6kernelc1SPi +// Marshalled kernel args should be: +// 1. offset 0, width 1 +// 2. offset 8 (because alignof(S) == 8), width 16 +// 3. offset 24, width 8 +// HOST: call i32 @cudaSetupArgument({{[^,]*}}, i64 1, i64 0) +// HOST: call i32 @cudaSetupArgument({{[^,]*}}, i64 16, i64 8) +// HOST: call i32 @cudaSetupArgument({{[^,]*}}, i64 8, i64 24) + +// DEVICE-LABEL: @_Z6kernelc1SPi +// DEVICE-SAME: i8{{[^,]*}}, %struct.S* byval align 8{{[^,]*}}, i32* +__global__ void kernel(char a, S s, int *b) {} diff --git a/test/CodeGenCUDA/launch-bounds.cu b/test/CodeGenCUDA/launch-bounds.cu index 6c369c6f3f0d..dda647ef3617 100644 --- a/test/CodeGenCUDA/launch-bounds.cu +++ b/test/CodeGenCUDA/launch-bounds.cu @@ -36,7 +36,7 @@ Kernel3() { } -template void Kernel3<MAX_THREADS_PER_BLOCK>(); +template __global__ void Kernel3<MAX_THREADS_PER_BLOCK>(); // CHECK: !{{[0-9]+}} = !{void ()* @{{.*}}Kernel3{{.*}}, !"maxntidx", i32 256} template <int max_threads_per_block, int min_blocks_per_mp> @@ -45,7 +45,7 @@ __launch_bounds__(max_threads_per_block, min_blocks_per_mp) Kernel4() { } -template void Kernel4<MAX_THREADS_PER_BLOCK, MIN_BLOCKS_PER_MP>(); +template __global__ void Kernel4<MAX_THREADS_PER_BLOCK, MIN_BLOCKS_PER_MP>(); // CHECK: !{{[0-9]+}} = !{void ()* @{{.*}}Kernel4{{.*}}, !"maxntidx", i32 256} // CHECK: !{{[0-9]+}} = !{void ()* @{{.*}}Kernel4{{.*}}, !"minctasm", i32 2} @@ -58,7 +58,7 @@ __launch_bounds__(max_threads_per_block + constint, Kernel5() { } -template void Kernel5<MAX_THREADS_PER_BLOCK, MIN_BLOCKS_PER_MP>(); +template __global__ void Kernel5<MAX_THREADS_PER_BLOCK, MIN_BLOCKS_PER_MP>(); // CHECK: !{{[0-9]+}} = !{void ()* @{{.*}}Kernel5{{.*}}, !"maxntidx", i32 356} // CHECK: !{{[0-9]+}} = !{void ()* @{{.*}}Kernel5{{.*}}, !"minctasm", i32 258} diff --git a/test/CodeGenCUDA/nothrow.cu b/test/CodeGenCUDA/nothrow.cu new file mode 100644 index 000000000000..f001b57981fa --- /dev/null +++ b/test/CodeGenCUDA/nothrow.cu @@ -0,0 +1,39 @@ +// RUN: %clang_cc1 -std=c++11 -fcxx-exceptions -fexceptions -fcuda-is-device \ +// RUN: -triple nvptx-nvidia-cuda -emit-llvm -disable-llvm-passes -o - %s | \ +// RUN: FileCheck -check-prefix DEVICE %s + +// RUN: %clang_cc1 -std=c++11 -fcxx-exceptions -fexceptions \ +// RUN: -triple x86_64-unknown-linux-gnu -emit-llvm -disable-llvm-passes -o - %s | \ +// RUN: FileCheck -check-prefix HOST %s + +#include "Inputs/cuda.h" + +__host__ __device__ void f(); + +// HOST: define void @_Z7host_fnv() [[HOST_ATTR:#[0-9]+]] +void host_fn() { f(); } + +// DEVICE: define void @_Z3foov() [[DEVICE_ATTR:#[0-9]+]] +__device__ void foo() { + // DEVICE: call void @_Z1fv + f(); +} + +// DEVICE: define void @_Z12foo_noexceptv() [[DEVICE_ATTR:#[0-9]+]] +__device__ void foo_noexcept() noexcept { + // DEVICE: call void @_Z1fv + f(); +} + +// This is nounwind only on the device side. +// CHECK: define void @_Z3foov() [[DEVICE_ATTR:#[0-9]+]] +__host__ __device__ void bar() { f(); } + +// DEVICE: define void @_Z3bazv() [[DEVICE_ATTR:#[0-9]+]] +__global__ void baz() { f(); } + +// DEVICE: attributes [[DEVICE_ATTR]] = { +// DEVICE-SAME: nounwind +// HOST: attributes [[HOST_ATTR]] = { +// HOST-NOT: nounwind +// HOST-SAME: } |
