diff options
| author | ruki <[email protected]> | 2019-06-11 14:21:01 +0800 |
|---|---|---|
| committer | GitHub <[email protected]> | 2019-06-11 14:21:01 +0800 |
| commit | ffb4f2c3cbad5fa1836923ee68d0f5322f3c92fa (patch) | |
| tree | 1567a894cd6cde9d1e84ffc43f8498c61d8ee763 /tests/projects | |
| parent | 62a558c85e80081f78d340410c4ba7185d42ca89 (diff) | |
| parent | fddedb324fbff0546c2cbe304fe95c8e254da6fb (diff) | |
Merge pull request #450 from OpportunityLiu/cuda-tests
Cuda tests & templates
Diffstat (limited to 'tests/projects')
| -rw-r--r-- | tests/projects/cuda/shared/inc/lib.cuh | 23 | ||||
| -rw-r--r-- | tests/projects/cuda/shared/src/lib.cu | 97 | ||||
| -rw-r--r-- | tests/projects/cuda/shared/src/main.cu | 34 | ||||
| -rw-r--r-- | tests/projects/cuda/shared/xmake.lua | 22 | ||||
| -rw-r--r-- | tests/projects/cuda/static/inc/lib.cuh | 3 | ||||
| -rw-r--r-- | tests/projects/cuda/static/src/lib.cu | 7 | ||||
| -rw-r--r-- | tests/projects/cuda/static/src/main.cu | 127 | ||||
| -rw-r--r-- | tests/projects/cuda/static/xmake.lua | 24 |
8 files changed, 337 insertions, 0 deletions
diff --git a/tests/projects/cuda/shared/inc/lib.cuh b/tests/projects/cuda/shared/inc/lib.cuh new file mode 100644 index 000000000..84d5db3ca --- /dev/null +++ b/tests/projects/cuda/shared/inc/lib.cuh @@ -0,0 +1,23 @@ +#pragma once + +#include "cuda_runtime.h" +#include "device_launch_parameters.h" + +#ifdef __cplusplus +extern "C" +{ +#endif + +#if defined(_WIN32) +#define __export __declspec(dllexport) +#elif defined(__GNUC__) && ((__GNUC__ >= 4) || (__GNUC__ == 3 && __GNUC_MINOR__ >= 3)) +#define __export __attribute__((visibility("default"))) +#else +#define __export +#endif + + __export cudaError_t addWithCuda(int *c, const int *a, const int *b, unsigned int size); + +#ifdef __cplusplus +} +#endif diff --git a/tests/projects/cuda/shared/src/lib.cu b/tests/projects/cuda/shared/src/lib.cu new file mode 100644 index 000000000..cd33b167b --- /dev/null +++ b/tests/projects/cuda/shared/src/lib.cu @@ -0,0 +1,97 @@ +#include <lib.cuh> +#include <stdio.h> + +__global__ void addKernel(int *c, const int *a, const int *b) +{ + int i = threadIdx.x; + c[i] = a[i] + b[i]; +} + +// Helper function for using CUDA to add vectors in parallel. +cudaError_t addWithCuda(int *c, const int *a, const int *b, unsigned int size) +{ + int *dev_a = 0; + int *dev_b = 0; + int *dev_c = 0; + cudaError_t cudaStatus; + + // Choose which GPU to run on, change this on a multi-GPU system. + cudaStatus = cudaSetDevice(0); + if (cudaStatus != cudaSuccess) + { + fprintf(stderr, "cudaSetDevice failed! Do you have a CUDA-capable GPU installed?"); + goto Error; + } + + // Allocate GPU buffers for three vectors (two input, one output) . + cudaStatus = cudaMalloc((void **)&dev_c, size * sizeof(int)); + if (cudaStatus != cudaSuccess) + { + fprintf(stderr, "cudaMalloc failed!"); + goto Error; + } + + cudaStatus = cudaMalloc((void **)&dev_a, size * sizeof(int)); + if (cudaStatus != cudaSuccess) + { + fprintf(stderr, "cudaMalloc failed!"); + goto Error; + } + + cudaStatus = cudaMalloc((void **)&dev_b, size * sizeof(int)); + if (cudaStatus != cudaSuccess) + { + fprintf(stderr, "cudaMalloc failed!"); + goto Error; + } + + // Copy input vectors from host memory to GPU buffers. + cudaStatus = cudaMemcpy(dev_a, a, size * sizeof(int), cudaMemcpyHostToDevice); + if (cudaStatus != cudaSuccess) + { + fprintf(stderr, "cudaMemcpy failed!"); + goto Error; + } + + cudaStatus = cudaMemcpy(dev_b, b, size * sizeof(int), cudaMemcpyHostToDevice); + if (cudaStatus != cudaSuccess) + { + fprintf(stderr, "cudaMemcpy failed!"); + goto Error; + } + + // Launch a kernel on the GPU with one thread for each element. + addKernel<<<1, size>>>(dev_c, dev_a, dev_b); + + // Check for any errors launching the kernel + cudaStatus = cudaGetLastError(); + if (cudaStatus != cudaSuccess) + { + fprintf(stderr, "addKernel launch failed: %s\n", cudaGetErrorString(cudaStatus)); + goto Error; + } + + // cudaDeviceSynchronize waits for the kernel to finish, and returns + // any errors encountered during the launch. + cudaStatus = cudaDeviceSynchronize(); + if (cudaStatus != cudaSuccess) + { + fprintf(stderr, "cudaDeviceSynchronize returned error code %d after launching addKernel!\n", cudaStatus); + goto Error; + } + + // Copy output vector from GPU buffer to host memory. + cudaStatus = cudaMemcpy(c, dev_c, size * sizeof(int), cudaMemcpyDeviceToHost); + if (cudaStatus != cudaSuccess) + { + fprintf(stderr, "cudaMemcpy failed!"); + goto Error; + } + +Error: + cudaFree(dev_c); + cudaFree(dev_a); + cudaFree(dev_b); + + return cudaStatus; +} diff --git a/tests/projects/cuda/shared/src/main.cu b/tests/projects/cuda/shared/src/main.cu new file mode 100644 index 000000000..a33ec5c3e --- /dev/null +++ b/tests/projects/cuda/shared/src/main.cu @@ -0,0 +1,34 @@ + +#include "cuda_runtime.h" +#include <stdio.h> +#include <lib.cuh> + +int main() +{ + const int arraySize = 5; + const int a[arraySize] = {1, 2, 3, 4, 5}; + const int b[arraySize] = {10, 20, 30, 40, 50}; + int c[arraySize] = {0}; + + // Add vectors in parallel. + cudaError_t cudaStatus = addWithCuda(c, a, b, arraySize); + if (cudaStatus != cudaSuccess) + { + fprintf(stderr, "addWithCuda failed!"); + return 1; + } + + printf("{1,2,3,4,5} + {10,20,30,40,50} = {%d,%d,%d,%d,%d}\n", + c[0], c[1], c[2], c[3], c[4]); + + // cudaDeviceReset must be called before exiting in order for profiling and + // tracing tools such as Nsight and Visual Profiler to show complete traces. + cudaStatus = cudaDeviceReset(); + if (cudaStatus != cudaSuccess) + { + fprintf(stderr, "cudaDeviceReset failed!"); + return 1; + } + + return 0; +} diff --git a/tests/projects/cuda/shared/xmake.lua b/tests/projects/cuda/shared/xmake.lua new file mode 100644 index 000000000..e96999a13 --- /dev/null +++ b/tests/projects/cuda/shared/xmake.lua @@ -0,0 +1,22 @@ + +-- add modes: debug and release +add_rules("mode.debug", "mode.release") + +-- generate PTX code for the virtual architecture to guarantee compatibility +add_cugencodes("compute_30") + +-- define target +target("lib") + + -- set kind + set_kind("shared") + + -- add files + add_files("src/lib.cu") + + add_includedirs("inc", {public = true}) + +target("bin") + add_deps("lib") + set_kind("binary") + add_files("src/main.cu") diff --git a/tests/projects/cuda/static/inc/lib.cuh b/tests/projects/cuda/static/inc/lib.cuh new file mode 100644 index 000000000..35255e31e --- /dev/null +++ b/tests/projects/cuda/static/inc/lib.cuh @@ -0,0 +1,3 @@ +#pragma once + +__global__ void addKernel(int *c, const int *a, const int *b);
\ No newline at end of file diff --git a/tests/projects/cuda/static/src/lib.cu b/tests/projects/cuda/static/src/lib.cu new file mode 100644 index 000000000..a5f054354 --- /dev/null +++ b/tests/projects/cuda/static/src/lib.cu @@ -0,0 +1,7 @@ +#include <lib.cuh> + +__global__ void addKernel(int *c, const int *a, const int *b) +{ + int i = threadIdx.x; + c[i] = a[i] + b[i]; +}
\ No newline at end of file diff --git a/tests/projects/cuda/static/src/main.cu b/tests/projects/cuda/static/src/main.cu new file mode 100644 index 000000000..32844cd56 --- /dev/null +++ b/tests/projects/cuda/static/src/main.cu @@ -0,0 +1,127 @@ + +#include "cuda_runtime.h" +#include "device_launch_parameters.h" + +#include <stdio.h> +#include <lib.cuh> + +cudaError_t addWithCuda(int *c, const int *a, const int *b, unsigned int size); + +int main() +{ + const int arraySize = 5; + const int a[arraySize] = {1, 2, 3, 4, 5}; + const int b[arraySize] = {10, 20, 30, 40, 50}; + int c[arraySize] = {0}; + + // Add vectors in parallel. + cudaError_t cudaStatus = addWithCuda(c, a, b, arraySize); + if (cudaStatus != cudaSuccess) + { + fprintf(stderr, "addWithCuda failed!"); + return 1; + } + + printf("{1,2,3,4,5} + {10,20,30,40,50} = {%d,%d,%d,%d,%d}\n", + c[0], c[1], c[2], c[3], c[4]); + + // cudaDeviceReset must be called before exiting in order for profiling and + // tracing tools such as Nsight and Visual Profiler to show complete traces. + cudaStatus = cudaDeviceReset(); + if (cudaStatus != cudaSuccess) + { + fprintf(stderr, "cudaDeviceReset failed!"); + return 1; + } + + return 0; +} + +// Helper function for using CUDA to add vectors in parallel. +cudaError_t addWithCuda(int *c, const int *a, const int *b, unsigned int size) +{ + int *dev_a = 0; + int *dev_b = 0; + int *dev_c = 0; + cudaError_t cudaStatus; + + // Choose which GPU to run on, change this on a multi-GPU system. + cudaStatus = cudaSetDevice(0); + if (cudaStatus != cudaSuccess) + { + fprintf(stderr, "cudaSetDevice failed! Do you have a CUDA-capable GPU installed?"); + goto Error; + } + + // Allocate GPU buffers for three vectors (two input, one output) . + cudaStatus = cudaMalloc((void **)&dev_c, size * sizeof(int)); + if (cudaStatus != cudaSuccess) + { + fprintf(stderr, "cudaMalloc failed!"); + goto Error; + } + + cudaStatus = cudaMalloc((void **)&dev_a, size * sizeof(int)); + if (cudaStatus != cudaSuccess) + { + fprintf(stderr, "cudaMalloc failed!"); + goto Error; + } + + cudaStatus = cudaMalloc((void **)&dev_b, size * sizeof(int)); + if (cudaStatus != cudaSuccess) + { + fprintf(stderr, "cudaMalloc failed!"); + goto Error; + } + + // Copy input vectors from host memory to GPU buffers. + cudaStatus = cudaMemcpy(dev_a, a, size * sizeof(int), cudaMemcpyHostToDevice); + if (cudaStatus != cudaSuccess) + { + fprintf(stderr, "cudaMemcpy failed!"); + goto Error; + } + + cudaStatus = cudaMemcpy(dev_b, b, size * sizeof(int), cudaMemcpyHostToDevice); + if (cudaStatus != cudaSuccess) + { + fprintf(stderr, "cudaMemcpy failed!"); + goto Error; + } + + // Launch a kernel on the GPU with one thread for each element. + addKernel<<<1, size>>>(dev_c, dev_a, dev_b); + + // Check for any errors launching the kernel + cudaStatus = cudaGetLastError(); + if (cudaStatus != cudaSuccess) + { + fprintf(stderr, "addKernel launch failed: %s\n", cudaGetErrorString(cudaStatus)); + goto Error; + } + + // cudaDeviceSynchronize waits for the kernel to finish, and returns + // any errors encountered during the launch. + cudaStatus = cudaDeviceSynchronize(); + if (cudaStatus != cudaSuccess) + { + fprintf(stderr, "cudaDeviceSynchronize returned error code %d after launching addKernel!\n", cudaStatus); + goto Error; + } + + // Copy output vector from GPU buffer to host memory. + cudaStatus = cudaMemcpy(c, dev_c, size * sizeof(int), cudaMemcpyDeviceToHost); + if (cudaStatus != cudaSuccess) + { + fprintf(stderr, "cudaMemcpy failed!"); + goto Error; + } + +Error: + cudaFree(dev_c); + cudaFree(dev_a); + cudaFree(dev_b); + + return cudaStatus; +} diff --git a/tests/projects/cuda/static/xmake.lua b/tests/projects/cuda/static/xmake.lua new file mode 100644 index 000000000..55e1e5fd9 --- /dev/null +++ b/tests/projects/cuda/static/xmake.lua @@ -0,0 +1,24 @@ + +-- add modes: debug and release +add_rules("mode.debug", "mode.release") + +-- generate PTX code for the virtual architecture to guarantee compatibility +add_cugencodes("compute_30") + +-- define target +target("lib") + + -- set kind + set_kind("static") + + add_cuflags("-rdc=true") + + add_includedirs("inc", {public = true}) + + -- add files + add_files("src/lib.cu") + +target("bin") + add_deps("lib") + set_kind("binary") + add_files("src/main.cu") |
