/********************************************************************
* CUDAWin32App10.cu
* This is a example of the CUDA program.
*********************************************************************/
#include
#include
#include
/************************************************************************/
/* Init CUDA */
/************************************************************************/
bool InitCUDA(void)
{
int count = 0;
int i = 0;
cudaGetDeviceCount(&count);
if(count == 0) {
fprintf(stderr, "There is no device.\n");
return false;
}
for(i = 0; i <>= 1) {
break;
}
}
}
if(i == count) {
fprintf(stderr, "There is no device supporting CUDA 1.x.\n");
return false;
}
cudaSetDevice(i);
return true;
}
/************************************************************************/
/* Example */
/************************************************************************/
__global__ static void HelloCUDA(char* result, int num, clock_t* time,int foo)
{
int i = 0;
char p_HelloCUDA[] = "Hello CUDA!";
clock_t start = clock();
for(i = 0; i < time =" clock()" device_result =" 0;" time =" 0;" time_used =" 0;">>>(device_result, 11 , time,1);
cudaMemcpy(&host_result, device_result, sizeof(char) * 11, cudaMemcpyDeviceToHost);
cudaMemcpy(&time_used, time, sizeof(clock_t), cudaMemcpyDeviceToHost);
cudaFree(device_result);
cudaFree(time);
printf("%s,%d\n", host_result, time_used);
return 0;
}
Showing posts with label cuda. Show all posts
Showing posts with label cuda. Show all posts
Wednesday, September 10, 2008
Cuda Hello World: Entire Code Listing
Cuda Hello World: part 2
In Part 1 we setup a Visual Studio project to run our first cuda program. In this part we will look deeper into the template and explain what it is doing.
A Cuda program is structured as followed
1) Define a Kernel
2) Copy system memory to GPU memory
3) Execute the Kernel
4) Copy results from GPU memory back to system memory
5) Print our results, and cleanup
We'll look at each part.
Define a Kernel
A kernel is a function that is executed. For our example will follow the sample template from the Visual Studio plugin and create a HelloCuda method.
The __global__ declaration specifier indicates that the procedure is a kernel entry point. Our function takes in an:
array of characters
size of the array
pointer to clock_t structure
The kernel doesn't do much it copies into our character array "Hello CUDA!". As you can probably infer from the assignment of time we are planning on only calling this function one time.
Copy system memory to GPU memory
Now that our kernel is defined we need to prep our data structures such that we can execute the kernel. For this example we'll need to allocate a block of memory to hold the "Hello CUDA!" string, and a clock_t structure for the elapsed time.
To allocate memory on the GPU we use cudaMalloc.
The memory has now been set aside in the GPU and were ready to execute the kernel
Execute the Kernel
The following code will execute our defined kernel passing in the arguments we just created.
At this stage the GPU will execute our program and will have our results, we need to copy those results back to the system so we can use them.
Copy results from GPU memory back to system memory
To get our results off of the GPU we use cudaMemcpy and copy the data back into our own storage. To do that we must define a char*, and a clock_t then memcpy the results back.
Print Results and Clean up
That's it the GPU has ran our program created a string called 'Hello CUDA!' copied that into the char * result that we have a pointer to called device_result. It has also filled in clock_t struct in gpu memory which we have a pointer to called time.
We've copied the gpu memory into local variables called host_result, and time_used and now can use it like an C program. So lets print it out.
The last thing we need to do is free the memory on the device we use cudaFree.
That's it! If you run the program you should get output similar to the following:
Click here for the entire code listing
A Cuda program is structured as followed
1) Define a Kernel
2) Copy system memory to GPU memory
3) Execute the Kernel
4) Copy results from GPU memory back to system memory
5) Print our results, and cleanup
We'll look at each part.
Define a Kernel
A kernel is a function that is executed. For our example will follow the sample template from the Visual Studio plugin and create a HelloCuda method.
__global__ static void HelloCUDA(char* result, int num, clock_t* time)
{
int i = 0;
char p_HelloCUDA[] = "Hello CUDA!";
clock_t start = clock();
for(i = 0; i < num; i++) {
result[i] = p_HelloCUDA[i];
}
*time = clock() - start;
}
The __global__ declaration specifier indicates that the procedure is a kernel entry point. Our function takes in an:
array of characters
size of the array
pointer to clock_t structure
The kernel doesn't do much it copies into our character array "Hello CUDA!". As you can probably infer from the assignment of time we are planning on only calling this function one time.
Copy system memory to GPU memory
Now that our kernel is defined we need to prep our data structures such that we can execute the kernel. For this example we'll need to allocate a block of memory to hold the "Hello CUDA!" string, and a clock_t structure for the elapsed time.
To allocate memory on the GPU we use cudaMalloc.
char *device_result = 0;
clock_t *time = 0;
cudaMalloc((void**) &device_result, sizeof(char) * 11);
cudaMalloc((void**) &time, sizeof(clock_t));
The memory has now been set aside in the GPU and were ready to execute the kernel
Execute the Kernel
The following code will execute our defined kernel passing in the arguments we just created.
HelloCUDA<<<1, 1, 0&rt;&rt;&rt;(device_result, 11 , time,1);
At this stage the GPU will execute our program and will have our results, we need to copy those results back to the system so we can use them.
Copy results from GPU memory back to system memory
To get our results off of the GPU we use cudaMemcpy and copy the data back into our own storage. To do that we must define a char*, and a clock_t then memcpy the results back.
char host_result[12] ={0};
clock_t time_used = 0;
cudaMemcpy(&host_result, device_result, sizeof(char) * 11, cudaMemcpyDeviceToHost);
cudaMemcpy(&time_used, time, sizeof(clock_t), cudaMemcpyDeviceToHost);
Print Results and Clean up
That's it the GPU has ran our program created a string called 'Hello CUDA!' copied that into the char * result that we have a pointer to called device_result. It has also filled in clock_t struct in gpu memory which we have a pointer to called time.
We've copied the gpu memory into local variables called host_result, and time_used and now can use it like an C program. So lets print it out.
cudaMemcpy(&host_result, device_result, sizeof(char) * 11, cudaMemcpyDeviceToHost);
cudaMemcpy(&time_used, time, sizeof(clock_t), cudaMemcpyDeviceToHost);
The last thing we need to do is free the memory on the device we use cudaFree.
cudaFree(device_result);
cudaFree(time);
That's it! If you run the program you should get output similar to the following:
CUDA initialized.
Hello CUDA!,0
Press any key to continue . . .
Click here for the entire code listing
Thursday, July 17, 2008
Cuda - Sieve of Eratosthenes
Here is an example of how to use cuda to find all prime numbers within a range.
Its uses the Sieve of Eratosthenes which creates an array initialized to 0 ( indicates prime) then you iterate over the array starting at zero and for each multiple of zero you mark it 1. You then move to the next element and repeat. When your down iterating all the primes are still marked as 0.
As an example cuda program I wrote the following kernel
The index is figured out by the special keywords blockidx, blockDim and threadIdx. I then go over the multiples for that index and mark them as not prime.
The code to call this looks like this:
For simplicity I simply create a new thread for each multiple. This is not very efficient as I may be using a value that is already found to not be prime.
Its uses the Sieve of Eratosthenes which creates an array initialized to 0 ( indicates prime) then you iterate over the array starting at zero and for each multiple of zero you mark it 1. You then move to the next element and repeat. When your down iterating all the primes are still marked as 0.
As an example cuda program I wrote the following kernel
__global__ static void Sieve(int * sieve,int sieve_size)
{
int idx = blockIdx.x * blockDim.x + threadIdx.x;
if (idx > 1) {
for(int i=idx+idx;i < sieve_size;i+=idx)
sieve[i] = 1;
}
}
The index is figured out by the special keywords blockidx, blockDim and threadIdx. I then go over the multiples for that index and mark them as not prime.
The code to call this looks like this:
/************************************************************************/
/* Init CUDA */
/************************************************************************/
bool InitCUDA(void)
{
int count = 0;
int i = 0;
cudaGetDeviceCount(&count);
if(count == 0) {
fprintf(stderr, "There is no device.\n");
return false;
}
for(i = 0; i < count; i++) {
cudaDeviceProp prop;
if(cudaGetDeviceProperties(&prop, i) == cudaSuccess) {
if(prop.major >= 1) {
break;
}
}
}
if(i == count) {
fprintf(stderr, "There is no device supporting CUDA 1.x.\n");
return false;
}
cudaSetDevice(i);
return true;
}
int main(int argc, char** argv)
{
int *device_sieve;
int host_sieve[1000];
double sieve_size = sizeof(host_sieve)/sizeof(int);
if(!InitCUDA()) {
return 0;
}
cudaMalloc((void**) &device_sieve, sizeof(int) * sieve_size);
Sieve<<<1, sqrt(sieve_size), 0>>>(device_sieve, sieve_size);
cudaThreadSynchronize();
cudaMemcpy(&host_sieve, device_sieve, sizeof(int) * sieve_size, cudaMemcpyDeviceToHost);
cudaFree(device_sieve);
for(int i=2;i < sieve_size;++i)
if (host_sieve[i] == 0)
printf("%d is prime\n",i);
return 0;
}
For simplicity I simply create a new thread for each multiple. This is not very efficient as I may be using a value that is already found to not be prime.
Thursday, June 26, 2008
CUDA Hello World
CUDA is an API to Nvidia's GPU's. It allows programmers access to a parallel machine that will run faster than cpu's for particular kinds of projects. In this 3 part tutorial will first setup Visual Studio to work with Cuda. If you are not using Visual Studio you can skip to Part 2
Step 1 is to download a new template type for Visual Studio. This will setup the compiler and other settings and get you ready to write code. To do this download following wizard http://forums.nvidia.com/index.php?showtopic=65111
Step 2 is optional, but I like to do it. This will set up syntax highlighting for your .cu (cuda) files.
In Visual Studio go to Tools->Text Editor->File Extensions and add .cu extensions to the list of cpp files.
Step 3, download the cuda SDK and driver if you have a cuda enabled card. Simply go to nvidia's webside and download what you need: http://www.nvidia.com/object/cuda_get.html
That's it, now you ready to create your first project. Continue To Part 2
Step 1 is to download a new template type for Visual Studio. This will setup the compiler and other settings and get you ready to write code. To do this download following wizard http://forums.nvidia.com/index.php?showtopic=65111
Step 2 is optional, but I like to do it. This will set up syntax highlighting for your .cu (cuda) files.
In Visual Studio go to Tools->Text Editor->File Extensions and add .cu extensions to the list of cpp files.
Step 3, download the cuda SDK and driver if you have a cuda enabled card. Simply go to nvidia's webside and download what you need: http://www.nvidia.com/object/cuda_get.html
That's it, now you ready to create your first project. Continue To Part 2
Tuesday, June 24, 2008
Python and CUDA
CUDA is Nvidia's api for leveraging the power of the GPU for parallel processing. The cuda api is in C and can be daunting to use. The following how to shows how to use PyCuda to access this powerful API from your python code.
First install PyCuda. You can fetch the latest package from http://pypi.python.org/pypi/pycuda.
Before you can use Cuda you must initialize the device the same way as you would in your C program.
For a cuda program the basic methodolgy is to copy from system memory to devices memory, perform processing, then copy data back from the device to the system. PyCuda provides facilities to do this.
First let's create a numpy array of data that we wish to transfer:
We now have our data on the device, we need to instruct the GPU to execute our Kernel. A Kernel, when talking about CUDA, is the actual code that will be executed on the GPU. PyCuda requires that you write the kernel in C and pass it to the device.
For example here is a kernel that adds one to the value of each element.
Now tell the device to execute our kernel.
Lastly we copy the contents from the device back to system memory and print the results.
Other Resources:
First install PyCuda. You can fetch the latest package from http://pypi.python.org/pypi/pycuda.
Before you can use Cuda you must initialize the device the same way as you would in your C program.
import pycuda.driver as pycuda
pycuda.init()
assert cuda.Device.count() >= 1
cudaDev = cuda.Device(0)
cudaCTX = dev.make_context()
For a cuda program the basic methodolgy is to copy from system memory to devices memory, perform processing, then copy data back from the device to the system. PyCuda provides facilities to do this.
First let's create a numpy array of data that we wish to transfer:
import numpy
a = numpy.random.randn(4,4)
a = a.astype(numpy.float32)
a_gpu = cuda.mem_alloc(a.size * a.dtype.itemsize)
pycuda.memcpy_htod(a_gpu, a)
We now have our data on the device, we need to instruct the GPU to execute our Kernel. A Kernel, when talking about CUDA, is the actual code that will be executed on the GPU. PyCuda requires that you write the kernel in C and pass it to the device.
For example here is a kernel that adds one to the value of each element.
mod = cuda.SourceModule("""
__global__ void addOne(float *a)
{
int idx = threadIdx.x + threadIdx.y*4;
a[idx]+= 1;
}
""")
Now tell the device to execute our kernel.
func = mod.get_function("addOne")
func(a_gpu, block=(4,4,1))
Lastly we copy the contents from the device back to system memory and print the results.
a_addOne = numpy.empty_like(a)
pycuda.memcpy_dtoh(a_doubled, a_gpu)
print a_doubled
print a
Other Resources:
Subscribe to:
Posts (Atom)