-
Notifications
You must be signed in to change notification settings - Fork 0
Expand file tree
/
Copy pathsummary_threads.cu
More file actions
57 lines (46 loc) · 1.81 KB
/
Copy pathsummary_threads.cu
File metadata and controls
57 lines (46 loc) · 1.81 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
#include <cuda.h>
#include <stdio.h>
#define CUDA_CHECK_RETURN(value) {\
cudaError_t _m_cudaStat = value;\
if (_m_cudaStat != cudaSuccess) {\
fprintf(stderr, "Error %s at line %d in file %s\n",\
cudaGetErrorString(_m_cudaStat), __LINE__, __FILE__);\
exit(1);\
}}
__global__ void gTest(float* a){
a[threadIdx.x+blockDim.x*blockIdx.x]=(float)(threadIdx.x+blockDim.x*blockIdx.x);
}
__global__ void sum (float* a, float* b)
{
a[threadIdx.x+blockDim.x*blockIdx.x]+=b[threadIdx.x+blockDim.x*blockIdx.x];
}
int main(int argc, char* argv[]){
float *da, *db, *ha;
int num_of_blocks=1<<2, threads_per_block=1<<2;
int N=num_of_blocks*threads_per_block;
float elapsedTime;
cudaEvent_t start,stop;
cudaEventCreate(&start);
cudaEventCreate(&stop);
ha=(float*)calloc(N, sizeof(float));
CUDA_CHECK_RETURN(cudaMalloc((void**)&da,N*sizeof(float)));
CUDA_CHECK_RETURN(cudaMalloc((void**)&db,N*sizeof(float)));
gTest<<<dim3(num_of_blocks), dim3(threads_per_block)>>>(da);
gTest<<<dim3(num_of_blocks), dim3(threads_per_block)>>>(db);
cudaEventRecord(start,0);
sum<<<dim3(num_of_blocks), dim3(threads_per_block)>>>(da,db);
cudaEventRecord(stop,0);
cudaEventSynchronize(stop);
CUDA_CHECK_RETURN(cudaGetLastError());
cudaEventElapsedTime(&elapsedTime,start,stop);
fprintf(stderr,"gTest took %g\n", elapsedTime);
cudaEventDestroy(start);
cudaEventDestroy(stop);
CUDA_CHECK_RETURN(cudaMemcpy(ha,da,N*sizeof(float), cudaMemcpyDeviceToHost));
for(int i=0;i<N;i++)
printf("%g\n",ha[i]);
free(ha);
cudaFree(da);
cudaFree(db);
return 0;
}