标签:代码 init tps tde sse 展示 host oid 引入
博客最后附上整体代码
如果有说的不对的地方还请前辈指出, 因为cuda真的接触没几天
#include <stdio.h>
#include <cuda_runtime.h>
__global__ void
vectorAdd(float *a, float *b, float *c, int num){
int i = blockDim.x * blockIdx.x + threadIdx.x; //vector is 1-dim, blockDim means the number of thread in a block
if(i < num){
c[i] = a[i] + b[i];
}
}
int i = blockDim.x * blockIdx.x + threadIdx.x;
这句代码解释一下:
blockDim.x 表示block的size行数(如果是一维的block的话,即一行有多少个thread)
blockIdx.x 表示当前运行到的第几个block(一维grid的话,即该grid中第几个block)
threadIdx.x 表示当前运行到的第几个thread (一维的block的话.即该block中第几个thread)
画个图解释一下
比如上面这个图的话, ABCDE各代表一个block, 总的为一个Grid, 每个block中有四个thread, 图中我花了箭头的也就是代表着第1个block中的第0个thread.
那么 i = blockDim.x * blockIdx.x + threadIdx.x 就是指 i = 4 * 1 + 0
host中申请内存
float *a = (float *)malloc(size);
float *b = (float *)malloc(size);
float *c = (float *)malloc(size);
free(a);
free(b);
free(c);
device中申请内存
float *da = NULL;
float *db = NULL;
float *dc = NULL;
cudaMalloc((void **)&da, size);
cudaMalloc((void **)&db, size);
cudaMalloc((void **)&dc, size);
cudaFree(da);
cudaFree(db);
cudaFree(dc);
cudaMemcpy(da,a,size,cudaMemcpyHostToDevice);
cudaMemcpy(db,b,size,cudaMemcpyHostToDevice);
cudaMemcpy(dc,c,size,cudaMemcpyHostToDevice);
上面的cudaMemcpyHostToDevice用于指定方向有四种关键词
cudaMemcpyHostToDevice | cudaMemcpyHostToHost | cudaMemcpyDeviceToDevice | cudaMemcpyDeviceToHost
int threadPerBlock = 256;
int blockPerGrid = (num + threadPerBlock - 1)/threadPerBlock;
vectorAdd <<< blockPerGrid, threadPerBlock >>> (da,db,dc,num)
此处确定了block中的thread数量以及一个grid中block数量
利用kernel function <<< blockPerGrid, threadPerBlock>>> (paras,...) 来实现在cuda中运算
#include <stdio.h>
#include <cuda_runtime.h>
// vectorAdd run in device
__global__ void
vectorAdd(float *a, float *b, float *c, int num){
int i = blockDim.x * blockIdx.x + threadIdx.x; //vector is 1-dim, blockDim means the number of thread in a block
if(i < num){
c[i] = a[i] + b[i];
}
}
// main run in host
int
main(void){
int num = 10000; // size of vector
size_t size = num * sizeof(float);
// host memery
float *a = (float *)malloc(size);
float *b = (float *)malloc(size);
float *c = (float *)malloc(size);
// init the vector
for(int i=1;i<num;++i){
a[i] = rand()/(float)RAND_MAX;
b[i] = rand()/(float)RAND_MAX;
}
// copy the host memery to device memery
float *da = NULL;
float *db = NULL;
float *dc = NULL;
cudaMalloc((void **)&da, size);
cudaMalloc((void **)&db, size);
cudaMalloc((void **)&dc, size);
cudaMemcpy(da,a,size,cudaMemcpyHostToDevice);
cudaMemcpy(db,b,size,cudaMemcpyHostToDevice);
cudaMemcpy(dc,c,size,cudaMemcpyHostToDevice);
// launch function add kernel
int threadPerBlock = 256;
int blockPerGrid = (num + threadPerBlock - 1)/threadPerBlock;
printf("threadPerBlock: %d \nblockPerGrid: %d \n",threadPerBlock,blockPerGrid);
vectorAdd <<< blockPerGrid, threadPerBlock >>> (da,db,dc,num);
//copy the device result to host
cudaMemcpy(c,dc,size,cudaMemcpyDeviceToHost);
// Verify that the result vector is correct
for (int i = 0; i < num; ++i){
if (fabs(a[i] + b[i] - c[i]) > 1e-5){
fprintf(stderr, "Result verification failed at element %d!\n", i);
return 0;
}
}
printf("Test PASSED\n");
// Free device global memory
cudaFree(da);
cudaFree(db);
cudaFree(dc);
// Free host memory
free(a);
free(b);
free(c);
printf("free is ok\n");
return 0;
}
标签:代码 init tps tde sse 展示 host oid 引入
原文地址:https://www.cnblogs.com/wangha/p/10803696.html