4.3 Reduction代码(Heterogeneous Parallel Programming class lab)

首先添加上Heterogeneous Parallel Programming class 中 lab: Reduction的代码：

myReduction.c

// MP Reduction

// Given a list (lst) of length n

// Output its sum = lst[0] + lst[1] + ... + lst[n-1];

#include    <wb.h>

#define BLOCK_SIZE 512 //@@ You can change this

#define wbCheck(stmt) do {                                                    \

        cudaError_t err = stmt;                                               \

        if (err != cudaSuccess) {                                             \

            wbLog(ERROR, "Failed to run stmt ", #stmt);                       \

            wbLog(ERROR, "Got CUDA error ...  ", cudaGetErrorString(err));    \

            return -;                                                        \

        }                                                                     \

    } while()

__global__ void reduction(float *g_idata, float *g_odata, unsigned int n){

    __shared__ float sdata[BLOCK_SIZE];

    // load shared mem

    unsigned int tid = threadIdx.x;

    unsigned int i = blockIdx.x*blockDim.x + threadIdx.x;

    sdata[tid] = (i < n) ? g_idata[i] : ;

    __syncthreads();

    // do reduction in shared mem, stride is divided by 2,

    for (unsigned int s=blockDim.x/; s>; s>>=)

    {

        //__syncthreads();

        if (tid < s)

        {

            sdata[tid] += sdata[tid + s];

        }

        __syncthreads();

    }

    // write result for this block to global mem

    if (tid == ) g_odata[blockIdx.x] = sdata[];

}

__global__ void total(float * input, float * output, int len) {

    //@@ Load a segment of the input vector into shared memory

    __shared__ float partialSum[ * BLOCK_SIZE];  //blockDim.x is not okay, compile fail

    unsigned int t = threadIdx.x;

    unsigned int start =  * blockIdx.x * blockDim.x;

    if (start + t < len)

       partialSum[t] = input[start + t];

    else

       partialSum[t] = ;

    if (start + blockDim.x + t < len)

       partialSum[blockDim.x + t] = input[start + blockDim.x + t];

    else

       partialSum[blockDim.x + t] = ;

    //@@ Traverse the reduction tree

    for (unsigned int stride = blockDim.x; stride >= ; stride >>= ) {

       __syncthreads();

       if (t < stride)

          partialSum[t] += partialSum[t+stride];

    }

    //@@ Write the computed sum of the block to the output vector at the

    //@@ correct index

    if (t == )

       output[blockIdx.x] = partialSum[];

}

int main(int argc, char ** argv) {

    int ii;

    wbArg_t args;

    float * hostInput; // The input 1D list

    float * hostOutput; // The output list

    float * deviceInput;

    float * deviceOutput;

    int numInputElements; // number of elements in the input list

    int numOutputElements; // number of elements in the output list

    args = wbArg_read(argc, argv);

    wbTime_start(Generic, "Importing data and creating memory on host");

    hostInput = (float *) wbImport(wbArg_getInputFile(args, ), &numInputElements);

    numOutputElements = numInputElements / (BLOCK_SIZE);

    if (numInputElements % (BLOCK_SIZE)) {

        numOutputElements++;

    }

    //This for kernel total

    /*numOutputElements = numInputElements / (BLOCK_SIZE <<1);

    if (numInputElements % (BLOCK_SIZE)<<1) {

        numOutputElements++;

    } */

    hostOutput = (float*) malloc(numOutputElements * sizeof(float));

    wbTime_stop(Generic, "Importing data and creating memory on host");

    wbLog(TRACE, "The number of input elements in the input is ", numInputElements);

    wbLog(TRACE, "The number of output elements in the input is ", numOutputElements);

    wbTime_start(GPU, "Allocating GPU memory.");

    //@@ Allocate GPU memory here

    cudaMalloc((void **) &deviceInput, numInputElements * sizeof(float));

    cudaMalloc((void **) &deviceOutput, numOutputElements * sizeof(float));

    wbTime_stop(GPU, "Allocating GPU memory.");

    wbTime_start(GPU, "Copying input memory to the GPU.");

    //@@ Copy memory to the GPU here

    cudaMemcpy(deviceInput,

               hostInput,

               numInputElements * sizeof(float),

               cudaMemcpyHostToDevice);

    wbTime_stop(GPU, "Copying input memory to the GPU.");

    //@@ Initialize the grid and block dimensions here

    dim3 dimGrid(numOutputElements, , );

    dim3 dimBlock(BLOCK_SIZE, , );

    wbTime_start(Compute, "Performing CUDA computation");

    //@@ Launch the GPU Kernel here

    reduction<<<dimGrid,dimBlock>>>(deviceInput, deviceOutput, numInputElements);

    //total<<<dimGrid, dimBlock>>>(deviceInput, deviceOutput, numInputElements);

    cudaDeviceSynchronize();

    wbTime_stop(Compute, "Performing CUDA computation");

    wbTime_start(Copy, "Copying output memory to the CPU");

    //@@ Copy the GPU memory back to the CPU here

    cudaMemcpy(hostOutput, deviceOutput, sizeof(float) * numOutputElements, cudaMemcpyDeviceToHost);

    wbTime_stop(Copy, "Copying output memory to the CPU");

    /********************************************************************

     * Reduce output vector on the host

     * NOTE: One could also perform the reduction of the output vector

     * recursively and support any size input. For simplicity, we do not

     * require that for this lab.

     ********************************************************************/

    for (ii = ; ii < numOutputElements; ii++) {

        hostOutput[] += hostOutput[ii];

    }

    wbTime_start(GPU, "Freeing GPU Memory");

    //@@ Free the GPU memory here

    cudaFree(deviceInput);

    cudaFree(deviceOutput);

    wbTime_stop(GPU, "Freeing GPU Memory");

    wbSolution(args, hostOutput, );

    free(hostInput);

    free(hostOutput);

    return ;

}

4.3 Reduction代码(Heterogeneous Parallel Programming class lab)的更多相关文章

PatentTips - Heterogeneous Parallel Primitives Programming Model
BACKGROUND 1. Field of the Invention The present invention relates generally to a programming model ...
Notes of Principles of Parallel Programming - TODO
0.1 TopicNotes of Lin C., Snyder L.. Principles of Parallel Programming. Beijing: China Machine Pres ...
Task Cancellation: Parallel Programming
http://beyondrelational.com/modules/2/blogs/79/posts/11524/task-cancellation-parallel-programming-ii ...
Samples for Parallel Programming with the .NET Framework
The .NET Framework 4 includes significant advancements for developers writing parallel and concurren ...
2018-12-09 疑似bug_中文代码示例之Programming in Scala笔记第九十章
续前文: 中文代码示例之Programming in Scala笔记第七八章源文档库: program-in-chinese/Programming_in_Scala_study_notes_zh ...
2018-11-27 中文代码示例之Programming in Scala笔记第七八章
续前文: 中文代码示例之Programming in Scala学习笔记第二三章中文代码示例之Programming in Scala笔记第四五六章. 同样仅节选有意思的例程部分作演示之用. 源文档 ...
2018-11-16 中文代码示例之Programming in Scala笔记第四五六章
续前文: 中文代码示例之Programming in Scala学习笔记第二三章. 同样仅节选有意思的例程部分作演示之用. 源文档仍在: program-in-chinese/Programming_ ...
Parallel Programming for FPGAs 学习笔记（1）
Parallel Programming for FPGAs 学习笔记(1)
Parallel Programming AND Asynchronous Programming
https://blogs.oracle.com/dave/ Java Memory Model...and the pragmatics of itAleksey Shipilevaleksey.s ...

随机推荐

[百度]数组A中任意两个相邻元素大小相差1，在其中查找某个数
一.问题来源及描述今天看了July的微博,发现了七月问题,有这个题,挺有意思的. 数组A中任意两个相邻元素大小相差1,现给定这样的数组A和目标整数t,找出t在数组A中的位置.如数组:[1,2,3,4 ...
荣誉，还是苦逼？| 也议全栈工程师和DevOps
引言全栈工程师(本文称「全栈」开发者)和 DevOps 无疑是近期最火的词汇,无论是国外还是国内.而且火爆程度远超于想象. 全栈和 DevOps,究竟是我们的新职业方向,还是仅仅创业公司老板的心头所 ...
【形式化方法：VDM++系列】2.VDMTools环境的搭建
接前文:http://www.cnblogs.com/Kassadin/p/3975853.html 上次讲了软件需求分析的演化过程,本次进入正题——VDM开发环境的搭建 (自从发现能打游戏以来,居然 ...
MyEclipse中创建maven工程
转载:http://blog.sina.com.cn/s/blog_4f925fc30102epdv.html 先要在MyEclipse中对Maven进行设置: 到此Maven对MyEclip ...
Qt之界面数据存储与获取（使用setUserData()和userData()）
在GUI开发中,往往需要在界面中存储一些有用的数据,这些数据可以来配置文件.注册表.数据库.或者是server. 无论来自哪里,这些数据对于用户来说都是至关重要的,它们在交互过程中大部分都会被用到,例 ...
Cinema 4D R16安装教程
CINEMA 4D_百度百科 http://baike.baidu.com/view/49453.htm?fr=aladdin 转自百度贴吧 [教程]Cinema 4D R16新功能介绍及安装教程_c ...
二维图形的矩阵变换（三）——在WPF中的应用矩阵变换
原文:二维图形的矩阵变换(三)--在WPF中的应用矩阵变换 UIElement和RenderTransform 首先,我们来看看什么样的对象可以进行变换.在WPF中,用于呈现给用户的对象的基类为Vis ...
搜索插件：ack.vim
ack.vim是Perl脚本ack的前端,对于Vim,也是grepprg和quickfix的简单封装,非常适合搜索 github地址为 https://github.com/mileszs/ack.v ...
bash把所有屏幕输出重定向到文件并保持屏幕输出的方法
输出到文件log中,并在屏幕上显示:#ls >&1 | tee log 追加输出到文件log中,并在屏幕上显示:#ls >&1 | tee -a log
poj3月题解
poj2110 二分答案+bfs判定 poj2112 二分答案+最大流判定(二分答案真乃USACO亲儿子) poj1986 裸的LCA,值得注意的是,树中任意两点的距离可以等于这两点到根的距离减去2* ...

4.3 Reduction代码(Heterogeneous Parallel Programming class lab)

4.3 Reduction代码(Heterogeneous Parallel Programming class lab)的更多相关文章

随机推荐

热门专题