4.3 Reduction代码(Heterogeneous Parallel Programming class lab)

首先添加上Heterogeneous Parallel Programming class 中 lab: Reduction的代码：

myReduction.c

// MP Reduction

// Given a list (lst) of length n

// Output its sum = lst[0] + lst[1] + ... + lst[n-1];

#include    <wb.h>

#define BLOCK_SIZE 512 //@@ You can change this

#define wbCheck(stmt) do {                                                    \

        cudaError_t err = stmt;                                               \

        if (err != cudaSuccess) {                                             \

            wbLog(ERROR, "Failed to run stmt ", #stmt);                       \

            wbLog(ERROR, "Got CUDA error ...  ", cudaGetErrorString(err));    \

            return -;                                                        \

        }                                                                     \

    } while()

__global__ void reduction(float *g_idata, float *g_odata, unsigned int n){

    __shared__ float sdata[BLOCK_SIZE];

    // load shared mem

    unsigned int tid = threadIdx.x;

    unsigned int i = blockIdx.x*blockDim.x + threadIdx.x;

    sdata[tid] = (i < n) ? g_idata[i] : ;

    __syncthreads();

    // do reduction in shared mem, stride is divided by 2,

    for (unsigned int s=blockDim.x/; s>; s>>=)

    {

        //__syncthreads();

        if (tid < s)

        {

            sdata[tid] += sdata[tid + s];

        }

        __syncthreads();

    }

    // write result for this block to global mem

    if (tid == ) g_odata[blockIdx.x] = sdata[];

}

__global__ void total(float * input, float * output, int len) {

    //@@ Load a segment of the input vector into shared memory

    __shared__ float partialSum[ * BLOCK_SIZE];  //blockDim.x is not okay, compile fail

    unsigned int t = threadIdx.x;

    unsigned int start =  * blockIdx.x * blockDim.x;

    if (start + t < len)

       partialSum[t] = input[start + t];

    else

       partialSum[t] = ;

    if (start + blockDim.x + t < len)

       partialSum[blockDim.x + t] = input[start + blockDim.x + t];

    else

       partialSum[blockDim.x + t] = ;

    //@@ Traverse the reduction tree

    for (unsigned int stride = blockDim.x; stride >= ; stride >>= ) {

       __syncthreads();

       if (t < stride)

          partialSum[t] += partialSum[t+stride];

    }

    //@@ Write the computed sum of the block to the output vector at the

    //@@ correct index

    if (t == )

       output[blockIdx.x] = partialSum[];

}

int main(int argc, char ** argv) {

    int ii;

    wbArg_t args;

    float * hostInput; // The input 1D list

    float * hostOutput; // The output list

    float * deviceInput;

    float * deviceOutput;

    int numInputElements; // number of elements in the input list

    int numOutputElements; // number of elements in the output list

    args = wbArg_read(argc, argv);

    wbTime_start(Generic, "Importing data and creating memory on host");

    hostInput = (float *) wbImport(wbArg_getInputFile(args, ), &numInputElements);

    numOutputElements = numInputElements / (BLOCK_SIZE);

    if (numInputElements % (BLOCK_SIZE)) {

        numOutputElements++;

    }

    //This for kernel total

    /*numOutputElements = numInputElements / (BLOCK_SIZE <<1);

    if (numInputElements % (BLOCK_SIZE)<<1) {

        numOutputElements++;

    } */

    hostOutput = (float*) malloc(numOutputElements * sizeof(float));

    wbTime_stop(Generic, "Importing data and creating memory on host");

    wbLog(TRACE, "The number of input elements in the input is ", numInputElements);

    wbLog(TRACE, "The number of output elements in the input is ", numOutputElements);

    wbTime_start(GPU, "Allocating GPU memory.");

    //@@ Allocate GPU memory here

    cudaMalloc((void **) &deviceInput, numInputElements * sizeof(float));

    cudaMalloc((void **) &deviceOutput, numOutputElements * sizeof(float));

    wbTime_stop(GPU, "Allocating GPU memory.");

    wbTime_start(GPU, "Copying input memory to the GPU.");

    //@@ Copy memory to the GPU here

    cudaMemcpy(deviceInput,

               hostInput,

               numInputElements * sizeof(float),

               cudaMemcpyHostToDevice);

    wbTime_stop(GPU, "Copying input memory to the GPU.");

    //@@ Initialize the grid and block dimensions here

    dim3 dimGrid(numOutputElements, , );

    dim3 dimBlock(BLOCK_SIZE, , );

    wbTime_start(Compute, "Performing CUDA computation");

    //@@ Launch the GPU Kernel here

    reduction<<<dimGrid,dimBlock>>>(deviceInput, deviceOutput, numInputElements);

    //total<<<dimGrid, dimBlock>>>(deviceInput, deviceOutput, numInputElements);

    cudaDeviceSynchronize();

    wbTime_stop(Compute, "Performing CUDA computation");

    wbTime_start(Copy, "Copying output memory to the CPU");

    //@@ Copy the GPU memory back to the CPU here

    cudaMemcpy(hostOutput, deviceOutput, sizeof(float) * numOutputElements, cudaMemcpyDeviceToHost);

    wbTime_stop(Copy, "Copying output memory to the CPU");

    /********************************************************************

     * Reduce output vector on the host

     * NOTE: One could also perform the reduction of the output vector

     * recursively and support any size input. For simplicity, we do not

     * require that for this lab.

     ********************************************************************/

    for (ii = ; ii < numOutputElements; ii++) {

        hostOutput[] += hostOutput[ii];

    }

    wbTime_start(GPU, "Freeing GPU Memory");

    //@@ Free the GPU memory here

    cudaFree(deviceInput);

    cudaFree(deviceOutput);

    wbTime_stop(GPU, "Freeing GPU Memory");

    wbSolution(args, hostOutput, );

    free(hostInput);

    free(hostOutput);

    return ;

}

4.3 Reduction代码(Heterogeneous Parallel Programming class lab)的更多相关文章

PatentTips - Heterogeneous Parallel Primitives Programming Model
BACKGROUND 1. Field of the Invention The present invention relates generally to a programming model ...
Notes of Principles of Parallel Programming - TODO
0.1 TopicNotes of Lin C., Snyder L.. Principles of Parallel Programming. Beijing: China Machine Pres ...
Task Cancellation: Parallel Programming
http://beyondrelational.com/modules/2/blogs/79/posts/11524/task-cancellation-parallel-programming-ii ...
Samples for Parallel Programming with the .NET Framework
The .NET Framework 4 includes significant advancements for developers writing parallel and concurren ...
2018-12-09 疑似bug_中文代码示例之Programming in Scala笔记第九十章
续前文: 中文代码示例之Programming in Scala笔记第七八章源文档库: program-in-chinese/Programming_in_Scala_study_notes_zh ...
2018-11-27 中文代码示例之Programming in Scala笔记第七八章
续前文: 中文代码示例之Programming in Scala学习笔记第二三章中文代码示例之Programming in Scala笔记第四五六章. 同样仅节选有意思的例程部分作演示之用. 源文档 ...
2018-11-16 中文代码示例之Programming in Scala笔记第四五六章
续前文: 中文代码示例之Programming in Scala学习笔记第二三章. 同样仅节选有意思的例程部分作演示之用. 源文档仍在: program-in-chinese/Programming_ ...
Parallel Programming for FPGAs 学习笔记（1）
Parallel Programming for FPGAs 学习笔记(1)
Parallel Programming AND Asynchronous Programming
https://blogs.oracle.com/dave/ Java Memory Model...and the pragmatics of itAleksey Shipilevaleksey.s ...

随机推荐

smarty 时间格式化date_format
代码如下:$smarty = new Smarty; $smarty->assign('yesterday', strtotime('-1 day')); $smarty->display ...
[Unity+Android]横版扫描二维码
原地址:http://blog.csdn.net/dingxiaowei2013/article/details/25086835 终于解决了一个忧伤好久的问题,严重拖了项目进度,深感惭愧!一直被一系 ...
用户自定义结构数据与VARIANT转换 .
用户自定义结构数据与VARIANT转换 cheungmine 将用户自定义的C结构数据存储成VARIANT类型,需要时再将VARIANT类型转为用户自定义的结构数据,有十分现实的意义,既然我们不想为这 ...
学习记录：浏览器JAVASCRIPT里的WINDOWS，DOCUMNET
看完以下这段话之后,就理解DOCUMNET.READY之类的说法了. 或是JAVASCRIPT的浏览器里更细致的操作DOCUMENT的东西了. DOCUMNET和WINDOWS谁大谁小, 立即执行的匿 ...
POJ2109——Power of Cryptography
Power of Cryptography DescriptionCurrent work in cryptography involves (among other things) large pr ...
P134、面试题22：栈的压入、弹出序列
题目:输入两个整数序列,第一个序列表示栈的压入顺序,请判断第二个序列是否为该栈的弹出顺序.假设压入栈的所有数字均不相等.例如序列1.2.3.4.5是某栈的压栈序列,序列4,5,3,2,1是该压栈序列对 ...
Oracle Exception 处理
1.问题来源Oracle中可以用dbms_output.put_line来打印提示信息,但是很容易缓冲区就溢出了.可以用DBMS_OUTPUT.ENABLE(1000000);来设置缓冲区的大小.但是 ...
android 电容屏（三）：驱动调试之驱动程序分析篇
平台信息: 内核:linux3.4.39系统:android4.4 平台:S5P4418(cortex a9) 作者:瘋耔(欢迎转载,请注明作者) 欢迎指正错误,共同学习.共同进步!! 关注博主新浪博 ...
Android 开发之 ---- 底层驱动开发(一)
驱动概述说到 android 驱动是离不开 Linux 驱动的.Android 内核采用的是 Linux2.6 内核 (最近Linux 3.3 已经包含了一些 Android 代码).但 Andro ...
poj3034Whac-a-Mole(dp)
链接状态转移好想不过有坑大家都犯的错误我也会犯很正常就是锤子可以移到n*n以外要命的是我只加了5 以为最多不会超过5 WA了N久才想到上下两方向都可以到5 所以最多加10 以时 ...

4.3 Reduction代码(Heterogeneous Parallel Programming class lab)

4.3 Reduction代码(Heterogeneous Parallel Programming class lab)的更多相关文章

随机推荐

热门专题