《GPU高性能编程CUDA实战》附录二散列表

▶ 使用CPU和GPU分别实现散列表

● CPU方法

 #include <stdio.h>

 #include <time.h>

 #include "cuda_runtime.h"

 #include "D:\Code\CUDA\book\common\book.h"

 #define SIZE            (100*1024*1024)

 #define ELEMENTS        (SIZE / sizeof(unsigned int))

 #define HASH_ENTRIES    (1024)

 struct Entry

 {

     unsigned int    key;

     void            *value;

     Entry           *next;

 };

 struct Table

 {

     size_t  count;

     Entry   **entries;

     Entry   *pool;

     Entry   *firstFree;

 };

 size_t hash(unsigned int key, size_t count)

 {

     return key % count;

 }

 void initialize_table(Table &table, int entries, int elements)

 {

     table.count = entries;

     table.entries = (Entry**)calloc(entries, sizeof(Entry*));

     table.pool = (Entry*)malloc(elements * sizeof(Entry));

     table.firstFree = table.pool;

 }

 void free_table(Table &table)

 {

     free(table.entries);

     free(table.pool);

 }

 void add_to_table(Table &table, unsigned int key, void *value)

 {

     size_t hashValue = hash(key, table.count);

     Entry *location = table.firstFree++;

     location->key = key;

     location->value = value;

     location->next = table.entries[hashValue];// 插到该分支的头部而不是尾部

     table.entries[hashValue] = location;

 }

 void verify_table(const Table &table)

 {

     int count = ;

     for (size_t i = ; i<table.count; i++)

     {

         Entry   *current = table.entries[i];

         while (current != NULL)

         {

             ++count;

             if (hash(current->key, table.count) != i)

                 printf("\n\t%d hashed to %ld, but was located at %ld\n", current->key, hash(current->key, table.count), i);

             current = current->next;

         }

     }

     if (count != ELEMENTS)

         printf("\n\t%d elements found in hash table.  Should be %ld\n",

             count, ELEMENTS);

     else

         printf("\n\tAll %d elements found in hash table.\n", count);

 }

 int main(void)

 {

     unsigned int *buffer =(unsigned int*)big_random_block(SIZE);

     Table table;

     clock_t start, stop;

     initialize_table(table, HASH_ENTRIES, ELEMENTS);

     start = clock();

     for (int i = ; i<ELEMENTS; i++)

         add_to_table(table, buffer[i], (void*)NULL);

     stop = clock();

     printf("\n\tBuilding the table:  %3.1f ms\n", (float)(stop - start) / (float)CLOCKS_PER_SEC * 1000.0f);

     verify_table(table);

     free_table(table);

     free(buffer);

     getchar();

     return ;

 }

● GPU方法（用到了前面的原子锁）

 #include <stdio.h>

 #include <time.h>

 #include "cuda_runtime.h"

 #include "device_launch_parameters.h"

 #include "cuda.h"

 #include "D:\Code\CUDA\book\common\book.h"

 #define SIZE            (100*1024*1024)

 #define ELEMENTS        (SIZE / sizeof(unsigned int))

 #define HASH_ENTRIES    (1024)

 struct Lock

 {

     int *mutex;

     Lock(void)

     {

         int state = ;

         cudaMalloc((void **)&mutex, sizeof(int));

         cudaMemcpy(mutex, &state, sizeof(int), cudaMemcpyHostToDevice);

     }

     ~Lock(void)

     {

         cudaFree(mutex);

     }

     __device__ void lock(void)

     {

         while (atomicCAS(mutex, , ) != );

     }

     __device__ void unlock(void)

     {

         atomicExch(mutex, );

     }

 };

 struct Entry

 {

     unsigned int    key;

     void            *value;

     Entry           *next;

 };

 struct Table

 {

     size_t  count;

     Entry   **entries;

     Entry   *pool;

     Entry   *firstFree;

 };

 __device__ __host__ size_t hash(unsigned int key, size_t count)

 {

     return key % count;

 }

 void initialize_table(Table &table, int entries, int elements)

 {

     table.count = entries;

     cudaMalloc((void**)&table.entries, entries * sizeof(Entry*));

     cudaMemset(table.entries, , entries * sizeof(Entry*));

     cudaMalloc((void**)&table.pool, elements * sizeof(Entry));

 }

 void free_table(Table &table)

 {

     cudaFree(table.entries);

     cudaFree(table.pool);

 }

 __global__ void add_to_table(unsigned int *keys, void **values, Table table, Lock *lock)

 // 锁数组用于锁定散列表中的每一个桶

 {

     int tid = threadIdx.x + blockIdx.x * blockDim.x;

     int stride = blockDim.x * gridDim.x;

     while (tid < ELEMENTS)

     {

         unsigned int key = keys[tid];

         size_t hashValue = hash(key, table.count);

         for (int i = ; i<; i++)// 利用循环来分散线程束，使同一线程束中的32个线程在循环的不同次数时进行写入

         {

             if ((tid % ) == i)

             {

                 Entry *location = &(table.pool[tid]);

                 location->key = key;

                 location->value = values[tid];

                 lock[hashValue].lock();

                 location->next = table.entries[hashValue];

                 table.entries[hashValue] = location;

                 lock[hashValue].unlock();

             }

         }

         tid += stride;

     }

 }

 void copy_table_to_host(const Table &table, Table &hostTable)

 {

     hostTable.count = table.count;

     hostTable.entries = (Entry**)calloc(table.count, sizeof(Entry*));

     hostTable.pool = (Entry*)malloc(ELEMENTS * sizeof(Entry));

     cudaMemcpy(hostTable.entries, table.entries, table.count * sizeof(Entry*), cudaMemcpyDeviceToHost);

     cudaMemcpy(hostTable.pool, table.pool, ELEMENTS * sizeof(Entry), cudaMemcpyDeviceToHost);

     for (int i = ; i < table.count; i++)

     {

         if (hostTable.entries[i] != NULL)

             hostTable.entries[i] = (Entry*)((size_t)hostTable.entries[i] - (size_t)table.pool + (size_t)hostTable.pool);

         // 从从显存到内存的地址线性偏移 x - adressGPU + addressCPU

     }

     for (int i = ; i < ELEMENTS; i++)

     {

         if (hostTable.pool[i].next != NULL)

             hostTable.pool[i].next = (Entry*)((size_t)hostTable.pool[i].next - (size_t)table.pool + (size_t)hostTable.pool);

         // 同样是做偏移，但是要找到下一个元素的地址

     }

 }

 void verify_table(const Table &dev_table)

 {

     Table   table;

     copy_table_to_host(dev_table, table);

     int count = ;

     for (size_t i = ; i < table.count; i++)

     {

         Entry   *current = table.entries[i];

         while (current != NULL)

         {

             ++count;

             if (hash(current->key, table.count) != i)

                 printf("%d hashed to %ld, but was located at %ld\n", current->key, hash(current->key, table.count), i);

             current = current->next;

         }

     }

     if (count != ELEMENTS)

         printf("%d elements found in hash table.  Should be %ld\n", count, ELEMENTS);

     else

         printf("All %d elements found in hash table.\n", count);

 }

 int main(void)

 {

     unsigned int *buffer = (unsigned int*)big_random_block(SIZE);

     unsigned int *dev_keys;

     void         **dev_values;

     cudaMalloc((void**)&dev_keys, SIZE);

     cudaMalloc((void**)&dev_values, SIZE);

     cudaMemcpy(dev_keys, buffer, SIZE, cudaMemcpyHostToDevice);

     Table table;

     initialize_table(table, HASH_ENTRIES, ELEMENTS);

     Lock    lock[HASH_ENTRIES];// 准备锁列表

     Lock    *dev_lock;

     cudaMalloc((void**)&dev_lock, HASH_ENTRIES * sizeof(Lock));

     cudaMemcpy(dev_lock, lock, HASH_ENTRIES * sizeof(Lock), cudaMemcpyHostToDevice);

     cudaEvent_t     start, stop;

     cudaEventCreate(&start);

     cudaEventCreate(&stop);

     cudaEventRecord(start, );

     add_to_table << <,  >> >(dev_keys, dev_values, table, dev_lock);

     cudaEventRecord(stop, );

     cudaEventSynchronize(stop);

     float   elapsedTime;

     cudaEventElapsedTime(&elapsedTime, start, stop);

     printf("Time to hash:  %3.1f ms\n", elapsedTime);

     verify_table(table);

     free_table(table);

     cudaEventDestroy(start);

     cudaEventDestroy(stop);

     free_table(table);

     cudaFree(dev_lock);

     cudaFree(dev_keys);

     cudaFree(dev_values);

     free(buffer);

     getchar();

     return ;

 }

《GPU高性能编程CUDA实战》附录二散列表的更多相关文章

[问题解决]《GPU高性能编程CUDA实战》中第4章Julia实例“显示器驱动已停止响应，并且已恢复”问题的解决方法
以下问题的出现及解决都基于"WIN7+CUDA7.5". 问题描述:当我编译运行<GPU高性能编程CUDA实战>中第4章所给Julia实例代码时,出现了显示器闪动的现象 ...
《GPU高性能编程CUDA实战》附录四其他头文件
▶ cpu_bitmap.h #ifndef __CPU_BITMAP_H__ #define __CPU_BITMAP_H__ #include "gl_helper.h" st ...
《GPU高性能编程CUDA实战》附录一高级原子操作
▶ 本章介绍了手动实现原子操作.重构了第五章向量点积的过程.核心是通过定义结构Lock及其运算,实现锁定,读写,解锁的过程. ● 章节代码 #include <stdio.h> #incl ...
《GPU高性能编程CUDA实战》附录三关于book.h
▶ 本书中用到的公用函数放到了头文件book.h中 #ifndef __BOOK_H__ #define __BOOK_H__ #include <stdio.h> #include &l ...
《GPU高性能编程CUDA实战》第五章线程并行
▶ 本章介绍了线程并行,并给出四个例子.长向量加法.波纹效果.点积和显示位图. ● 长向量加法(线程块并行 + 线程并行) #include <stdio.h> #include &quo ...
《GPU高性能编程CUDA实战》第十一章多GPU系统的CUDA C
▶ 本章介绍了多设备胸膛下的 CUDA 编程,以及一些特殊存储类型对计算速度的影响 ● 显存和零拷贝内存的拷贝与计算对比 #include <stdio.h> #include " ...
《GPU高性能编程CUDA实战》第七章纹理内存
▶ 本章介绍了纹理内存的使用,并给出了热传导的两个个例子.分别使用了一维和二维纹理单元. ● 热传导(使用一维纹理) #include <stdio.h> #include "c ...
《GPU高性能编程CUDA实战》第四章简单的线程块并行
▶ 本章介绍了线程块并行,并给出两个例子:长向量加法和绘制julia集. ● 长向量加法,中规中矩的GPU加法,包含申请内存和显存,赋值,显存传入,计算,显存传出,处理结果,清理内存和显存.用到了 t ...
《GPU高性能编程CUDA实战》第八章图形互操作性
▶ OpenGL与DirectX,等待填坑. ● basic_interop #include <stdio.h> #include "cuda_runtime.h" ...

随机推荐

POJ 3254 Corn Fields状态压缩DP
下面有别人的题解报告,并且不止这一个状态压缩题的哦···· http://blog.csdn.net/accry/article/details/6607703 下面是我的代码,代码很挫,绝对有很大的 ...
hdu2085-2086
hdu2085 模拟 #include<stdio.h> ][]; void fun(){ a[][]=; a[][]=; ;i<=;i++){ a[i][]=*a[i-][]+*a ...
python open和file的区别
opne和file都是用来对文件的操作 open:内置函数,使用方式是open('file_name', mode, buffering),返回值是一个file对象,以写模式打开文件如果不存在会被创建 ...
FineUI Grid中WindowField根据列数据决定是否Enalble
前台页面Grid控件中设置OnPreRowDataBound属性,windowFile控件设置ID protected void Grid1_PreRowDataBound(object sender ...
Angular 4 路由介绍
Angular 4 路由 1. 创建工程 ng new router --routing 2. 创建home和product组件 ng g component home ng g component ...
JS 响应式布局
1.media 效果为屏幕宽度变化时,背景颜色也变化 <!DOCTYPE html> <html lang="en"> <head> <m ...
HttpPostedFile类
在研究HttpRequest的时候,搞文件上传的时候,经常碰到返回HttpPostedFile对象的情况,这个对象才是真正包含文件内容的东西. 经常要获取的最重要的内容是FileName属性与Sava ...
剑指offer-python面试篇第一部分
互联网协议定义(分别有4层.5层及7层协议的说法,以下从上层向下层介绍)? a) 四层协议:应用层.传输层.网络层.网络接口层 a) 五层协议: 应用层:用户使用的应用程序都归属于应用层,作用为规定应 ...
ASP.NET ASHX中获得Session
有时候需要在ASHX中获取Session,可是一般是获取不到的,如何解决? 1-在 aspx和aspx.cs中,都是以Session["xxx"]="aaa"和 ...
go的module用法
新版不需要项目放在GOPATH里面了,这个恶心的机制之前还被n多人捧臭脚.简单列一下用法新建项目 cd 项目目录go mod init 项目名写好代码 go build 或者 go mod tid ...

《GPU高性能编程CUDA实战》附录二 散列表

《GPU高性能编程CUDA实战》附录二 散列表的更多相关文章

随机推荐

热门专题

《GPU高性能编程CUDA实战》附录二散列表

《GPU高性能编程CUDA实战》附录二散列表的更多相关文章