added implementation of one thread per element for example01 - it's the fastest thus far
This commit is contained in:
@@ -10,11 +10,15 @@ timing results of vectors added together on a CPU vs GPU. the size of the vector
|
|||||||
- CPU code
|
- CPU code
|
||||||
- GPU code equivalent
|
- GPU code equivalent
|
||||||
- GPU code where the buffers are created and destroyed each iteration (this is guaranteed to be slower, of course, but it would be nice to see the amount of overhead that occurs with inefficient copying...)
|
- GPU code where the buffers are created and destroyed each iteration (this is guaranteed to be slower, of course, but it would be nice to see the amount of overhead that occurs with inefficient copying...)
|
||||||
|
- GPU code where the number of threads spawned is equal to the size of the vector, and each thread adds one element
|
||||||
|
|
||||||
todo:
|
todo:
|
||||||
|
|
||||||
- GPU code where a work-group barrier is initiated after each thread is done its work (equivalent to the regular GPU code, but with one extra line...)
|
- GPU code where a work-group barrier is initiated after each thread is done its work (equivalent to the regular GPU code, but with one extra line...)
|
||||||
|
|
||||||
|
### results
|
||||||
|
so far, the third version (each thread doing only one element) is the fastest, and I'm trying to understand why.
|
||||||
|
|
||||||
## TODO
|
## TODO
|
||||||
|
|
||||||
- figure out how OpenCL manages memory (when are buffers cleared on the GPU?)
|
- figure out how OpenCL manages memory (when are buffers cleared on the GPU?)
|
||||||
|
|||||||
@@ -3,6 +3,19 @@
|
|||||||
#include <ctime>
|
#include <ctime>
|
||||||
|
|
||||||
|
|
||||||
|
void compareResults (double CPUtime, double GPUtime, int trial) {
|
||||||
|
double time_ratio = (CPUtime / GPUtime);
|
||||||
|
std::cout << "VERSION " << trial << " -----------" << std::endl;
|
||||||
|
std::cout << "CPU time: " << CPUtime << std::endl;
|
||||||
|
std::cout << "GPU time: " << GPUtime << std::endl;
|
||||||
|
std::cout << "GPU is ";
|
||||||
|
if (time_ratio > 1)
|
||||||
|
std::cout << time_ratio << " times faster!" << std::endl;
|
||||||
|
else
|
||||||
|
std::cout << (1/time_ratio) << " times slower :(" << std::endl;
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
double timeAddVectorsCPU(int n, int k) {
|
double timeAddVectorsCPU(int n, int k) {
|
||||||
// adds two vectors of size n, k times, returns total duration
|
// adds two vectors of size n, k times, returns total duration
|
||||||
std::clock_t start;
|
std::clock_t start;
|
||||||
@@ -53,6 +66,9 @@ int main() {
|
|||||||
cl::Context context({default_device});
|
cl::Context context({default_device});
|
||||||
cl::Program::Sources sources;
|
cl::Program::Sources sources;
|
||||||
|
|
||||||
|
std::cout << CL_DEVICE_MAX_WORK_ITEM_SIZES << std::endl;
|
||||||
|
exit(1);
|
||||||
|
|
||||||
// calculates for each element; C = A + B
|
// calculates for each element; C = A + B
|
||||||
std::string kernel_code=
|
std::string kernel_code=
|
||||||
// is equivalent to the host's "time_add_vectors" function, except the
|
// is equivalent to the host's "time_add_vectors" function, except the
|
||||||
@@ -89,6 +105,14 @@ int main() {
|
|||||||
""
|
""
|
||||||
" for (int i=start; i<stop; i++)"
|
" for (int i=start; i<stop; i++)"
|
||||||
" v3[i] = v1[i] + v2[i];"
|
" v3[i] = v1[i] + v2[i];"
|
||||||
|
" }"
|
||||||
|
""
|
||||||
|
" void kernel add_single(global const int* v1, global const int* v2, global int* v3, "
|
||||||
|
" global const int* constants) { "
|
||||||
|
" int k = constants[1];"
|
||||||
|
" int ID = get_global_id(0);"
|
||||||
|
" for (int i=0; i<k; i++)"
|
||||||
|
" v3[ID] = v1[ID] + v2[ID];"
|
||||||
" }";
|
" }";
|
||||||
sources.push_back({kernel_code.c_str(), kernel_code.length()});
|
sources.push_back({kernel_code.c_str(), kernel_code.length()});
|
||||||
|
|
||||||
@@ -99,9 +123,9 @@ int main() {
|
|||||||
}
|
}
|
||||||
|
|
||||||
int n, k, Nthreads;
|
int n, k, Nthreads;
|
||||||
n = 100000; // size of vectors
|
n = 128000; // size of vectors
|
||||||
k = 1000; // number of loop iterations
|
k = 1000; // number of loop iterations
|
||||||
Nthreads = 10;
|
Nthreads = 128;
|
||||||
int constants[2] = {n, k};
|
int constants[2] = {n, k};
|
||||||
|
|
||||||
// run the CPU code
|
// run the CPU code
|
||||||
@@ -111,8 +135,9 @@ int main() {
|
|||||||
// adds the same two vectors multiple times -- i.e. it's equivalent to the
|
// adds the same two vectors multiple times -- i.e. it's equivalent to the
|
||||||
// host (CPU) code above, but with some necessary overhead.
|
// host (CPU) code above, but with some necessary overhead.
|
||||||
cl::CommandQueue queue(context, default_device);
|
cl::CommandQueue queue(context, default_device);
|
||||||
cl::KernelFunctor looped_add(cl::Kernel(program, "looped_add"), queue, cl::NullRange, cl::NDRange(Nthreads), cl::NullRange);
|
|
||||||
cl::KernelFunctor add(cl::Kernel(program, "add"), queue, cl::NullRange, cl::NDRange(Nthreads), cl::NullRange);
|
cl::KernelFunctor add(cl::Kernel(program, "add"), queue, cl::NullRange, cl::NDRange(Nthreads), cl::NullRange);
|
||||||
|
cl::Kernel looped_add = cl::Kernel(program, "looped_add");
|
||||||
|
cl::Kernel add_single = cl::Kernel(program, "add_single");
|
||||||
|
|
||||||
// construct vectors
|
// construct vectors
|
||||||
int A[n], B[n], C[n];
|
int A[n], B[n], C[n];
|
||||||
@@ -139,12 +164,17 @@ int main() {
|
|||||||
queue.enqueueWriteBuffer(buffer_constants, CL_TRUE, 0, sizeof(int)*2, constants);
|
queue.enqueueWriteBuffer(buffer_constants, CL_TRUE, 0, sizeof(int)*2, constants);
|
||||||
|
|
||||||
// RUN ZE KERNEL
|
// RUN ZE KERNEL
|
||||||
looped_add(buffer_A, buffer_B, buffer_C, buffer_constants);
|
looped_add.setArg(0, buffer_A);
|
||||||
|
looped_add.setArg(1, buffer_B);
|
||||||
|
looped_add.setArg(2, buffer_C);
|
||||||
|
looped_add.setArg(3, buffer_constants);
|
||||||
|
queue.enqueueNDRangeKernel(looped_add, cl::NullRange, 32, cl::NullRange);
|
||||||
|
|
||||||
// read result from GPU to here; including for the sake of timing
|
// read result from GPU to here; including for the sake of timing
|
||||||
queue.enqueueReadBuffer(buffer_C, CL_TRUE, 0, sizeof(int)*n, C);
|
queue.enqueueReadBuffer(buffer_C, CL_TRUE, 0, sizeof(int)*n, C);
|
||||||
GPUtime1 = (std::clock() - start_time) / (double) CLOCKS_PER_SEC;
|
GPUtime1 = (std::clock() - start_time) / (double) CLOCKS_PER_SEC;
|
||||||
|
|
||||||
|
|
||||||
// do the same thing, except copy the arrays over every iteration
|
// do the same thing, except copy the arrays over every iteration
|
||||||
double GPUtime2;
|
double GPUtime2;
|
||||||
start_time = std::clock();
|
start_time = std::clock();
|
||||||
@@ -163,26 +193,38 @@ int main() {
|
|||||||
queue.enqueueReadBuffer(buffer_C2, CL_TRUE, 0, sizeof(int)*n, C);
|
queue.enqueueReadBuffer(buffer_C2, CL_TRUE, 0, sizeof(int)*n, C);
|
||||||
GPUtime2 = (std::clock() - start_time) / (double) CLOCKS_PER_SEC;
|
GPUtime2 = (std::clock() - start_time) / (double) CLOCKS_PER_SEC;
|
||||||
|
|
||||||
// let's compare!
|
|
||||||
double time_ratio = (CPUtime / GPUtime1);
|
|
||||||
std::cout << "VERSION 1 -----------" << std::endl;
|
|
||||||
std::cout << "CPU time: " << CPUtime << std::endl;
|
|
||||||
std::cout << "GPU time: " << GPUtime1 << std::endl;
|
|
||||||
std::cout << "GPU is ";
|
|
||||||
if (time_ratio > 1)
|
|
||||||
std::cout << time_ratio << " times faster!" << std::endl;
|
|
||||||
else
|
|
||||||
std::cout << time_ratio << " times slower :(" << std::endl;
|
|
||||||
|
|
||||||
time_ratio = (CPUtime / GPUtime2);
|
// similar to the first trial, except that each element is done
|
||||||
std::cout << "\nVERSION 2 -----------" << std::endl;
|
// by one thread, instead of multiple elements per thread
|
||||||
std::cout << "CPU time: " << CPUtime << std::endl;
|
|
||||||
std::cout << "GPU time: " << GPUtime2 << std::endl;
|
|
||||||
std::cout << "GPU is ";
|
// start timer
|
||||||
if (time_ratio > 1)
|
double GPUtime3;
|
||||||
std::cout << time_ratio << " times faster!" << std::endl;
|
start_time = std::clock();
|
||||||
else
|
|
||||||
std::cout << time_ratio << " times slower :(" << std::endl;
|
cl::Buffer buffer_A3(context, CL_MEM_READ_WRITE, sizeof(int)*n);
|
||||||
|
cl::Buffer buffer_B3(context, CL_MEM_READ_WRITE, sizeof(int)*n);
|
||||||
|
cl::Buffer buffer_C3(context, CL_MEM_READ_WRITE, sizeof(int)*n);
|
||||||
|
cl::Buffer buffer_constants3(context, CL_MEM_READ_ONLY, sizeof(int)*2);
|
||||||
|
queue.enqueueWriteBuffer(buffer_A3, CL_TRUE, 0, sizeof(int)*n, A);
|
||||||
|
queue.enqueueWriteBuffer(buffer_B3, CL_TRUE, 0, sizeof(int)*n, B);
|
||||||
|
queue.enqueueWriteBuffer(buffer_constants3, CL_TRUE, 0, sizeof(int)*2, constants);
|
||||||
|
|
||||||
|
// run it
|
||||||
|
add_single.setArg(0, buffer_A3);
|
||||||
|
add_single.setArg(1, buffer_B3);
|
||||||
|
add_single.setArg(2, buffer_C3);
|
||||||
|
add_single.setArg(3, buffer_constants3);
|
||||||
|
queue.enqueueNDRangeKernel(add_single, cl::NDRange(n), cl::NDRange(32), cl::NullRange);
|
||||||
|
|
||||||
|
// end timer
|
||||||
|
queue.enqueueReadBuffer(buffer_C3, CL_TRUE, 0, sizeof(int)*n, C);
|
||||||
|
GPUtime3 = (std::clock() - start_time) / (double) CLOCKS_PER_SEC;
|
||||||
|
|
||||||
|
// let's compare!
|
||||||
|
compareResults(CPUtime, GPUtime1, 1);
|
||||||
|
compareResults(CPUtime, GPUtime2, 2);
|
||||||
|
compareResults(CPUtime, GPUtime3, 3);
|
||||||
return 0;
|
return 0;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|||||||
Reference in New Issue
Block a user