use system header files instead of local ones

This commit is contained in:
Dakota St. Laurent
2015-07-07 17:14:07 -04:00
parent d954c321a0
commit 15d0ffaac2
6 changed files with 39 additions and 25 deletions

3
.gitignore vendored
View File

@@ -1,6 +1,3 @@
# openCL C++ headers
CL/
# compiled files # compiled files
*.out *.out

View File

@@ -2,13 +2,29 @@
here is my feeble attempt at learning OpenCL, please don't make fun of me too much :hamburger: here is my feeble attempt at learning OpenCL, please don't make fun of me too much :hamburger:
## Configuration ## Configuration
This currently runs on OS X, and I'm using local header files instead of global header files because I'm unfamiliar with C++. Deal with it. Run the following in a terminal to set up: This code uses OpenCL 1.1 on a NVIDIA GPU.
### Linux
(Only tested on Ubuntu). For NVIDIA GPUs, I've installed the following packages: `nvidia-346 nvidia-346-dev nvidia-346-uvm nvidia-libopencl1-346 nvidia-modprobe nvidia-opencl-icd-346 nvidia-settings`. Since the `opencl-headers` package in the main repository is for OpenCL 1.2, you can get the OpenCL 1.1 header files from [here](http://packages.ubuntu.com/precise/opencl-headers).
Then to compile:
``` ```
git clone git@github.com:SaintDako/OpenCL-examples.git g++ -std=c++0x main.cpp -o main.out -lOpenCL
cd OpenCL-examples ```
mkdir CL
curl https://www.khronos.org/registry/cl/api/1.2/cl.hpp -o CL/cl.hpp ### OS X
OpenCL is installed on OS X by default, but since this code uses the C++ bindings, you'll need to get that too. Get the [official C++ bindings from the OpenCL registr](https://www.khronos.org/registry/cl/api/1.1/cl.hpp) and copy it to the OpenCL framework directory, or do the following:
```
wget https://www.khronos.org/registry/cl/api/1.1/cl.hpp
sudo cp cl.hpp /System/Library/Frameworks/OpenCL.framework/Headers/
```
To compile:
```
clang++ -std=c++0x -framework OpenCL main.cpp -o main.out
``` ```
## example 00 ## example 00

View File

@@ -1,5 +1,9 @@
#include <iostream> #include <iostream>
#include "../CL/cl.hpp" #ifdef __APPLE__
#include <OpenCL/cl.hpp>
#else
#include <CL/cl.hpp>
#endif
int main() { int main() {
// get all platforms (drivers), e.g. NVIDIA // get all platforms (drivers), e.g. NVIDIA

View File

@@ -1,16 +1,6 @@
# Example 01 # Example 01
This example compares the timings of adding vectors on the CPU versus adding vectors on the GPU, the latter of which has different implementations. This example compares the timings of adding vectors on the CPU versus adding vectors on the GPU, the latter of which has different implementations.
## Compiling
```
clang++ -std=c++0x -framework OpenCL main.cpp -o main.out
```
To ignore deprecation warnings, add the flag `-Wno-deprecated-declarations`.
Run from this directory, as a relative path is used for the OpenCL header file (for now).
## About ## About
The code runs the following implementations of adding large vectors (131072 elements; 8 * 32 * 512). The vectors are added together 10000 times. The code runs the following implementations of adding large vectors (131072 elements; 8 * 32 * 512). The vectors are added together 10000 times.

View File

@@ -1,6 +1,10 @@
#include <iostream> #include <iostream>
#include <ctime> #include <ctime>
#include "../CL/cl.hpp" #ifdef __APPLE__
#include <OpenCL/cl.hpp>
#else
#include <CL/cl.hpp>
#endif
#define NUM_GLOBAL_WITEMS 1024 #define NUM_GLOBAL_WITEMS 1024

View File

@@ -1,7 +1,11 @@
#include <iostream> #include <iostream>
#include <algorithm> #include <algorithm>
#include <iterator> #include <iterator>
#include "../CL/cl.hpp" #ifdef __APPLE__
#include <OpenCL/cl.hpp>
#else
#include <CL/cl.hpp>
#endif
using namespace std; using namespace std;
using namespace cl; using namespace cl;
@@ -25,7 +29,7 @@ Platform getPlatform() {
} }
Device getDevice(cl::Platform platform, int i, bool display=false) { Device getDevice(Platform platform, int i, bool display=false) {
/* Returns the deviced specified by the index i on platform. /* Returns the deviced specified by the index i on platform.
* If display is true, then all of the platforms are listed. * If display is true, then all of the platforms are listed.
*/ */
@@ -59,7 +63,6 @@ int main() {
Context context({default_device}); Context context({default_device});
Program::Sources sources; Program::Sources sources;
// calculates for each element; C = A + B
std::string kernel_code= std::string kernel_code=
"void kernel multiply_by(global int* A, const int c) {" "void kernel multiply_by(global int* A, const int c) {"
" A[get_global_id(0)] = c * A[get_global_id(0)];" " A[get_global_id(0)] = c * A[get_global_id(0)];"
@@ -76,12 +79,12 @@ int main() {
CommandQueue queue(context, default_device); CommandQueue queue(context, default_device);
queue.enqueueWriteBuffer(buffer_A, CL_TRUE, 0, sizeof(int)*n, A); queue.enqueueWriteBuffer(buffer_A, CL_TRUE, 0, sizeof(int)*n, A);
Kernel multiply_by = cl::Kernel(program, "multiply_by"); Kernel multiply_by = Kernel(program, "multiply_by");
multiply_by.setArg(0, buffer_A); multiply_by.setArg(0, buffer_A);
for (int c=2; c<=c_max; c++) { for (int c=2; c<=c_max; c++) {
multiply_by.setArg(1, c); multiply_by.setArg(1, c);
queue.enqueueNDRangeKernel(multiply_by, cl::NullRange, cl::NDRange(n), cl::NDRange(32)); queue.enqueueNDRangeKernel(multiply_by, NullRange, NDRange(n), NDRange(32));
} }
queue.enqueueReadBuffer(buffer_A, CL_TRUE, 0, sizeof(int)*n, B); queue.enqueueReadBuffer(buffer_A, CL_TRUE, 0, sizeof(int)*n, B);