use system header files instead of local ones
This commit is contained in:
3
.gitignore
vendored
3
.gitignore
vendored
@@ -1,6 +1,3 @@
|
|||||||
# openCL C++ headers
|
|
||||||
CL/
|
|
||||||
|
|
||||||
# compiled files
|
# compiled files
|
||||||
*.out
|
*.out
|
||||||
|
|
||||||
|
|||||||
26
README.md
26
README.md
@@ -2,13 +2,29 @@
|
|||||||
here is my feeble attempt at learning OpenCL, please don't make fun of me too much :hamburger:
|
here is my feeble attempt at learning OpenCL, please don't make fun of me too much :hamburger:
|
||||||
|
|
||||||
## Configuration
|
## Configuration
|
||||||
This currently runs on OS X, and I'm using local header files instead of global header files because I'm unfamiliar with C++. Deal with it. Run the following in a terminal to set up:
|
This code uses OpenCL 1.1 on a NVIDIA GPU.
|
||||||
|
|
||||||
|
### Linux
|
||||||
|
(Only tested on Ubuntu). For NVIDIA GPUs, I've installed the following packages: `nvidia-346 nvidia-346-dev nvidia-346-uvm nvidia-libopencl1-346 nvidia-modprobe nvidia-opencl-icd-346 nvidia-settings`. Since the `opencl-headers` package in the main repository is for OpenCL 1.2, you can get the OpenCL 1.1 header files from [here](http://packages.ubuntu.com/precise/opencl-headers).
|
||||||
|
|
||||||
|
Then to compile:
|
||||||
|
|
||||||
```
|
```
|
||||||
git clone git@github.com:SaintDako/OpenCL-examples.git
|
g++ -std=c++0x main.cpp -o main.out -lOpenCL
|
||||||
cd OpenCL-examples
|
```
|
||||||
mkdir CL
|
|
||||||
curl https://www.khronos.org/registry/cl/api/1.2/cl.hpp -o CL/cl.hpp
|
### OS X
|
||||||
|
OpenCL is installed on OS X by default, but since this code uses the C++ bindings, you'll need to get that too. Get the [official C++ bindings from the OpenCL registr](https://www.khronos.org/registry/cl/api/1.1/cl.hpp) and copy it to the OpenCL framework directory, or do the following:
|
||||||
|
|
||||||
|
```
|
||||||
|
wget https://www.khronos.org/registry/cl/api/1.1/cl.hpp
|
||||||
|
sudo cp cl.hpp /System/Library/Frameworks/OpenCL.framework/Headers/
|
||||||
|
```
|
||||||
|
|
||||||
|
To compile:
|
||||||
|
|
||||||
|
```
|
||||||
|
clang++ -std=c++0x -framework OpenCL main.cpp -o main.out
|
||||||
```
|
```
|
||||||
|
|
||||||
## example 00
|
## example 00
|
||||||
|
|||||||
@@ -1,5 +1,9 @@
|
|||||||
#include <iostream>
|
#include <iostream>
|
||||||
#include "../CL/cl.hpp"
|
#ifdef __APPLE__
|
||||||
|
#include <OpenCL/cl.hpp>
|
||||||
|
#else
|
||||||
|
#include <CL/cl.hpp>
|
||||||
|
#endif
|
||||||
|
|
||||||
int main() {
|
int main() {
|
||||||
// get all platforms (drivers), e.g. NVIDIA
|
// get all platforms (drivers), e.g. NVIDIA
|
||||||
|
|||||||
@@ -1,16 +1,6 @@
|
|||||||
# Example 01
|
# Example 01
|
||||||
This example compares the timings of adding vectors on the CPU versus adding vectors on the GPU, the latter of which has different implementations.
|
This example compares the timings of adding vectors on the CPU versus adding vectors on the GPU, the latter of which has different implementations.
|
||||||
|
|
||||||
## Compiling
|
|
||||||
|
|
||||||
```
|
|
||||||
clang++ -std=c++0x -framework OpenCL main.cpp -o main.out
|
|
||||||
```
|
|
||||||
|
|
||||||
To ignore deprecation warnings, add the flag `-Wno-deprecated-declarations`.
|
|
||||||
|
|
||||||
Run from this directory, as a relative path is used for the OpenCL header file (for now).
|
|
||||||
|
|
||||||
## About
|
## About
|
||||||
The code runs the following implementations of adding large vectors (131072 elements; 8 * 32 * 512). The vectors are added together 10000 times.
|
The code runs the following implementations of adding large vectors (131072 elements; 8 * 32 * 512). The vectors are added together 10000 times.
|
||||||
|
|
||||||
|
|||||||
@@ -1,6 +1,10 @@
|
|||||||
#include <iostream>
|
#include <iostream>
|
||||||
#include <ctime>
|
#include <ctime>
|
||||||
#include "../CL/cl.hpp"
|
#ifdef __APPLE__
|
||||||
|
#include <OpenCL/cl.hpp>
|
||||||
|
#else
|
||||||
|
#include <CL/cl.hpp>
|
||||||
|
#endif
|
||||||
|
|
||||||
#define NUM_GLOBAL_WITEMS 1024
|
#define NUM_GLOBAL_WITEMS 1024
|
||||||
|
|
||||||
|
|||||||
@@ -1,7 +1,11 @@
|
|||||||
#include <iostream>
|
#include <iostream>
|
||||||
#include <algorithm>
|
#include <algorithm>
|
||||||
#include <iterator>
|
#include <iterator>
|
||||||
#include "../CL/cl.hpp"
|
#ifdef __APPLE__
|
||||||
|
#include <OpenCL/cl.hpp>
|
||||||
|
#else
|
||||||
|
#include <CL/cl.hpp>
|
||||||
|
#endif
|
||||||
|
|
||||||
using namespace std;
|
using namespace std;
|
||||||
using namespace cl;
|
using namespace cl;
|
||||||
@@ -25,7 +29,7 @@ Platform getPlatform() {
|
|||||||
}
|
}
|
||||||
|
|
||||||
|
|
||||||
Device getDevice(cl::Platform platform, int i, bool display=false) {
|
Device getDevice(Platform platform, int i, bool display=false) {
|
||||||
/* Returns the deviced specified by the index i on platform.
|
/* Returns the deviced specified by the index i on platform.
|
||||||
* If display is true, then all of the platforms are listed.
|
* If display is true, then all of the platforms are listed.
|
||||||
*/
|
*/
|
||||||
@@ -59,7 +63,6 @@ int main() {
|
|||||||
Context context({default_device});
|
Context context({default_device});
|
||||||
Program::Sources sources;
|
Program::Sources sources;
|
||||||
|
|
||||||
// calculates for each element; C = A + B
|
|
||||||
std::string kernel_code=
|
std::string kernel_code=
|
||||||
"void kernel multiply_by(global int* A, const int c) {"
|
"void kernel multiply_by(global int* A, const int c) {"
|
||||||
" A[get_global_id(0)] = c * A[get_global_id(0)];"
|
" A[get_global_id(0)] = c * A[get_global_id(0)];"
|
||||||
@@ -76,12 +79,12 @@ int main() {
|
|||||||
CommandQueue queue(context, default_device);
|
CommandQueue queue(context, default_device);
|
||||||
queue.enqueueWriteBuffer(buffer_A, CL_TRUE, 0, sizeof(int)*n, A);
|
queue.enqueueWriteBuffer(buffer_A, CL_TRUE, 0, sizeof(int)*n, A);
|
||||||
|
|
||||||
Kernel multiply_by = cl::Kernel(program, "multiply_by");
|
Kernel multiply_by = Kernel(program, "multiply_by");
|
||||||
multiply_by.setArg(0, buffer_A);
|
multiply_by.setArg(0, buffer_A);
|
||||||
|
|
||||||
for (int c=2; c<=c_max; c++) {
|
for (int c=2; c<=c_max; c++) {
|
||||||
multiply_by.setArg(1, c);
|
multiply_by.setArg(1, c);
|
||||||
queue.enqueueNDRangeKernel(multiply_by, cl::NullRange, cl::NDRange(n), cl::NDRange(32));
|
queue.enqueueNDRangeKernel(multiply_by, NullRange, NDRange(n), NDRange(32));
|
||||||
}
|
}
|
||||||
|
|
||||||
queue.enqueueReadBuffer(buffer_A, CL_TRUE, 0, sizeof(int)*n, B);
|
queue.enqueueReadBuffer(buffer_A, CL_TRUE, 0, sizeof(int)*n, B);
|
||||||
|
|||||||
Reference in New Issue
Block a user