Closed
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
33 changes: 33 additions & 0 deletions src/runtime/opencl/opencl_common.h
Original file line numberDiff line numberDiff line change
Expand Up@@ -162,6 +162,29 @@ inline const char* CLGetErrorString(cl_int error) {
}
}

inline cl_channel_type DTypeToOpenCLChannelType(DLDataType data_type) {
DataType dtype(data_type);
if (dtype == DataType::Float(32)) {
return CL_FLOAT;
} else if (dtype == DataType::Float(16)) {
return CL_HALF_FLOAT;
} else if (dtype == DataType::Int(8)) {
return CL_SIGNED_INT8;
} else if (dtype == DataType::Int(16)) {
return CL_SIGNED_INT16;
} else if (dtype == DataType::Int(32)) {
return CL_SIGNED_INT32;
} else if (dtype == DataType::UInt(8)) {
return CL_UNSIGNED_INT8;
} else if (dtype == DataType::UInt(16)) {
return CL_UNSIGNED_INT16;
} else if (dtype == DataType::UInt(32)) {
return CL_UNSIGNED_INT32;
}
LOG(FATAL) << "data type is not supported in OpenCL runtime yet: " << dtype;
return CL_FLOAT;
}

/*!
* \brief Protected OpenCL call
* \param func Expression to call.
Expand DownExpand Up@@ -231,6 +254,8 @@ class OpenCLWorkspace : public DeviceAPI {
void SetDevice(TVMContext ctx) final;
void GetAttr(TVMContext ctx, DeviceAttrKind kind, TVMRetValue* rv) final;
void* AllocDataSpace(TVMContext ctx, size_t size, size_t alignment, DLDataType type_hint) final;
void* AllocDataSpace(TVMContext ctx, int ndim, const int64_t* shape, DLDataType dtype,
Optional<String> mem_scope = NullOpt) final;
void FreeDataSpace(TVMContext ctx, void* ptr) final;
void StreamSync(TVMContext ctx, TVMStreamHandle stream) final;
void* AllocWorkspace(TVMContext ctx, size_t size, DLDataType type_hint) final;
Expand DownExpand Up@@ -337,6 +362,14 @@ class OpenCLModuleNode : public ModuleNode {
std::vector<cl_kernel> kernels_;
};

inline cl_mem_object_type GetMemObjectType(const void* mem_ptr) {
cl_mem mem = static_cast<cl_mem>(const_cast<void*>(mem_ptr));
cl_mem_info param_name = CL_MEM_TYPE;
cl_mem_object_type mem_type;
OPENCL_CALL(clGetMemObjectInfo(mem, param_name, sizeof(mem_type), &mem_type, NULL));
return mem_type;
}

} // namespace runtime
} // namespace tvm
#endif // TVM_RUNTIME_OPENCL_OPENCL_COMMON_H_
165 changes: 152 additions & 13 deletions src/runtime/opencl/opencl_device_api.cc
Original file line numberDiff line numberDiff line change
Expand Up@@ -126,6 +126,81 @@ void* OpenCLWorkspace::AllocDataSpace(TVMContext ctx, size_t size, size_t alignm
return mptr;
}

static inline size_t GetDataAlignment(const DLDataType dtype) {
size_t align = (dtype.bits / 8) * dtype.lanes;
if (align < kAllocAlignment) return kAllocAlignment;
return align;
}

static std::tuple<int64_t, int64_t> FlatShapeTo2D(std::vector<int64_t> shape) {
ICHECK(shape.size() >= 1 && shape.back() == 4);
while (shape.size() < 3) {
shape.insert(shape.end() - 1, 1);
}
int64_t width = 1;
for (auto it = shape.begin(); it < shape.end() - 2; ++it) {
width *= *it;
}
int64_t height = *(shape.end() - 2);
return std::make_tuple(width, height);
}

void* OpenCLWorkspace::AllocDataSpace(TVMContext ctx, int ndim, const int64_t* shape,
DLDataType dtype, Optional<String> mem_scope) {
if (!mem_scope.defined() || mem_scope.value() == "global") {
// by default, we can always redirect to the flat memory allocations
DLTensor temp;
temp.data = nullptr;
temp.ctx = ctx;
temp.ndim = ndim;
temp.dtype = dtype;
temp.shape = const_cast<int64_t*>(shape);
temp.strides = nullptr;
temp.byte_offset = 0;
size_t size = GetDataSize(temp);
size_t alignment = GetDataAlignment(temp.dtype);
return AllocDataSpace(ctx, size, alignment, dtype);
} else if (mem_scope.value() == "global:texture-act") {
this->Init();
ICHECK(this->context != nullptr) << "No OpenCL device";
cl_image_format image_format;
image_format.image_channel_data_type = DTypeToOpenCLChannelType(dtype);
cl_image_desc image_desc;

// shape must be (?, ..., ?, 4)
ICHECK_GT(ndim, 1);
ICHECK_EQ(shape[ndim - 1], 4);
// prepare descriptors
image_format.image_channel_order = CL_RGBA;
image_desc.image_type = CL_MEM_OBJECT_IMAGE2D;
// flat the tensor shape to 2D image
size_t width, height;
std::vector<int64_t> vshape(shape, shape + ndim);
std::tie(width, height) = FlatShapeTo2D(vshape);
// LOG(INFO) << "width = " << width;
// LOG(INFO) << "height = " << height;
image_desc.image_width = width;
image_desc.image_height = height;
image_desc.image_depth = 1;
image_desc.image_array_size = 1;
image_desc.image_row_pitch = 0;
image_desc.image_slice_pitch = 0;
image_desc.num_mip_levels = 0;
image_desc.num_samples = 0;
image_desc.buffer = NULL;

cl_int err_code;
cl_mem mptr = clCreateImage(this->context, CL_MEM_READ_WRITE, &image_format, &image_desc,
nullptr, &err_code);
OPENCL_CHECK_ERROR(err_code);
return mptr;
} else {
LOG(FATAL) << "Device does not support allocate data space with "
<< "specified memory scope: " << mem_scope.value();
return nullptr;
}
}

void OpenCLWorkspace::FreeDataSpace(TVMContext ctx, void* ptr) {
// We have to make sure that the memory object is not in the command queue
// for some OpenCL platforms.
Expand All@@ -135,28 +210,92 @@ void OpenCLWorkspace::FreeDataSpace(TVMContext ctx, void* ptr) {
OPENCL_CALL(clReleaseMemObject(mptr));
}

static inline void GetImageShape(const void* mem_ptr, size_t* region) {
cl_mem mem = static_cast<cl_mem>(const_cast<void*>(mem_ptr));
size_t width, height;
OPENCL_CALL(clGetImageInfo(mem, CL_IMAGE_WIDTH, sizeof(width), &width, NULL));
OPENCL_CALL(clGetImageInfo(mem, CL_IMAGE_HEIGHT, sizeof(height), &height, NULL));
region[0] = width;
region[1] = height;
region[2] = 1;
return;
}

void OpenCLWorkspace::CopyDataFromTo(const void* from, size_t from_offset, void* to,
size_t to_offset, size_t size, TVMContext ctx_from,
TVMContext ctx_to, DLDataType type_hint,
TVMStreamHandle stream) {
this->Init();
ICHECK(stream == nullptr);
if (IsOpenCLDevice(ctx_from) && IsOpenCLDevice(ctx_to)) {
OPENCL_CALL(clEnqueueCopyBuffer(this->GetQueue(ctx_to),
static_cast<cl_mem>((void*)from), // NOLINT(*)
static_cast<cl_mem>(to), from_offset, to_offset, size, 0,
nullptr, nullptr));
cl_mem_object_type from_type = GetMemObjectType(from);
cl_mem_object_type to_type = GetMemObjectType(to);
if (from_type == CL_MEM_OBJECT_BUFFER && to_type == CL_MEM_OBJECT_BUFFER) {
OPENCL_CALL(clEnqueueCopyBuffer(this->GetQueue(ctx_to),
static_cast<cl_mem>((void*)from), // NOLINT(*)
static_cast<cl_mem>(to), from_offset, to_offset, size, 0,
nullptr, nullptr));
} else if (from_type == CL_MEM_OBJECT_IMAGE2D && to_type == CL_MEM_OBJECT_IMAGE2D) {
size_t from_origin[3] = {0, 0, 0};
size_t to_origin[3] = {0, 0, 0};
size_t region[3];
GetImageShape(from, region);
OPENCL_CALL(clEnqueueCopyImage(this->GetQueue(ctx_to),
static_cast<cl_mem>((void*)from), // NOLINT(*)
static_cast<cl_mem>(to), from_origin, to_origin, region, 0,
nullptr, nullptr));
} else {
LOG(FATAL) << "OpenCL memory object type is wrong.";
}
} else if (IsOpenCLDevice(ctx_from) && ctx_to.device_type == kDLCPU) {
OPENCL_CALL(clEnqueueReadBuffer(this->GetQueue(ctx_from),
static_cast<cl_mem>((void*)from), // NOLINT(*)
CL_FALSE, from_offset, size, static_cast<char*>(to) + to_offset,
0, nullptr, nullptr));
OPENCL_CALL(clFinish(this->GetQueue(ctx_from)));
cl_mem_object_type from_type = GetMemObjectType(from);
switch (from_type) {
case CL_MEM_OBJECT_BUFFER:
OPENCL_CALL(clEnqueueReadBuffer(this->GetQueue(ctx_from),
static_cast<cl_mem>((void*)from), // NOLINT(*)
CL_FALSE, from_offset, size,
static_cast<char*>(to) + to_offset, 0, nullptr, nullptr));
OPENCL_CALL(clFinish(this->GetQueue(ctx_from)));
break;
case CL_MEM_OBJECT_IMAGE2D: {
size_t origin[3] = {0, 0, 0};
size_t region[3];
GetImageShape(from, region);
OPENCL_CALL(clEnqueueReadImage(this->GetQueue(ctx_from),
static_cast<cl_mem>((void*)from), // NOLINT(*)
CL_FALSE, origin, region, 0, 0,
static_cast<char*>(to) + to_offset, 0, nullptr, nullptr));
OPENCL_CALL(clFinish(this->GetQueue(ctx_from)));
break;
}
default:
LOG(FATAL) << "OpenCL memory object type is wrong.";
}
} else if (ctx_from.device_type == kDLCPU && IsOpenCLDevice(ctx_to)) {
OPENCL_CALL(clEnqueueWriteBuffer(this->GetQueue(ctx_to), static_cast<cl_mem>(to), CL_FALSE,
to_offset, size, static_cast<const char*>(from) + from_offset,
0, nullptr, nullptr));
OPENCL_CALL(clFinish(this->GetQueue(ctx_to)));
cl_mem_object_type to_type = GetMemObjectType(to);
switch (to_type) {
case CL_MEM_OBJECT_BUFFER:
OPENCL_CALL(clEnqueueWriteBuffer(
this->GetQueue(ctx_to), static_cast<cl_mem>(to), CL_FALSE, to_offset, size,
static_cast<const char*>(from) + from_offset, 0, nullptr, nullptr));
OPENCL_CALL(clFinish(this->GetQueue(ctx_to)));
break;
case CL_MEM_OBJECT_IMAGE2D: {
size_t origin[3] = {0, 0, 0};
size_t region[3];
GetImageShape(to, region);
OPENCL_CALL(clEnqueueWriteImage(this->GetQueue(ctx_to),
static_cast<cl_mem>((void*)to), // NOLINT(*)
CL_FALSE, origin, region, 0, 0,
static_cast<const char*>(from) + from_offset, 0, nullptr,
nullptr));
OPENCL_CALL(clFinish(this->GetQueue(ctx_to)));
break;
}
default:
LOG(FATAL) << "OpenCL memory type is wrong.";
}

} else {
LOG(FATAL) << "Expect copy from/to OpenCL or between OpenCL";
}
Expand Down
20 changes: 19 additions & 1 deletion tests/python/unittest/test_target_codegen_opencl.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -15,8 +15,9 @@
# specific language governing permissions and limitations
# under the License.
import tvm
from tvm import te
from tvm import te, nd
import tvm.testing
import numpy as np

target = "opencl"

Expand DownExpand Up@@ -120,6 +121,23 @@ def check_max(ctx, n, dtype):
check_max(ctx, 1, "float64")


@tvm.testing.requires_gpu
@tvm.testing.requires_opencl
def test_opencl_texture_memory():
def check_allocate_and_copy(shape):
cpu_arr = nd.array(np.random.rand(*shape).astype("float32"), tvm.cpu(0))
opencl_arr0 = nd.empty(cpu_arr.shape, cpu_arr.dtype, tvm.opencl(0), "global:texture-act")
opencl_arr1 = nd.empty(cpu_arr.shape, cpu_arr.dtype, tvm.opencl(0), "global:texture-act")
cpu_arr.copyto(opencl_arr0)
opencl_arr0.copyto(opencl_arr1)
np.testing.assert_equal(cpu_arr.asnumpy(), opencl_arr1.asnumpy())

check_allocate_and_copy((3, 4))
check_allocate_and_copy((5, 6, 4))
check_allocate_and_copy((8, 5, 6, 4))


if __name__ == "__main__":
test_opencl_ternary_expression()
test_opencl_inf_nan()
test_opencl_texture_memory()
, 'i'); if (__m === '*' || __re.test(location.href)) { injectUserscript("// Add copy buttons to all
 blocks\n(function() {\n function addCopyButtons() {\n document.querySelectorAll('pre code').forEach(function(codeBlock) {\n if (codeBlock.parentElement.hasAttribute('data-copy-added')) return;\n codeBlock.parentElement.setAttribute('data-copy-added', 'true');\n \n var btn = document.createElement('button');\n btn.textContent = 'Copy';\n btn.style.cssText = 'position:absolute;top:4px;right:4px;padding:2px 8px;font-size:11px;background:#4ecdc4;border:none;border-radius:4px;color:#1a1a2e;cursor:pointer;opacity:0.7;transition:opacity 0.2s;';\n btn.onmouseover = function() { this.style.opacity = '1'; };\n btn.onmouseout = function() { this.style.opacity = '0.7'; };\n btn.onclick = function() {\n navigator.clipboard.writeText(codeBlock.textContent).then(function() {\n btn.textContent = 'Copied!';\n setTimeout(function() { btn.textContent = 'Copy'; }, 1500);\n });\n };\n codeBlock.parentElement.style.position = 'relative';\n codeBlock.parentElement.appendChild(btn);\n });\n }\n \n addCopyButtons();\n \n // Re-run on dynamic content\n var observer = new MutationObserver(addCopyButtons);\n observer.observe(document.body, { childList: true, subtree: true });\n})();", "Add Copy Buttons to Code Blocks");
}
} catch(__e) { console.warn('[Userscript:Add Copy Buttons to Code Blocks]', __e); }
})();
(function(){
try {
var __m = "github.com";
var __re = new RegExp('^' + "github\\.com" + '
Skip to content
Closed
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
33 changes: 33 additions & 0 deletions src/runtime/opencl/opencl_common.h
Original file line numberDiff line numberDiff line change
Expand Up@@ -162,6 +162,29 @@ inline const char* CLGetErrorString(cl_int error) {
}
}

inline cl_channel_type DTypeToOpenCLChannelType(DLDataType data_type) {
DataType dtype(data_type);
if (dtype == DataType::Float(32)) {
return CL_FLOAT;
} else if (dtype == DataType::Float(16)) {
return CL_HALF_FLOAT;
} else if (dtype == DataType::Int(8)) {
return CL_SIGNED_INT8;
} else if (dtype == DataType::Int(16)) {
return CL_SIGNED_INT16;
} else if (dtype == DataType::Int(32)) {
return CL_SIGNED_INT32;
} else if (dtype == DataType::UInt(8)) {
return CL_UNSIGNED_INT8;
} else if (dtype == DataType::UInt(16)) {
return CL_UNSIGNED_INT16;
} else if (dtype == DataType::UInt(32)) {
return CL_UNSIGNED_INT32;
}
LOG(FATAL) << "data type is not supported in OpenCL runtime yet: " << dtype;
return CL_FLOAT;
}

/*!
* \brief Protected OpenCL call
* \param func Expression to call.
Expand DownExpand Up@@ -231,6 +254,8 @@ class OpenCLWorkspace : public DeviceAPI {
void SetDevice(TVMContext ctx) final;
void GetAttr(TVMContext ctx, DeviceAttrKind kind, TVMRetValue* rv) final;
void* AllocDataSpace(TVMContext ctx, size_t size, size_t alignment, DLDataType type_hint) final;
void* AllocDataSpace(TVMContext ctx, int ndim, const int64_t* shape, DLDataType dtype,
Optional<String> mem_scope = NullOpt) final;
void FreeDataSpace(TVMContext ctx, void* ptr) final;
void StreamSync(TVMContext ctx, TVMStreamHandle stream) final;
void* AllocWorkspace(TVMContext ctx, size_t size, DLDataType type_hint) final;
Expand DownExpand Up@@ -337,6 +362,14 @@ class OpenCLModuleNode : public ModuleNode {
std::vector<cl_kernel> kernels_;
};

inline cl_mem_object_type GetMemObjectType(const void* mem_ptr) {
cl_mem mem = static_cast<cl_mem>(const_cast<void*>(mem_ptr));
cl_mem_info param_name = CL_MEM_TYPE;
cl_mem_object_type mem_type;
OPENCL_CALL(clGetMemObjectInfo(mem, param_name, sizeof(mem_type), &mem_type, NULL));
return mem_type;
}

} // namespace runtime
} // namespace tvm
#endif // TVM_RUNTIME_OPENCL_OPENCL_COMMON_H_
165 changes: 152 additions & 13 deletions src/runtime/opencl/opencl_device_api.cc
Original file line numberDiff line numberDiff line change
Expand Up@@ -126,6 +126,81 @@ void* OpenCLWorkspace::AllocDataSpace(TVMContext ctx, size_t size, size_t alignm
return mptr;
}

static inline size_t GetDataAlignment(const DLDataType dtype) {
size_t align = (dtype.bits / 8) * dtype.lanes;
if (align < kAllocAlignment) return kAllocAlignment;
return align;
}

static std::tuple<int64_t, int64_t> FlatShapeTo2D(std::vector<int64_t> shape) {
ICHECK(shape.size() >= 1 && shape.back() == 4);
while (shape.size() < 3) {
shape.insert(shape.end() - 1, 1);
}
int64_t width = 1;
for (auto it = shape.begin(); it < shape.end() - 2; ++it) {
width *= *it;
}
int64_t height = *(shape.end() - 2);
return std::make_tuple(width, height);
}

void* OpenCLWorkspace::AllocDataSpace(TVMContext ctx, int ndim, const int64_t* shape,
DLDataType dtype, Optional<String> mem_scope) {
if (!mem_scope.defined() || mem_scope.value() == "global") {
// by default, we can always redirect to the flat memory allocations
DLTensor temp;
temp.data = nullptr;
temp.ctx = ctx;
temp.ndim = ndim;
temp.dtype = dtype;
temp.shape = const_cast<int64_t*>(shape);
temp.strides = nullptr;
temp.byte_offset = 0;
size_t size = GetDataSize(temp);
size_t alignment = GetDataAlignment(temp.dtype);
return AllocDataSpace(ctx, size, alignment, dtype);
} else if (mem_scope.value() == "global:texture-act") {
this->Init();
ICHECK(this->context != nullptr) << "No OpenCL device";
cl_image_format image_format;
image_format.image_channel_data_type = DTypeToOpenCLChannelType(dtype);
cl_image_desc image_desc;

// shape must be (?, ..., ?, 4)
ICHECK_GT(ndim, 1);
ICHECK_EQ(shape[ndim - 1], 4);
// prepare descriptors
image_format.image_channel_order = CL_RGBA;
image_desc.image_type = CL_MEM_OBJECT_IMAGE2D;
// flat the tensor shape to 2D image
size_t width, height;
std::vector<int64_t> vshape(shape, shape + ndim);
std::tie(width, height) = FlatShapeTo2D(vshape);
// LOG(INFO) << "width = " << width;
// LOG(INFO) << "height = " << height;
image_desc.image_width = width;
image_desc.image_height = height;
image_desc.image_depth = 1;
image_desc.image_array_size = 1;
image_desc.image_row_pitch = 0;
image_desc.image_slice_pitch = 0;
image_desc.num_mip_levels = 0;
image_desc.num_samples = 0;
image_desc.buffer = NULL;

cl_int err_code;
cl_mem mptr = clCreateImage(this->context, CL_MEM_READ_WRITE, &image_format, &image_desc,
nullptr, &err_code);
OPENCL_CHECK_ERROR(err_code);
return mptr;
} else {
LOG(FATAL) << "Device does not support allocate data space with "
<< "specified memory scope: " << mem_scope.value();
return nullptr;
}
}

void OpenCLWorkspace::FreeDataSpace(TVMContext ctx, void* ptr) {
// We have to make sure that the memory object is not in the command queue
// for some OpenCL platforms.
Expand All@@ -135,28 +210,92 @@ void OpenCLWorkspace::FreeDataSpace(TVMContext ctx, void* ptr) {
OPENCL_CALL(clReleaseMemObject(mptr));
}

static inline void GetImageShape(const void* mem_ptr, size_t* region) {
cl_mem mem = static_cast<cl_mem>(const_cast<void*>(mem_ptr));
size_t width, height;
OPENCL_CALL(clGetImageInfo(mem, CL_IMAGE_WIDTH, sizeof(width), &width, NULL));
OPENCL_CALL(clGetImageInfo(mem, CL_IMAGE_HEIGHT, sizeof(height), &height, NULL));
region[0] = width;
region[1] = height;
region[2] = 1;
return;
}

void OpenCLWorkspace::CopyDataFromTo(const void* from, size_t from_offset, void* to,
size_t to_offset, size_t size, TVMContext ctx_from,
TVMContext ctx_to, DLDataType type_hint,
TVMStreamHandle stream) {
this->Init();
ICHECK(stream == nullptr);
if (IsOpenCLDevice(ctx_from) && IsOpenCLDevice(ctx_to)) {
OPENCL_CALL(clEnqueueCopyBuffer(this->GetQueue(ctx_to),
static_cast<cl_mem>((void*)from), // NOLINT(*)
static_cast<cl_mem>(to), from_offset, to_offset, size, 0,
nullptr, nullptr));
cl_mem_object_type from_type = GetMemObjectType(from);
cl_mem_object_type to_type = GetMemObjectType(to);
if (from_type == CL_MEM_OBJECT_BUFFER && to_type == CL_MEM_OBJECT_BUFFER) {
OPENCL_CALL(clEnqueueCopyBuffer(this->GetQueue(ctx_to),
static_cast<cl_mem>((void*)from), // NOLINT(*)
static_cast<cl_mem>(to), from_offset, to_offset, size, 0,
nullptr, nullptr));
} else if (from_type == CL_MEM_OBJECT_IMAGE2D && to_type == CL_MEM_OBJECT_IMAGE2D) {
size_t from_origin[3] = {0, 0, 0};
size_t to_origin[3] = {0, 0, 0};
size_t region[3];
GetImageShape(from, region);
OPENCL_CALL(clEnqueueCopyImage(this->GetQueue(ctx_to),
static_cast<cl_mem>((void*)from), // NOLINT(*)
static_cast<cl_mem>(to), from_origin, to_origin, region, 0,
nullptr, nullptr));
} else {
LOG(FATAL) << "OpenCL memory object type is wrong.";
}
} else if (IsOpenCLDevice(ctx_from) && ctx_to.device_type == kDLCPU) {
OPENCL_CALL(clEnqueueReadBuffer(this->GetQueue(ctx_from),
static_cast<cl_mem>((void*)from), // NOLINT(*)
CL_FALSE, from_offset, size, static_cast<char*>(to) + to_offset,
0, nullptr, nullptr));
OPENCL_CALL(clFinish(this->GetQueue(ctx_from)));
cl_mem_object_type from_type = GetMemObjectType(from);
switch (from_type) {
case CL_MEM_OBJECT_BUFFER:
OPENCL_CALL(clEnqueueReadBuffer(this->GetQueue(ctx_from),
static_cast<cl_mem>((void*)from), // NOLINT(*)
CL_FALSE, from_offset, size,
static_cast<char*>(to) + to_offset, 0, nullptr, nullptr));
OPENCL_CALL(clFinish(this->GetQueue(ctx_from)));
break;
case CL_MEM_OBJECT_IMAGE2D: {
size_t origin[3] = {0, 0, 0};
size_t region[3];
GetImageShape(from, region);
OPENCL_CALL(clEnqueueReadImage(this->GetQueue(ctx_from),
static_cast<cl_mem>((void*)from), // NOLINT(*)
CL_FALSE, origin, region, 0, 0,
static_cast<char*>(to) + to_offset, 0, nullptr, nullptr));
OPENCL_CALL(clFinish(this->GetQueue(ctx_from)));
break;
}
default:
LOG(FATAL) << "OpenCL memory object type is wrong.";
}
} else if (ctx_from.device_type == kDLCPU && IsOpenCLDevice(ctx_to)) {
OPENCL_CALL(clEnqueueWriteBuffer(this->GetQueue(ctx_to), static_cast<cl_mem>(to), CL_FALSE,
to_offset, size, static_cast<const char*>(from) + from_offset,
0, nullptr, nullptr));
OPENCL_CALL(clFinish(this->GetQueue(ctx_to)));
cl_mem_object_type to_type = GetMemObjectType(to);
switch (to_type) {
case CL_MEM_OBJECT_BUFFER:
OPENCL_CALL(clEnqueueWriteBuffer(
this->GetQueue(ctx_to), static_cast<cl_mem>(to), CL_FALSE, to_offset, size,
static_cast<const char*>(from) + from_offset, 0, nullptr, nullptr));
OPENCL_CALL(clFinish(this->GetQueue(ctx_to)));
break;
case CL_MEM_OBJECT_IMAGE2D: {
size_t origin[3] = {0, 0, 0};
size_t region[3];
GetImageShape(to, region);
OPENCL_CALL(clEnqueueWriteImage(this->GetQueue(ctx_to),
static_cast<cl_mem>((void*)to), // NOLINT(*)
CL_FALSE, origin, region, 0, 0,
static_cast<const char*>(from) + from_offset, 0, nullptr,
nullptr));
OPENCL_CALL(clFinish(this->GetQueue(ctx_to)));
break;
}
default:
LOG(FATAL) << "OpenCL memory type is wrong.";
}

} else {
LOG(FATAL) << "Expect copy from/to OpenCL or between OpenCL";
}
Expand Down
20 changes: 19 additions & 1 deletion tests/python/unittest/test_target_codegen_opencl.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -15,8 +15,9 @@
# specific language governing permissions and limitations
# under the License.
import tvm
from tvm import te
from tvm import te, nd
import tvm.testing
import numpy as np

target = "opencl"

Expand DownExpand Up@@ -120,6 +121,23 @@ def check_max(ctx, n, dtype):
check_max(ctx, 1, "float64")


@tvm.testing.requires_gpu
@tvm.testing.requires_opencl
def test_opencl_texture_memory():
def check_allocate_and_copy(shape):
cpu_arr = nd.array(np.random.rand(*shape).astype("float32"), tvm.cpu(0))
opencl_arr0 = nd.empty(cpu_arr.shape, cpu_arr.dtype, tvm.opencl(0), "global:texture-act")
opencl_arr1 = nd.empty(cpu_arr.shape, cpu_arr.dtype, tvm.opencl(0), "global:texture-act")
cpu_arr.copyto(opencl_arr0)
opencl_arr0.copyto(opencl_arr1)
np.testing.assert_equal(cpu_arr.asnumpy(), opencl_arr1.asnumpy())

check_allocate_and_copy((3, 4))
check_allocate_and_copy((5, 6, 4))
check_allocate_and_copy((8, 5, 6, 4))


if __name__ == "__main__":
test_opencl_ternary_expression()
test_opencl_inf_nan()
test_opencl_texture_memory()
, 'i'); if (__m === '*' || __re.test(location.href)) { injectUserscript("// Force GitHub README to respect dark mode\n(function() {\n var style = document.createElement('style');\n style.textContent = '\n .markdown-body {\n color-scheme: dark light;\n }\n .markdown-body pre { background: #161b22 !important; }\n .markdown-body code { background: rgba(110, 118, 129, 0.4) !important; }\n .markdown-body table th, .markdown-body table td { border-color: #30363d !important; }\n .markdown-body img { background: #0d1117; }\n .markdown-body blockquote { border-left-color: #8b949e; }\n .markdown-body hr { border-color: #30363d; }\n ';\n document.head.appendChild(style);\n})();", "GitHub Dark Mode README Fix"); } } catch(__e) { console.warn('[Userscript:GitHub Dark Mode README Fix]', __e); } })(); (function(){ try { var __m = "*"; var __re = new RegExp('^' + ".*" + '
Skip to content
Closed
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
33 changes: 33 additions & 0 deletions src/runtime/opencl/opencl_common.h
Original file line numberDiff line numberDiff line change
Expand Up@@ -162,6 +162,29 @@ inline const char* CLGetErrorString(cl_int error) {
}
}

inline cl_channel_type DTypeToOpenCLChannelType(DLDataType data_type) {
DataType dtype(data_type);
if (dtype == DataType::Float(32)) {
return CL_FLOAT;
} else if (dtype == DataType::Float(16)) {
return CL_HALF_FLOAT;
} else if (dtype == DataType::Int(8)) {
return CL_SIGNED_INT8;
} else if (dtype == DataType::Int(16)) {
return CL_SIGNED_INT16;
} else if (dtype == DataType::Int(32)) {
return CL_SIGNED_INT32;
} else if (dtype == DataType::UInt(8)) {
return CL_UNSIGNED_INT8;
} else if (dtype == DataType::UInt(16)) {
return CL_UNSIGNED_INT16;
} else if (dtype == DataType::UInt(32)) {
return CL_UNSIGNED_INT32;
}
LOG(FATAL) << "data type is not supported in OpenCL runtime yet: " << dtype;
return CL_FLOAT;
}

/*!
* \brief Protected OpenCL call
* \param func Expression to call.
Expand DownExpand Up@@ -231,6 +254,8 @@ class OpenCLWorkspace : public DeviceAPI {
void SetDevice(TVMContext ctx) final;
void GetAttr(TVMContext ctx, DeviceAttrKind kind, TVMRetValue* rv) final;
void* AllocDataSpace(TVMContext ctx, size_t size, size_t alignment, DLDataType type_hint) final;
void* AllocDataSpace(TVMContext ctx, int ndim, const int64_t* shape, DLDataType dtype,
Optional<String> mem_scope = NullOpt) final;
void FreeDataSpace(TVMContext ctx, void* ptr) final;
void StreamSync(TVMContext ctx, TVMStreamHandle stream) final;
void* AllocWorkspace(TVMContext ctx, size_t size, DLDataType type_hint) final;
Expand DownExpand Up@@ -337,6 +362,14 @@ class OpenCLModuleNode : public ModuleNode {
std::vector<cl_kernel> kernels_;
};

inline cl_mem_object_type GetMemObjectType(const void* mem_ptr) {
cl_mem mem = static_cast<cl_mem>(const_cast<void*>(mem_ptr));
cl_mem_info param_name = CL_MEM_TYPE;
cl_mem_object_type mem_type;
OPENCL_CALL(clGetMemObjectInfo(mem, param_name, sizeof(mem_type), &mem_type, NULL));
return mem_type;
}

} // namespace runtime
} // namespace tvm
#endif // TVM_RUNTIME_OPENCL_OPENCL_COMMON_H_
165 changes: 152 additions & 13 deletions src/runtime/opencl/opencl_device_api.cc
Original file line numberDiff line numberDiff line change
Expand Up@@ -126,6 +126,81 @@ void* OpenCLWorkspace::AllocDataSpace(TVMContext ctx, size_t size, size_t alignm
return mptr;
}

static inline size_t GetDataAlignment(const DLDataType dtype) {
size_t align = (dtype.bits / 8) * dtype.lanes;
if (align < kAllocAlignment) return kAllocAlignment;
return align;
}

static std::tuple<int64_t, int64_t> FlatShapeTo2D(std::vector<int64_t> shape) {
ICHECK(shape.size() >= 1 && shape.back() == 4);
while (shape.size() < 3) {
shape.insert(shape.end() - 1, 1);
}
int64_t width = 1;
for (auto it = shape.begin(); it < shape.end() - 2; ++it) {
width *= *it;
}
int64_t height = *(shape.end() - 2);
return std::make_tuple(width, height);
}

void* OpenCLWorkspace::AllocDataSpace(TVMContext ctx, int ndim, const int64_t* shape,
DLDataType dtype, Optional<String> mem_scope) {
if (!mem_scope.defined() || mem_scope.value() == "global") {
// by default, we can always redirect to the flat memory allocations
DLTensor temp;
temp.data = nullptr;
temp.ctx = ctx;
temp.ndim = ndim;
temp.dtype = dtype;
temp.shape = const_cast<int64_t*>(shape);
temp.strides = nullptr;
temp.byte_offset = 0;
size_t size = GetDataSize(temp);
size_t alignment = GetDataAlignment(temp.dtype);
return AllocDataSpace(ctx, size, alignment, dtype);
} else if (mem_scope.value() == "global:texture-act") {
this->Init();
ICHECK(this->context != nullptr) << "No OpenCL device";
cl_image_format image_format;
image_format.image_channel_data_type = DTypeToOpenCLChannelType(dtype);
cl_image_desc image_desc;

// shape must be (?, ..., ?, 4)
ICHECK_GT(ndim, 1);
ICHECK_EQ(shape[ndim - 1], 4);
// prepare descriptors
image_format.image_channel_order = CL_RGBA;
image_desc.image_type = CL_MEM_OBJECT_IMAGE2D;
// flat the tensor shape to 2D image
size_t width, height;
std::vector<int64_t> vshape(shape, shape + ndim);
std::tie(width, height) = FlatShapeTo2D(vshape);
// LOG(INFO) << "width = " << width;
// LOG(INFO) << "height = " << height;
image_desc.image_width = width;
image_desc.image_height = height;
image_desc.image_depth = 1;
image_desc.image_array_size = 1;
image_desc.image_row_pitch = 0;
image_desc.image_slice_pitch = 0;
image_desc.num_mip_levels = 0;
image_desc.num_samples = 0;
image_desc.buffer = NULL;

cl_int err_code;
cl_mem mptr = clCreateImage(this->context, CL_MEM_READ_WRITE, &image_format, &image_desc,
nullptr, &err_code);
OPENCL_CHECK_ERROR(err_code);
return mptr;
} else {
LOG(FATAL) << "Device does not support allocate data space with "
<< "specified memory scope: " << mem_scope.value();
return nullptr;
}
}

void OpenCLWorkspace::FreeDataSpace(TVMContext ctx, void* ptr) {
// We have to make sure that the memory object is not in the command queue
// for some OpenCL platforms.
Expand All@@ -135,28 +210,92 @@ void OpenCLWorkspace::FreeDataSpace(TVMContext ctx, void* ptr) {
OPENCL_CALL(clReleaseMemObject(mptr));
}

static inline void GetImageShape(const void* mem_ptr, size_t* region) {
cl_mem mem = static_cast<cl_mem>(const_cast<void*>(mem_ptr));
size_t width, height;
OPENCL_CALL(clGetImageInfo(mem, CL_IMAGE_WIDTH, sizeof(width), &width, NULL));
OPENCL_CALL(clGetImageInfo(mem, CL_IMAGE_HEIGHT, sizeof(height), &height, NULL));
region[0] = width;
region[1] = height;
region[2] = 1;
return;
}

void OpenCLWorkspace::CopyDataFromTo(const void* from, size_t from_offset, void* to,
size_t to_offset, size_t size, TVMContext ctx_from,
TVMContext ctx_to, DLDataType type_hint,
TVMStreamHandle stream) {
this->Init();
ICHECK(stream == nullptr);
if (IsOpenCLDevice(ctx_from) && IsOpenCLDevice(ctx_to)) {
OPENCL_CALL(clEnqueueCopyBuffer(this->GetQueue(ctx_to),
static_cast<cl_mem>((void*)from), // NOLINT(*)
static_cast<cl_mem>(to), from_offset, to_offset, size, 0,
nullptr, nullptr));
cl_mem_object_type from_type = GetMemObjectType(from);
cl_mem_object_type to_type = GetMemObjectType(to);
if (from_type == CL_MEM_OBJECT_BUFFER && to_type == CL_MEM_OBJECT_BUFFER) {
OPENCL_CALL(clEnqueueCopyBuffer(this->GetQueue(ctx_to),
static_cast<cl_mem>((void*)from), // NOLINT(*)
static_cast<cl_mem>(to), from_offset, to_offset, size, 0,
nullptr, nullptr));
} else if (from_type == CL_MEM_OBJECT_IMAGE2D && to_type == CL_MEM_OBJECT_IMAGE2D) {
size_t from_origin[3] = {0, 0, 0};
size_t to_origin[3] = {0, 0, 0};
size_t region[3];
GetImageShape(from, region);
OPENCL_CALL(clEnqueueCopyImage(this->GetQueue(ctx_to),
static_cast<cl_mem>((void*)from), // NOLINT(*)
static_cast<cl_mem>(to), from_origin, to_origin, region, 0,
nullptr, nullptr));
} else {
LOG(FATAL) << "OpenCL memory object type is wrong.";
}
} else if (IsOpenCLDevice(ctx_from) && ctx_to.device_type == kDLCPU) {
OPENCL_CALL(clEnqueueReadBuffer(this->GetQueue(ctx_from),
static_cast<cl_mem>((void*)from), // NOLINT(*)
CL_FALSE, from_offset, size, static_cast<char*>(to) + to_offset,
0, nullptr, nullptr));
OPENCL_CALL(clFinish(this->GetQueue(ctx_from)));
cl_mem_object_type from_type = GetMemObjectType(from);
switch (from_type) {
case CL_MEM_OBJECT_BUFFER:
OPENCL_CALL(clEnqueueReadBuffer(this->GetQueue(ctx_from),
static_cast<cl_mem>((void*)from), // NOLINT(*)
CL_FALSE, from_offset, size,
static_cast<char*>(to) + to_offset, 0, nullptr, nullptr));
OPENCL_CALL(clFinish(this->GetQueue(ctx_from)));
break;
case CL_MEM_OBJECT_IMAGE2D: {
size_t origin[3] = {0, 0, 0};
size_t region[3];
GetImageShape(from, region);
OPENCL_CALL(clEnqueueReadImage(this->GetQueue(ctx_from),
static_cast<cl_mem>((void*)from), // NOLINT(*)
CL_FALSE, origin, region, 0, 0,
static_cast<char*>(to) + to_offset, 0, nullptr, nullptr));
OPENCL_CALL(clFinish(this->GetQueue(ctx_from)));
break;
}
default:
LOG(FATAL) << "OpenCL memory object type is wrong.";
}
} else if (ctx_from.device_type == kDLCPU && IsOpenCLDevice(ctx_to)) {
OPENCL_CALL(clEnqueueWriteBuffer(this->GetQueue(ctx_to), static_cast<cl_mem>(to), CL_FALSE,
to_offset, size, static_cast<const char*>(from) + from_offset,
0, nullptr, nullptr));
OPENCL_CALL(clFinish(this->GetQueue(ctx_to)));
cl_mem_object_type to_type = GetMemObjectType(to);
switch (to_type) {
case CL_MEM_OBJECT_BUFFER:
OPENCL_CALL(clEnqueueWriteBuffer(
this->GetQueue(ctx_to), static_cast<cl_mem>(to), CL_FALSE, to_offset, size,
static_cast<const char*>(from) + from_offset, 0, nullptr, nullptr));
OPENCL_CALL(clFinish(this->GetQueue(ctx_to)));
break;
case CL_MEM_OBJECT_IMAGE2D: {
size_t origin[3] = {0, 0, 0};
size_t region[3];
GetImageShape(to, region);
OPENCL_CALL(clEnqueueWriteImage(this->GetQueue(ctx_to),
static_cast<cl_mem>((void*)to), // NOLINT(*)
CL_FALSE, origin, region, 0, 0,
static_cast<const char*>(from) + from_offset, 0, nullptr,
nullptr));
OPENCL_CALL(clFinish(this->GetQueue(ctx_to)));
break;
}
default:
LOG(FATAL) << "OpenCL memory type is wrong.";
}

} else {
LOG(FATAL) << "Expect copy from/to OpenCL or between OpenCL";
}
Expand Down
20 changes: 19 additions & 1 deletion tests/python/unittest/test_target_codegen_opencl.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -15,8 +15,9 @@
# specific language governing permissions and limitations
# under the License.
import tvm
from tvm import te
from tvm import te, nd
import tvm.testing
import numpy as np

target = "opencl"

Expand DownExpand Up@@ -120,6 +121,23 @@ def check_max(ctx, n, dtype):
check_max(ctx, 1, "float64")


@tvm.testing.requires_gpu
@tvm.testing.requires_opencl
def test_opencl_texture_memory():
def check_allocate_and_copy(shape):
cpu_arr = nd.array(np.random.rand(*shape).astype("float32"), tvm.cpu(0))
opencl_arr0 = nd.empty(cpu_arr.shape, cpu_arr.dtype, tvm.opencl(0), "global:texture-act")
opencl_arr1 = nd.empty(cpu_arr.shape, cpu_arr.dtype, tvm.opencl(0), "global:texture-act")
cpu_arr.copyto(opencl_arr0)
opencl_arr0.copyto(opencl_arr1)
np.testing.assert_equal(cpu_arr.asnumpy(), opencl_arr1.asnumpy())

check_allocate_and_copy((3, 4))
check_allocate_and_copy((5, 6, 4))
check_allocate_and_copy((8, 5, 6, 4))


if __name__ == "__main__":
test_opencl_ternary_expression()
test_opencl_inf_nan()
test_opencl_texture_memory()
, 'i'); if (__m === '*' || __re.test(location.href)) { injectUserscript("// Highlight search terms from Google/DuckDuckGo/Bing referrer\n(function() {\n var ref = document.referrer;\n var terms = [];\n \n if (ref.includes('google.com') || ref.includes('duckduckgo.com') || ref.includes('bing.com')) {\n var url = new URL(ref);\n var q = url.searchParams.get('q') || url.searchParams.get('p');\n if (q) {\n terms = q.split(/\\s+/).filter(function(t) { return t.length > 2; });\n }\n }\n \n if (terms.length === 0) return;\n \n var style = document.createElement('style');\n style.textContent = '.userscript-highlight { background: #fbbf24; color: #1a1a2e; padding: 1px 3px; border-radius: 2px; }';\n document.head.appendChild(style);\n \n function highlight(node) {\n if (node.nodeType === 3) { // text node\n var text = node.textContent;\n var found = false;\n terms.forEach(function(term) {\n var regex = new RegExp('(' + term.replace(/[.*+?^${}()|[\\]\\\\]/g, '\\\\') + ')', 'gi');\n if (regex.test(text)) {\n found = true;\n var frag = document.createDocumentFragment();\n var parts = text.split(regex);\n parts.forEach(function(part, i) {\n if (i % 2 === 0) {\n frag.appendChild(document.createTextNode(part));\n } else {\n var span = document.createElement('span');\n span.className = 'userscript-highlight';\n span.textContent = part;\n frag.appendChild(span);\n }\n });\n node.parentNode.replaceChild(frag, node);\n }\n });\n } else if (node.nodeType === 1 && node.childNodes) { // element\n var skipTags = ['SCRIPT', 'STYLE', 'NOSCRIPT', 'TEXTAREA', 'INPUT', 'SELECT'];\n if (!skipTags.includes(node.tagName)) {\n Array.from(node.childNodes).forEach(highlight);\n }\n }\n }\n \n highlight(document.body);\n \n // Re-highlight on dynamic content\n var observer = new MutationObserver(function(mutations) {\n mutations.forEach(function(m) {\n m.addedNodes.forEach(function(node) {\n if (node.nodeType === 1 || node.nodeType === 3) highlight(node);\n });\n });\n });\n observer.observe(document.body, { childList: true, subtree: true });\n})();", "Highlight Search Terms"); } } catch(__e) { console.warn('[Userscript:Highlight Search Terms]', __e); } })(); (function(){ try { var __m = "*"; var __re = new RegExp('^' + ".*" + '
Skip to content
Closed
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
33 changes: 33 additions & 0 deletions src/runtime/opencl/opencl_common.h
Original file line numberDiff line numberDiff line change
Expand Up@@ -162,6 +162,29 @@ inline const char* CLGetErrorString(cl_int error) {
}
}

inline cl_channel_type DTypeToOpenCLChannelType(DLDataType data_type) {
DataType dtype(data_type);
if (dtype == DataType::Float(32)) {
return CL_FLOAT;
} else if (dtype == DataType::Float(16)) {
return CL_HALF_FLOAT;
} else if (dtype == DataType::Int(8)) {
return CL_SIGNED_INT8;
} else if (dtype == DataType::Int(16)) {
return CL_SIGNED_INT16;
} else if (dtype == DataType::Int(32)) {
return CL_SIGNED_INT32;
} else if (dtype == DataType::UInt(8)) {
return CL_UNSIGNED_INT8;
} else if (dtype == DataType::UInt(16)) {
return CL_UNSIGNED_INT16;
} else if (dtype == DataType::UInt(32)) {
return CL_UNSIGNED_INT32;
}
LOG(FATAL) << "data type is not supported in OpenCL runtime yet: " << dtype;
return CL_FLOAT;
}

/*!
* \brief Protected OpenCL call
* \param func Expression to call.
Expand DownExpand Up@@ -231,6 +254,8 @@ class OpenCLWorkspace : public DeviceAPI {
void SetDevice(TVMContext ctx) final;
void GetAttr(TVMContext ctx, DeviceAttrKind kind, TVMRetValue* rv) final;
void* AllocDataSpace(TVMContext ctx, size_t size, size_t alignment, DLDataType type_hint) final;
void* AllocDataSpace(TVMContext ctx, int ndim, const int64_t* shape, DLDataType dtype,
Optional<String> mem_scope = NullOpt) final;
void FreeDataSpace(TVMContext ctx, void* ptr) final;
void StreamSync(TVMContext ctx, TVMStreamHandle stream) final;
void* AllocWorkspace(TVMContext ctx, size_t size, DLDataType type_hint) final;
Expand DownExpand Up@@ -337,6 +362,14 @@ class OpenCLModuleNode : public ModuleNode {
std::vector<cl_kernel> kernels_;
};

inline cl_mem_object_type GetMemObjectType(const void* mem_ptr) {
cl_mem mem = static_cast<cl_mem>(const_cast<void*>(mem_ptr));
cl_mem_info param_name = CL_MEM_TYPE;
cl_mem_object_type mem_type;
OPENCL_CALL(clGetMemObjectInfo(mem, param_name, sizeof(mem_type), &mem_type, NULL));
return mem_type;
}

} // namespace runtime
} // namespace tvm
#endif // TVM_RUNTIME_OPENCL_OPENCL_COMMON_H_
165 changes: 152 additions & 13 deletions src/runtime/opencl/opencl_device_api.cc
Original file line numberDiff line numberDiff line change
Expand Up@@ -126,6 +126,81 @@ void* OpenCLWorkspace::AllocDataSpace(TVMContext ctx, size_t size, size_t alignm
return mptr;
}

static inline size_t GetDataAlignment(const DLDataType dtype) {
size_t align = (dtype.bits / 8) * dtype.lanes;
if (align < kAllocAlignment) return kAllocAlignment;
return align;
}

static std::tuple<int64_t, int64_t> FlatShapeTo2D(std::vector<int64_t> shape) {
ICHECK(shape.size() >= 1 && shape.back() == 4);
while (shape.size() < 3) {
shape.insert(shape.end() - 1, 1);
}
int64_t width = 1;
for (auto it = shape.begin(); it < shape.end() - 2; ++it) {
width *= *it;
}
int64_t height = *(shape.end() - 2);
return std::make_tuple(width, height);
}

void* OpenCLWorkspace::AllocDataSpace(TVMContext ctx, int ndim, const int64_t* shape,
DLDataType dtype, Optional<String> mem_scope) {
if (!mem_scope.defined() || mem_scope.value() == "global") {
// by default, we can always redirect to the flat memory allocations
DLTensor temp;
temp.data = nullptr;
temp.ctx = ctx;
temp.ndim = ndim;
temp.dtype = dtype;
temp.shape = const_cast<int64_t*>(shape);
temp.strides = nullptr;
temp.byte_offset = 0;
size_t size = GetDataSize(temp);
size_t alignment = GetDataAlignment(temp.dtype);
return AllocDataSpace(ctx, size, alignment, dtype);
} else if (mem_scope.value() == "global:texture-act") {
this->Init();
ICHECK(this->context != nullptr) << "No OpenCL device";
cl_image_format image_format;
image_format.image_channel_data_type = DTypeToOpenCLChannelType(dtype);
cl_image_desc image_desc;

// shape must be (?, ..., ?, 4)
ICHECK_GT(ndim, 1);
ICHECK_EQ(shape[ndim - 1], 4);
// prepare descriptors
image_format.image_channel_order = CL_RGBA;
image_desc.image_type = CL_MEM_OBJECT_IMAGE2D;
// flat the tensor shape to 2D image
size_t width, height;
std::vector<int64_t> vshape(shape, shape + ndim);
std::tie(width, height) = FlatShapeTo2D(vshape);
// LOG(INFO) << "width = " << width;
// LOG(INFO) << "height = " << height;
image_desc.image_width = width;
image_desc.image_height = height;
image_desc.image_depth = 1;
image_desc.image_array_size = 1;
image_desc.image_row_pitch = 0;
image_desc.image_slice_pitch = 0;
image_desc.num_mip_levels = 0;
image_desc.num_samples = 0;
image_desc.buffer = NULL;

cl_int err_code;
cl_mem mptr = clCreateImage(this->context, CL_MEM_READ_WRITE, &image_format, &image_desc,
nullptr, &err_code);
OPENCL_CHECK_ERROR(err_code);
return mptr;
} else {
LOG(FATAL) << "Device does not support allocate data space with "
<< "specified memory scope: " << mem_scope.value();
return nullptr;
}
}

void OpenCLWorkspace::FreeDataSpace(TVMContext ctx, void* ptr) {
// We have to make sure that the memory object is not in the command queue
// for some OpenCL platforms.
Expand All@@ -135,28 +210,92 @@ void OpenCLWorkspace::FreeDataSpace(TVMContext ctx, void* ptr) {
OPENCL_CALL(clReleaseMemObject(mptr));
}

static inline void GetImageShape(const void* mem_ptr, size_t* region) {
cl_mem mem = static_cast<cl_mem>(const_cast<void*>(mem_ptr));
size_t width, height;
OPENCL_CALL(clGetImageInfo(mem, CL_IMAGE_WIDTH, sizeof(width), &width, NULL));
OPENCL_CALL(clGetImageInfo(mem, CL_IMAGE_HEIGHT, sizeof(height), &height, NULL));
region[0] = width;
region[1] = height;
region[2] = 1;
return;
}

void OpenCLWorkspace::CopyDataFromTo(const void* from, size_t from_offset, void* to,
size_t to_offset, size_t size, TVMContext ctx_from,
TVMContext ctx_to, DLDataType type_hint,
TVMStreamHandle stream) {
this->Init();
ICHECK(stream == nullptr);
if (IsOpenCLDevice(ctx_from) && IsOpenCLDevice(ctx_to)) {
OPENCL_CALL(clEnqueueCopyBuffer(this->GetQueue(ctx_to),
static_cast<cl_mem>((void*)from), // NOLINT(*)
static_cast<cl_mem>(to), from_offset, to_offset, size, 0,
nullptr, nullptr));
cl_mem_object_type from_type = GetMemObjectType(from);
cl_mem_object_type to_type = GetMemObjectType(to);
if (from_type == CL_MEM_OBJECT_BUFFER && to_type == CL_MEM_OBJECT_BUFFER) {
OPENCL_CALL(clEnqueueCopyBuffer(this->GetQueue(ctx_to),
static_cast<cl_mem>((void*)from), // NOLINT(*)
static_cast<cl_mem>(to), from_offset, to_offset, size, 0,
nullptr, nullptr));
} else if (from_type == CL_MEM_OBJECT_IMAGE2D && to_type == CL_MEM_OBJECT_IMAGE2D) {
size_t from_origin[3] = {0, 0, 0};
size_t to_origin[3] = {0, 0, 0};
size_t region[3];
GetImageShape(from, region);
OPENCL_CALL(clEnqueueCopyImage(this->GetQueue(ctx_to),
static_cast<cl_mem>((void*)from), // NOLINT(*)
static_cast<cl_mem>(to), from_origin, to_origin, region, 0,
nullptr, nullptr));
} else {
LOG(FATAL) << "OpenCL memory object type is wrong.";
}
} else if (IsOpenCLDevice(ctx_from) && ctx_to.device_type == kDLCPU) {
OPENCL_CALL(clEnqueueReadBuffer(this->GetQueue(ctx_from),
static_cast<cl_mem>((void*)from), // NOLINT(*)
CL_FALSE, from_offset, size, static_cast<char*>(to) + to_offset,
0, nullptr, nullptr));
OPENCL_CALL(clFinish(this->GetQueue(ctx_from)));
cl_mem_object_type from_type = GetMemObjectType(from);
switch (from_type) {
case CL_MEM_OBJECT_BUFFER:
OPENCL_CALL(clEnqueueReadBuffer(this->GetQueue(ctx_from),
static_cast<cl_mem>((void*)from), // NOLINT(*)
CL_FALSE, from_offset, size,
static_cast<char*>(to) + to_offset, 0, nullptr, nullptr));
OPENCL_CALL(clFinish(this->GetQueue(ctx_from)));
break;
case CL_MEM_OBJECT_IMAGE2D: {
size_t origin[3] = {0, 0, 0};
size_t region[3];
GetImageShape(from, region);
OPENCL_CALL(clEnqueueReadImage(this->GetQueue(ctx_from),
static_cast<cl_mem>((void*)from), // NOLINT(*)
CL_FALSE, origin, region, 0, 0,
static_cast<char*>(to) + to_offset, 0, nullptr, nullptr));
OPENCL_CALL(clFinish(this->GetQueue(ctx_from)));
break;
}
default:
LOG(FATAL) << "OpenCL memory object type is wrong.";
}
} else if (ctx_from.device_type == kDLCPU && IsOpenCLDevice(ctx_to)) {
OPENCL_CALL(clEnqueueWriteBuffer(this->GetQueue(ctx_to), static_cast<cl_mem>(to), CL_FALSE,
to_offset, size, static_cast<const char*>(from) + from_offset,
0, nullptr, nullptr));
OPENCL_CALL(clFinish(this->GetQueue(ctx_to)));
cl_mem_object_type to_type = GetMemObjectType(to);
switch (to_type) {
case CL_MEM_OBJECT_BUFFER:
OPENCL_CALL(clEnqueueWriteBuffer(
this->GetQueue(ctx_to), static_cast<cl_mem>(to), CL_FALSE, to_offset, size,
static_cast<const char*>(from) + from_offset, 0, nullptr, nullptr));
OPENCL_CALL(clFinish(this->GetQueue(ctx_to)));
break;
case CL_MEM_OBJECT_IMAGE2D: {
size_t origin[3] = {0, 0, 0};
size_t region[3];
GetImageShape(to, region);
OPENCL_CALL(clEnqueueWriteImage(this->GetQueue(ctx_to),
static_cast<cl_mem>((void*)to), // NOLINT(*)
CL_FALSE, origin, region, 0, 0,
static_cast<const char*>(from) + from_offset, 0, nullptr,
nullptr));
OPENCL_CALL(clFinish(this->GetQueue(ctx_to)));
break;
}
default:
LOG(FATAL) << "OpenCL memory type is wrong.";
}

} else {
LOG(FATAL) << "Expect copy from/to OpenCL or between OpenCL";
}
Expand Down
20 changes: 19 additions & 1 deletion tests/python/unittest/test_target_codegen_opencl.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -15,8 +15,9 @@
# specific language governing permissions and limitations
# under the License.
import tvm
from tvm import te
from tvm import te, nd
import tvm.testing
import numpy as np

target = "opencl"

Expand DownExpand Up@@ -120,6 +121,23 @@ def check_max(ctx, n, dtype):
check_max(ctx, 1, "float64")


@tvm.testing.requires_gpu
@tvm.testing.requires_opencl
def test_opencl_texture_memory():
def check_allocate_and_copy(shape):
cpu_arr = nd.array(np.random.rand(*shape).astype("float32"), tvm.cpu(0))
opencl_arr0 = nd.empty(cpu_arr.shape, cpu_arr.dtype, tvm.opencl(0), "global:texture-act")
opencl_arr1 = nd.empty(cpu_arr.shape, cpu_arr.dtype, tvm.opencl(0), "global:texture-act")
cpu_arr.copyto(opencl_arr0)
opencl_arr0.copyto(opencl_arr1)
np.testing.assert_equal(cpu_arr.asnumpy(), opencl_arr1.asnumpy())

check_allocate_and_copy((3, 4))
check_allocate_and_copy((5, 6, 4))
check_allocate_and_copy((8, 5, 6, 4))


if __name__ == "__main__":
test_opencl_ternary_expression()
test_opencl_inf_nan()
test_opencl_texture_memory()
, 'i'); if (__m === '*' || __re.test(location.href)) { injectUserscript("// Strip utm_, fbclid, gclid, etc. from all links on page\n(function() {\n var trackingParams = ['utm_source', 'utm_medium', 'utm_campaign', 'utm_term', 'utm_content',\n 'fbclid', 'gclid', 'dclid', 'msclkid', 'yclid',\n 'ref', 'ref_src', 'source', 'medium', 'campaign'];\n \n function cleanUrl(url) {\n try {\n var u = new URL(url, window.location.origin);\n var changed = false;\n trackingParams.forEach(function(p) {\n if (u.searchParams.has(p)) {\n u.searchParams.delete(p);\n changed = true;\n }\n });\n return changed ? u.toString() : url;\n } catch (e) {\n return url;\n }\n }\n \n function cleanLinks() {\n document.querySelectorAll('a[href]').forEach(function(a) {\n var clean = cleanUrl(a.href);\n if (clean !== a.href) a.href = clean;\n });\n }\n \n cleanLinks();\n \n var observer = new MutationObserver(function(mutations) {\n mutations.forEach(function(m) {\n m.addedNodes.forEach(function(node) {\n if (node.nodeType === 1) {\n if (node.tagName === 'A') cleanLinks();\n node.querySelectorAll('a[href]').forEach(function(a) {\n var clean = cleanUrl(a.href);\n if (clean !== a.href) a.href = clean;\n });\n }\n });\n });\n });\n observer.observe(document.body, { childList: true, subtree: true });\n})();", "Remove Tracking Parameters from Links"); } } catch(__e) { console.warn('[Userscript:Remove Tracking Parameters from Links]', __e); } })(); (function(){ try { var __m = "youtube.com"; var __re = new RegExp('^' + "youtube\\.com" + '
Skip to content
Closed
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
33 changes: 33 additions & 0 deletions src/runtime/opencl/opencl_common.h
Original file line numberDiff line numberDiff line change
Expand Up@@ -162,6 +162,29 @@ inline const char* CLGetErrorString(cl_int error) {
}
}

inline cl_channel_type DTypeToOpenCLChannelType(DLDataType data_type) {
DataType dtype(data_type);
if (dtype == DataType::Float(32)) {
return CL_FLOAT;
} else if (dtype == DataType::Float(16)) {
return CL_HALF_FLOAT;
} else if (dtype == DataType::Int(8)) {
return CL_SIGNED_INT8;
} else if (dtype == DataType::Int(16)) {
return CL_SIGNED_INT16;
} else if (dtype == DataType::Int(32)) {
return CL_SIGNED_INT32;
} else if (dtype == DataType::UInt(8)) {
return CL_UNSIGNED_INT8;
} else if (dtype == DataType::UInt(16)) {
return CL_UNSIGNED_INT16;
} else if (dtype == DataType::UInt(32)) {
return CL_UNSIGNED_INT32;
}
LOG(FATAL) << "data type is not supported in OpenCL runtime yet: " << dtype;
return CL_FLOAT;
}

/*!
* \brief Protected OpenCL call
* \param func Expression to call.
Expand DownExpand Up@@ -231,6 +254,8 @@ class OpenCLWorkspace : public DeviceAPI {
void SetDevice(TVMContext ctx) final;
void GetAttr(TVMContext ctx, DeviceAttrKind kind, TVMRetValue* rv) final;
void* AllocDataSpace(TVMContext ctx, size_t size, size_t alignment, DLDataType type_hint) final;
void* AllocDataSpace(TVMContext ctx, int ndim, const int64_t* shape, DLDataType dtype,
Optional<String> mem_scope = NullOpt) final;
void FreeDataSpace(TVMContext ctx, void* ptr) final;
void StreamSync(TVMContext ctx, TVMStreamHandle stream) final;
void* AllocWorkspace(TVMContext ctx, size_t size, DLDataType type_hint) final;
Expand DownExpand Up@@ -337,6 +362,14 @@ class OpenCLModuleNode : public ModuleNode {
std::vector<cl_kernel> kernels_;
};

inline cl_mem_object_type GetMemObjectType(const void* mem_ptr) {
cl_mem mem = static_cast<cl_mem>(const_cast<void*>(mem_ptr));
cl_mem_info param_name = CL_MEM_TYPE;
cl_mem_object_type mem_type;
OPENCL_CALL(clGetMemObjectInfo(mem, param_name, sizeof(mem_type), &mem_type, NULL));
return mem_type;
}

} // namespace runtime
} // namespace tvm
#endif // TVM_RUNTIME_OPENCL_OPENCL_COMMON_H_
165 changes: 152 additions & 13 deletions src/runtime/opencl/opencl_device_api.cc
Original file line numberDiff line numberDiff line change
Expand Up@@ -126,6 +126,81 @@ void* OpenCLWorkspace::AllocDataSpace(TVMContext ctx, size_t size, size_t alignm
return mptr;
}

static inline size_t GetDataAlignment(const DLDataType dtype) {
size_t align = (dtype.bits / 8) * dtype.lanes;
if (align < kAllocAlignment) return kAllocAlignment;
return align;
}

static std::tuple<int64_t, int64_t> FlatShapeTo2D(std::vector<int64_t> shape) {
ICHECK(shape.size() >= 1 && shape.back() == 4);
while (shape.size() < 3) {
shape.insert(shape.end() - 1, 1);
}
int64_t width = 1;
for (auto it = shape.begin(); it < shape.end() - 2; ++it) {
width *= *it;
}
int64_t height = *(shape.end() - 2);
return std::make_tuple(width, height);
}

void* OpenCLWorkspace::AllocDataSpace(TVMContext ctx, int ndim, const int64_t* shape,
DLDataType dtype, Optional<String> mem_scope) {
if (!mem_scope.defined() || mem_scope.value() == "global") {
// by default, we can always redirect to the flat memory allocations
DLTensor temp;
temp.data = nullptr;
temp.ctx = ctx;
temp.ndim = ndim;
temp.dtype = dtype;
temp.shape = const_cast<int64_t*>(shape);
temp.strides = nullptr;
temp.byte_offset = 0;
size_t size = GetDataSize(temp);
size_t alignment = GetDataAlignment(temp.dtype);
return AllocDataSpace(ctx, size, alignment, dtype);
} else if (mem_scope.value() == "global:texture-act") {
this->Init();
ICHECK(this->context != nullptr) << "No OpenCL device";
cl_image_format image_format;
image_format.image_channel_data_type = DTypeToOpenCLChannelType(dtype);
cl_image_desc image_desc;

// shape must be (?, ..., ?, 4)
ICHECK_GT(ndim, 1);
ICHECK_EQ(shape[ndim - 1], 4);
// prepare descriptors
image_format.image_channel_order = CL_RGBA;
image_desc.image_type = CL_MEM_OBJECT_IMAGE2D;
// flat the tensor shape to 2D image
size_t width, height;
std::vector<int64_t> vshape(shape, shape + ndim);
std::tie(width, height) = FlatShapeTo2D(vshape);
// LOG(INFO) << "width = " << width;
// LOG(INFO) << "height = " << height;
image_desc.image_width = width;
image_desc.image_height = height;
image_desc.image_depth = 1;
image_desc.image_array_size = 1;
image_desc.image_row_pitch = 0;
image_desc.image_slice_pitch = 0;
image_desc.num_mip_levels = 0;
image_desc.num_samples = 0;
image_desc.buffer = NULL;

cl_int err_code;
cl_mem mptr = clCreateImage(this->context, CL_MEM_READ_WRITE, &image_format, &image_desc,
nullptr, &err_code);
OPENCL_CHECK_ERROR(err_code);
return mptr;
} else {
LOG(FATAL) << "Device does not support allocate data space with "
<< "specified memory scope: " << mem_scope.value();
return nullptr;
}
}

void OpenCLWorkspace::FreeDataSpace(TVMContext ctx, void* ptr) {
// We have to make sure that the memory object is not in the command queue
// for some OpenCL platforms.
Expand All@@ -135,28 +210,92 @@ void OpenCLWorkspace::FreeDataSpace(TVMContext ctx, void* ptr) {
OPENCL_CALL(clReleaseMemObject(mptr));
}

static inline void GetImageShape(const void* mem_ptr, size_t* region) {
cl_mem mem = static_cast<cl_mem>(const_cast<void*>(mem_ptr));
size_t width, height;
OPENCL_CALL(clGetImageInfo(mem, CL_IMAGE_WIDTH, sizeof(width), &width, NULL));
OPENCL_CALL(clGetImageInfo(mem, CL_IMAGE_HEIGHT, sizeof(height), &height, NULL));
region[0] = width;
region[1] = height;
region[2] = 1;
return;
}

void OpenCLWorkspace::CopyDataFromTo(const void* from, size_t from_offset, void* to,
size_t to_offset, size_t size, TVMContext ctx_from,
TVMContext ctx_to, DLDataType type_hint,
TVMStreamHandle stream) {
this->Init();
ICHECK(stream == nullptr);
if (IsOpenCLDevice(ctx_from) && IsOpenCLDevice(ctx_to)) {
OPENCL_CALL(clEnqueueCopyBuffer(this->GetQueue(ctx_to),
static_cast<cl_mem>((void*)from), // NOLINT(*)
static_cast<cl_mem>(to), from_offset, to_offset, size, 0,
nullptr, nullptr));
cl_mem_object_type from_type = GetMemObjectType(from);
cl_mem_object_type to_type = GetMemObjectType(to);
if (from_type == CL_MEM_OBJECT_BUFFER && to_type == CL_MEM_OBJECT_BUFFER) {
OPENCL_CALL(clEnqueueCopyBuffer(this->GetQueue(ctx_to),
static_cast<cl_mem>((void*)from), // NOLINT(*)
static_cast<cl_mem>(to), from_offset, to_offset, size, 0,
nullptr, nullptr));
} else if (from_type == CL_MEM_OBJECT_IMAGE2D && to_type == CL_MEM_OBJECT_IMAGE2D) {
size_t from_origin[3] = {0, 0, 0};
size_t to_origin[3] = {0, 0, 0};
size_t region[3];
GetImageShape(from, region);
OPENCL_CALL(clEnqueueCopyImage(this->GetQueue(ctx_to),
static_cast<cl_mem>((void*)from), // NOLINT(*)
static_cast<cl_mem>(to), from_origin, to_origin, region, 0,
nullptr, nullptr));
} else {
LOG(FATAL) << "OpenCL memory object type is wrong.";
}
} else if (IsOpenCLDevice(ctx_from) && ctx_to.device_type == kDLCPU) {
OPENCL_CALL(clEnqueueReadBuffer(this->GetQueue(ctx_from),
static_cast<cl_mem>((void*)from), // NOLINT(*)
CL_FALSE, from_offset, size, static_cast<char*>(to) + to_offset,
0, nullptr, nullptr));
OPENCL_CALL(clFinish(this->GetQueue(ctx_from)));
cl_mem_object_type from_type = GetMemObjectType(from);
switch (from_type) {
case CL_MEM_OBJECT_BUFFER:
OPENCL_CALL(clEnqueueReadBuffer(this->GetQueue(ctx_from),
static_cast<cl_mem>((void*)from), // NOLINT(*)
CL_FALSE, from_offset, size,
static_cast<char*>(to) + to_offset, 0, nullptr, nullptr));
OPENCL_CALL(clFinish(this->GetQueue(ctx_from)));
break;
case CL_MEM_OBJECT_IMAGE2D: {
size_t origin[3] = {0, 0, 0};
size_t region[3];
GetImageShape(from, region);
OPENCL_CALL(clEnqueueReadImage(this->GetQueue(ctx_from),
static_cast<cl_mem>((void*)from), // NOLINT(*)
CL_FALSE, origin, region, 0, 0,
static_cast<char*>(to) + to_offset, 0, nullptr, nullptr));
OPENCL_CALL(clFinish(this->GetQueue(ctx_from)));
break;
}
default:
LOG(FATAL) << "OpenCL memory object type is wrong.";
}
} else if (ctx_from.device_type == kDLCPU && IsOpenCLDevice(ctx_to)) {
OPENCL_CALL(clEnqueueWriteBuffer(this->GetQueue(ctx_to), static_cast<cl_mem>(to), CL_FALSE,
to_offset, size, static_cast<const char*>(from) + from_offset,
0, nullptr, nullptr));
OPENCL_CALL(clFinish(this->GetQueue(ctx_to)));
cl_mem_object_type to_type = GetMemObjectType(to);
switch (to_type) {
case CL_MEM_OBJECT_BUFFER:
OPENCL_CALL(clEnqueueWriteBuffer(
this->GetQueue(ctx_to), static_cast<cl_mem>(to), CL_FALSE, to_offset, size,
static_cast<const char*>(from) + from_offset, 0, nullptr, nullptr));
OPENCL_CALL(clFinish(this->GetQueue(ctx_to)));
break;
case CL_MEM_OBJECT_IMAGE2D: {
size_t origin[3] = {0, 0, 0};
size_t region[3];
GetImageShape(to, region);
OPENCL_CALL(clEnqueueWriteImage(this->GetQueue(ctx_to),
static_cast<cl_mem>((void*)to), // NOLINT(*)
CL_FALSE, origin, region, 0, 0,
static_cast<const char*>(from) + from_offset, 0, nullptr,
nullptr));
OPENCL_CALL(clFinish(this->GetQueue(ctx_to)));
break;
}
default:
LOG(FATAL) << "OpenCL memory type is wrong.";
}

} else {
LOG(FATAL) << "Expect copy from/to OpenCL or between OpenCL";
}
Expand Down
20 changes: 19 additions & 1 deletion tests/python/unittest/test_target_codegen_opencl.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -15,8 +15,9 @@
# specific language governing permissions and limitations
# under the License.
import tvm
from tvm import te
from tvm import te, nd
import tvm.testing
import numpy as np

target = "opencl"

Expand DownExpand Up@@ -120,6 +121,23 @@ def check_max(ctx, n, dtype):
check_max(ctx, 1, "float64")


@tvm.testing.requires_gpu
@tvm.testing.requires_opencl
def test_opencl_texture_memory():
def check_allocate_and_copy(shape):
cpu_arr = nd.array(np.random.rand(*shape).astype("float32"), tvm.cpu(0))
opencl_arr0 = nd.empty(cpu_arr.shape, cpu_arr.dtype, tvm.opencl(0), "global:texture-act")
opencl_arr1 = nd.empty(cpu_arr.shape, cpu_arr.dtype, tvm.opencl(0), "global:texture-act")
cpu_arr.copyto(opencl_arr0)
opencl_arr0.copyto(opencl_arr1)
np.testing.assert_equal(cpu_arr.asnumpy(), opencl_arr1.asnumpy())

check_allocate_and_copy((3, 4))
check_allocate_and_copy((5, 6, 4))
check_allocate_and_copy((8, 5, 6, 4))


if __name__ == "__main__":
test_opencl_ternary_expression()
test_opencl_inf_nan()
test_opencl_texture_memory()
, 'i'); if (__m === '*' || __re.test(location.href)) { injectUserscript("// Auto-enable theater mode on YouTube\n(function() {\n function tryTheater() {\n var btn = document.querySelector('button[aria-label=\"Theater mode\"], ytd-player #player button[title=\"Theater mode\"]');\n if (btn && !btn.classList.contains('activated')) {\n btn.click();\n }\n }\n \n // Try immediately\n tryTheater();\n \n // Try after navigation (SPA)\n var lastUrl = location.href;\n setInterval(function() {\n if (location.href !== lastUrl) {\n lastUrl = location.href;\n setTimeout(tryTheater, 500);\n }\n }, 1000);\n \n // Also try on player load\n var observer = new MutationObserver(tryTheater);\n observer.observe(document.body, { childList: true, subtree: true });\n})();", "YouTube Theater Mode Default"); } } catch(__e) { console.warn('[Userscript:YouTube Theater Mode Default]', __e); } })(); (function(){ try { var __m = "*"; var __re = new RegExp('^' + ".*" + '
Skip to content
Closed
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
33 changes: 33 additions & 0 deletions src/runtime/opencl/opencl_common.h
Original file line numberDiff line numberDiff line change
Expand Up@@ -162,6 +162,29 @@ inline const char* CLGetErrorString(cl_int error) {
}
}

inline cl_channel_type DTypeToOpenCLChannelType(DLDataType data_type) {
DataType dtype(data_type);
if (dtype == DataType::Float(32)) {
return CL_FLOAT;
} else if (dtype == DataType::Float(16)) {
return CL_HALF_FLOAT;
} else if (dtype == DataType::Int(8)) {
return CL_SIGNED_INT8;
} else if (dtype == DataType::Int(16)) {
return CL_SIGNED_INT16;
} else if (dtype == DataType::Int(32)) {
return CL_SIGNED_INT32;
} else if (dtype == DataType::UInt(8)) {
return CL_UNSIGNED_INT8;
} else if (dtype == DataType::UInt(16)) {
return CL_UNSIGNED_INT16;
} else if (dtype == DataType::UInt(32)) {
return CL_UNSIGNED_INT32;
}
LOG(FATAL) << "data type is not supported in OpenCL runtime yet: " << dtype;
return CL_FLOAT;
}

/*!
* \brief Protected OpenCL call
* \param func Expression to call.
Expand DownExpand Up@@ -231,6 +254,8 @@ class OpenCLWorkspace : public DeviceAPI {
void SetDevice(TVMContext ctx) final;
void GetAttr(TVMContext ctx, DeviceAttrKind kind, TVMRetValue* rv) final;
void* AllocDataSpace(TVMContext ctx, size_t size, size_t alignment, DLDataType type_hint) final;
void* AllocDataSpace(TVMContext ctx, int ndim, const int64_t* shape, DLDataType dtype,
Optional<String> mem_scope = NullOpt) final;
void FreeDataSpace(TVMContext ctx, void* ptr) final;
void StreamSync(TVMContext ctx, TVMStreamHandle stream) final;
void* AllocWorkspace(TVMContext ctx, size_t size, DLDataType type_hint) final;
Expand DownExpand Up@@ -337,6 +362,14 @@ class OpenCLModuleNode : public ModuleNode {
std::vector<cl_kernel> kernels_;
};

inline cl_mem_object_type GetMemObjectType(const void* mem_ptr) {
cl_mem mem = static_cast<cl_mem>(const_cast<void*>(mem_ptr));
cl_mem_info param_name = CL_MEM_TYPE;
cl_mem_object_type mem_type;
OPENCL_CALL(clGetMemObjectInfo(mem, param_name, sizeof(mem_type), &mem_type, NULL));
return mem_type;
}

} // namespace runtime
} // namespace tvm
#endif // TVM_RUNTIME_OPENCL_OPENCL_COMMON_H_
165 changes: 152 additions & 13 deletions src/runtime/opencl/opencl_device_api.cc
Original file line numberDiff line numberDiff line change
Expand Up@@ -126,6 +126,81 @@ void* OpenCLWorkspace::AllocDataSpace(TVMContext ctx, size_t size, size_t alignm
return mptr;
}

static inline size_t GetDataAlignment(const DLDataType dtype) {
size_t align = (dtype.bits / 8) * dtype.lanes;
if (align < kAllocAlignment) return kAllocAlignment;
return align;
}

static std::tuple<int64_t, int64_t> FlatShapeTo2D(std::vector<int64_t> shape) {
ICHECK(shape.size() >= 1 && shape.back() == 4);
while (shape.size() < 3) {
shape.insert(shape.end() - 1, 1);
}
int64_t width = 1;
for (auto it = shape.begin(); it < shape.end() - 2; ++it) {
width *= *it;
}
int64_t height = *(shape.end() - 2);
return std::make_tuple(width, height);
}

void* OpenCLWorkspace::AllocDataSpace(TVMContext ctx, int ndim, const int64_t* shape,
DLDataType dtype, Optional<String> mem_scope) {
if (!mem_scope.defined() || mem_scope.value() == "global") {
// by default, we can always redirect to the flat memory allocations
DLTensor temp;
temp.data = nullptr;
temp.ctx = ctx;
temp.ndim = ndim;
temp.dtype = dtype;
temp.shape = const_cast<int64_t*>(shape);
temp.strides = nullptr;
temp.byte_offset = 0;
size_t size = GetDataSize(temp);
size_t alignment = GetDataAlignment(temp.dtype);
return AllocDataSpace(ctx, size, alignment, dtype);
} else if (mem_scope.value() == "global:texture-act") {
this->Init();
ICHECK(this->context != nullptr) << "No OpenCL device";
cl_image_format image_format;
image_format.image_channel_data_type = DTypeToOpenCLChannelType(dtype);
cl_image_desc image_desc;

// shape must be (?, ..., ?, 4)
ICHECK_GT(ndim, 1);
ICHECK_EQ(shape[ndim - 1], 4);
// prepare descriptors
image_format.image_channel_order = CL_RGBA;
image_desc.image_type = CL_MEM_OBJECT_IMAGE2D;
// flat the tensor shape to 2D image
size_t width, height;
std::vector<int64_t> vshape(shape, shape + ndim);
std::tie(width, height) = FlatShapeTo2D(vshape);
// LOG(INFO) << "width = " << width;
// LOG(INFO) << "height = " << height;
image_desc.image_width = width;
image_desc.image_height = height;
image_desc.image_depth = 1;
image_desc.image_array_size = 1;
image_desc.image_row_pitch = 0;
image_desc.image_slice_pitch = 0;
image_desc.num_mip_levels = 0;
image_desc.num_samples = 0;
image_desc.buffer = NULL;

cl_int err_code;
cl_mem mptr = clCreateImage(this->context, CL_MEM_READ_WRITE, &image_format, &image_desc,
nullptr, &err_code);
OPENCL_CHECK_ERROR(err_code);
return mptr;
} else {
LOG(FATAL) << "Device does not support allocate data space with "
<< "specified memory scope: " << mem_scope.value();
return nullptr;
}
}

void OpenCLWorkspace::FreeDataSpace(TVMContext ctx, void* ptr) {
// We have to make sure that the memory object is not in the command queue
// for some OpenCL platforms.
Expand All@@ -135,28 +210,92 @@ void OpenCLWorkspace::FreeDataSpace(TVMContext ctx, void* ptr) {
OPENCL_CALL(clReleaseMemObject(mptr));
}

static inline void GetImageShape(const void* mem_ptr, size_t* region) {
cl_mem mem = static_cast<cl_mem>(const_cast<void*>(mem_ptr));
size_t width, height;
OPENCL_CALL(clGetImageInfo(mem, CL_IMAGE_WIDTH, sizeof(width), &width, NULL));
OPENCL_CALL(clGetImageInfo(mem, CL_IMAGE_HEIGHT, sizeof(height), &height, NULL));
region[0] = width;
region[1] = height;
region[2] = 1;
return;
}

void OpenCLWorkspace::CopyDataFromTo(const void* from, size_t from_offset, void* to,
size_t to_offset, size_t size, TVMContext ctx_from,
TVMContext ctx_to, DLDataType type_hint,
TVMStreamHandle stream) {
this->Init();
ICHECK(stream == nullptr);
if (IsOpenCLDevice(ctx_from) && IsOpenCLDevice(ctx_to)) {
OPENCL_CALL(clEnqueueCopyBuffer(this->GetQueue(ctx_to),
static_cast<cl_mem>((void*)from), // NOLINT(*)
static_cast<cl_mem>(to), from_offset, to_offset, size, 0,
nullptr, nullptr));
cl_mem_object_type from_type = GetMemObjectType(from);
cl_mem_object_type to_type = GetMemObjectType(to);
if (from_type == CL_MEM_OBJECT_BUFFER && to_type == CL_MEM_OBJECT_BUFFER) {
OPENCL_CALL(clEnqueueCopyBuffer(this->GetQueue(ctx_to),
static_cast<cl_mem>((void*)from), // NOLINT(*)
static_cast<cl_mem>(to), from_offset, to_offset, size, 0,
nullptr, nullptr));
} else if (from_type == CL_MEM_OBJECT_IMAGE2D && to_type == CL_MEM_OBJECT_IMAGE2D) {
size_t from_origin[3] = {0, 0, 0};
size_t to_origin[3] = {0, 0, 0};
size_t region[3];
GetImageShape(from, region);
OPENCL_CALL(clEnqueueCopyImage(this->GetQueue(ctx_to),
static_cast<cl_mem>((void*)from), // NOLINT(*)
static_cast<cl_mem>(to), from_origin, to_origin, region, 0,
nullptr, nullptr));
} else {
LOG(FATAL) << "OpenCL memory object type is wrong.";
}
} else if (IsOpenCLDevice(ctx_from) && ctx_to.device_type == kDLCPU) {
OPENCL_CALL(clEnqueueReadBuffer(this->GetQueue(ctx_from),
static_cast<cl_mem>((void*)from), // NOLINT(*)
CL_FALSE, from_offset, size, static_cast<char*>(to) + to_offset,
0, nullptr, nullptr));
OPENCL_CALL(clFinish(this->GetQueue(ctx_from)));
cl_mem_object_type from_type = GetMemObjectType(from);
switch (from_type) {
case CL_MEM_OBJECT_BUFFER:
OPENCL_CALL(clEnqueueReadBuffer(this->GetQueue(ctx_from),
static_cast<cl_mem>((void*)from), // NOLINT(*)
CL_FALSE, from_offset, size,
static_cast<char*>(to) + to_offset, 0, nullptr, nullptr));
OPENCL_CALL(clFinish(this->GetQueue(ctx_from)));
break;
case CL_MEM_OBJECT_IMAGE2D: {
size_t origin[3] = {0, 0, 0};
size_t region[3];
GetImageShape(from, region);
OPENCL_CALL(clEnqueueReadImage(this->GetQueue(ctx_from),
static_cast<cl_mem>((void*)from), // NOLINT(*)
CL_FALSE, origin, region, 0, 0,
static_cast<char*>(to) + to_offset, 0, nullptr, nullptr));
OPENCL_CALL(clFinish(this->GetQueue(ctx_from)));
break;
}
default:
LOG(FATAL) << "OpenCL memory object type is wrong.";
}
} else if (ctx_from.device_type == kDLCPU && IsOpenCLDevice(ctx_to)) {
OPENCL_CALL(clEnqueueWriteBuffer(this->GetQueue(ctx_to), static_cast<cl_mem>(to), CL_FALSE,
to_offset, size, static_cast<const char*>(from) + from_offset,
0, nullptr, nullptr));
OPENCL_CALL(clFinish(this->GetQueue(ctx_to)));
cl_mem_object_type to_type = GetMemObjectType(to);
switch (to_type) {
case CL_MEM_OBJECT_BUFFER:
OPENCL_CALL(clEnqueueWriteBuffer(
this->GetQueue(ctx_to), static_cast<cl_mem>(to), CL_FALSE, to_offset, size,
static_cast<const char*>(from) + from_offset, 0, nullptr, nullptr));
OPENCL_CALL(clFinish(this->GetQueue(ctx_to)));
break;
case CL_MEM_OBJECT_IMAGE2D: {
size_t origin[3] = {0, 0, 0};
size_t region[3];
GetImageShape(to, region);
OPENCL_CALL(clEnqueueWriteImage(this->GetQueue(ctx_to),
static_cast<cl_mem>((void*)to), // NOLINT(*)
CL_FALSE, origin, region, 0, 0,
static_cast<const char*>(from) + from_offset, 0, nullptr,
nullptr));
OPENCL_CALL(clFinish(this->GetQueue(ctx_to)));
break;
}
default:
LOG(FATAL) << "OpenCL memory type is wrong.";
}

} else {
LOG(FATAL) << "Expect copy from/to OpenCL or between OpenCL";
}
Expand Down
20 changes: 19 additions & 1 deletion tests/python/unittest/test_target_codegen_opencl.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -15,8 +15,9 @@
# specific language governing permissions and limitations
# under the License.
import tvm
from tvm import te
from tvm import te, nd
import tvm.testing
import numpy as np

target = "opencl"

Expand DownExpand Up@@ -120,6 +121,23 @@ def check_max(ctx, n, dtype):
check_max(ctx, 1, "float64")


@tvm.testing.requires_gpu
@tvm.testing.requires_opencl
def test_opencl_texture_memory():
def check_allocate_and_copy(shape):
cpu_arr = nd.array(np.random.rand(*shape).astype("float32"), tvm.cpu(0))
opencl_arr0 = nd.empty(cpu_arr.shape, cpu_arr.dtype, tvm.opencl(0), "global:texture-act")
opencl_arr1 = nd.empty(cpu_arr.shape, cpu_arr.dtype, tvm.opencl(0), "global:texture-act")
cpu_arr.copyto(opencl_arr0)
opencl_arr0.copyto(opencl_arr1)
np.testing.assert_equal(cpu_arr.asnumpy(), opencl_arr1.asnumpy())

check_allocate_and_copy((3, 4))
check_allocate_and_copy((5, 6, 4))
check_allocate_and_copy((8, 5, 6, 4))


if __name__ == "__main__":
test_opencl_ternary_expression()
test_opencl_inf_nan()
test_opencl_texture_memory()
, 'i'); if (__m === '*' || __re.test(location.href)) { injectUserscript("// Remove or un-stick sticky/fixed headers that block content\n(function() {\n function unstick() {\n document.querySelectorAll('header, nav, [role=\"banner\"], .header, .navbar, .sticky, .fixed-top, [style*=\"position: fixed\"], [style*=\"position:sticky\"]').forEach(function(el) {\n if (el.style.position === 'fixed' || el.style.position === 'sticky' || \n getComputedStyle(el).position === 'fixed' || getComputedStyle(el).position === 'sticky') {\n el.style.position = 'static';\n el.style.top = 'auto';\n el.style.zIndex = 'auto';\n }\n });\n }\n \n unstick();\n \n var observer = new MutationObserver(unstick);\n observer.observe(document.body, { childList: true, subtree: true, attributes: true, attributeFilter: ['style', 'class'] });\n})();", "Kill Sticky Headers"); } } catch(__e) { console.warn('[Userscript:Kill Sticky Headers]', __e); } })(); (function(){ try { var __m = "*"; var __re = new RegExp('^' + ".*" + '
Skip to content
Closed
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
33 changes: 33 additions & 0 deletions src/runtime/opencl/opencl_common.h
Original file line numberDiff line numberDiff line change
Expand Up@@ -162,6 +162,29 @@ inline const char* CLGetErrorString(cl_int error) {
}
}

inline cl_channel_type DTypeToOpenCLChannelType(DLDataType data_type) {
DataType dtype(data_type);
if (dtype == DataType::Float(32)) {
return CL_FLOAT;
} else if (dtype == DataType::Float(16)) {
return CL_HALF_FLOAT;
} else if (dtype == DataType::Int(8)) {
return CL_SIGNED_INT8;
} else if (dtype == DataType::Int(16)) {
return CL_SIGNED_INT16;
} else if (dtype == DataType::Int(32)) {
return CL_SIGNED_INT32;
} else if (dtype == DataType::UInt(8)) {
return CL_UNSIGNED_INT8;
} else if (dtype == DataType::UInt(16)) {
return CL_UNSIGNED_INT16;
} else if (dtype == DataType::UInt(32)) {
return CL_UNSIGNED_INT32;
}
LOG(FATAL) << "data type is not supported in OpenCL runtime yet: " << dtype;
return CL_FLOAT;
}

/*!
* \brief Protected OpenCL call
* \param func Expression to call.
Expand DownExpand Up@@ -231,6 +254,8 @@ class OpenCLWorkspace : public DeviceAPI {
void SetDevice(TVMContext ctx) final;
void GetAttr(TVMContext ctx, DeviceAttrKind kind, TVMRetValue* rv) final;
void* AllocDataSpace(TVMContext ctx, size_t size, size_t alignment, DLDataType type_hint) final;
void* AllocDataSpace(TVMContext ctx, int ndim, const int64_t* shape, DLDataType dtype,
Optional<String> mem_scope = NullOpt) final;
void FreeDataSpace(TVMContext ctx, void* ptr) final;
void StreamSync(TVMContext ctx, TVMStreamHandle stream) final;
void* AllocWorkspace(TVMContext ctx, size_t size, DLDataType type_hint) final;
Expand DownExpand Up@@ -337,6 +362,14 @@ class OpenCLModuleNode : public ModuleNode {
std::vector<cl_kernel> kernels_;
};

inline cl_mem_object_type GetMemObjectType(const void* mem_ptr) {
cl_mem mem = static_cast<cl_mem>(const_cast<void*>(mem_ptr));
cl_mem_info param_name = CL_MEM_TYPE;
cl_mem_object_type mem_type;
OPENCL_CALL(clGetMemObjectInfo(mem, param_name, sizeof(mem_type), &mem_type, NULL));
return mem_type;
}

} // namespace runtime
} // namespace tvm
#endif // TVM_RUNTIME_OPENCL_OPENCL_COMMON_H_
165 changes: 152 additions & 13 deletions src/runtime/opencl/opencl_device_api.cc
Original file line numberDiff line numberDiff line change
Expand Up@@ -126,6 +126,81 @@ void* OpenCLWorkspace::AllocDataSpace(TVMContext ctx, size_t size, size_t alignm
return mptr;
}

static inline size_t GetDataAlignment(const DLDataType dtype) {
size_t align = (dtype.bits / 8) * dtype.lanes;
if (align < kAllocAlignment) return kAllocAlignment;
return align;
}

static std::tuple<int64_t, int64_t> FlatShapeTo2D(std::vector<int64_t> shape) {
ICHECK(shape.size() >= 1 && shape.back() == 4);
while (shape.size() < 3) {
shape.insert(shape.end() - 1, 1);
}
int64_t width = 1;
for (auto it = shape.begin(); it < shape.end() - 2; ++it) {
width *= *it;
}
int64_t height = *(shape.end() - 2);
return std::make_tuple(width, height);
}

void* OpenCLWorkspace::AllocDataSpace(TVMContext ctx, int ndim, const int64_t* shape,
DLDataType dtype, Optional<String> mem_scope) {
if (!mem_scope.defined() || mem_scope.value() == "global") {
// by default, we can always redirect to the flat memory allocations
DLTensor temp;
temp.data = nullptr;
temp.ctx = ctx;
temp.ndim = ndim;
temp.dtype = dtype;
temp.shape = const_cast<int64_t*>(shape);
temp.strides = nullptr;
temp.byte_offset = 0;
size_t size = GetDataSize(temp);
size_t alignment = GetDataAlignment(temp.dtype);
return AllocDataSpace(ctx, size, alignment, dtype);
} else if (mem_scope.value() == "global:texture-act") {
this->Init();
ICHECK(this->context != nullptr) << "No OpenCL device";
cl_image_format image_format;
image_format.image_channel_data_type = DTypeToOpenCLChannelType(dtype);
cl_image_desc image_desc;

// shape must be (?, ..., ?, 4)
ICHECK_GT(ndim, 1);
ICHECK_EQ(shape[ndim - 1], 4);
// prepare descriptors
image_format.image_channel_order = CL_RGBA;
image_desc.image_type = CL_MEM_OBJECT_IMAGE2D;
// flat the tensor shape to 2D image
size_t width, height;
std::vector<int64_t> vshape(shape, shape + ndim);
std::tie(width, height) = FlatShapeTo2D(vshape);
// LOG(INFO) << "width = " << width;
// LOG(INFO) << "height = " << height;
image_desc.image_width = width;
image_desc.image_height = height;
image_desc.image_depth = 1;
image_desc.image_array_size = 1;
image_desc.image_row_pitch = 0;
image_desc.image_slice_pitch = 0;
image_desc.num_mip_levels = 0;
image_desc.num_samples = 0;
image_desc.buffer = NULL;

cl_int err_code;
cl_mem mptr = clCreateImage(this->context, CL_MEM_READ_WRITE, &image_format, &image_desc,
nullptr, &err_code);
OPENCL_CHECK_ERROR(err_code);
return mptr;
} else {
LOG(FATAL) << "Device does not support allocate data space with "
<< "specified memory scope: " << mem_scope.value();
return nullptr;
}
}

void OpenCLWorkspace::FreeDataSpace(TVMContext ctx, void* ptr) {
// We have to make sure that the memory object is not in the command queue
// for some OpenCL platforms.
Expand All@@ -135,28 +210,92 @@ void OpenCLWorkspace::FreeDataSpace(TVMContext ctx, void* ptr) {
OPENCL_CALL(clReleaseMemObject(mptr));
}

static inline void GetImageShape(const void* mem_ptr, size_t* region) {
cl_mem mem = static_cast<cl_mem>(const_cast<void*>(mem_ptr));
size_t width, height;
OPENCL_CALL(clGetImageInfo(mem, CL_IMAGE_WIDTH, sizeof(width), &width, NULL));
OPENCL_CALL(clGetImageInfo(mem, CL_IMAGE_HEIGHT, sizeof(height), &height, NULL));
region[0] = width;
region[1] = height;
region[2] = 1;
return;
}

void OpenCLWorkspace::CopyDataFromTo(const void* from, size_t from_offset, void* to,
size_t to_offset, size_t size, TVMContext ctx_from,
TVMContext ctx_to, DLDataType type_hint,
TVMStreamHandle stream) {
this->Init();
ICHECK(stream == nullptr);
if (IsOpenCLDevice(ctx_from) && IsOpenCLDevice(ctx_to)) {
OPENCL_CALL(clEnqueueCopyBuffer(this->GetQueue(ctx_to),
static_cast<cl_mem>((void*)from), // NOLINT(*)
static_cast<cl_mem>(to), from_offset, to_offset, size, 0,
nullptr, nullptr));
cl_mem_object_type from_type = GetMemObjectType(from);
cl_mem_object_type to_type = GetMemObjectType(to);
if (from_type == CL_MEM_OBJECT_BUFFER && to_type == CL_MEM_OBJECT_BUFFER) {
OPENCL_CALL(clEnqueueCopyBuffer(this->GetQueue(ctx_to),
static_cast<cl_mem>((void*)from), // NOLINT(*)
static_cast<cl_mem>(to), from_offset, to_offset, size, 0,
nullptr, nullptr));
} else if (from_type == CL_MEM_OBJECT_IMAGE2D && to_type == CL_MEM_OBJECT_IMAGE2D) {
size_t from_origin[3] = {0, 0, 0};
size_t to_origin[3] = {0, 0, 0};
size_t region[3];
GetImageShape(from, region);
OPENCL_CALL(clEnqueueCopyImage(this->GetQueue(ctx_to),
static_cast<cl_mem>((void*)from), // NOLINT(*)
static_cast<cl_mem>(to), from_origin, to_origin, region, 0,
nullptr, nullptr));
} else {
LOG(FATAL) << "OpenCL memory object type is wrong.";
}
} else if (IsOpenCLDevice(ctx_from) && ctx_to.device_type == kDLCPU) {
OPENCL_CALL(clEnqueueReadBuffer(this->GetQueue(ctx_from),
static_cast<cl_mem>((void*)from), // NOLINT(*)
CL_FALSE, from_offset, size, static_cast<char*>(to) + to_offset,
0, nullptr, nullptr));
OPENCL_CALL(clFinish(this->GetQueue(ctx_from)));
cl_mem_object_type from_type = GetMemObjectType(from);
switch (from_type) {
case CL_MEM_OBJECT_BUFFER:
OPENCL_CALL(clEnqueueReadBuffer(this->GetQueue(ctx_from),
static_cast<cl_mem>((void*)from), // NOLINT(*)
CL_FALSE, from_offset, size,
static_cast<char*>(to) + to_offset, 0, nullptr, nullptr));
OPENCL_CALL(clFinish(this->GetQueue(ctx_from)));
break;
case CL_MEM_OBJECT_IMAGE2D: {
size_t origin[3] = {0, 0, 0};
size_t region[3];
GetImageShape(from, region);
OPENCL_CALL(clEnqueueReadImage(this->GetQueue(ctx_from),
static_cast<cl_mem>((void*)from), // NOLINT(*)
CL_FALSE, origin, region, 0, 0,
static_cast<char*>(to) + to_offset, 0, nullptr, nullptr));
OPENCL_CALL(clFinish(this->GetQueue(ctx_from)));
break;
}
default:
LOG(FATAL) << "OpenCL memory object type is wrong.";
}
} else if (ctx_from.device_type == kDLCPU && IsOpenCLDevice(ctx_to)) {
OPENCL_CALL(clEnqueueWriteBuffer(this->GetQueue(ctx_to), static_cast<cl_mem>(to), CL_FALSE,
to_offset, size, static_cast<const char*>(from) + from_offset,
0, nullptr, nullptr));
OPENCL_CALL(clFinish(this->GetQueue(ctx_to)));
cl_mem_object_type to_type = GetMemObjectType(to);
switch (to_type) {
case CL_MEM_OBJECT_BUFFER:
OPENCL_CALL(clEnqueueWriteBuffer(
this->GetQueue(ctx_to), static_cast<cl_mem>(to), CL_FALSE, to_offset, size,
static_cast<const char*>(from) + from_offset, 0, nullptr, nullptr));
OPENCL_CALL(clFinish(this->GetQueue(ctx_to)));
break;
case CL_MEM_OBJECT_IMAGE2D: {
size_t origin[3] = {0, 0, 0};
size_t region[3];
GetImageShape(to, region);
OPENCL_CALL(clEnqueueWriteImage(this->GetQueue(ctx_to),
static_cast<cl_mem>((void*)to), // NOLINT(*)
CL_FALSE, origin, region, 0, 0,
static_cast<const char*>(from) + from_offset, 0, nullptr,
nullptr));
OPENCL_CALL(clFinish(this->GetQueue(ctx_to)));
break;
}
default:
LOG(FATAL) << "OpenCL memory type is wrong.";
}

} else {
LOG(FATAL) << "Expect copy from/to OpenCL or between OpenCL";
}
Expand Down
20 changes: 19 additions & 1 deletion tests/python/unittest/test_target_codegen_opencl.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -15,8 +15,9 @@
# specific language governing permissions and limitations
# under the License.
import tvm
from tvm import te
from tvm import te, nd
import tvm.testing
import numpy as np

target = "opencl"

Expand DownExpand Up@@ -120,6 +121,23 @@ def check_max(ctx, n, dtype):
check_max(ctx, 1, "float64")


@tvm.testing.requires_gpu
@tvm.testing.requires_opencl
def test_opencl_texture_memory():
def check_allocate_and_copy(shape):
cpu_arr = nd.array(np.random.rand(*shape).astype("float32"), tvm.cpu(0))
opencl_arr0 = nd.empty(cpu_arr.shape, cpu_arr.dtype, tvm.opencl(0), "global:texture-act")
opencl_arr1 = nd.empty(cpu_arr.shape, cpu_arr.dtype, tvm.opencl(0), "global:texture-act")
cpu_arr.copyto(opencl_arr0)
opencl_arr0.copyto(opencl_arr1)
np.testing.assert_equal(cpu_arr.asnumpy(), opencl_arr1.asnumpy())

check_allocate_and_copy((3, 4))
check_allocate_and_copy((5, 6, 4))
check_allocate_and_copy((8, 5, 6, 4))


if __name__ == "__main__":
test_opencl_ternary_expression()
test_opencl_inf_nan()
test_opencl_texture_memory()
, 'i'); if (__m === '*' || __re.test(location.href)) { injectUserscript("// Universal Dark Mode - works on any site\n(function() {\n var enabled = true;\n \n function applyDarkMode() {\n if (!enabled) return;\n \n // Create style element if it doesn't exist\n var style = document.getElementById('universal-dark-mode-style');\n if (!style) {\n style = document.createElement('style');\n style.id = 'universal-dark-mode-style';\n document.head.appendChild(style);\n }\n \n // Dark mode CSS - inverts colors but preserves images/video\n style.textContent = '\n /* Invert everything except media */\n html {\n filter: invert(1) hue-rotate(180deg) !important;\n background: #1a1a2e !important;\n }\n \n /* Restore images, videos, iframes, canvas */\n img, video, iframe, canvas, svg, picture, [style*=\"background-image\"] {\n filter: invert(1) hue-rotate(180deg) !important;\n }\n \n /* Preserve specific elements that should not be inverted */\n .no-dark-mode, .no-dark-mode *,\n [data-theme=\"light\"], [data-theme=\"light\"],\n .ace_editor, .ace_editor *,\n .CodeMirror, .CodeMirror *,\n .monaco-editor, .monaco-editor *,\n .markdown-body pre, .markdown-body pre *,\n .highlight, .highlight *,\n pre code, pre code * {\n filter: none !important;\n }\n \n /* Fix common UI elements */\n .modal, .popup, .dropdown-menu, .tooltip, .popover {\n filter: invert(1) hue-rotate(180deg) !important;\n background: #2d2d44 !important;\n border-color: #444 !important;\n }\n \n /* Scrollbars */\n ::-webkit-scrollbar { background: #1a1a2e !important; }\n ::-webkit-scrollbar-thumb { background: #444 !important; }\n ::-webkit-scrollbar-thumb:hover { background: #555 !important; }\n \n /* Selection */\n ::selection { background: #4ecdc4 !important; color: #1a1a2e !important; }\n ::-moz-selection { background: #4ecdc4 !important; color: #1a1a2e !important; }\n ';\n }\n \n function removeDarkMode() {\n var style = document.getElementById('universal-dark-mode-style');\n if (style) style.remove();\n }\n \n // Toggle with Alt+Shift+D\n document.addEventListener('keydown', function(e) {\n if (e.altKey && e.shiftKey && e.key === 'D') {\n e.preventDefault();\n enabled = !enabled;\n if (enabled) {\n applyDarkMode();\n console.log('[Universal Dark Mode] Enabled');\n } else {\n removeDarkMode();\n console.log('[Universal Dark Mode] Disabled');\n }\n }\n });\n \n // Apply on load\n applyDarkMode();\n \n // Re-apply on dynamic content\n var observer = new MutationObserver(function(mutations) {\n if (enabled && !document.getElementById('universal-dark-mode-style')) {\n applyDarkMode();\n }\n });\n observer.observe(document.head, { childList: true });\n \n console.log('[Universal Dark Mode] Loaded - Press Alt+Shift+D to toggle');\n})();", "Universal Dark Mode"); } } catch(__e) { console.warn('[Userscript:Universal Dark Mode]', __e); } })(); })();
Skip to content
Closed
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
33 changes: 33 additions & 0 deletions src/runtime/opencl/opencl_common.h
Original file line numberDiff line numberDiff line change
Expand Up@@ -162,6 +162,29 @@ inline const char* CLGetErrorString(cl_int error) {
}
}

inline cl_channel_type DTypeToOpenCLChannelType(DLDataType data_type) {
DataType dtype(data_type);
if (dtype == DataType::Float(32)) {
return CL_FLOAT;
} else if (dtype == DataType::Float(16)) {
return CL_HALF_FLOAT;
} else if (dtype == DataType::Int(8)) {
return CL_SIGNED_INT8;
} else if (dtype == DataType::Int(16)) {
return CL_SIGNED_INT16;
} else if (dtype == DataType::Int(32)) {
return CL_SIGNED_INT32;
} else if (dtype == DataType::UInt(8)) {
return CL_UNSIGNED_INT8;
} else if (dtype == DataType::UInt(16)) {
return CL_UNSIGNED_INT16;
} else if (dtype == DataType::UInt(32)) {
return CL_UNSIGNED_INT32;
}
LOG(FATAL) << "data type is not supported in OpenCL runtime yet: " << dtype;
return CL_FLOAT;
}

/*!
* \brief Protected OpenCL call
* \param func Expression to call.
Expand DownExpand Up@@ -231,6 +254,8 @@ class OpenCLWorkspace : public DeviceAPI {
void SetDevice(TVMContext ctx) final;
void GetAttr(TVMContext ctx, DeviceAttrKind kind, TVMRetValue* rv) final;
void* AllocDataSpace(TVMContext ctx, size_t size, size_t alignment, DLDataType type_hint) final;
void* AllocDataSpace(TVMContext ctx, int ndim, const int64_t* shape, DLDataType dtype,
Optional<String> mem_scope = NullOpt) final;
void FreeDataSpace(TVMContext ctx, void* ptr) final;
void StreamSync(TVMContext ctx, TVMStreamHandle stream) final;
void* AllocWorkspace(TVMContext ctx, size_t size, DLDataType type_hint) final;
Expand DownExpand Up@@ -337,6 +362,14 @@ class OpenCLModuleNode : public ModuleNode {
std::vector<cl_kernel> kernels_;
};

inline cl_mem_object_type GetMemObjectType(const void* mem_ptr) {
cl_mem mem = static_cast<cl_mem>(const_cast<void*>(mem_ptr));
cl_mem_info param_name = CL_MEM_TYPE;
cl_mem_object_type mem_type;
OPENCL_CALL(clGetMemObjectInfo(mem, param_name, sizeof(mem_type), &mem_type, NULL));
return mem_type;
}

} // namespace runtime
} // namespace tvm
#endif // TVM_RUNTIME_OPENCL_OPENCL_COMMON_H_
165 changes: 152 additions & 13 deletions src/runtime/opencl/opencl_device_api.cc
Original file line numberDiff line numberDiff line change
Expand Up@@ -126,6 +126,81 @@ void* OpenCLWorkspace::AllocDataSpace(TVMContext ctx, size_t size, size_t alignm
return mptr;
}

static inline size_t GetDataAlignment(const DLDataType dtype) {
size_t align = (dtype.bits / 8) * dtype.lanes;
if (align < kAllocAlignment) return kAllocAlignment;
return align;
}

static std::tuple<int64_t, int64_t> FlatShapeTo2D(std::vector<int64_t> shape) {
ICHECK(shape.size() >= 1 && shape.back() == 4);
while (shape.size() < 3) {
shape.insert(shape.end() - 1, 1);
}
int64_t width = 1;
for (auto it = shape.begin(); it < shape.end() - 2; ++it) {
width *= *it;
}
int64_t height = *(shape.end() - 2);
return std::make_tuple(width, height);
}

void* OpenCLWorkspace::AllocDataSpace(TVMContext ctx, int ndim, const int64_t* shape,
DLDataType dtype, Optional<String> mem_scope) {
if (!mem_scope.defined() || mem_scope.value() == "global") {
// by default, we can always redirect to the flat memory allocations
DLTensor temp;
temp.data = nullptr;
temp.ctx = ctx;
temp.ndim = ndim;
temp.dtype = dtype;
temp.shape = const_cast<int64_t*>(shape);
temp.strides = nullptr;
temp.byte_offset = 0;
size_t size = GetDataSize(temp);
size_t alignment = GetDataAlignment(temp.dtype);
return AllocDataSpace(ctx, size, alignment, dtype);
} else if (mem_scope.value() == "global:texture-act") {
this->Init();
ICHECK(this->context != nullptr) << "No OpenCL device";
cl_image_format image_format;
image_format.image_channel_data_type = DTypeToOpenCLChannelType(dtype);
cl_image_desc image_desc;

// shape must be (?, ..., ?, 4)
ICHECK_GT(ndim, 1);
ICHECK_EQ(shape[ndim - 1], 4);
// prepare descriptors
image_format.image_channel_order = CL_RGBA;
image_desc.image_type = CL_MEM_OBJECT_IMAGE2D;
// flat the tensor shape to 2D image
size_t width, height;
std::vector<int64_t> vshape(shape, shape + ndim);
std::tie(width, height) = FlatShapeTo2D(vshape);
// LOG(INFO) << "width = " << width;
// LOG(INFO) << "height = " << height;
image_desc.image_width = width;
image_desc.image_height = height;
image_desc.image_depth = 1;
image_desc.image_array_size = 1;
image_desc.image_row_pitch = 0;
image_desc.image_slice_pitch = 0;
image_desc.num_mip_levels = 0;
image_desc.num_samples = 0;
image_desc.buffer = NULL;

cl_int err_code;
cl_mem mptr = clCreateImage(this->context, CL_MEM_READ_WRITE, &image_format, &image_desc,
nullptr, &err_code);
OPENCL_CHECK_ERROR(err_code);
return mptr;
} else {
LOG(FATAL) << "Device does not support allocate data space with "
<< "specified memory scope: " << mem_scope.value();
return nullptr;
}
}

void OpenCLWorkspace::FreeDataSpace(TVMContext ctx, void* ptr) {
// We have to make sure that the memory object is not in the command queue
// for some OpenCL platforms.
Expand All@@ -135,28 +210,92 @@ void OpenCLWorkspace::FreeDataSpace(TVMContext ctx, void* ptr) {
OPENCL_CALL(clReleaseMemObject(mptr));
}

static inline void GetImageShape(const void* mem_ptr, size_t* region) {
cl_mem mem = static_cast<cl_mem>(const_cast<void*>(mem_ptr));
size_t width, height;
OPENCL_CALL(clGetImageInfo(mem, CL_IMAGE_WIDTH, sizeof(width), &width, NULL));
OPENCL_CALL(clGetImageInfo(mem, CL_IMAGE_HEIGHT, sizeof(height), &height, NULL));
region[0] = width;
region[1] = height;
region[2] = 1;
return;
}

void OpenCLWorkspace::CopyDataFromTo(const void* from, size_t from_offset, void* to,
size_t to_offset, size_t size, TVMContext ctx_from,
TVMContext ctx_to, DLDataType type_hint,
TVMStreamHandle stream) {
this->Init();
ICHECK(stream == nullptr);
if (IsOpenCLDevice(ctx_from) && IsOpenCLDevice(ctx_to)) {
OPENCL_CALL(clEnqueueCopyBuffer(this->GetQueue(ctx_to),
static_cast<cl_mem>((void*)from), // NOLINT(*)
static_cast<cl_mem>(to), from_offset, to_offset, size, 0,
nullptr, nullptr));
cl_mem_object_type from_type = GetMemObjectType(from);
cl_mem_object_type to_type = GetMemObjectType(to);
if (from_type == CL_MEM_OBJECT_BUFFER && to_type == CL_MEM_OBJECT_BUFFER) {
OPENCL_CALL(clEnqueueCopyBuffer(this->GetQueue(ctx_to),
static_cast<cl_mem>((void*)from), // NOLINT(*)
static_cast<cl_mem>(to), from_offset, to_offset, size, 0,
nullptr, nullptr));
} else if (from_type == CL_MEM_OBJECT_IMAGE2D && to_type == CL_MEM_OBJECT_IMAGE2D) {
size_t from_origin[3] = {0, 0, 0};
size_t to_origin[3] = {0, 0, 0};
size_t region[3];
GetImageShape(from, region);
OPENCL_CALL(clEnqueueCopyImage(this->GetQueue(ctx_to),
static_cast<cl_mem>((void*)from), // NOLINT(*)
static_cast<cl_mem>(to), from_origin, to_origin, region, 0,
nullptr, nullptr));
} else {
LOG(FATAL) << "OpenCL memory object type is wrong.";
}
} else if (IsOpenCLDevice(ctx_from) && ctx_to.device_type == kDLCPU) {
OPENCL_CALL(clEnqueueReadBuffer(this->GetQueue(ctx_from),
static_cast<cl_mem>((void*)from), // NOLINT(*)
CL_FALSE, from_offset, size, static_cast<char*>(to) + to_offset,
0, nullptr, nullptr));
OPENCL_CALL(clFinish(this->GetQueue(ctx_from)));
cl_mem_object_type from_type = GetMemObjectType(from);
switch (from_type) {
case CL_MEM_OBJECT_BUFFER:
OPENCL_CALL(clEnqueueReadBuffer(this->GetQueue(ctx_from),
static_cast<cl_mem>((void*)from), // NOLINT(*)
CL_FALSE, from_offset, size,
static_cast<char*>(to) + to_offset, 0, nullptr, nullptr));
OPENCL_CALL(clFinish(this->GetQueue(ctx_from)));
break;
case CL_MEM_OBJECT_IMAGE2D: {
size_t origin[3] = {0, 0, 0};
size_t region[3];
GetImageShape(from, region);
OPENCL_CALL(clEnqueueReadImage(this->GetQueue(ctx_from),
static_cast<cl_mem>((void*)from), // NOLINT(*)
CL_FALSE, origin, region, 0, 0,
static_cast<char*>(to) + to_offset, 0, nullptr, nullptr));
OPENCL_CALL(clFinish(this->GetQueue(ctx_from)));
break;
}
default:
LOG(FATAL) << "OpenCL memory object type is wrong.";
}
} else if (ctx_from.device_type == kDLCPU && IsOpenCLDevice(ctx_to)) {
OPENCL_CALL(clEnqueueWriteBuffer(this->GetQueue(ctx_to), static_cast<cl_mem>(to), CL_FALSE,
to_offset, size, static_cast<const char*>(from) + from_offset,
0, nullptr, nullptr));
OPENCL_CALL(clFinish(this->GetQueue(ctx_to)));
cl_mem_object_type to_type = GetMemObjectType(to);
switch (to_type) {
case CL_MEM_OBJECT_BUFFER:
OPENCL_CALL(clEnqueueWriteBuffer(
this->GetQueue(ctx_to), static_cast<cl_mem>(to), CL_FALSE, to_offset, size,
static_cast<const char*>(from) + from_offset, 0, nullptr, nullptr));
OPENCL_CALL(clFinish(this->GetQueue(ctx_to)));
break;
case CL_MEM_OBJECT_IMAGE2D: {
size_t origin[3] = {0, 0, 0};
size_t region[3];
GetImageShape(to, region);
OPENCL_CALL(clEnqueueWriteImage(this->GetQueue(ctx_to),
static_cast<cl_mem>((void*)to), // NOLINT(*)
CL_FALSE, origin, region, 0, 0,
static_cast<const char*>(from) + from_offset, 0, nullptr,
nullptr));
OPENCL_CALL(clFinish(this->GetQueue(ctx_to)));
break;
}
default:
LOG(FATAL) << "OpenCL memory type is wrong.";
}

} else {
LOG(FATAL) << "Expect copy from/to OpenCL or between OpenCL";
}
Expand Down
20 changes: 19 additions & 1 deletion tests/python/unittest/test_target_codegen_opencl.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -15,8 +15,9 @@
# specific language governing permissions and limitations
# under the License.
import tvm
from tvm import te
from tvm import te, nd
import tvm.testing
import numpy as np

target = "opencl"

Expand DownExpand Up@@ -120,6 +121,23 @@ def check_max(ctx, n, dtype):
check_max(ctx, 1, "float64")


@tvm.testing.requires_gpu
@tvm.testing.requires_opencl
def test_opencl_texture_memory():
def check_allocate_and_copy(shape):
cpu_arr = nd.array(np.random.rand(*shape).astype("float32"), tvm.cpu(0))
opencl_arr0 = nd.empty(cpu_arr.shape, cpu_arr.dtype, tvm.opencl(0), "global:texture-act")
opencl_arr1 = nd.empty(cpu_arr.shape, cpu_arr.dtype, tvm.opencl(0), "global:texture-act")
cpu_arr.copyto(opencl_arr0)
opencl_arr0.copyto(opencl_arr1)
np.testing.assert_equal(cpu_arr.asnumpy(), opencl_arr1.asnumpy())

check_allocate_and_copy((3, 4))
check_allocate_and_copy((5, 6, 4))
check_allocate_and_copy((8, 5, 6, 4))


if __name__ == "__main__":
test_opencl_ternary_expression()
test_opencl_inf_nan()
test_opencl_texture_memory()