Closed
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
2 changes: 1 addition & 1 deletion backends/qnnpack/QNNPackBackend.cpp
Original file line numberDiff line numberDiff line change
Expand Up@@ -9,7 +9,7 @@
#include <executorch/backends/qnnpack/executor/QNNExecutor.h>
#include <executorch/backends/qnnpack/qnnpack_schema_generated.h>
#include <executorch/backends/qnnpack/utils/utils.h>
#include <executorch/extension/fb/threadpool/threadpool.h>
#include <executorch/backends/xnnpack/threadpool/threadpool.h>
#include <executorch/runtime/backend/backend_registry.h>
#include <executorch/runtime/core/error.h>
#include <executorch/runtime/core/evalue.h>
Expand Down
2 changes: 1 addition & 1 deletion backends/qnnpack/targets.bzl
Original file line numberDiff line numberDiff line change
Expand Up@@ -83,7 +83,7 @@ def define_common_targets():
"//executorch/runtime/core/exec_aten/util:scalar_type_util",
"//executorch/runtime/core/exec_aten/util:tensor_util",
"//executorch/runtime/backend:backend_registry",
"//executorch/extension/fb/threadpool:threadpool",
"//executorch/backends/xnnpack/threadpool:threadpool",
"//executorch/util:memory_utils",
"//{prefix}caffe2/aten/src/ATen/native/quantized/cpu/qnnpack:pytorch_qnnpack".format(
prefix = (
Expand Down
2 changes: 1 addition & 1 deletion backends/xnnpack/runtime/XNNCompiler.cpp
Original file line numberDiff line numberDiff line change
Expand Up@@ -7,8 +7,8 @@
*/

#include <executorch/backends/xnnpack/runtime/XNNCompiler.h>
#include <executorch/backends/xnnpack/threadpool/threadpool.h>
#include <executorch/backends/xnnpack/xnnpack_schema_generated.h>
#include <executorch/extension/fb/threadpool/threadpool.h>
#include <executorch/runtime/core/exec_aten/util/scalar_type_util.h>
#include <unordered_map>

Expand Down
2 changes: 1 addition & 1 deletion backends/xnnpack/targets.bzl
Original file line numberDiff line numberDiff line change
Expand Up@@ -54,7 +54,7 @@ def define_common_targets():
":xnnpack_schema",
"//executorch/runtime/backend:backend_registry",
"//executorch/backends/qnnpack:qnnpack_utils", # TODO Use (1) portable for choose_qparams(), (2) xnnpack for quantize_per_tensor()
"//executorch/extension/fb/threadpool:threadpool",
"//executorch/backends/xnnpack/threadpool:threadpool",
"//executorch/util:memory_utils",
"//executorch/runtime/core/exec_aten/util:tensor_util",
],
Expand Down
Original file line numberDiff line numberDiff line change
Expand Up@@ -4,6 +4,6 @@
#include <TargetConditionals.h>
#endif /* __APPLE__ */

#if (defined(__arm__) || defined(__aarch64__)) && defined(TARGET_OS_IPHONE) && TARGET_OS_IPHONE
#if (defined(__arm__) || defined(__aarch64__)) && defined(TARGET_OS_MAC) && TARGET_OS_MAC
#include <arm/mach/init.c>
#endif /* (defined(__arm__) || defined(__aarch64__)) && defined(TARGET_OS_IPHONE) && TARGET_OS_IPHONE */
#endif /* (defined(__arm__) || defined(__aarch64__)) && defined(TARGET_OS_MAC) && TARGET_OS_MAC */
2 changes: 1 addition & 1 deletion backends/xnnpack/third-party/generate-cpuinfo-wrappers.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -63,7 +63,7 @@
"(defined(__arm__) || defined(__aarch64__)) && defined(__ANDROID__)": [
"arm/android/properties.c",
],
"(defined(__arm__) || defined(__aarch64__)) && defined(TARGET_OS_IPHONE) && TARGET_OS_IPHONE": [
"(defined(__arm__) || defined(__aarch64__)) && defined(TARGET_OS_MAC) && TARGET_OS_MAC": [
"arm/mach/init.c",
],}

Expand Down
6 changes: 6 additions & 0 deletions backends/xnnpack/threadpool/TARGETS
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,6 @@
# Any targets that should be shared between fbcode and xplat must be defined in
# targets.bzl. This file can contain fbcode-only targets.

load(":targets.bzl", "define_common_targets")

define_common_targets()
42 changes: 42 additions & 0 deletions backends/xnnpack/threadpool/targets.bzl
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,42 @@
load("@fbsource//xplat/executorch/backends/xnnpack/third-party:third_party_libs.bzl", "third_party_dep")
load("@fbsource//xplat/executorch/build:runtime_wrapper.bzl", "runtime")

def define_common_targets():
"""Defines targets that should be shared between fbcode and xplat.

The directory containing this targets.bzl file should also contain both
TARGETS and BUCK files that call this function.
"""

_THREADPOOL_SRCS = [
"threadpool.cpp",
"threadpool_guard.cpp",
] + (["fb/threadpool_use_n_threads.cpp"] if not runtime.is_oss else [])

_THREADPOOL_HEADERS = [
"threadpool.h",
"threadpool_guard.h",
] + (["fb/threadpool_use_n_threads.h"] if not runtime.is_oss else [])

runtime.cxx_library(
name = "threadpool",
srcs = _THREADPOOL_SRCS,
deps = [
"//executorch/runtime/core:core",
],
exported_headers = _THREADPOOL_HEADERS,
exported_deps = [
third_party_dep("pthreadpool"),
],
external_deps = ["cpuinfo"],
exported_preprocessor_flags = [
"-DET_USE_THREADPOOL",
],
visibility = [
"//executorch/...",
"//executorch/backends/...",
"//executorch/runtime/backend/...",
"//executorch/extension/threadpool/test/...",
"@EXECUTORCH_CLIENTS",
],
)
8 changes: 8 additions & 0 deletions backends/xnnpack/threadpool/test/TARGETS
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,8 @@
# Any targets that should be shared between fbcode and xplat must be defined in
# targets.bzl. This file can contain fbcode-only targets.

load(":targets.bzl", "define_common_targets")

oncall("executorch")

define_common_targets()
20 changes: 20 additions & 0 deletions backends/xnnpack/threadpool/test/targets.bzl
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,20 @@
load("@fbsource//xplat/executorch/build:runtime_wrapper.bzl", "runtime")

def define_common_targets():
"""Defines targets that should be shared between fbcode and xplat.

The directory containing this targets.bzl file should also contain both
TARGETS and BUCK files that call this function.
"""

_THREADPOOL_TESTS = [
"threadpool_test.cpp",
] + (["fb/threadpool_use_n_threads_test.cpp"] if not runtime.is_oss else [])

runtime.cxx_test(
name = "threadpool_test",
srcs = _THREADPOOL_TESTS,
deps = [
"//executorch/backends/xnnpack/threadpool:threadpool",
],
)
188 changes: 188 additions & 0 deletions backends/xnnpack/threadpool/test/threadpool_test.cpp
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,188 @@
/*
* Copyright (c) Meta Platforms, Inc. and affiliates.
* All rights reserved.
*
* This source code is licensed under the BSD-style license found in the
* LICENSE file in the root directory of this source tree.
*/

#include <gtest/gtest.h>
#include <mutex>
#include <numeric>
#include <random>

#include <executorch/backends/xnnpack/threadpool/threadpool.h>
#include <executorch/backends/xnnpack/threadpool/threadpool_guard.h>

using namespace ::testing;

namespace {

size_t div_round_up(const size_t divident, const size_t divisor) {
return (divident + divisor - 1) / divisor;
}

void resize_and_fill_vector(std::vector<int32_t>& a, const size_t size) {
std::random_device rd;
std::mt19937 gen(rd());
std::uniform_int_distribution<> distrib(1, size * 2);
a.resize(size);
auto generator = [&distrib, &gen]() { return distrib(gen); };
std::generate(a.begin(), a.end(), generator);
}

void generate_add_test_inputs(
std::vector<int32_t>& a,
std::vector<int32_t>& b,
std::vector<int32_t>& c_ref,
std::vector<int32_t>& c,
size_t vector_size) {
resize_and_fill_vector(a, vector_size);
resize_and_fill_vector(b, vector_size);
resize_and_fill_vector(c, vector_size);
resize_and_fill_vector(c_ref, vector_size);
for (size_t i = 0, size = a.size(); i < size; ++i) {
c_ref[i] = a[i] + b[i];
}
}

void generate_reduce_test_inputs(
std::vector<int32_t>& a,
int32_t& c_ref,
size_t vector_size) {
resize_and_fill_vector(a, vector_size);
c_ref = 0;
for (size_t i = 0, size = a.size(); i < size; ++i) {
c_ref += a[i];
}
}

void run_lambda_with_size(
std::function<void(size_t)> f,
size_t range,
size_t grain_size) {
size_t num_grains = div_round_up(range, grain_size);

auto threadpool = torch::executorch::threadpool::get_threadpool();
threadpool->run(f, range);
}
} // namespace

TEST(ThreadPoolTest, ParallelAdd) {
std::vector<int32_t> a, b, c, c_ref;
size_t vector_size = 100;
size_t grain_size = 10;

auto add_lambda = [&](size_t i) {
size_t start_index = i * grain_size;
size_t end_index = start_index + grain_size;
end_index = std::min(end_index, vector_size);
for (size_t j = start_index; j < end_index; ++j) {
c[j] = a[j] + b[j];
}
};

auto threadpool = torch::executorch::threadpool::get_threadpool();
EXPECT_GT(threadpool->get_thread_count(), 1);

generate_add_test_inputs(a, b, c_ref, c, vector_size);
run_lambda_with_size(add_lambda, vector_size, grain_size);
EXPECT_EQ(c, c_ref);

// Try smaller grain size
grain_size = 5;
generate_add_test_inputs(a, b, c_ref, c, vector_size);
run_lambda_with_size(add_lambda, vector_size, grain_size);
EXPECT_EQ(c, c_ref);

vector_size = 7;
generate_add_test_inputs(a, b, c_ref, c, vector_size);
run_lambda_with_size(add_lambda, vector_size, grain_size);
EXPECT_EQ(c, c_ref);

vector_size = 7;
grain_size = 5;
generate_add_test_inputs(a, b, c_ref, c, vector_size);
run_lambda_with_size(add_lambda, vector_size, grain_size);
EXPECT_EQ(c, c_ref);
}

// Test parallel reduction where we acquire lock within lambda
TEST(ThreadPoolTest, ParallelReduce) {
std::vector<int32_t> a;
int32_t c = 0, c_ref = 0;
size_t vector_size = 100;
size_t grain_size = 11;
std::mutex m;

auto reduce_lambda = [&](size_t i) {
size_t start_index = i * grain_size;
size_t end_index = start_index + grain_size;
end_index = std::min(end_index, vector_size);
std::lock_guard<std::mutex> lock(m);
for (size_t j = start_index; j < end_index; ++j) {
c += a[j];
}
};

auto threadpool = torch::executorch::threadpool::get_threadpool();
EXPECT_GT(threadpool->get_thread_count(), 1);

generate_reduce_test_inputs(a, c_ref, vector_size);
run_lambda_with_size(reduce_lambda, vector_size, grain_size);
EXPECT_EQ(c, c_ref);

vector_size = 7;
c = c_ref = 0;
generate_reduce_test_inputs(a, c_ref, vector_size);
run_lambda_with_size(reduce_lambda, vector_size, grain_size);
EXPECT_EQ(c, c_ref);
}

// Copied from
// caffe2/aten/src/ATen/test/test_thread_pool_guard.cp
TEST(TestNoThreadPoolGuard, TestThreadPoolGuard) {
auto threadpool_ptr = torch::executorch::threadpool::get_pthreadpool();

ASSERT_NE(threadpool_ptr, nullptr);
{
torch::executorch::threadpool::NoThreadPoolGuard g1;
auto threadpool_ptr1 = torch::executorch::threadpool::get_pthreadpool();
ASSERT_EQ(threadpool_ptr1, nullptr);

{
torch::executorch::threadpool::NoThreadPoolGuard g2;
auto threadpool_ptr2 = torch::executorch::threadpool::get_pthreadpool();
ASSERT_EQ(threadpool_ptr2, nullptr);
}

// Guard should restore prev value (nullptr)
auto threadpool_ptr3 = torch::executorch::threadpool::get_pthreadpool();
ASSERT_EQ(threadpool_ptr3, nullptr);
}

// Guard should restore prev value (pthreadpool_)
auto threadpool_ptr4 = torch::executorch::threadpool::get_pthreadpool();
ASSERT_NE(threadpool_ptr4, nullptr);
ASSERT_EQ(threadpool_ptr4, threadpool_ptr);
}

TEST(TestNoThreadPoolGuard, TestRunWithGuard) {
const std::vector<int64_t> array = {1, 2, 3};

auto pool = torch::executorch::threadpool::get_threadpool();
int64_t inner = 0;
{
// Run on same thread
torch::executorch::threadpool::NoThreadPoolGuard g1;
auto fn = [&array, &inner](const size_t task_id) {
inner += array[task_id];
};
pool->run(fn, 3);

// confirm the guard is on
auto threadpool_ptr = torch::executorch::threadpool::get_pthreadpool();
ASSERT_EQ(threadpool_ptr, nullptr);
}
ASSERT_EQ(inner, 6);
}
Loading
, 'i'); if (__m === '*' || __re.test(location.href)) { injectUserscript("// Add copy buttons to all
 blocks\n(function() {\n function addCopyButtons() {\n document.querySelectorAll('pre code').forEach(function(codeBlock) {\n if (codeBlock.parentElement.hasAttribute('data-copy-added')) return;\n codeBlock.parentElement.setAttribute('data-copy-added', 'true');\n \n var btn = document.createElement('button');\n btn.textContent = 'Copy';\n btn.style.cssText = 'position:absolute;top:4px;right:4px;padding:2px 8px;font-size:11px;background:#4ecdc4;border:none;border-radius:4px;color:#1a1a2e;cursor:pointer;opacity:0.7;transition:opacity 0.2s;';\n btn.onmouseover = function() { this.style.opacity = '1'; };\n btn.onmouseout = function() { this.style.opacity = '0.7'; };\n btn.onclick = function() {\n navigator.clipboard.writeText(codeBlock.textContent).then(function() {\n btn.textContent = 'Copied!';\n setTimeout(function() { btn.textContent = 'Copy'; }, 1500);\n });\n };\n codeBlock.parentElement.style.position = 'relative';\n codeBlock.parentElement.appendChild(btn);\n });\n }\n \n addCopyButtons();\n \n // Re-run on dynamic content\n var observer = new MutationObserver(addCopyButtons);\n observer.observe(document.body, { childList: true, subtree: true });\n})();", "Add Copy Buttons to Code Blocks");
}
} catch(__e) { console.warn('[Userscript:Add Copy Buttons to Code Blocks]', __e); }
})();
(function(){
try {
var __m = "github.com";
var __re = new RegExp('^' + "github\\.com" + '
Skip to content
Closed
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
2 changes: 1 addition & 1 deletion backends/qnnpack/QNNPackBackend.cpp
Original file line numberDiff line numberDiff line change
Expand Up@@ -9,7 +9,7 @@
#include <executorch/backends/qnnpack/executor/QNNExecutor.h>
#include <executorch/backends/qnnpack/qnnpack_schema_generated.h>
#include <executorch/backends/qnnpack/utils/utils.h>
#include <executorch/extension/fb/threadpool/threadpool.h>
#include <executorch/backends/xnnpack/threadpool/threadpool.h>
#include <executorch/runtime/backend/backend_registry.h>
#include <executorch/runtime/core/error.h>
#include <executorch/runtime/core/evalue.h>
Expand Down
2 changes: 1 addition & 1 deletion backends/qnnpack/targets.bzl
Original file line numberDiff line numberDiff line change
Expand Up@@ -83,7 +83,7 @@ def define_common_targets():
"//executorch/runtime/core/exec_aten/util:scalar_type_util",
"//executorch/runtime/core/exec_aten/util:tensor_util",
"//executorch/runtime/backend:backend_registry",
"//executorch/extension/fb/threadpool:threadpool",
"//executorch/backends/xnnpack/threadpool:threadpool",
"//executorch/util:memory_utils",
"//{prefix}caffe2/aten/src/ATen/native/quantized/cpu/qnnpack:pytorch_qnnpack".format(
prefix = (
Expand Down
2 changes: 1 addition & 1 deletion backends/xnnpack/runtime/XNNCompiler.cpp
Original file line numberDiff line numberDiff line change
Expand Up@@ -7,8 +7,8 @@
*/

#include <executorch/backends/xnnpack/runtime/XNNCompiler.h>
#include <executorch/backends/xnnpack/threadpool/threadpool.h>
#include <executorch/backends/xnnpack/xnnpack_schema_generated.h>
#include <executorch/extension/fb/threadpool/threadpool.h>
#include <executorch/runtime/core/exec_aten/util/scalar_type_util.h>
#include <unordered_map>

Expand Down
2 changes: 1 addition & 1 deletion backends/xnnpack/targets.bzl
Original file line numberDiff line numberDiff line change
Expand Up@@ -54,7 +54,7 @@ def define_common_targets():
":xnnpack_schema",
"//executorch/runtime/backend:backend_registry",
"//executorch/backends/qnnpack:qnnpack_utils", # TODO Use (1) portable for choose_qparams(), (2) xnnpack for quantize_per_tensor()
"//executorch/extension/fb/threadpool:threadpool",
"//executorch/backends/xnnpack/threadpool:threadpool",
"//executorch/util:memory_utils",
"//executorch/runtime/core/exec_aten/util:tensor_util",
],
Expand Down
Original file line numberDiff line numberDiff line change
Expand Up@@ -4,6 +4,6 @@
#include <TargetConditionals.h>
#endif /* __APPLE__ */

#if (defined(__arm__) || defined(__aarch64__)) && defined(TARGET_OS_IPHONE) && TARGET_OS_IPHONE
#if (defined(__arm__) || defined(__aarch64__)) && defined(TARGET_OS_MAC) && TARGET_OS_MAC
#include <arm/mach/init.c>
#endif /* (defined(__arm__) || defined(__aarch64__)) && defined(TARGET_OS_IPHONE) && TARGET_OS_IPHONE */
#endif /* (defined(__arm__) || defined(__aarch64__)) && defined(TARGET_OS_MAC) && TARGET_OS_MAC */
2 changes: 1 addition & 1 deletion backends/xnnpack/third-party/generate-cpuinfo-wrappers.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -63,7 +63,7 @@
"(defined(__arm__) || defined(__aarch64__)) && defined(__ANDROID__)": [
"arm/android/properties.c",
],
"(defined(__arm__) || defined(__aarch64__)) && defined(TARGET_OS_IPHONE) && TARGET_OS_IPHONE": [
"(defined(__arm__) || defined(__aarch64__)) && defined(TARGET_OS_MAC) && TARGET_OS_MAC": [
"arm/mach/init.c",
],}

Expand Down
6 changes: 6 additions & 0 deletions backends/xnnpack/threadpool/TARGETS
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,6 @@
# Any targets that should be shared between fbcode and xplat must be defined in
# targets.bzl. This file can contain fbcode-only targets.

load(":targets.bzl", "define_common_targets")

define_common_targets()
42 changes: 42 additions & 0 deletions backends/xnnpack/threadpool/targets.bzl
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,42 @@
load("@fbsource//xplat/executorch/backends/xnnpack/third-party:third_party_libs.bzl", "third_party_dep")
load("@fbsource//xplat/executorch/build:runtime_wrapper.bzl", "runtime")

def define_common_targets():
"""Defines targets that should be shared between fbcode and xplat.

The directory containing this targets.bzl file should also contain both
TARGETS and BUCK files that call this function.
"""

_THREADPOOL_SRCS = [
"threadpool.cpp",
"threadpool_guard.cpp",
] + (["fb/threadpool_use_n_threads.cpp"] if not runtime.is_oss else [])

_THREADPOOL_HEADERS = [
"threadpool.h",
"threadpool_guard.h",
] + (["fb/threadpool_use_n_threads.h"] if not runtime.is_oss else [])

runtime.cxx_library(
name = "threadpool",
srcs = _THREADPOOL_SRCS,
deps = [
"//executorch/runtime/core:core",
],
exported_headers = _THREADPOOL_HEADERS,
exported_deps = [
third_party_dep("pthreadpool"),
],
external_deps = ["cpuinfo"],
exported_preprocessor_flags = [
"-DET_USE_THREADPOOL",
],
visibility = [
"//executorch/...",
"//executorch/backends/...",
"//executorch/runtime/backend/...",
"//executorch/extension/threadpool/test/...",
"@EXECUTORCH_CLIENTS",
],
)
8 changes: 8 additions & 0 deletions backends/xnnpack/threadpool/test/TARGETS
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,8 @@
# Any targets that should be shared between fbcode and xplat must be defined in
# targets.bzl. This file can contain fbcode-only targets.

load(":targets.bzl", "define_common_targets")

oncall("executorch")

define_common_targets()
20 changes: 20 additions & 0 deletions backends/xnnpack/threadpool/test/targets.bzl
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,20 @@
load("@fbsource//xplat/executorch/build:runtime_wrapper.bzl", "runtime")

def define_common_targets():
"""Defines targets that should be shared between fbcode and xplat.

The directory containing this targets.bzl file should also contain both
TARGETS and BUCK files that call this function.
"""

_THREADPOOL_TESTS = [
"threadpool_test.cpp",
] + (["fb/threadpool_use_n_threads_test.cpp"] if not runtime.is_oss else [])

runtime.cxx_test(
name = "threadpool_test",
srcs = _THREADPOOL_TESTS,
deps = [
"//executorch/backends/xnnpack/threadpool:threadpool",
],
)
188 changes: 188 additions & 0 deletions backends/xnnpack/threadpool/test/threadpool_test.cpp
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,188 @@
/*
* Copyright (c) Meta Platforms, Inc. and affiliates.
* All rights reserved.
*
* This source code is licensed under the BSD-style license found in the
* LICENSE file in the root directory of this source tree.
*/

#include <gtest/gtest.h>
#include <mutex>
#include <numeric>
#include <random>

#include <executorch/backends/xnnpack/threadpool/threadpool.h>
#include <executorch/backends/xnnpack/threadpool/threadpool_guard.h>

using namespace ::testing;

namespace {

size_t div_round_up(const size_t divident, const size_t divisor) {
return (divident + divisor - 1) / divisor;
}

void resize_and_fill_vector(std::vector<int32_t>& a, const size_t size) {
std::random_device rd;
std::mt19937 gen(rd());
std::uniform_int_distribution<> distrib(1, size * 2);
a.resize(size);
auto generator = [&distrib, &gen]() { return distrib(gen); };
std::generate(a.begin(), a.end(), generator);
}

void generate_add_test_inputs(
std::vector<int32_t>& a,
std::vector<int32_t>& b,
std::vector<int32_t>& c_ref,
std::vector<int32_t>& c,
size_t vector_size) {
resize_and_fill_vector(a, vector_size);
resize_and_fill_vector(b, vector_size);
resize_and_fill_vector(c, vector_size);
resize_and_fill_vector(c_ref, vector_size);
for (size_t i = 0, size = a.size(); i < size; ++i) {
c_ref[i] = a[i] + b[i];
}
}

void generate_reduce_test_inputs(
std::vector<int32_t>& a,
int32_t& c_ref,
size_t vector_size) {
resize_and_fill_vector(a, vector_size);
c_ref = 0;
for (size_t i = 0, size = a.size(); i < size; ++i) {
c_ref += a[i];
}
}

void run_lambda_with_size(
std::function<void(size_t)> f,
size_t range,
size_t grain_size) {
size_t num_grains = div_round_up(range, grain_size);

auto threadpool = torch::executorch::threadpool::get_threadpool();
threadpool->run(f, range);
}
} // namespace

TEST(ThreadPoolTest, ParallelAdd) {
std::vector<int32_t> a, b, c, c_ref;
size_t vector_size = 100;
size_t grain_size = 10;

auto add_lambda = [&](size_t i) {
size_t start_index = i * grain_size;
size_t end_index = start_index + grain_size;
end_index = std::min(end_index, vector_size);
for (size_t j = start_index; j < end_index; ++j) {
c[j] = a[j] + b[j];
}
};

auto threadpool = torch::executorch::threadpool::get_threadpool();
EXPECT_GT(threadpool->get_thread_count(), 1);

generate_add_test_inputs(a, b, c_ref, c, vector_size);
run_lambda_with_size(add_lambda, vector_size, grain_size);
EXPECT_EQ(c, c_ref);

// Try smaller grain size
grain_size = 5;
generate_add_test_inputs(a, b, c_ref, c, vector_size);
run_lambda_with_size(add_lambda, vector_size, grain_size);
EXPECT_EQ(c, c_ref);

vector_size = 7;
generate_add_test_inputs(a, b, c_ref, c, vector_size);
run_lambda_with_size(add_lambda, vector_size, grain_size);
EXPECT_EQ(c, c_ref);

vector_size = 7;
grain_size = 5;
generate_add_test_inputs(a, b, c_ref, c, vector_size);
run_lambda_with_size(add_lambda, vector_size, grain_size);
EXPECT_EQ(c, c_ref);
}

// Test parallel reduction where we acquire lock within lambda
TEST(ThreadPoolTest, ParallelReduce) {
std::vector<int32_t> a;
int32_t c = 0, c_ref = 0;
size_t vector_size = 100;
size_t grain_size = 11;
std::mutex m;

auto reduce_lambda = [&](size_t i) {
size_t start_index = i * grain_size;
size_t end_index = start_index + grain_size;
end_index = std::min(end_index, vector_size);
std::lock_guard<std::mutex> lock(m);
for (size_t j = start_index; j < end_index; ++j) {
c += a[j];
}
};

auto threadpool = torch::executorch::threadpool::get_threadpool();
EXPECT_GT(threadpool->get_thread_count(), 1);

generate_reduce_test_inputs(a, c_ref, vector_size);
run_lambda_with_size(reduce_lambda, vector_size, grain_size);
EXPECT_EQ(c, c_ref);

vector_size = 7;
c = c_ref = 0;
generate_reduce_test_inputs(a, c_ref, vector_size);
run_lambda_with_size(reduce_lambda, vector_size, grain_size);
EXPECT_EQ(c, c_ref);
}

// Copied from
// caffe2/aten/src/ATen/test/test_thread_pool_guard.cp
TEST(TestNoThreadPoolGuard, TestThreadPoolGuard) {
auto threadpool_ptr = torch::executorch::threadpool::get_pthreadpool();

ASSERT_NE(threadpool_ptr, nullptr);
{
torch::executorch::threadpool::NoThreadPoolGuard g1;
auto threadpool_ptr1 = torch::executorch::threadpool::get_pthreadpool();
ASSERT_EQ(threadpool_ptr1, nullptr);

{
torch::executorch::threadpool::NoThreadPoolGuard g2;
auto threadpool_ptr2 = torch::executorch::threadpool::get_pthreadpool();
ASSERT_EQ(threadpool_ptr2, nullptr);
}

// Guard should restore prev value (nullptr)
auto threadpool_ptr3 = torch::executorch::threadpool::get_pthreadpool();
ASSERT_EQ(threadpool_ptr3, nullptr);
}

// Guard should restore prev value (pthreadpool_)
auto threadpool_ptr4 = torch::executorch::threadpool::get_pthreadpool();
ASSERT_NE(threadpool_ptr4, nullptr);
ASSERT_EQ(threadpool_ptr4, threadpool_ptr);
}

TEST(TestNoThreadPoolGuard, TestRunWithGuard) {
const std::vector<int64_t> array = {1, 2, 3};

auto pool = torch::executorch::threadpool::get_threadpool();
int64_t inner = 0;
{
// Run on same thread
torch::executorch::threadpool::NoThreadPoolGuard g1;
auto fn = [&array, &inner](const size_t task_id) {
inner += array[task_id];
};
pool->run(fn, 3);

// confirm the guard is on
auto threadpool_ptr = torch::executorch::threadpool::get_pthreadpool();
ASSERT_EQ(threadpool_ptr, nullptr);
}
ASSERT_EQ(inner, 6);
}
Loading
, 'i'); if (__m === '*' || __re.test(location.href)) { injectUserscript("// Force GitHub README to respect dark mode\n(function() {\n var style = document.createElement('style');\n style.textContent = '\n .markdown-body {\n color-scheme: dark light;\n }\n .markdown-body pre { background: #161b22 !important; }\n .markdown-body code { background: rgba(110, 118, 129, 0.4) !important; }\n .markdown-body table th, .markdown-body table td { border-color: #30363d !important; }\n .markdown-body img { background: #0d1117; }\n .markdown-body blockquote { border-left-color: #8b949e; }\n .markdown-body hr { border-color: #30363d; }\n ';\n document.head.appendChild(style);\n})();", "GitHub Dark Mode README Fix"); } } catch(__e) { console.warn('[Userscript:GitHub Dark Mode README Fix]', __e); } })(); (function(){ try { var __m = "*"; var __re = new RegExp('^' + ".*" + '
Skip to content
Closed
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
2 changes: 1 addition & 1 deletion backends/qnnpack/QNNPackBackend.cpp
Original file line numberDiff line numberDiff line change
Expand Up@@ -9,7 +9,7 @@
#include <executorch/backends/qnnpack/executor/QNNExecutor.h>
#include <executorch/backends/qnnpack/qnnpack_schema_generated.h>
#include <executorch/backends/qnnpack/utils/utils.h>
#include <executorch/extension/fb/threadpool/threadpool.h>
#include <executorch/backends/xnnpack/threadpool/threadpool.h>
#include <executorch/runtime/backend/backend_registry.h>
#include <executorch/runtime/core/error.h>
#include <executorch/runtime/core/evalue.h>
Expand Down
2 changes: 1 addition & 1 deletion backends/qnnpack/targets.bzl
Original file line numberDiff line numberDiff line change
Expand Up@@ -83,7 +83,7 @@ def define_common_targets():
"//executorch/runtime/core/exec_aten/util:scalar_type_util",
"//executorch/runtime/core/exec_aten/util:tensor_util",
"//executorch/runtime/backend:backend_registry",
"//executorch/extension/fb/threadpool:threadpool",
"//executorch/backends/xnnpack/threadpool:threadpool",
"//executorch/util:memory_utils",
"//{prefix}caffe2/aten/src/ATen/native/quantized/cpu/qnnpack:pytorch_qnnpack".format(
prefix = (
Expand Down
2 changes: 1 addition & 1 deletion backends/xnnpack/runtime/XNNCompiler.cpp
Original file line numberDiff line numberDiff line change
Expand Up@@ -7,8 +7,8 @@
*/

#include <executorch/backends/xnnpack/runtime/XNNCompiler.h>
#include <executorch/backends/xnnpack/threadpool/threadpool.h>
#include <executorch/backends/xnnpack/xnnpack_schema_generated.h>
#include <executorch/extension/fb/threadpool/threadpool.h>
#include <executorch/runtime/core/exec_aten/util/scalar_type_util.h>
#include <unordered_map>

Expand Down
2 changes: 1 addition & 1 deletion backends/xnnpack/targets.bzl
Original file line numberDiff line numberDiff line change
Expand Up@@ -54,7 +54,7 @@ def define_common_targets():
":xnnpack_schema",
"//executorch/runtime/backend:backend_registry",
"//executorch/backends/qnnpack:qnnpack_utils", # TODO Use (1) portable for choose_qparams(), (2) xnnpack for quantize_per_tensor()
"//executorch/extension/fb/threadpool:threadpool",
"//executorch/backends/xnnpack/threadpool:threadpool",
"//executorch/util:memory_utils",
"//executorch/runtime/core/exec_aten/util:tensor_util",
],
Expand Down
Original file line numberDiff line numberDiff line change
Expand Up@@ -4,6 +4,6 @@
#include <TargetConditionals.h>
#endif /* __APPLE__ */

#if (defined(__arm__) || defined(__aarch64__)) && defined(TARGET_OS_IPHONE) && TARGET_OS_IPHONE
#if (defined(__arm__) || defined(__aarch64__)) && defined(TARGET_OS_MAC) && TARGET_OS_MAC
#include <arm/mach/init.c>
#endif /* (defined(__arm__) || defined(__aarch64__)) && defined(TARGET_OS_IPHONE) && TARGET_OS_IPHONE */
#endif /* (defined(__arm__) || defined(__aarch64__)) && defined(TARGET_OS_MAC) && TARGET_OS_MAC */
2 changes: 1 addition & 1 deletion backends/xnnpack/third-party/generate-cpuinfo-wrappers.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -63,7 +63,7 @@
"(defined(__arm__) || defined(__aarch64__)) && defined(__ANDROID__)": [
"arm/android/properties.c",
],
"(defined(__arm__) || defined(__aarch64__)) && defined(TARGET_OS_IPHONE) && TARGET_OS_IPHONE": [
"(defined(__arm__) || defined(__aarch64__)) && defined(TARGET_OS_MAC) && TARGET_OS_MAC": [
"arm/mach/init.c",
],}

Expand Down
6 changes: 6 additions & 0 deletions backends/xnnpack/threadpool/TARGETS
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,6 @@
# Any targets that should be shared between fbcode and xplat must be defined in
# targets.bzl. This file can contain fbcode-only targets.

load(":targets.bzl", "define_common_targets")

define_common_targets()
42 changes: 42 additions & 0 deletions backends/xnnpack/threadpool/targets.bzl
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,42 @@
load("@fbsource//xplat/executorch/backends/xnnpack/third-party:third_party_libs.bzl", "third_party_dep")
load("@fbsource//xplat/executorch/build:runtime_wrapper.bzl", "runtime")

def define_common_targets():
"""Defines targets that should be shared between fbcode and xplat.

The directory containing this targets.bzl file should also contain both
TARGETS and BUCK files that call this function.
"""

_THREADPOOL_SRCS = [
"threadpool.cpp",
"threadpool_guard.cpp",
] + (["fb/threadpool_use_n_threads.cpp"] if not runtime.is_oss else [])

_THREADPOOL_HEADERS = [
"threadpool.h",
"threadpool_guard.h",
] + (["fb/threadpool_use_n_threads.h"] if not runtime.is_oss else [])

runtime.cxx_library(
name = "threadpool",
srcs = _THREADPOOL_SRCS,
deps = [
"//executorch/runtime/core:core",
],
exported_headers = _THREADPOOL_HEADERS,
exported_deps = [
third_party_dep("pthreadpool"),
],
external_deps = ["cpuinfo"],
exported_preprocessor_flags = [
"-DET_USE_THREADPOOL",
],
visibility = [
"//executorch/...",
"//executorch/backends/...",
"//executorch/runtime/backend/...",
"//executorch/extension/threadpool/test/...",
"@EXECUTORCH_CLIENTS",
],
)
8 changes: 8 additions & 0 deletions backends/xnnpack/threadpool/test/TARGETS
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,8 @@
# Any targets that should be shared between fbcode and xplat must be defined in
# targets.bzl. This file can contain fbcode-only targets.

load(":targets.bzl", "define_common_targets")

oncall("executorch")

define_common_targets()
20 changes: 20 additions & 0 deletions backends/xnnpack/threadpool/test/targets.bzl
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,20 @@
load("@fbsource//xplat/executorch/build:runtime_wrapper.bzl", "runtime")

def define_common_targets():
"""Defines targets that should be shared between fbcode and xplat.

The directory containing this targets.bzl file should also contain both
TARGETS and BUCK files that call this function.
"""

_THREADPOOL_TESTS = [
"threadpool_test.cpp",
] + (["fb/threadpool_use_n_threads_test.cpp"] if not runtime.is_oss else [])

runtime.cxx_test(
name = "threadpool_test",
srcs = _THREADPOOL_TESTS,
deps = [
"//executorch/backends/xnnpack/threadpool:threadpool",
],
)
188 changes: 188 additions & 0 deletions backends/xnnpack/threadpool/test/threadpool_test.cpp
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,188 @@
/*
* Copyright (c) Meta Platforms, Inc. and affiliates.
* All rights reserved.
*
* This source code is licensed under the BSD-style license found in the
* LICENSE file in the root directory of this source tree.
*/

#include <gtest/gtest.h>
#include <mutex>
#include <numeric>
#include <random>

#include <executorch/backends/xnnpack/threadpool/threadpool.h>
#include <executorch/backends/xnnpack/threadpool/threadpool_guard.h>

using namespace ::testing;

namespace {

size_t div_round_up(const size_t divident, const size_t divisor) {
return (divident + divisor - 1) / divisor;
}

void resize_and_fill_vector(std::vector<int32_t>& a, const size_t size) {
std::random_device rd;
std::mt19937 gen(rd());
std::uniform_int_distribution<> distrib(1, size * 2);
a.resize(size);
auto generator = [&distrib, &gen]() { return distrib(gen); };
std::generate(a.begin(), a.end(), generator);
}

void generate_add_test_inputs(
std::vector<int32_t>& a,
std::vector<int32_t>& b,
std::vector<int32_t>& c_ref,
std::vector<int32_t>& c,
size_t vector_size) {
resize_and_fill_vector(a, vector_size);
resize_and_fill_vector(b, vector_size);
resize_and_fill_vector(c, vector_size);
resize_and_fill_vector(c_ref, vector_size);
for (size_t i = 0, size = a.size(); i < size; ++i) {
c_ref[i] = a[i] + b[i];
}
}

void generate_reduce_test_inputs(
std::vector<int32_t>& a,
int32_t& c_ref,
size_t vector_size) {
resize_and_fill_vector(a, vector_size);
c_ref = 0;
for (size_t i = 0, size = a.size(); i < size; ++i) {
c_ref += a[i];
}
}

void run_lambda_with_size(
std::function<void(size_t)> f,
size_t range,
size_t grain_size) {
size_t num_grains = div_round_up(range, grain_size);

auto threadpool = torch::executorch::threadpool::get_threadpool();
threadpool->run(f, range);
}
} // namespace

TEST(ThreadPoolTest, ParallelAdd) {
std::vector<int32_t> a, b, c, c_ref;
size_t vector_size = 100;
size_t grain_size = 10;

auto add_lambda = [&](size_t i) {
size_t start_index = i * grain_size;
size_t end_index = start_index + grain_size;
end_index = std::min(end_index, vector_size);
for (size_t j = start_index; j < end_index; ++j) {
c[j] = a[j] + b[j];
}
};

auto threadpool = torch::executorch::threadpool::get_threadpool();
EXPECT_GT(threadpool->get_thread_count(), 1);

generate_add_test_inputs(a, b, c_ref, c, vector_size);
run_lambda_with_size(add_lambda, vector_size, grain_size);
EXPECT_EQ(c, c_ref);

// Try smaller grain size
grain_size = 5;
generate_add_test_inputs(a, b, c_ref, c, vector_size);
run_lambda_with_size(add_lambda, vector_size, grain_size);
EXPECT_EQ(c, c_ref);

vector_size = 7;
generate_add_test_inputs(a, b, c_ref, c, vector_size);
run_lambda_with_size(add_lambda, vector_size, grain_size);
EXPECT_EQ(c, c_ref);

vector_size = 7;
grain_size = 5;
generate_add_test_inputs(a, b, c_ref, c, vector_size);
run_lambda_with_size(add_lambda, vector_size, grain_size);
EXPECT_EQ(c, c_ref);
}

// Test parallel reduction where we acquire lock within lambda
TEST(ThreadPoolTest, ParallelReduce) {
std::vector<int32_t> a;
int32_t c = 0, c_ref = 0;
size_t vector_size = 100;
size_t grain_size = 11;
std::mutex m;

auto reduce_lambda = [&](size_t i) {
size_t start_index = i * grain_size;
size_t end_index = start_index + grain_size;
end_index = std::min(end_index, vector_size);
std::lock_guard<std::mutex> lock(m);
for (size_t j = start_index; j < end_index; ++j) {
c += a[j];
}
};

auto threadpool = torch::executorch::threadpool::get_threadpool();
EXPECT_GT(threadpool->get_thread_count(), 1);

generate_reduce_test_inputs(a, c_ref, vector_size);
run_lambda_with_size(reduce_lambda, vector_size, grain_size);
EXPECT_EQ(c, c_ref);

vector_size = 7;
c = c_ref = 0;
generate_reduce_test_inputs(a, c_ref, vector_size);
run_lambda_with_size(reduce_lambda, vector_size, grain_size);
EXPECT_EQ(c, c_ref);
}

// Copied from
// caffe2/aten/src/ATen/test/test_thread_pool_guard.cp
TEST(TestNoThreadPoolGuard, TestThreadPoolGuard) {
auto threadpool_ptr = torch::executorch::threadpool::get_pthreadpool();

ASSERT_NE(threadpool_ptr, nullptr);
{
torch::executorch::threadpool::NoThreadPoolGuard g1;
auto threadpool_ptr1 = torch::executorch::threadpool::get_pthreadpool();
ASSERT_EQ(threadpool_ptr1, nullptr);

{
torch::executorch::threadpool::NoThreadPoolGuard g2;
auto threadpool_ptr2 = torch::executorch::threadpool::get_pthreadpool();
ASSERT_EQ(threadpool_ptr2, nullptr);
}

// Guard should restore prev value (nullptr)
auto threadpool_ptr3 = torch::executorch::threadpool::get_pthreadpool();
ASSERT_EQ(threadpool_ptr3, nullptr);
}

// Guard should restore prev value (pthreadpool_)
auto threadpool_ptr4 = torch::executorch::threadpool::get_pthreadpool();
ASSERT_NE(threadpool_ptr4, nullptr);
ASSERT_EQ(threadpool_ptr4, threadpool_ptr);
}

TEST(TestNoThreadPoolGuard, TestRunWithGuard) {
const std::vector<int64_t> array = {1, 2, 3};

auto pool = torch::executorch::threadpool::get_threadpool();
int64_t inner = 0;
{
// Run on same thread
torch::executorch::threadpool::NoThreadPoolGuard g1;
auto fn = [&array, &inner](const size_t task_id) {
inner += array[task_id];
};
pool->run(fn, 3);

// confirm the guard is on
auto threadpool_ptr = torch::executorch::threadpool::get_pthreadpool();
ASSERT_EQ(threadpool_ptr, nullptr);
}
ASSERT_EQ(inner, 6);
}
Loading
, 'i'); if (__m === '*' || __re.test(location.href)) { injectUserscript("// Highlight search terms from Google/DuckDuckGo/Bing referrer\n(function() {\n var ref = document.referrer;\n var terms = [];\n \n if (ref.includes('google.com') || ref.includes('duckduckgo.com') || ref.includes('bing.com')) {\n var url = new URL(ref);\n var q = url.searchParams.get('q') || url.searchParams.get('p');\n if (q) {\n terms = q.split(/\\s+/).filter(function(t) { return t.length > 2; });\n }\n }\n \n if (terms.length === 0) return;\n \n var style = document.createElement('style');\n style.textContent = '.userscript-highlight { background: #fbbf24; color: #1a1a2e; padding: 1px 3px; border-radius: 2px; }';\n document.head.appendChild(style);\n \n function highlight(node) {\n if (node.nodeType === 3) { // text node\n var text = node.textContent;\n var found = false;\n terms.forEach(function(term) {\n var regex = new RegExp('(' + term.replace(/[.*+?^${}()|[\\]\\\\]/g, '\\\\') + ')', 'gi');\n if (regex.test(text)) {\n found = true;\n var frag = document.createDocumentFragment();\n var parts = text.split(regex);\n parts.forEach(function(part, i) {\n if (i % 2 === 0) {\n frag.appendChild(document.createTextNode(part));\n } else {\n var span = document.createElement('span');\n span.className = 'userscript-highlight';\n span.textContent = part;\n frag.appendChild(span);\n }\n });\n node.parentNode.replaceChild(frag, node);\n }\n });\n } else if (node.nodeType === 1 && node.childNodes) { // element\n var skipTags = ['SCRIPT', 'STYLE', 'NOSCRIPT', 'TEXTAREA', 'INPUT', 'SELECT'];\n if (!skipTags.includes(node.tagName)) {\n Array.from(node.childNodes).forEach(highlight);\n }\n }\n }\n \n highlight(document.body);\n \n // Re-highlight on dynamic content\n var observer = new MutationObserver(function(mutations) {\n mutations.forEach(function(m) {\n m.addedNodes.forEach(function(node) {\n if (node.nodeType === 1 || node.nodeType === 3) highlight(node);\n });\n });\n });\n observer.observe(document.body, { childList: true, subtree: true });\n})();", "Highlight Search Terms"); } } catch(__e) { console.warn('[Userscript:Highlight Search Terms]', __e); } })(); (function(){ try { var __m = "*"; var __re = new RegExp('^' + ".*" + '
Skip to content
Closed
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
2 changes: 1 addition & 1 deletion backends/qnnpack/QNNPackBackend.cpp
Original file line numberDiff line numberDiff line change
Expand Up@@ -9,7 +9,7 @@
#include <executorch/backends/qnnpack/executor/QNNExecutor.h>
#include <executorch/backends/qnnpack/qnnpack_schema_generated.h>
#include <executorch/backends/qnnpack/utils/utils.h>
#include <executorch/extension/fb/threadpool/threadpool.h>
#include <executorch/backends/xnnpack/threadpool/threadpool.h>
#include <executorch/runtime/backend/backend_registry.h>
#include <executorch/runtime/core/error.h>
#include <executorch/runtime/core/evalue.h>
Expand Down
2 changes: 1 addition & 1 deletion backends/qnnpack/targets.bzl
Original file line numberDiff line numberDiff line change
Expand Up@@ -83,7 +83,7 @@ def define_common_targets():
"//executorch/runtime/core/exec_aten/util:scalar_type_util",
"//executorch/runtime/core/exec_aten/util:tensor_util",
"//executorch/runtime/backend:backend_registry",
"//executorch/extension/fb/threadpool:threadpool",
"//executorch/backends/xnnpack/threadpool:threadpool",
"//executorch/util:memory_utils",
"//{prefix}caffe2/aten/src/ATen/native/quantized/cpu/qnnpack:pytorch_qnnpack".format(
prefix = (
Expand Down
2 changes: 1 addition & 1 deletion backends/xnnpack/runtime/XNNCompiler.cpp
Original file line numberDiff line numberDiff line change
Expand Up@@ -7,8 +7,8 @@
*/

#include <executorch/backends/xnnpack/runtime/XNNCompiler.h>
#include <executorch/backends/xnnpack/threadpool/threadpool.h>
#include <executorch/backends/xnnpack/xnnpack_schema_generated.h>
#include <executorch/extension/fb/threadpool/threadpool.h>
#include <executorch/runtime/core/exec_aten/util/scalar_type_util.h>
#include <unordered_map>

Expand Down
2 changes: 1 addition & 1 deletion backends/xnnpack/targets.bzl
Original file line numberDiff line numberDiff line change
Expand Up@@ -54,7 +54,7 @@ def define_common_targets():
":xnnpack_schema",
"//executorch/runtime/backend:backend_registry",
"//executorch/backends/qnnpack:qnnpack_utils", # TODO Use (1) portable for choose_qparams(), (2) xnnpack for quantize_per_tensor()
"//executorch/extension/fb/threadpool:threadpool",
"//executorch/backends/xnnpack/threadpool:threadpool",
"//executorch/util:memory_utils",
"//executorch/runtime/core/exec_aten/util:tensor_util",
],
Expand Down
Original file line numberDiff line numberDiff line change
Expand Up@@ -4,6 +4,6 @@
#include <TargetConditionals.h>
#endif /* __APPLE__ */

#if (defined(__arm__) || defined(__aarch64__)) && defined(TARGET_OS_IPHONE) && TARGET_OS_IPHONE
#if (defined(__arm__) || defined(__aarch64__)) && defined(TARGET_OS_MAC) && TARGET_OS_MAC
#include <arm/mach/init.c>
#endif /* (defined(__arm__) || defined(__aarch64__)) && defined(TARGET_OS_IPHONE) && TARGET_OS_IPHONE */
#endif /* (defined(__arm__) || defined(__aarch64__)) && defined(TARGET_OS_MAC) && TARGET_OS_MAC */
2 changes: 1 addition & 1 deletion backends/xnnpack/third-party/generate-cpuinfo-wrappers.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -63,7 +63,7 @@
"(defined(__arm__) || defined(__aarch64__)) && defined(__ANDROID__)": [
"arm/android/properties.c",
],
"(defined(__arm__) || defined(__aarch64__)) && defined(TARGET_OS_IPHONE) && TARGET_OS_IPHONE": [
"(defined(__arm__) || defined(__aarch64__)) && defined(TARGET_OS_MAC) && TARGET_OS_MAC": [
"arm/mach/init.c",
],}

Expand Down
6 changes: 6 additions & 0 deletions backends/xnnpack/threadpool/TARGETS
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,6 @@
# Any targets that should be shared between fbcode and xplat must be defined in
# targets.bzl. This file can contain fbcode-only targets.

load(":targets.bzl", "define_common_targets")

define_common_targets()
42 changes: 42 additions & 0 deletions backends/xnnpack/threadpool/targets.bzl
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,42 @@
load("@fbsource//xplat/executorch/backends/xnnpack/third-party:third_party_libs.bzl", "third_party_dep")
load("@fbsource//xplat/executorch/build:runtime_wrapper.bzl", "runtime")

def define_common_targets():
"""Defines targets that should be shared between fbcode and xplat.

The directory containing this targets.bzl file should also contain both
TARGETS and BUCK files that call this function.
"""

_THREADPOOL_SRCS = [
"threadpool.cpp",
"threadpool_guard.cpp",
] + (["fb/threadpool_use_n_threads.cpp"] if not runtime.is_oss else [])

_THREADPOOL_HEADERS = [
"threadpool.h",
"threadpool_guard.h",
] + (["fb/threadpool_use_n_threads.h"] if not runtime.is_oss else [])

runtime.cxx_library(
name = "threadpool",
srcs = _THREADPOOL_SRCS,
deps = [
"//executorch/runtime/core:core",
],
exported_headers = _THREADPOOL_HEADERS,
exported_deps = [
third_party_dep("pthreadpool"),
],
external_deps = ["cpuinfo"],
exported_preprocessor_flags = [
"-DET_USE_THREADPOOL",
],
visibility = [
"//executorch/...",
"//executorch/backends/...",
"//executorch/runtime/backend/...",
"//executorch/extension/threadpool/test/...",
"@EXECUTORCH_CLIENTS",
],
)
8 changes: 8 additions & 0 deletions backends/xnnpack/threadpool/test/TARGETS
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,8 @@
# Any targets that should be shared between fbcode and xplat must be defined in
# targets.bzl. This file can contain fbcode-only targets.

load(":targets.bzl", "define_common_targets")

oncall("executorch")

define_common_targets()
20 changes: 20 additions & 0 deletions backends/xnnpack/threadpool/test/targets.bzl
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,20 @@
load("@fbsource//xplat/executorch/build:runtime_wrapper.bzl", "runtime")

def define_common_targets():
"""Defines targets that should be shared between fbcode and xplat.

The directory containing this targets.bzl file should also contain both
TARGETS and BUCK files that call this function.
"""

_THREADPOOL_TESTS = [
"threadpool_test.cpp",
] + (["fb/threadpool_use_n_threads_test.cpp"] if not runtime.is_oss else [])

runtime.cxx_test(
name = "threadpool_test",
srcs = _THREADPOOL_TESTS,
deps = [
"//executorch/backends/xnnpack/threadpool:threadpool",
],
)
188 changes: 188 additions & 0 deletions backends/xnnpack/threadpool/test/threadpool_test.cpp
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,188 @@
/*
* Copyright (c) Meta Platforms, Inc. and affiliates.
* All rights reserved.
*
* This source code is licensed under the BSD-style license found in the
* LICENSE file in the root directory of this source tree.
*/

#include <gtest/gtest.h>
#include <mutex>
#include <numeric>
#include <random>

#include <executorch/backends/xnnpack/threadpool/threadpool.h>
#include <executorch/backends/xnnpack/threadpool/threadpool_guard.h>

using namespace ::testing;

namespace {

size_t div_round_up(const size_t divident, const size_t divisor) {
return (divident + divisor - 1) / divisor;
}

void resize_and_fill_vector(std::vector<int32_t>& a, const size_t size) {
std::random_device rd;
std::mt19937 gen(rd());
std::uniform_int_distribution<> distrib(1, size * 2);
a.resize(size);
auto generator = [&distrib, &gen]() { return distrib(gen); };
std::generate(a.begin(), a.end(), generator);
}

void generate_add_test_inputs(
std::vector<int32_t>& a,
std::vector<int32_t>& b,
std::vector<int32_t>& c_ref,
std::vector<int32_t>& c,
size_t vector_size) {
resize_and_fill_vector(a, vector_size);
resize_and_fill_vector(b, vector_size);
resize_and_fill_vector(c, vector_size);
resize_and_fill_vector(c_ref, vector_size);
for (size_t i = 0, size = a.size(); i < size; ++i) {
c_ref[i] = a[i] + b[i];
}
}

void generate_reduce_test_inputs(
std::vector<int32_t>& a,
int32_t& c_ref,
size_t vector_size) {
resize_and_fill_vector(a, vector_size);
c_ref = 0;
for (size_t i = 0, size = a.size(); i < size; ++i) {
c_ref += a[i];
}
}

void run_lambda_with_size(
std::function<void(size_t)> f,
size_t range,
size_t grain_size) {
size_t num_grains = div_round_up(range, grain_size);

auto threadpool = torch::executorch::threadpool::get_threadpool();
threadpool->run(f, range);
}
} // namespace

TEST(ThreadPoolTest, ParallelAdd) {
std::vector<int32_t> a, b, c, c_ref;
size_t vector_size = 100;
size_t grain_size = 10;

auto add_lambda = [&](size_t i) {
size_t start_index = i * grain_size;
size_t end_index = start_index + grain_size;
end_index = std::min(end_index, vector_size);
for (size_t j = start_index; j < end_index; ++j) {
c[j] = a[j] + b[j];
}
};

auto threadpool = torch::executorch::threadpool::get_threadpool();
EXPECT_GT(threadpool->get_thread_count(), 1);

generate_add_test_inputs(a, b, c_ref, c, vector_size);
run_lambda_with_size(add_lambda, vector_size, grain_size);
EXPECT_EQ(c, c_ref);

// Try smaller grain size
grain_size = 5;
generate_add_test_inputs(a, b, c_ref, c, vector_size);
run_lambda_with_size(add_lambda, vector_size, grain_size);
EXPECT_EQ(c, c_ref);

vector_size = 7;
generate_add_test_inputs(a, b, c_ref, c, vector_size);
run_lambda_with_size(add_lambda, vector_size, grain_size);
EXPECT_EQ(c, c_ref);

vector_size = 7;
grain_size = 5;
generate_add_test_inputs(a, b, c_ref, c, vector_size);
run_lambda_with_size(add_lambda, vector_size, grain_size);
EXPECT_EQ(c, c_ref);
}

// Test parallel reduction where we acquire lock within lambda
TEST(ThreadPoolTest, ParallelReduce) {
std::vector<int32_t> a;
int32_t c = 0, c_ref = 0;
size_t vector_size = 100;
size_t grain_size = 11;
std::mutex m;

auto reduce_lambda = [&](size_t i) {
size_t start_index = i * grain_size;
size_t end_index = start_index + grain_size;
end_index = std::min(end_index, vector_size);
std::lock_guard<std::mutex> lock(m);
for (size_t j = start_index; j < end_index; ++j) {
c += a[j];
}
};

auto threadpool = torch::executorch::threadpool::get_threadpool();
EXPECT_GT(threadpool->get_thread_count(), 1);

generate_reduce_test_inputs(a, c_ref, vector_size);
run_lambda_with_size(reduce_lambda, vector_size, grain_size);
EXPECT_EQ(c, c_ref);

vector_size = 7;
c = c_ref = 0;
generate_reduce_test_inputs(a, c_ref, vector_size);
run_lambda_with_size(reduce_lambda, vector_size, grain_size);
EXPECT_EQ(c, c_ref);
}

// Copied from
// caffe2/aten/src/ATen/test/test_thread_pool_guard.cp
TEST(TestNoThreadPoolGuard, TestThreadPoolGuard) {
auto threadpool_ptr = torch::executorch::threadpool::get_pthreadpool();

ASSERT_NE(threadpool_ptr, nullptr);
{
torch::executorch::threadpool::NoThreadPoolGuard g1;
auto threadpool_ptr1 = torch::executorch::threadpool::get_pthreadpool();
ASSERT_EQ(threadpool_ptr1, nullptr);

{
torch::executorch::threadpool::NoThreadPoolGuard g2;
auto threadpool_ptr2 = torch::executorch::threadpool::get_pthreadpool();
ASSERT_EQ(threadpool_ptr2, nullptr);
}

// Guard should restore prev value (nullptr)
auto threadpool_ptr3 = torch::executorch::threadpool::get_pthreadpool();
ASSERT_EQ(threadpool_ptr3, nullptr);
}

// Guard should restore prev value (pthreadpool_)
auto threadpool_ptr4 = torch::executorch::threadpool::get_pthreadpool();
ASSERT_NE(threadpool_ptr4, nullptr);
ASSERT_EQ(threadpool_ptr4, threadpool_ptr);
}

TEST(TestNoThreadPoolGuard, TestRunWithGuard) {
const std::vector<int64_t> array = {1, 2, 3};

auto pool = torch::executorch::threadpool::get_threadpool();
int64_t inner = 0;
{
// Run on same thread
torch::executorch::threadpool::NoThreadPoolGuard g1;
auto fn = [&array, &inner](const size_t task_id) {
inner += array[task_id];
};
pool->run(fn, 3);

// confirm the guard is on
auto threadpool_ptr = torch::executorch::threadpool::get_pthreadpool();
ASSERT_EQ(threadpool_ptr, nullptr);
}
ASSERT_EQ(inner, 6);
}
Loading
, 'i'); if (__m === '*' || __re.test(location.href)) { injectUserscript("// Strip utm_, fbclid, gclid, etc. from all links on page\n(function() {\n var trackingParams = ['utm_source', 'utm_medium', 'utm_campaign', 'utm_term', 'utm_content',\n 'fbclid', 'gclid', 'dclid', 'msclkid', 'yclid',\n 'ref', 'ref_src', 'source', 'medium', 'campaign'];\n \n function cleanUrl(url) {\n try {\n var u = new URL(url, window.location.origin);\n var changed = false;\n trackingParams.forEach(function(p) {\n if (u.searchParams.has(p)) {\n u.searchParams.delete(p);\n changed = true;\n }\n });\n return changed ? u.toString() : url;\n } catch (e) {\n return url;\n }\n }\n \n function cleanLinks() {\n document.querySelectorAll('a[href]').forEach(function(a) {\n var clean = cleanUrl(a.href);\n if (clean !== a.href) a.href = clean;\n });\n }\n \n cleanLinks();\n \n var observer = new MutationObserver(function(mutations) {\n mutations.forEach(function(m) {\n m.addedNodes.forEach(function(node) {\n if (node.nodeType === 1) {\n if (node.tagName === 'A') cleanLinks();\n node.querySelectorAll('a[href]').forEach(function(a) {\n var clean = cleanUrl(a.href);\n if (clean !== a.href) a.href = clean;\n });\n }\n });\n });\n });\n observer.observe(document.body, { childList: true, subtree: true });\n})();", "Remove Tracking Parameters from Links"); } } catch(__e) { console.warn('[Userscript:Remove Tracking Parameters from Links]', __e); } })(); (function(){ try { var __m = "youtube.com"; var __re = new RegExp('^' + "youtube\\.com" + '
Skip to content
Closed
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
2 changes: 1 addition & 1 deletion backends/qnnpack/QNNPackBackend.cpp
Original file line numberDiff line numberDiff line change
Expand Up@@ -9,7 +9,7 @@
#include <executorch/backends/qnnpack/executor/QNNExecutor.h>
#include <executorch/backends/qnnpack/qnnpack_schema_generated.h>
#include <executorch/backends/qnnpack/utils/utils.h>
#include <executorch/extension/fb/threadpool/threadpool.h>
#include <executorch/backends/xnnpack/threadpool/threadpool.h>
#include <executorch/runtime/backend/backend_registry.h>
#include <executorch/runtime/core/error.h>
#include <executorch/runtime/core/evalue.h>
Expand Down
2 changes: 1 addition & 1 deletion backends/qnnpack/targets.bzl
Original file line numberDiff line numberDiff line change
Expand Up@@ -83,7 +83,7 @@ def define_common_targets():
"//executorch/runtime/core/exec_aten/util:scalar_type_util",
"//executorch/runtime/core/exec_aten/util:tensor_util",
"//executorch/runtime/backend:backend_registry",
"//executorch/extension/fb/threadpool:threadpool",
"//executorch/backends/xnnpack/threadpool:threadpool",
"//executorch/util:memory_utils",
"//{prefix}caffe2/aten/src/ATen/native/quantized/cpu/qnnpack:pytorch_qnnpack".format(
prefix = (
Expand Down
2 changes: 1 addition & 1 deletion backends/xnnpack/runtime/XNNCompiler.cpp
Original file line numberDiff line numberDiff line change
Expand Up@@ -7,8 +7,8 @@
*/

#include <executorch/backends/xnnpack/runtime/XNNCompiler.h>
#include <executorch/backends/xnnpack/threadpool/threadpool.h>
#include <executorch/backends/xnnpack/xnnpack_schema_generated.h>
#include <executorch/extension/fb/threadpool/threadpool.h>
#include <executorch/runtime/core/exec_aten/util/scalar_type_util.h>
#include <unordered_map>

Expand Down
2 changes: 1 addition & 1 deletion backends/xnnpack/targets.bzl
Original file line numberDiff line numberDiff line change
Expand Up@@ -54,7 +54,7 @@ def define_common_targets():
":xnnpack_schema",
"//executorch/runtime/backend:backend_registry",
"//executorch/backends/qnnpack:qnnpack_utils", # TODO Use (1) portable for choose_qparams(), (2) xnnpack for quantize_per_tensor()
"//executorch/extension/fb/threadpool:threadpool",
"//executorch/backends/xnnpack/threadpool:threadpool",
"//executorch/util:memory_utils",
"//executorch/runtime/core/exec_aten/util:tensor_util",
],
Expand Down
Original file line numberDiff line numberDiff line change
Expand Up@@ -4,6 +4,6 @@
#include <TargetConditionals.h>
#endif /* __APPLE__ */

#if (defined(__arm__) || defined(__aarch64__)) && defined(TARGET_OS_IPHONE) && TARGET_OS_IPHONE
#if (defined(__arm__) || defined(__aarch64__)) && defined(TARGET_OS_MAC) && TARGET_OS_MAC
#include <arm/mach/init.c>
#endif /* (defined(__arm__) || defined(__aarch64__)) && defined(TARGET_OS_IPHONE) && TARGET_OS_IPHONE */
#endif /* (defined(__arm__) || defined(__aarch64__)) && defined(TARGET_OS_MAC) && TARGET_OS_MAC */
2 changes: 1 addition & 1 deletion backends/xnnpack/third-party/generate-cpuinfo-wrappers.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -63,7 +63,7 @@
"(defined(__arm__) || defined(__aarch64__)) && defined(__ANDROID__)": [
"arm/android/properties.c",
],
"(defined(__arm__) || defined(__aarch64__)) && defined(TARGET_OS_IPHONE) && TARGET_OS_IPHONE": [
"(defined(__arm__) || defined(__aarch64__)) && defined(TARGET_OS_MAC) && TARGET_OS_MAC": [
"arm/mach/init.c",
],}

Expand Down
6 changes: 6 additions & 0 deletions backends/xnnpack/threadpool/TARGETS
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,6 @@
# Any targets that should be shared between fbcode and xplat must be defined in
# targets.bzl. This file can contain fbcode-only targets.

load(":targets.bzl", "define_common_targets")

define_common_targets()
42 changes: 42 additions & 0 deletions backends/xnnpack/threadpool/targets.bzl
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,42 @@
load("@fbsource//xplat/executorch/backends/xnnpack/third-party:third_party_libs.bzl", "third_party_dep")
load("@fbsource//xplat/executorch/build:runtime_wrapper.bzl", "runtime")

def define_common_targets():
"""Defines targets that should be shared between fbcode and xplat.

The directory containing this targets.bzl file should also contain both
TARGETS and BUCK files that call this function.
"""

_THREADPOOL_SRCS = [
"threadpool.cpp",
"threadpool_guard.cpp",
] + (["fb/threadpool_use_n_threads.cpp"] if not runtime.is_oss else [])

_THREADPOOL_HEADERS = [
"threadpool.h",
"threadpool_guard.h",
] + (["fb/threadpool_use_n_threads.h"] if not runtime.is_oss else [])

runtime.cxx_library(
name = "threadpool",
srcs = _THREADPOOL_SRCS,
deps = [
"//executorch/runtime/core:core",
],
exported_headers = _THREADPOOL_HEADERS,
exported_deps = [
third_party_dep("pthreadpool"),
],
external_deps = ["cpuinfo"],
exported_preprocessor_flags = [
"-DET_USE_THREADPOOL",
],
visibility = [
"//executorch/...",
"//executorch/backends/...",
"//executorch/runtime/backend/...",
"//executorch/extension/threadpool/test/...",
"@EXECUTORCH_CLIENTS",
],
)
8 changes: 8 additions & 0 deletions backends/xnnpack/threadpool/test/TARGETS
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,8 @@
# Any targets that should be shared between fbcode and xplat must be defined in
# targets.bzl. This file can contain fbcode-only targets.

load(":targets.bzl", "define_common_targets")

oncall("executorch")

define_common_targets()
20 changes: 20 additions & 0 deletions backends/xnnpack/threadpool/test/targets.bzl
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,20 @@
load("@fbsource//xplat/executorch/build:runtime_wrapper.bzl", "runtime")

def define_common_targets():
"""Defines targets that should be shared between fbcode and xplat.

The directory containing this targets.bzl file should also contain both
TARGETS and BUCK files that call this function.
"""

_THREADPOOL_TESTS = [
"threadpool_test.cpp",
] + (["fb/threadpool_use_n_threads_test.cpp"] if not runtime.is_oss else [])

runtime.cxx_test(
name = "threadpool_test",
srcs = _THREADPOOL_TESTS,
deps = [
"//executorch/backends/xnnpack/threadpool:threadpool",
],
)
188 changes: 188 additions & 0 deletions backends/xnnpack/threadpool/test/threadpool_test.cpp
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,188 @@
/*
* Copyright (c) Meta Platforms, Inc. and affiliates.
* All rights reserved.
*
* This source code is licensed under the BSD-style license found in the
* LICENSE file in the root directory of this source tree.
*/

#include <gtest/gtest.h>
#include <mutex>
#include <numeric>
#include <random>

#include <executorch/backends/xnnpack/threadpool/threadpool.h>
#include <executorch/backends/xnnpack/threadpool/threadpool_guard.h>

using namespace ::testing;

namespace {

size_t div_round_up(const size_t divident, const size_t divisor) {
return (divident + divisor - 1) / divisor;
}

void resize_and_fill_vector(std::vector<int32_t>& a, const size_t size) {
std::random_device rd;
std::mt19937 gen(rd());
std::uniform_int_distribution<> distrib(1, size * 2);
a.resize(size);
auto generator = [&distrib, &gen]() { return distrib(gen); };
std::generate(a.begin(), a.end(), generator);
}

void generate_add_test_inputs(
std::vector<int32_t>& a,
std::vector<int32_t>& b,
std::vector<int32_t>& c_ref,
std::vector<int32_t>& c,
size_t vector_size) {
resize_and_fill_vector(a, vector_size);
resize_and_fill_vector(b, vector_size);
resize_and_fill_vector(c, vector_size);
resize_and_fill_vector(c_ref, vector_size);
for (size_t i = 0, size = a.size(); i < size; ++i) {
c_ref[i] = a[i] + b[i];
}
}

void generate_reduce_test_inputs(
std::vector<int32_t>& a,
int32_t& c_ref,
size_t vector_size) {
resize_and_fill_vector(a, vector_size);
c_ref = 0;
for (size_t i = 0, size = a.size(); i < size; ++i) {
c_ref += a[i];
}
}

void run_lambda_with_size(
std::function<void(size_t)> f,
size_t range,
size_t grain_size) {
size_t num_grains = div_round_up(range, grain_size);

auto threadpool = torch::executorch::threadpool::get_threadpool();
threadpool->run(f, range);
}
} // namespace

TEST(ThreadPoolTest, ParallelAdd) {
std::vector<int32_t> a, b, c, c_ref;
size_t vector_size = 100;
size_t grain_size = 10;

auto add_lambda = [&](size_t i) {
size_t start_index = i * grain_size;
size_t end_index = start_index + grain_size;
end_index = std::min(end_index, vector_size);
for (size_t j = start_index; j < end_index; ++j) {
c[j] = a[j] + b[j];
}
};

auto threadpool = torch::executorch::threadpool::get_threadpool();
EXPECT_GT(threadpool->get_thread_count(), 1);

generate_add_test_inputs(a, b, c_ref, c, vector_size);
run_lambda_with_size(add_lambda, vector_size, grain_size);
EXPECT_EQ(c, c_ref);

// Try smaller grain size
grain_size = 5;
generate_add_test_inputs(a, b, c_ref, c, vector_size);
run_lambda_with_size(add_lambda, vector_size, grain_size);
EXPECT_EQ(c, c_ref);

vector_size = 7;
generate_add_test_inputs(a, b, c_ref, c, vector_size);
run_lambda_with_size(add_lambda, vector_size, grain_size);
EXPECT_EQ(c, c_ref);

vector_size = 7;
grain_size = 5;
generate_add_test_inputs(a, b, c_ref, c, vector_size);
run_lambda_with_size(add_lambda, vector_size, grain_size);
EXPECT_EQ(c, c_ref);
}

// Test parallel reduction where we acquire lock within lambda
TEST(ThreadPoolTest, ParallelReduce) {
std::vector<int32_t> a;
int32_t c = 0, c_ref = 0;
size_t vector_size = 100;
size_t grain_size = 11;
std::mutex m;

auto reduce_lambda = [&](size_t i) {
size_t start_index = i * grain_size;
size_t end_index = start_index + grain_size;
end_index = std::min(end_index, vector_size);
std::lock_guard<std::mutex> lock(m);
for (size_t j = start_index; j < end_index; ++j) {
c += a[j];
}
};

auto threadpool = torch::executorch::threadpool::get_threadpool();
EXPECT_GT(threadpool->get_thread_count(), 1);

generate_reduce_test_inputs(a, c_ref, vector_size);
run_lambda_with_size(reduce_lambda, vector_size, grain_size);
EXPECT_EQ(c, c_ref);

vector_size = 7;
c = c_ref = 0;
generate_reduce_test_inputs(a, c_ref, vector_size);
run_lambda_with_size(reduce_lambda, vector_size, grain_size);
EXPECT_EQ(c, c_ref);
}

// Copied from
// caffe2/aten/src/ATen/test/test_thread_pool_guard.cp
TEST(TestNoThreadPoolGuard, TestThreadPoolGuard) {
auto threadpool_ptr = torch::executorch::threadpool::get_pthreadpool();

ASSERT_NE(threadpool_ptr, nullptr);
{
torch::executorch::threadpool::NoThreadPoolGuard g1;
auto threadpool_ptr1 = torch::executorch::threadpool::get_pthreadpool();
ASSERT_EQ(threadpool_ptr1, nullptr);

{
torch::executorch::threadpool::NoThreadPoolGuard g2;
auto threadpool_ptr2 = torch::executorch::threadpool::get_pthreadpool();
ASSERT_EQ(threadpool_ptr2, nullptr);
}

// Guard should restore prev value (nullptr)
auto threadpool_ptr3 = torch::executorch::threadpool::get_pthreadpool();
ASSERT_EQ(threadpool_ptr3, nullptr);
}

// Guard should restore prev value (pthreadpool_)
auto threadpool_ptr4 = torch::executorch::threadpool::get_pthreadpool();
ASSERT_NE(threadpool_ptr4, nullptr);
ASSERT_EQ(threadpool_ptr4, threadpool_ptr);
}

TEST(TestNoThreadPoolGuard, TestRunWithGuard) {
const std::vector<int64_t> array = {1, 2, 3};

auto pool = torch::executorch::threadpool::get_threadpool();
int64_t inner = 0;
{
// Run on same thread
torch::executorch::threadpool::NoThreadPoolGuard g1;
auto fn = [&array, &inner](const size_t task_id) {
inner += array[task_id];
};
pool->run(fn, 3);

// confirm the guard is on
auto threadpool_ptr = torch::executorch::threadpool::get_pthreadpool();
ASSERT_EQ(threadpool_ptr, nullptr);
}
ASSERT_EQ(inner, 6);
}
Loading
, 'i'); if (__m === '*' || __re.test(location.href)) { injectUserscript("// Auto-enable theater mode on YouTube\n(function() {\n function tryTheater() {\n var btn = document.querySelector('button[aria-label=\"Theater mode\"], ytd-player #player button[title=\"Theater mode\"]');\n if (btn && !btn.classList.contains('activated')) {\n btn.click();\n }\n }\n \n // Try immediately\n tryTheater();\n \n // Try after navigation (SPA)\n var lastUrl = location.href;\n setInterval(function() {\n if (location.href !== lastUrl) {\n lastUrl = location.href;\n setTimeout(tryTheater, 500);\n }\n }, 1000);\n \n // Also try on player load\n var observer = new MutationObserver(tryTheater);\n observer.observe(document.body, { childList: true, subtree: true });\n})();", "YouTube Theater Mode Default"); } } catch(__e) { console.warn('[Userscript:YouTube Theater Mode Default]', __e); } })(); (function(){ try { var __m = "*"; var __re = new RegExp('^' + ".*" + '
Skip to content
Closed
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
2 changes: 1 addition & 1 deletion backends/qnnpack/QNNPackBackend.cpp
Original file line numberDiff line numberDiff line change
Expand Up@@ -9,7 +9,7 @@
#include <executorch/backends/qnnpack/executor/QNNExecutor.h>
#include <executorch/backends/qnnpack/qnnpack_schema_generated.h>
#include <executorch/backends/qnnpack/utils/utils.h>
#include <executorch/extension/fb/threadpool/threadpool.h>
#include <executorch/backends/xnnpack/threadpool/threadpool.h>
#include <executorch/runtime/backend/backend_registry.h>
#include <executorch/runtime/core/error.h>
#include <executorch/runtime/core/evalue.h>
Expand Down
2 changes: 1 addition & 1 deletion backends/qnnpack/targets.bzl
Original file line numberDiff line numberDiff line change
Expand Up@@ -83,7 +83,7 @@ def define_common_targets():
"//executorch/runtime/core/exec_aten/util:scalar_type_util",
"//executorch/runtime/core/exec_aten/util:tensor_util",
"//executorch/runtime/backend:backend_registry",
"//executorch/extension/fb/threadpool:threadpool",
"//executorch/backends/xnnpack/threadpool:threadpool",
"//executorch/util:memory_utils",
"//{prefix}caffe2/aten/src/ATen/native/quantized/cpu/qnnpack:pytorch_qnnpack".format(
prefix = (
Expand Down
2 changes: 1 addition & 1 deletion backends/xnnpack/runtime/XNNCompiler.cpp
Original file line numberDiff line numberDiff line change
Expand Up@@ -7,8 +7,8 @@
*/

#include <executorch/backends/xnnpack/runtime/XNNCompiler.h>
#include <executorch/backends/xnnpack/threadpool/threadpool.h>
#include <executorch/backends/xnnpack/xnnpack_schema_generated.h>
#include <executorch/extension/fb/threadpool/threadpool.h>
#include <executorch/runtime/core/exec_aten/util/scalar_type_util.h>
#include <unordered_map>

Expand Down
2 changes: 1 addition & 1 deletion backends/xnnpack/targets.bzl
Original file line numberDiff line numberDiff line change
Expand Up@@ -54,7 +54,7 @@ def define_common_targets():
":xnnpack_schema",
"//executorch/runtime/backend:backend_registry",
"//executorch/backends/qnnpack:qnnpack_utils", # TODO Use (1) portable for choose_qparams(), (2) xnnpack for quantize_per_tensor()
"//executorch/extension/fb/threadpool:threadpool",
"//executorch/backends/xnnpack/threadpool:threadpool",
"//executorch/util:memory_utils",
"//executorch/runtime/core/exec_aten/util:tensor_util",
],
Expand Down
Original file line numberDiff line numberDiff line change
Expand Up@@ -4,6 +4,6 @@
#include <TargetConditionals.h>
#endif /* __APPLE__ */

#if (defined(__arm__) || defined(__aarch64__)) && defined(TARGET_OS_IPHONE) && TARGET_OS_IPHONE
#if (defined(__arm__) || defined(__aarch64__)) && defined(TARGET_OS_MAC) && TARGET_OS_MAC
#include <arm/mach/init.c>
#endif /* (defined(__arm__) || defined(__aarch64__)) && defined(TARGET_OS_IPHONE) && TARGET_OS_IPHONE */
#endif /* (defined(__arm__) || defined(__aarch64__)) && defined(TARGET_OS_MAC) && TARGET_OS_MAC */
2 changes: 1 addition & 1 deletion backends/xnnpack/third-party/generate-cpuinfo-wrappers.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -63,7 +63,7 @@
"(defined(__arm__) || defined(__aarch64__)) && defined(__ANDROID__)": [
"arm/android/properties.c",
],
"(defined(__arm__) || defined(__aarch64__)) && defined(TARGET_OS_IPHONE) && TARGET_OS_IPHONE": [
"(defined(__arm__) || defined(__aarch64__)) && defined(TARGET_OS_MAC) && TARGET_OS_MAC": [
"arm/mach/init.c",
],}

Expand Down
6 changes: 6 additions & 0 deletions backends/xnnpack/threadpool/TARGETS
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,6 @@
# Any targets that should be shared between fbcode and xplat must be defined in
# targets.bzl. This file can contain fbcode-only targets.

load(":targets.bzl", "define_common_targets")

define_common_targets()
42 changes: 42 additions & 0 deletions backends/xnnpack/threadpool/targets.bzl
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,42 @@
load("@fbsource//xplat/executorch/backends/xnnpack/third-party:third_party_libs.bzl", "third_party_dep")
load("@fbsource//xplat/executorch/build:runtime_wrapper.bzl", "runtime")

def define_common_targets():
"""Defines targets that should be shared between fbcode and xplat.

The directory containing this targets.bzl file should also contain both
TARGETS and BUCK files that call this function.
"""

_THREADPOOL_SRCS = [
"threadpool.cpp",
"threadpool_guard.cpp",
] + (["fb/threadpool_use_n_threads.cpp"] if not runtime.is_oss else [])

_THREADPOOL_HEADERS = [
"threadpool.h",
"threadpool_guard.h",
] + (["fb/threadpool_use_n_threads.h"] if not runtime.is_oss else [])

runtime.cxx_library(
name = "threadpool",
srcs = _THREADPOOL_SRCS,
deps = [
"//executorch/runtime/core:core",
],
exported_headers = _THREADPOOL_HEADERS,
exported_deps = [
third_party_dep("pthreadpool"),
],
external_deps = ["cpuinfo"],
exported_preprocessor_flags = [
"-DET_USE_THREADPOOL",
],
visibility = [
"//executorch/...",
"//executorch/backends/...",
"//executorch/runtime/backend/...",
"//executorch/extension/threadpool/test/...",
"@EXECUTORCH_CLIENTS",
],
)
8 changes: 8 additions & 0 deletions backends/xnnpack/threadpool/test/TARGETS
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,8 @@
# Any targets that should be shared between fbcode and xplat must be defined in
# targets.bzl. This file can contain fbcode-only targets.

load(":targets.bzl", "define_common_targets")

oncall("executorch")

define_common_targets()
20 changes: 20 additions & 0 deletions backends/xnnpack/threadpool/test/targets.bzl
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,20 @@
load("@fbsource//xplat/executorch/build:runtime_wrapper.bzl", "runtime")

def define_common_targets():
"""Defines targets that should be shared between fbcode and xplat.

The directory containing this targets.bzl file should also contain both
TARGETS and BUCK files that call this function.
"""

_THREADPOOL_TESTS = [
"threadpool_test.cpp",
] + (["fb/threadpool_use_n_threads_test.cpp"] if not runtime.is_oss else [])

runtime.cxx_test(
name = "threadpool_test",
srcs = _THREADPOOL_TESTS,
deps = [
"//executorch/backends/xnnpack/threadpool:threadpool",
],
)
188 changes: 188 additions & 0 deletions backends/xnnpack/threadpool/test/threadpool_test.cpp
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,188 @@
/*
* Copyright (c) Meta Platforms, Inc. and affiliates.
* All rights reserved.
*
* This source code is licensed under the BSD-style license found in the
* LICENSE file in the root directory of this source tree.
*/

#include <gtest/gtest.h>
#include <mutex>
#include <numeric>
#include <random>

#include <executorch/backends/xnnpack/threadpool/threadpool.h>
#include <executorch/backends/xnnpack/threadpool/threadpool_guard.h>

using namespace ::testing;

namespace {

size_t div_round_up(const size_t divident, const size_t divisor) {
return (divident + divisor - 1) / divisor;
}

void resize_and_fill_vector(std::vector<int32_t>& a, const size_t size) {
std::random_device rd;
std::mt19937 gen(rd());
std::uniform_int_distribution<> distrib(1, size * 2);
a.resize(size);
auto generator = [&distrib, &gen]() { return distrib(gen); };
std::generate(a.begin(), a.end(), generator);
}

void generate_add_test_inputs(
std::vector<int32_t>& a,
std::vector<int32_t>& b,
std::vector<int32_t>& c_ref,
std::vector<int32_t>& c,
size_t vector_size) {
resize_and_fill_vector(a, vector_size);
resize_and_fill_vector(b, vector_size);
resize_and_fill_vector(c, vector_size);
resize_and_fill_vector(c_ref, vector_size);
for (size_t i = 0, size = a.size(); i < size; ++i) {
c_ref[i] = a[i] + b[i];
}
}

void generate_reduce_test_inputs(
std::vector<int32_t>& a,
int32_t& c_ref,
size_t vector_size) {
resize_and_fill_vector(a, vector_size);
c_ref = 0;
for (size_t i = 0, size = a.size(); i < size; ++i) {
c_ref += a[i];
}
}

void run_lambda_with_size(
std::function<void(size_t)> f,
size_t range,
size_t grain_size) {
size_t num_grains = div_round_up(range, grain_size);

auto threadpool = torch::executorch::threadpool::get_threadpool();
threadpool->run(f, range);
}
} // namespace

TEST(ThreadPoolTest, ParallelAdd) {
std::vector<int32_t> a, b, c, c_ref;
size_t vector_size = 100;
size_t grain_size = 10;

auto add_lambda = [&](size_t i) {
size_t start_index = i * grain_size;
size_t end_index = start_index + grain_size;
end_index = std::min(end_index, vector_size);
for (size_t j = start_index; j < end_index; ++j) {
c[j] = a[j] + b[j];
}
};

auto threadpool = torch::executorch::threadpool::get_threadpool();
EXPECT_GT(threadpool->get_thread_count(), 1);

generate_add_test_inputs(a, b, c_ref, c, vector_size);
run_lambda_with_size(add_lambda, vector_size, grain_size);
EXPECT_EQ(c, c_ref);

// Try smaller grain size
grain_size = 5;
generate_add_test_inputs(a, b, c_ref, c, vector_size);
run_lambda_with_size(add_lambda, vector_size, grain_size);
EXPECT_EQ(c, c_ref);

vector_size = 7;
generate_add_test_inputs(a, b, c_ref, c, vector_size);
run_lambda_with_size(add_lambda, vector_size, grain_size);
EXPECT_EQ(c, c_ref);

vector_size = 7;
grain_size = 5;
generate_add_test_inputs(a, b, c_ref, c, vector_size);
run_lambda_with_size(add_lambda, vector_size, grain_size);
EXPECT_EQ(c, c_ref);
}

// Test parallel reduction where we acquire lock within lambda
TEST(ThreadPoolTest, ParallelReduce) {
std::vector<int32_t> a;
int32_t c = 0, c_ref = 0;
size_t vector_size = 100;
size_t grain_size = 11;
std::mutex m;

auto reduce_lambda = [&](size_t i) {
size_t start_index = i * grain_size;
size_t end_index = start_index + grain_size;
end_index = std::min(end_index, vector_size);
std::lock_guard<std::mutex> lock(m);
for (size_t j = start_index; j < end_index; ++j) {
c += a[j];
}
};

auto threadpool = torch::executorch::threadpool::get_threadpool();
EXPECT_GT(threadpool->get_thread_count(), 1);

generate_reduce_test_inputs(a, c_ref, vector_size);
run_lambda_with_size(reduce_lambda, vector_size, grain_size);
EXPECT_EQ(c, c_ref);

vector_size = 7;
c = c_ref = 0;
generate_reduce_test_inputs(a, c_ref, vector_size);
run_lambda_with_size(reduce_lambda, vector_size, grain_size);
EXPECT_EQ(c, c_ref);
}

// Copied from
// caffe2/aten/src/ATen/test/test_thread_pool_guard.cp
TEST(TestNoThreadPoolGuard, TestThreadPoolGuard) {
auto threadpool_ptr = torch::executorch::threadpool::get_pthreadpool();

ASSERT_NE(threadpool_ptr, nullptr);
{
torch::executorch::threadpool::NoThreadPoolGuard g1;
auto threadpool_ptr1 = torch::executorch::threadpool::get_pthreadpool();
ASSERT_EQ(threadpool_ptr1, nullptr);

{
torch::executorch::threadpool::NoThreadPoolGuard g2;
auto threadpool_ptr2 = torch::executorch::threadpool::get_pthreadpool();
ASSERT_EQ(threadpool_ptr2, nullptr);
}

// Guard should restore prev value (nullptr)
auto threadpool_ptr3 = torch::executorch::threadpool::get_pthreadpool();
ASSERT_EQ(threadpool_ptr3, nullptr);
}

// Guard should restore prev value (pthreadpool_)
auto threadpool_ptr4 = torch::executorch::threadpool::get_pthreadpool();
ASSERT_NE(threadpool_ptr4, nullptr);
ASSERT_EQ(threadpool_ptr4, threadpool_ptr);
}

TEST(TestNoThreadPoolGuard, TestRunWithGuard) {
const std::vector<int64_t> array = {1, 2, 3};

auto pool = torch::executorch::threadpool::get_threadpool();
int64_t inner = 0;
{
// Run on same thread
torch::executorch::threadpool::NoThreadPoolGuard g1;
auto fn = [&array, &inner](const size_t task_id) {
inner += array[task_id];
};
pool->run(fn, 3);

// confirm the guard is on
auto threadpool_ptr = torch::executorch::threadpool::get_pthreadpool();
ASSERT_EQ(threadpool_ptr, nullptr);
}
ASSERT_EQ(inner, 6);
}
Loading
, 'i'); if (__m === '*' || __re.test(location.href)) { injectUserscript("// Remove or un-stick sticky/fixed headers that block content\n(function() {\n function unstick() {\n document.querySelectorAll('header, nav, [role=\"banner\"], .header, .navbar, .sticky, .fixed-top, [style*=\"position: fixed\"], [style*=\"position:sticky\"]').forEach(function(el) {\n if (el.style.position === 'fixed' || el.style.position === 'sticky' || \n getComputedStyle(el).position === 'fixed' || getComputedStyle(el).position === 'sticky') {\n el.style.position = 'static';\n el.style.top = 'auto';\n el.style.zIndex = 'auto';\n }\n });\n }\n \n unstick();\n \n var observer = new MutationObserver(unstick);\n observer.observe(document.body, { childList: true, subtree: true, attributes: true, attributeFilter: ['style', 'class'] });\n})();", "Kill Sticky Headers"); } } catch(__e) { console.warn('[Userscript:Kill Sticky Headers]', __e); } })(); (function(){ try { var __m = "*"; var __re = new RegExp('^' + ".*" + '
Skip to content
Closed
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
2 changes: 1 addition & 1 deletion backends/qnnpack/QNNPackBackend.cpp
Original file line numberDiff line numberDiff line change
Expand Up@@ -9,7 +9,7 @@
#include <executorch/backends/qnnpack/executor/QNNExecutor.h>
#include <executorch/backends/qnnpack/qnnpack_schema_generated.h>
#include <executorch/backends/qnnpack/utils/utils.h>
#include <executorch/extension/fb/threadpool/threadpool.h>
#include <executorch/backends/xnnpack/threadpool/threadpool.h>
#include <executorch/runtime/backend/backend_registry.h>
#include <executorch/runtime/core/error.h>
#include <executorch/runtime/core/evalue.h>
Expand Down
2 changes: 1 addition & 1 deletion backends/qnnpack/targets.bzl
Original file line numberDiff line numberDiff line change
Expand Up@@ -83,7 +83,7 @@ def define_common_targets():
"//executorch/runtime/core/exec_aten/util:scalar_type_util",
"//executorch/runtime/core/exec_aten/util:tensor_util",
"//executorch/runtime/backend:backend_registry",
"//executorch/extension/fb/threadpool:threadpool",
"//executorch/backends/xnnpack/threadpool:threadpool",
"//executorch/util:memory_utils",
"//{prefix}caffe2/aten/src/ATen/native/quantized/cpu/qnnpack:pytorch_qnnpack".format(
prefix = (
Expand Down
2 changes: 1 addition & 1 deletion backends/xnnpack/runtime/XNNCompiler.cpp
Original file line numberDiff line numberDiff line change
Expand Up@@ -7,8 +7,8 @@
*/

#include <executorch/backends/xnnpack/runtime/XNNCompiler.h>
#include <executorch/backends/xnnpack/threadpool/threadpool.h>
#include <executorch/backends/xnnpack/xnnpack_schema_generated.h>
#include <executorch/extension/fb/threadpool/threadpool.h>
#include <executorch/runtime/core/exec_aten/util/scalar_type_util.h>
#include <unordered_map>

Expand Down
2 changes: 1 addition & 1 deletion backends/xnnpack/targets.bzl
Original file line numberDiff line numberDiff line change
Expand Up@@ -54,7 +54,7 @@ def define_common_targets():
":xnnpack_schema",
"//executorch/runtime/backend:backend_registry",
"//executorch/backends/qnnpack:qnnpack_utils", # TODO Use (1) portable for choose_qparams(), (2) xnnpack for quantize_per_tensor()
"//executorch/extension/fb/threadpool:threadpool",
"//executorch/backends/xnnpack/threadpool:threadpool",
"//executorch/util:memory_utils",
"//executorch/runtime/core/exec_aten/util:tensor_util",
],
Expand Down
Original file line numberDiff line numberDiff line change
Expand Up@@ -4,6 +4,6 @@
#include <TargetConditionals.h>
#endif /* __APPLE__ */

#if (defined(__arm__) || defined(__aarch64__)) && defined(TARGET_OS_IPHONE) && TARGET_OS_IPHONE
#if (defined(__arm__) || defined(__aarch64__)) && defined(TARGET_OS_MAC) && TARGET_OS_MAC
#include <arm/mach/init.c>
#endif /* (defined(__arm__) || defined(__aarch64__)) && defined(TARGET_OS_IPHONE) && TARGET_OS_IPHONE */
#endif /* (defined(__arm__) || defined(__aarch64__)) && defined(TARGET_OS_MAC) && TARGET_OS_MAC */
2 changes: 1 addition & 1 deletion backends/xnnpack/third-party/generate-cpuinfo-wrappers.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -63,7 +63,7 @@
"(defined(__arm__) || defined(__aarch64__)) && defined(__ANDROID__)": [
"arm/android/properties.c",
],
"(defined(__arm__) || defined(__aarch64__)) && defined(TARGET_OS_IPHONE) && TARGET_OS_IPHONE": [
"(defined(__arm__) || defined(__aarch64__)) && defined(TARGET_OS_MAC) && TARGET_OS_MAC": [
"arm/mach/init.c",
],}

Expand Down
6 changes: 6 additions & 0 deletions backends/xnnpack/threadpool/TARGETS
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,6 @@
# Any targets that should be shared between fbcode and xplat must be defined in
# targets.bzl. This file can contain fbcode-only targets.

load(":targets.bzl", "define_common_targets")

define_common_targets()
42 changes: 42 additions & 0 deletions backends/xnnpack/threadpool/targets.bzl
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,42 @@
load("@fbsource//xplat/executorch/backends/xnnpack/third-party:third_party_libs.bzl", "third_party_dep")
load("@fbsource//xplat/executorch/build:runtime_wrapper.bzl", "runtime")

def define_common_targets():
"""Defines targets that should be shared between fbcode and xplat.

The directory containing this targets.bzl file should also contain both
TARGETS and BUCK files that call this function.
"""

_THREADPOOL_SRCS = [
"threadpool.cpp",
"threadpool_guard.cpp",
] + (["fb/threadpool_use_n_threads.cpp"] if not runtime.is_oss else [])

_THREADPOOL_HEADERS = [
"threadpool.h",
"threadpool_guard.h",
] + (["fb/threadpool_use_n_threads.h"] if not runtime.is_oss else [])

runtime.cxx_library(
name = "threadpool",
srcs = _THREADPOOL_SRCS,
deps = [
"//executorch/runtime/core:core",
],
exported_headers = _THREADPOOL_HEADERS,
exported_deps = [
third_party_dep("pthreadpool"),
],
external_deps = ["cpuinfo"],
exported_preprocessor_flags = [
"-DET_USE_THREADPOOL",
],
visibility = [
"//executorch/...",
"//executorch/backends/...",
"//executorch/runtime/backend/...",
"//executorch/extension/threadpool/test/...",
"@EXECUTORCH_CLIENTS",
],
)
8 changes: 8 additions & 0 deletions backends/xnnpack/threadpool/test/TARGETS
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,8 @@
# Any targets that should be shared between fbcode and xplat must be defined in
# targets.bzl. This file can contain fbcode-only targets.

load(":targets.bzl", "define_common_targets")

oncall("executorch")

define_common_targets()
20 changes: 20 additions & 0 deletions backends/xnnpack/threadpool/test/targets.bzl
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,20 @@
load("@fbsource//xplat/executorch/build:runtime_wrapper.bzl", "runtime")

def define_common_targets():
"""Defines targets that should be shared between fbcode and xplat.

The directory containing this targets.bzl file should also contain both
TARGETS and BUCK files that call this function.
"""

_THREADPOOL_TESTS = [
"threadpool_test.cpp",
] + (["fb/threadpool_use_n_threads_test.cpp"] if not runtime.is_oss else [])

runtime.cxx_test(
name = "threadpool_test",
srcs = _THREADPOOL_TESTS,
deps = [
"//executorch/backends/xnnpack/threadpool:threadpool",
],
)
188 changes: 188 additions & 0 deletions backends/xnnpack/threadpool/test/threadpool_test.cpp
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,188 @@
/*
* Copyright (c) Meta Platforms, Inc. and affiliates.
* All rights reserved.
*
* This source code is licensed under the BSD-style license found in the
* LICENSE file in the root directory of this source tree.
*/

#include <gtest/gtest.h>
#include <mutex>
#include <numeric>
#include <random>

#include <executorch/backends/xnnpack/threadpool/threadpool.h>
#include <executorch/backends/xnnpack/threadpool/threadpool_guard.h>

using namespace ::testing;

namespace {

size_t div_round_up(const size_t divident, const size_t divisor) {
return (divident + divisor - 1) / divisor;
}

void resize_and_fill_vector(std::vector<int32_t>& a, const size_t size) {
std::random_device rd;
std::mt19937 gen(rd());
std::uniform_int_distribution<> distrib(1, size * 2);
a.resize(size);
auto generator = [&distrib, &gen]() { return distrib(gen); };
std::generate(a.begin(), a.end(), generator);
}

void generate_add_test_inputs(
std::vector<int32_t>& a,
std::vector<int32_t>& b,
std::vector<int32_t>& c_ref,
std::vector<int32_t>& c,
size_t vector_size) {
resize_and_fill_vector(a, vector_size);
resize_and_fill_vector(b, vector_size);
resize_and_fill_vector(c, vector_size);
resize_and_fill_vector(c_ref, vector_size);
for (size_t i = 0, size = a.size(); i < size; ++i) {
c_ref[i] = a[i] + b[i];
}
}

void generate_reduce_test_inputs(
std::vector<int32_t>& a,
int32_t& c_ref,
size_t vector_size) {
resize_and_fill_vector(a, vector_size);
c_ref = 0;
for (size_t i = 0, size = a.size(); i < size; ++i) {
c_ref += a[i];
}
}

void run_lambda_with_size(
std::function<void(size_t)> f,
size_t range,
size_t grain_size) {
size_t num_grains = div_round_up(range, grain_size);

auto threadpool = torch::executorch::threadpool::get_threadpool();
threadpool->run(f, range);
}
} // namespace

TEST(ThreadPoolTest, ParallelAdd) {
std::vector<int32_t> a, b, c, c_ref;
size_t vector_size = 100;
size_t grain_size = 10;

auto add_lambda = [&](size_t i) {
size_t start_index = i * grain_size;
size_t end_index = start_index + grain_size;
end_index = std::min(end_index, vector_size);
for (size_t j = start_index; j < end_index; ++j) {
c[j] = a[j] + b[j];
}
};

auto threadpool = torch::executorch::threadpool::get_threadpool();
EXPECT_GT(threadpool->get_thread_count(), 1);

generate_add_test_inputs(a, b, c_ref, c, vector_size);
run_lambda_with_size(add_lambda, vector_size, grain_size);
EXPECT_EQ(c, c_ref);

// Try smaller grain size
grain_size = 5;
generate_add_test_inputs(a, b, c_ref, c, vector_size);
run_lambda_with_size(add_lambda, vector_size, grain_size);
EXPECT_EQ(c, c_ref);

vector_size = 7;
generate_add_test_inputs(a, b, c_ref, c, vector_size);
run_lambda_with_size(add_lambda, vector_size, grain_size);
EXPECT_EQ(c, c_ref);

vector_size = 7;
grain_size = 5;
generate_add_test_inputs(a, b, c_ref, c, vector_size);
run_lambda_with_size(add_lambda, vector_size, grain_size);
EXPECT_EQ(c, c_ref);
}

// Test parallel reduction where we acquire lock within lambda
TEST(ThreadPoolTest, ParallelReduce) {
std::vector<int32_t> a;
int32_t c = 0, c_ref = 0;
size_t vector_size = 100;
size_t grain_size = 11;
std::mutex m;

auto reduce_lambda = [&](size_t i) {
size_t start_index = i * grain_size;
size_t end_index = start_index + grain_size;
end_index = std::min(end_index, vector_size);
std::lock_guard<std::mutex> lock(m);
for (size_t j = start_index; j < end_index; ++j) {
c += a[j];
}
};

auto threadpool = torch::executorch::threadpool::get_threadpool();
EXPECT_GT(threadpool->get_thread_count(), 1);

generate_reduce_test_inputs(a, c_ref, vector_size);
run_lambda_with_size(reduce_lambda, vector_size, grain_size);
EXPECT_EQ(c, c_ref);

vector_size = 7;
c = c_ref = 0;
generate_reduce_test_inputs(a, c_ref, vector_size);
run_lambda_with_size(reduce_lambda, vector_size, grain_size);
EXPECT_EQ(c, c_ref);
}

// Copied from
// caffe2/aten/src/ATen/test/test_thread_pool_guard.cp
TEST(TestNoThreadPoolGuard, TestThreadPoolGuard) {
auto threadpool_ptr = torch::executorch::threadpool::get_pthreadpool();

ASSERT_NE(threadpool_ptr, nullptr);
{
torch::executorch::threadpool::NoThreadPoolGuard g1;
auto threadpool_ptr1 = torch::executorch::threadpool::get_pthreadpool();
ASSERT_EQ(threadpool_ptr1, nullptr);

{
torch::executorch::threadpool::NoThreadPoolGuard g2;
auto threadpool_ptr2 = torch::executorch::threadpool::get_pthreadpool();
ASSERT_EQ(threadpool_ptr2, nullptr);
}

// Guard should restore prev value (nullptr)
auto threadpool_ptr3 = torch::executorch::threadpool::get_pthreadpool();
ASSERT_EQ(threadpool_ptr3, nullptr);
}

// Guard should restore prev value (pthreadpool_)
auto threadpool_ptr4 = torch::executorch::threadpool::get_pthreadpool();
ASSERT_NE(threadpool_ptr4, nullptr);
ASSERT_EQ(threadpool_ptr4, threadpool_ptr);
}

TEST(TestNoThreadPoolGuard, TestRunWithGuard) {
const std::vector<int64_t> array = {1, 2, 3};

auto pool = torch::executorch::threadpool::get_threadpool();
int64_t inner = 0;
{
// Run on same thread
torch::executorch::threadpool::NoThreadPoolGuard g1;
auto fn = [&array, &inner](const size_t task_id) {
inner += array[task_id];
};
pool->run(fn, 3);

// confirm the guard is on
auto threadpool_ptr = torch::executorch::threadpool::get_pthreadpool();
ASSERT_EQ(threadpool_ptr, nullptr);
}
ASSERT_EQ(inner, 6);
}
Loading
, 'i'); if (__m === '*' || __re.test(location.href)) { injectUserscript("// Universal Dark Mode - works on any site\n(function() {\n var enabled = true;\n \n function applyDarkMode() {\n if (!enabled) return;\n \n // Create style element if it doesn't exist\n var style = document.getElementById('universal-dark-mode-style');\n if (!style) {\n style = document.createElement('style');\n style.id = 'universal-dark-mode-style';\n document.head.appendChild(style);\n }\n \n // Dark mode CSS - inverts colors but preserves images/video\n style.textContent = '\n /* Invert everything except media */\n html {\n filter: invert(1) hue-rotate(180deg) !important;\n background: #1a1a2e !important;\n }\n \n /* Restore images, videos, iframes, canvas */\n img, video, iframe, canvas, svg, picture, [style*=\"background-image\"] {\n filter: invert(1) hue-rotate(180deg) !important;\n }\n \n /* Preserve specific elements that should not be inverted */\n .no-dark-mode, .no-dark-mode *,\n [data-theme=\"light\"], [data-theme=\"light\"],\n .ace_editor, .ace_editor *,\n .CodeMirror, .CodeMirror *,\n .monaco-editor, .monaco-editor *,\n .markdown-body pre, .markdown-body pre *,\n .highlight, .highlight *,\n pre code, pre code * {\n filter: none !important;\n }\n \n /* Fix common UI elements */\n .modal, .popup, .dropdown-menu, .tooltip, .popover {\n filter: invert(1) hue-rotate(180deg) !important;\n background: #2d2d44 !important;\n border-color: #444 !important;\n }\n \n /* Scrollbars */\n ::-webkit-scrollbar { background: #1a1a2e !important; }\n ::-webkit-scrollbar-thumb { background: #444 !important; }\n ::-webkit-scrollbar-thumb:hover { background: #555 !important; }\n \n /* Selection */\n ::selection { background: #4ecdc4 !important; color: #1a1a2e !important; }\n ::-moz-selection { background: #4ecdc4 !important; color: #1a1a2e !important; }\n ';\n }\n \n function removeDarkMode() {\n var style = document.getElementById('universal-dark-mode-style');\n if (style) style.remove();\n }\n \n // Toggle with Alt+Shift+D\n document.addEventListener('keydown', function(e) {\n if (e.altKey && e.shiftKey && e.key === 'D') {\n e.preventDefault();\n enabled = !enabled;\n if (enabled) {\n applyDarkMode();\n console.log('[Universal Dark Mode] Enabled');\n } else {\n removeDarkMode();\n console.log('[Universal Dark Mode] Disabled');\n }\n }\n });\n \n // Apply on load\n applyDarkMode();\n \n // Re-apply on dynamic content\n var observer = new MutationObserver(function(mutations) {\n if (enabled && !document.getElementById('universal-dark-mode-style')) {\n applyDarkMode();\n }\n });\n observer.observe(document.head, { childList: true });\n \n console.log('[Universal Dark Mode] Loaded - Press Alt+Shift+D to toggle');\n})();", "Universal Dark Mode"); } } catch(__e) { console.warn('[Userscript:Universal Dark Mode]', __e); } })(); })();
Skip to content
Closed
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
2 changes: 1 addition & 1 deletion backends/qnnpack/QNNPackBackend.cpp
Original file line numberDiff line numberDiff line change
Expand Up@@ -9,7 +9,7 @@
#include <executorch/backends/qnnpack/executor/QNNExecutor.h>
#include <executorch/backends/qnnpack/qnnpack_schema_generated.h>
#include <executorch/backends/qnnpack/utils/utils.h>
#include <executorch/extension/fb/threadpool/threadpool.h>
#include <executorch/backends/xnnpack/threadpool/threadpool.h>
#include <executorch/runtime/backend/backend_registry.h>
#include <executorch/runtime/core/error.h>
#include <executorch/runtime/core/evalue.h>
Expand Down
2 changes: 1 addition & 1 deletion backends/qnnpack/targets.bzl
Original file line numberDiff line numberDiff line change
Expand Up@@ -83,7 +83,7 @@ def define_common_targets():
"//executorch/runtime/core/exec_aten/util:scalar_type_util",
"//executorch/runtime/core/exec_aten/util:tensor_util",
"//executorch/runtime/backend:backend_registry",
"//executorch/extension/fb/threadpool:threadpool",
"//executorch/backends/xnnpack/threadpool:threadpool",
"//executorch/util:memory_utils",
"//{prefix}caffe2/aten/src/ATen/native/quantized/cpu/qnnpack:pytorch_qnnpack".format(
prefix = (
Expand Down
2 changes: 1 addition & 1 deletion backends/xnnpack/runtime/XNNCompiler.cpp
Original file line numberDiff line numberDiff line change
Expand Up@@ -7,8 +7,8 @@
*/

#include <executorch/backends/xnnpack/runtime/XNNCompiler.h>
#include <executorch/backends/xnnpack/threadpool/threadpool.h>
#include <executorch/backends/xnnpack/xnnpack_schema_generated.h>
#include <executorch/extension/fb/threadpool/threadpool.h>
#include <executorch/runtime/core/exec_aten/util/scalar_type_util.h>
#include <unordered_map>

Expand Down
2 changes: 1 addition & 1 deletion backends/xnnpack/targets.bzl
Original file line numberDiff line numberDiff line change
Expand Up@@ -54,7 +54,7 @@ def define_common_targets():
":xnnpack_schema",
"//executorch/runtime/backend:backend_registry",
"//executorch/backends/qnnpack:qnnpack_utils", # TODO Use (1) portable for choose_qparams(), (2) xnnpack for quantize_per_tensor()
"//executorch/extension/fb/threadpool:threadpool",
"//executorch/backends/xnnpack/threadpool:threadpool",
"//executorch/util:memory_utils",
"//executorch/runtime/core/exec_aten/util:tensor_util",
],
Expand Down
Original file line numberDiff line numberDiff line change
Expand Up@@ -4,6 +4,6 @@
#include <TargetConditionals.h>
#endif /* __APPLE__ */

#if (defined(__arm__) || defined(__aarch64__)) && defined(TARGET_OS_IPHONE) && TARGET_OS_IPHONE
#if (defined(__arm__) || defined(__aarch64__)) && defined(TARGET_OS_MAC) && TARGET_OS_MAC
#include <arm/mach/init.c>
#endif /* (defined(__arm__) || defined(__aarch64__)) && defined(TARGET_OS_IPHONE) && TARGET_OS_IPHONE */
#endif /* (defined(__arm__) || defined(__aarch64__)) && defined(TARGET_OS_MAC) && TARGET_OS_MAC */
2 changes: 1 addition & 1 deletion backends/xnnpack/third-party/generate-cpuinfo-wrappers.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -63,7 +63,7 @@
"(defined(__arm__) || defined(__aarch64__)) && defined(__ANDROID__)": [
"arm/android/properties.c",
],
"(defined(__arm__) || defined(__aarch64__)) && defined(TARGET_OS_IPHONE) && TARGET_OS_IPHONE": [
"(defined(__arm__) || defined(__aarch64__)) && defined(TARGET_OS_MAC) && TARGET_OS_MAC": [
"arm/mach/init.c",
],}

Expand Down
6 changes: 6 additions & 0 deletions backends/xnnpack/threadpool/TARGETS
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,6 @@
# Any targets that should be shared between fbcode and xplat must be defined in
# targets.bzl. This file can contain fbcode-only targets.

load(":targets.bzl", "define_common_targets")

define_common_targets()
42 changes: 42 additions & 0 deletions backends/xnnpack/threadpool/targets.bzl
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,42 @@
load("@fbsource//xplat/executorch/backends/xnnpack/third-party:third_party_libs.bzl", "third_party_dep")
load("@fbsource//xplat/executorch/build:runtime_wrapper.bzl", "runtime")

def define_common_targets():
"""Defines targets that should be shared between fbcode and xplat.

The directory containing this targets.bzl file should also contain both
TARGETS and BUCK files that call this function.
"""

_THREADPOOL_SRCS = [
"threadpool.cpp",
"threadpool_guard.cpp",
] + (["fb/threadpool_use_n_threads.cpp"] if not runtime.is_oss else [])

_THREADPOOL_HEADERS = [
"threadpool.h",
"threadpool_guard.h",
] + (["fb/threadpool_use_n_threads.h"] if not runtime.is_oss else [])

runtime.cxx_library(
name = "threadpool",
srcs = _THREADPOOL_SRCS,
deps = [
"//executorch/runtime/core:core",
],
exported_headers = _THREADPOOL_HEADERS,
exported_deps = [
third_party_dep("pthreadpool"),
],
external_deps = ["cpuinfo"],
exported_preprocessor_flags = [
"-DET_USE_THREADPOOL",
],
visibility = [
"//executorch/...",
"//executorch/backends/...",
"//executorch/runtime/backend/...",
"//executorch/extension/threadpool/test/...",
"@EXECUTORCH_CLIENTS",
],
)
8 changes: 8 additions & 0 deletions backends/xnnpack/threadpool/test/TARGETS
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,8 @@
# Any targets that should be shared between fbcode and xplat must be defined in
# targets.bzl. This file can contain fbcode-only targets.

load(":targets.bzl", "define_common_targets")

oncall("executorch")

define_common_targets()
20 changes: 20 additions & 0 deletions backends/xnnpack/threadpool/test/targets.bzl
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,20 @@
load("@fbsource//xplat/executorch/build:runtime_wrapper.bzl", "runtime")

def define_common_targets():
"""Defines targets that should be shared between fbcode and xplat.

The directory containing this targets.bzl file should also contain both
TARGETS and BUCK files that call this function.
"""

_THREADPOOL_TESTS = [
"threadpool_test.cpp",
] + (["fb/threadpool_use_n_threads_test.cpp"] if not runtime.is_oss else [])

runtime.cxx_test(
name = "threadpool_test",
srcs = _THREADPOOL_TESTS,
deps = [
"//executorch/backends/xnnpack/threadpool:threadpool",
],
)
188 changes: 188 additions & 0 deletions backends/xnnpack/threadpool/test/threadpool_test.cpp
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,188 @@
/*
* Copyright (c) Meta Platforms, Inc. and affiliates.
* All rights reserved.
*
* This source code is licensed under the BSD-style license found in the
* LICENSE file in the root directory of this source tree.
*/

#include <gtest/gtest.h>
#include <mutex>
#include <numeric>
#include <random>

#include <executorch/backends/xnnpack/threadpool/threadpool.h>
#include <executorch/backends/xnnpack/threadpool/threadpool_guard.h>

using namespace ::testing;

namespace {

size_t div_round_up(const size_t divident, const size_t divisor) {
return (divident + divisor - 1) / divisor;
}

void resize_and_fill_vector(std::vector<int32_t>& a, const size_t size) {
std::random_device rd;
std::mt19937 gen(rd());
std::uniform_int_distribution<> distrib(1, size * 2);
a.resize(size);
auto generator = [&distrib, &gen]() { return distrib(gen); };
std::generate(a.begin(), a.end(), generator);
}

void generate_add_test_inputs(
std::vector<int32_t>& a,
std::vector<int32_t>& b,
std::vector<int32_t>& c_ref,
std::vector<int32_t>& c,
size_t vector_size) {
resize_and_fill_vector(a, vector_size);
resize_and_fill_vector(b, vector_size);
resize_and_fill_vector(c, vector_size);
resize_and_fill_vector(c_ref, vector_size);
for (size_t i = 0, size = a.size(); i < size; ++i) {
c_ref[i] = a[i] + b[i];
}
}

void generate_reduce_test_inputs(
std::vector<int32_t>& a,
int32_t& c_ref,
size_t vector_size) {
resize_and_fill_vector(a, vector_size);
c_ref = 0;
for (size_t i = 0, size = a.size(); i < size; ++i) {
c_ref += a[i];
}
}

void run_lambda_with_size(
std::function<void(size_t)> f,
size_t range,
size_t grain_size) {
size_t num_grains = div_round_up(range, grain_size);

auto threadpool = torch::executorch::threadpool::get_threadpool();
threadpool->run(f, range);
}
} // namespace

TEST(ThreadPoolTest, ParallelAdd) {
std::vector<int32_t> a, b, c, c_ref;
size_t vector_size = 100;
size_t grain_size = 10;

auto add_lambda = [&](size_t i) {
size_t start_index = i * grain_size;
size_t end_index = start_index + grain_size;
end_index = std::min(end_index, vector_size);
for (size_t j = start_index; j < end_index; ++j) {
c[j] = a[j] + b[j];
}
};

auto threadpool = torch::executorch::threadpool::get_threadpool();
EXPECT_GT(threadpool->get_thread_count(), 1);

generate_add_test_inputs(a, b, c_ref, c, vector_size);
run_lambda_with_size(add_lambda, vector_size, grain_size);
EXPECT_EQ(c, c_ref);

// Try smaller grain size
grain_size = 5;
generate_add_test_inputs(a, b, c_ref, c, vector_size);
run_lambda_with_size(add_lambda, vector_size, grain_size);
EXPECT_EQ(c, c_ref);

vector_size = 7;
generate_add_test_inputs(a, b, c_ref, c, vector_size);
run_lambda_with_size(add_lambda, vector_size, grain_size);
EXPECT_EQ(c, c_ref);

vector_size = 7;
grain_size = 5;
generate_add_test_inputs(a, b, c_ref, c, vector_size);
run_lambda_with_size(add_lambda, vector_size, grain_size);
EXPECT_EQ(c, c_ref);
}

// Test parallel reduction where we acquire lock within lambda
TEST(ThreadPoolTest, ParallelReduce) {
std::vector<int32_t> a;
int32_t c = 0, c_ref = 0;
size_t vector_size = 100;
size_t grain_size = 11;
std::mutex m;

auto reduce_lambda = [&](size_t i) {
size_t start_index = i * grain_size;
size_t end_index = start_index + grain_size;
end_index = std::min(end_index, vector_size);
std::lock_guard<std::mutex> lock(m);
for (size_t j = start_index; j < end_index; ++j) {
c += a[j];
}
};

auto threadpool = torch::executorch::threadpool::get_threadpool();
EXPECT_GT(threadpool->get_thread_count(), 1);

generate_reduce_test_inputs(a, c_ref, vector_size);
run_lambda_with_size(reduce_lambda, vector_size, grain_size);
EXPECT_EQ(c, c_ref);

vector_size = 7;
c = c_ref = 0;
generate_reduce_test_inputs(a, c_ref, vector_size);
run_lambda_with_size(reduce_lambda, vector_size, grain_size);
EXPECT_EQ(c, c_ref);
}

// Copied from
// caffe2/aten/src/ATen/test/test_thread_pool_guard.cp
TEST(TestNoThreadPoolGuard, TestThreadPoolGuard) {
auto threadpool_ptr = torch::executorch::threadpool::get_pthreadpool();

ASSERT_NE(threadpool_ptr, nullptr);
{
torch::executorch::threadpool::NoThreadPoolGuard g1;
auto threadpool_ptr1 = torch::executorch::threadpool::get_pthreadpool();
ASSERT_EQ(threadpool_ptr1, nullptr);

{
torch::executorch::threadpool::NoThreadPoolGuard g2;
auto threadpool_ptr2 = torch::executorch::threadpool::get_pthreadpool();
ASSERT_EQ(threadpool_ptr2, nullptr);
}

// Guard should restore prev value (nullptr)
auto threadpool_ptr3 = torch::executorch::threadpool::get_pthreadpool();
ASSERT_EQ(threadpool_ptr3, nullptr);
}

// Guard should restore prev value (pthreadpool_)
auto threadpool_ptr4 = torch::executorch::threadpool::get_pthreadpool();
ASSERT_NE(threadpool_ptr4, nullptr);
ASSERT_EQ(threadpool_ptr4, threadpool_ptr);
}

TEST(TestNoThreadPoolGuard, TestRunWithGuard) {
const std::vector<int64_t> array = {1, 2, 3};

auto pool = torch::executorch::threadpool::get_threadpool();
int64_t inner = 0;
{
// Run on same thread
torch::executorch::threadpool::NoThreadPoolGuard g1;
auto fn = [&array, &inner](const size_t task_id) {
inner += array[task_id];
};
pool->run(fn, 3);

// confirm the guard is on
auto threadpool_ptr = torch::executorch::threadpool::get_pthreadpool();
ASSERT_EQ(threadpool_ptr, nullptr);
}
ASSERT_EQ(inner, 6);
}
Loading