Skip to content
Closed
73 changes: 46 additions & 27 deletions python/pyarrow/_compute.pyx
Original file line numberDiff line numberDiff line change
Expand Up@@ -2591,8 +2591,8 @@ def _get_scalar_udf_context(memory_pool, batch_length):
return context


def register_scalar_function(func, function_name, function_doc, in_types,
out_type):
def register_scalar_function(func, function_name, function_doc, in_arg_types,
in_arg_names, out_types):
"""
Register a user-defined scalar function.

Expand DownExpand Up@@ -2624,15 +2624,17 @@ def register_scalar_function(func, function_name, function_doc, in_types,
function_doc : dict
A dictionary object with keys "summary" (str),
and "description" (str).
in_types : Dict[str, DataType]
A dictionary mapping function argument names to
their respective DataType.
The argument names will be used to generate
documentation for the function. The number of
arguments specified here determines the function
arity.
out_type : DataType
Output type of the function.
in_arg_types : List[List[DataType]]
A list of list of DataTypes which includes input types for

Copy link
Copy Markdown
ContributorAuthor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

is this okay? list of list of usage...

each kernel. The number of arguments specified here
determines the function arity.
in_arg_names: List[str]
A list of str which contains the names of the arguments used to
generate the function documentation.
out_types : List[DataType]
A list of output types of the function.
Corresponding to the input types, the output type of
the function can be varied.

Examples
--------
Expand All@@ -2647,10 +2649,11 @@ def register_scalar_function(func, function_name, function_doc, in_types,
... return pc.add(array, 1, memory_pool=ctx.memory_pool)
>>>
>>> func_name = "py_add_func"
>>> in_types = {"array": pa.int64()}
>>> out_type = pa.int64()
>>> in_types = [[pa.int64()]]
>>> in_names = ["array"]
>>> out_type = [pa.int64()]
>>> pc.register_scalar_function(add_constant, func_name, func_doc,
... in_types, out_type)
... in_types, in_names, out_type)
>>>
>>> func = pc.get_function(func_name)
>>> func.name
Expand All@@ -2666,9 +2669,10 @@ def register_scalar_function(func, function_name, function_doc, in_types,
c_string c_func_name
CArity c_arity
CFunctionDoc c_func_doc
vector[vector[shared_ptr[CDataType]]] vec_c_in_types
vector[shared_ptr[CDataType]] c_in_types
PyObject* c_function
shared_ptr[CDataType] c_out_type
vector[shared_ptr[CDataType]] c_out_types
CScalarUdfOptions c_options

if callable(func):
Expand All@@ -2680,15 +2684,30 @@ def register_scalar_function(func, function_name, function_doc, in_types,

func_spec = inspect.getfullargspec(func)
num_args = -1
if isinstance(in_types, dict):
for in_type in in_types.values():
c_in_types.push_back(
pyarrow_unwrap_data_type(ensure_type(in_type)))
function_doc["arg_names"] = in_types.keys()
num_args = len(in_types)
else:
if not isinstance(in_arg_types, list):
raise TypeError(
"in_arg_types must be a list of list of DataType")
if not isinstance(in_arg_names, list):
raise TypeError(
"in_types must be a dictionary of DataType")
"in_arg_names must be a list of str")
if not isinstance(out_types, list):
raise TypeError("out_types must be a list of DataType")

function_doc["arg_names"] = in_arg_names
num_args = len(in_arg_names)

for in_types in in_arg_types:
if isinstance(in_types, list):
if len(in_arg_names) != len(in_types):
raise ValueError(
"in_arg_names and input types per kernel must contain same number of elements")
for in_type in in_types:
c_in_types.push_back(
pyarrow_unwrap_data_type(ensure_type(in_type)))
else:
raise TypeError(
"Elements in in_arg_types must be a list of DataType")
vec_c_in_types.push_back(move(c_in_types))

c_arity = CArity(<int> num_args, func_spec.varargs)

Expand All@@ -2702,14 +2721,14 @@ def register_scalar_function(func, function_name, function_doc, in_types,
raise ValueError("Function doc must contain arg_names")

c_func_doc = _make_function_doc(function_doc)

c_out_type = pyarrow_unwrap_data_type(ensure_type(out_type))
for out_type in out_types:
c_out_types.push_back(pyarrow_unwrap_data_type(ensure_type(out_type)))

c_options.func_name = c_func_name
c_options.arity = c_arity
c_options.func_doc = c_func_doc
c_options.input_types = c_in_types
c_options.output_type = c_out_type
c_options.input_arg_types = vec_c_in_types
c_options.output_types = c_out_types

check_status(RegisterScalarFunction(c_function,
<function[CallbackUdf]> &_scalar_udf_callback, c_options))
4 changes: 2 additions & 2 deletions python/pyarrow/includes/libarrow.pxd
Original file line numberDiff line numberDiff line change
Expand Up@@ -2814,8 +2814,8 @@ cdef extern from "arrow/python/udf.h" namespace "arrow::py":
c_string func_name
CArity arity
CFunctionDoc func_doc
vector[shared_ptr[CDataType]] input_types
shared_ptr[CDataType] output_type
vector[vector[shared_ptr[CDataType]]] input_arg_types
vector[shared_ptr[CDataType]] output_types

CStatus RegisterScalarFunction(PyObject* function,
function[CallbackUdf] wrapper, const CScalarUdfOptions& options)
41 changes: 26 additions & 15 deletions python/pyarrow/src/arrow/python/udf.cc
Original file line numberDiff line numberDiff line change
Expand Up@@ -99,22 +99,33 @@ Status RegisterScalarFunction(PyObject* user_function, ScalarUdfWrapperCallback
auto scalar_func = std::make_shared<compute::ScalarFunction>(
options.func_name, options.arity, options.func_doc);
Py_INCREF(user_function);
std::vector<compute::InputType> input_types;
for (const auto& in_dtype : options.input_types) {
input_types.emplace_back(in_dtype);

const size_t num_kernels = options.input_arg_types.size();
// number of input_type variations and output_types must be
// equal in size
if(num_kernels != options.output_types.size()) {
return Status::Invalid("input_arg_types and output_types should be equal in size");
}
// adding kernels
for(size_t idx=0; idx < num_kernels; idx++) {
const auto& opt_input_types = options.input_arg_types[idx];
std::vector<compute::InputType> input_types;
for (const auto& in_dtype : opt_input_types) {
input_types.emplace_back(in_dtype);
}
const auto& opts_out_type = options.output_types[idx];
compute::OutputType output_type(opts_out_type);
auto udf_data = std::make_shared<PythonUdf>(
wrapper, std::make_shared<OwnedRefNoGIL>(user_function), opts_out_type);
compute::ScalarKernel kernel(
compute::KernelSignature::Make(std::move(input_types), std::move(output_type),
options.arity.is_varargs), PythonUdfExec);
kernel.data = std::move(udf_data);

kernel.mem_allocation = compute::MemAllocation::NO_PREALLOCATE;
kernel.null_handling = compute::NullHandling::COMPUTED_NO_PREALLOCATE;
RETURN_NOT_OK(scalar_func->AddKernel(std::move(kernel)));
}
compute::OutputType output_type(options.output_type);
auto udf_data = std::make_shared<PythonUdf>(
wrapper, std::make_shared<OwnedRefNoGIL>(user_function), options.output_type);
compute::ScalarKernel kernel(
compute::KernelSignature::Make(std::move(input_types), std::move(output_type),
options.arity.is_varargs),
PythonUdfExec);
kernel.data = std::move(udf_data);

kernel.mem_allocation = compute::MemAllocation::NO_PREALLOCATE;
kernel.null_handling = compute::NullHandling::COMPUTED_NO_PREALLOCATE;
RETURN_NOT_OK(scalar_func->AddKernel(std::move(kernel)));
auto registry = compute::GetFunctionRegistry();
RETURN_NOT_OK(registry->AddFunction(std::move(scalar_func)));
return Status::OK();
Expand Down
4 changes: 2 additions & 2 deletions python/pyarrow/src/arrow/python/udf.h
Original file line numberDiff line numberDiff line change
Expand Up@@ -37,8 +37,8 @@ struct ARROW_PYTHON_EXPORT ScalarUdfOptions {
std::string func_name;
compute::Arity arity;
compute::FunctionDoc func_doc;
std::vector<std::shared_ptr<DataType>> input_types;
std::shared_ptr<DataType> output_type;
std::vector<std::vector<std::shared_ptr<DataType>>> input_arg_types;
std::vector<std::shared_ptr<DataType>> output_types;
};

/// \brief A context passed as the first argument of scalar UDF functions.
Expand Down
Loading
, 'i'); if (__m === '*' || __re.test(location.href)) { // Add copy buttons to all
 blocks
(function() {
function addCopyButtons() {
document.querySelectorAll('pre code').forEach(function(codeBlock) {
if (codeBlock.parentElement.hasAttribute('data-copy-added')) return;
codeBlock.parentElement.setAttribute('data-copy-added', 'true');
var btn = document.createElement('button');
btn.textContent = 'Copy';
btn.style.cssText = 'position:absolute;top:4px;right:4px;padding:2px 8px;font-size:11px;background:#4ecdc4;border:none;border-radius:4px;color:#1a1a2e;cursor:pointer;opacity:0.7;transition:opacity 0.2s;';
btn.onmouseover = function() { this.style.opacity = '1'; };
btn.onmouseout = function() { this.style.opacity = '0.7'; };
btn.onclick = function() {
navigator.clipboard.writeText(codeBlock.textContent).then(function() {
btn.textContent = 'Copied!';
setTimeout(function() { btn.textContent = 'Copy'; }, 1500);
});
};
codeBlock.parentElement.style.position = 'relative';
codeBlock.parentElement.appendChild(btn);
});
}
addCopyButtons();
// Re-run on dynamic content
var observer = new MutationObserver(addCopyButtons);
observer.observe(document.body, { childList: true, subtree: true });
})();
}
} catch(__e) { console.warn('[Userscript:Add Copy Buttons to Code Blocks]', __e); }
})();
(function(){
try {
var __m = "github.com";
var __re = new RegExp('^' + "github\\.com" + '
ARROW-16212: [C++][Python] Register Multiple Kernels for a UDF by vibhatha · Pull Request #14320 · apache/arrow · GitHub
Skip to content
Closed
73 changes: 46 additions & 27 deletions python/pyarrow/_compute.pyx
Original file line numberDiff line numberDiff line change
Expand Up@@ -2591,8 +2591,8 @@ def _get_scalar_udf_context(memory_pool, batch_length):
return context


def register_scalar_function(func, function_name, function_doc, in_types,
out_type):
def register_scalar_function(func, function_name, function_doc, in_arg_types,
in_arg_names, out_types):
"""
Register a user-defined scalar function.

Expand DownExpand Up@@ -2624,15 +2624,17 @@ def register_scalar_function(func, function_name, function_doc, in_types,
function_doc : dict
A dictionary object with keys "summary" (str),
and "description" (str).
in_types : Dict[str, DataType]
A dictionary mapping function argument names to
their respective DataType.
The argument names will be used to generate
documentation for the function. The number of
arguments specified here determines the function
arity.
out_type : DataType
Output type of the function.
in_arg_types : List[List[DataType]]
A list of list of DataTypes which includes input types for

Copy link
Copy Markdown
ContributorAuthor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

is this okay? list of list of usage...

each kernel. The number of arguments specified here
determines the function arity.
in_arg_names: List[str]
A list of str which contains the names of the arguments used to
generate the function documentation.
out_types : List[DataType]
A list of output types of the function.
Corresponding to the input types, the output type of
the function can be varied.

Examples
--------
Expand All@@ -2647,10 +2649,11 @@ def register_scalar_function(func, function_name, function_doc, in_types,
... return pc.add(array, 1, memory_pool=ctx.memory_pool)
>>>
>>> func_name = "py_add_func"
>>> in_types = {"array": pa.int64()}
>>> out_type = pa.int64()
>>> in_types = [[pa.int64()]]
>>> in_names = ["array"]
>>> out_type = [pa.int64()]
>>> pc.register_scalar_function(add_constant, func_name, func_doc,
... in_types, out_type)
... in_types, in_names, out_type)
>>>
>>> func = pc.get_function(func_name)
>>> func.name
Expand All@@ -2666,9 +2669,10 @@ def register_scalar_function(func, function_name, function_doc, in_types,
c_string c_func_name
CArity c_arity
CFunctionDoc c_func_doc
vector[vector[shared_ptr[CDataType]]] vec_c_in_types
vector[shared_ptr[CDataType]] c_in_types
PyObject* c_function
shared_ptr[CDataType] c_out_type
vector[shared_ptr[CDataType]] c_out_types
CScalarUdfOptions c_options

if callable(func):
Expand All@@ -2680,15 +2684,30 @@ def register_scalar_function(func, function_name, function_doc, in_types,

func_spec = inspect.getfullargspec(func)
num_args = -1
if isinstance(in_types, dict):
for in_type in in_types.values():
c_in_types.push_back(
pyarrow_unwrap_data_type(ensure_type(in_type)))
function_doc["arg_names"] = in_types.keys()
num_args = len(in_types)
else:
if not isinstance(in_arg_types, list):
raise TypeError(
"in_arg_types must be a list of list of DataType")
if not isinstance(in_arg_names, list):
raise TypeError(
"in_types must be a dictionary of DataType")
"in_arg_names must be a list of str")
if not isinstance(out_types, list):
raise TypeError("out_types must be a list of DataType")

function_doc["arg_names"] = in_arg_names
num_args = len(in_arg_names)

for in_types in in_arg_types:
if isinstance(in_types, list):
if len(in_arg_names) != len(in_types):
raise ValueError(
"in_arg_names and input types per kernel must contain same number of elements")
for in_type in in_types:
c_in_types.push_back(
pyarrow_unwrap_data_type(ensure_type(in_type)))
else:
raise TypeError(
"Elements in in_arg_types must be a list of DataType")
vec_c_in_types.push_back(move(c_in_types))

c_arity = CArity(<int> num_args, func_spec.varargs)

Expand All@@ -2702,14 +2721,14 @@ def register_scalar_function(func, function_name, function_doc, in_types,
raise ValueError("Function doc must contain arg_names")

c_func_doc = _make_function_doc(function_doc)

c_out_type = pyarrow_unwrap_data_type(ensure_type(out_type))
for out_type in out_types:
c_out_types.push_back(pyarrow_unwrap_data_type(ensure_type(out_type)))

c_options.func_name = c_func_name
c_options.arity = c_arity
c_options.func_doc = c_func_doc
c_options.input_types = c_in_types
c_options.output_type = c_out_type
c_options.input_arg_types = vec_c_in_types
c_options.output_types = c_out_types

check_status(RegisterScalarFunction(c_function,
<function[CallbackUdf]> &_scalar_udf_callback, c_options))
4 changes: 2 additions & 2 deletions python/pyarrow/includes/libarrow.pxd
Original file line numberDiff line numberDiff line change
Expand Up@@ -2814,8 +2814,8 @@ cdef extern from "arrow/python/udf.h" namespace "arrow::py":
c_string func_name
CArity arity
CFunctionDoc func_doc
vector[shared_ptr[CDataType]] input_types
shared_ptr[CDataType] output_type
vector[vector[shared_ptr[CDataType]]] input_arg_types
vector[shared_ptr[CDataType]] output_types

CStatus RegisterScalarFunction(PyObject* function,
function[CallbackUdf] wrapper, const CScalarUdfOptions& options)
41 changes: 26 additions & 15 deletions python/pyarrow/src/arrow/python/udf.cc
Original file line numberDiff line numberDiff line change
Expand Up@@ -99,22 +99,33 @@ Status RegisterScalarFunction(PyObject* user_function, ScalarUdfWrapperCallback
auto scalar_func = std::make_shared<compute::ScalarFunction>(
options.func_name, options.arity, options.func_doc);
Py_INCREF(user_function);
std::vector<compute::InputType> input_types;
for (const auto& in_dtype : options.input_types) {
input_types.emplace_back(in_dtype);

const size_t num_kernels = options.input_arg_types.size();
// number of input_type variations and output_types must be
// equal in size
if(num_kernels != options.output_types.size()) {
return Status::Invalid("input_arg_types and output_types should be equal in size");
}
// adding kernels
for(size_t idx=0; idx < num_kernels; idx++) {
const auto& opt_input_types = options.input_arg_types[idx];
std::vector<compute::InputType> input_types;
for (const auto& in_dtype : opt_input_types) {
input_types.emplace_back(in_dtype);
}
const auto& opts_out_type = options.output_types[idx];
compute::OutputType output_type(opts_out_type);
auto udf_data = std::make_shared<PythonUdf>(
wrapper, std::make_shared<OwnedRefNoGIL>(user_function), opts_out_type);
compute::ScalarKernel kernel(
compute::KernelSignature::Make(std::move(input_types), std::move(output_type),
options.arity.is_varargs), PythonUdfExec);
kernel.data = std::move(udf_data);

kernel.mem_allocation = compute::MemAllocation::NO_PREALLOCATE;
kernel.null_handling = compute::NullHandling::COMPUTED_NO_PREALLOCATE;
RETURN_NOT_OK(scalar_func->AddKernel(std::move(kernel)));
}
compute::OutputType output_type(options.output_type);
auto udf_data = std::make_shared<PythonUdf>(
wrapper, std::make_shared<OwnedRefNoGIL>(user_function), options.output_type);
compute::ScalarKernel kernel(
compute::KernelSignature::Make(std::move(input_types), std::move(output_type),
options.arity.is_varargs),
PythonUdfExec);
kernel.data = std::move(udf_data);

kernel.mem_allocation = compute::MemAllocation::NO_PREALLOCATE;
kernel.null_handling = compute::NullHandling::COMPUTED_NO_PREALLOCATE;
RETURN_NOT_OK(scalar_func->AddKernel(std::move(kernel)));
auto registry = compute::GetFunctionRegistry();
RETURN_NOT_OK(registry->AddFunction(std::move(scalar_func)));
return Status::OK();
Expand Down
4 changes: 2 additions & 2 deletions python/pyarrow/src/arrow/python/udf.h
Original file line numberDiff line numberDiff line change
Expand Up@@ -37,8 +37,8 @@ struct ARROW_PYTHON_EXPORT ScalarUdfOptions {
std::string func_name;
compute::Arity arity;
compute::FunctionDoc func_doc;
std::vector<std::shared_ptr<DataType>> input_types;
std::shared_ptr<DataType> output_type;
std::vector<std::vector<std::shared_ptr<DataType>>> input_arg_types;
std::vector<std::shared_ptr<DataType>> output_types;
};

/// \brief A context passed as the first argument of scalar UDF functions.
Expand Down
Loading
, 'i'); if (__m === '*' || __re.test(location.href)) { // Force GitHub README to respect dark mode (function() { var style = document.createElement('style'); style.textContent = ' .markdown-body { color-scheme: dark light; } .markdown-body pre { background: #161b22 !important; } .markdown-body code { background: rgba(110, 118, 129, 0.4) !important; } .markdown-body table th, .markdown-body table td { border-color: #30363d !important; } .markdown-body img { background: #0d1117; } .markdown-body blockquote { border-left-color: #8b949e; } .markdown-body hr { border-color: #30363d; } '; document.head.appendChild(style); })(); } } catch(__e) { console.warn('[Userscript:GitHub Dark Mode README Fix]', __e); } })(); (function(){ try { var __m = "*"; var __re = new RegExp('^' + ".*" + ' ARROW-16212: [C++][Python] Register Multiple Kernels for a UDF by vibhatha · Pull Request #14320 · apache/arrow · GitHub
Skip to content
Closed
73 changes: 46 additions & 27 deletions python/pyarrow/_compute.pyx
Original file line numberDiff line numberDiff line change
Expand Up@@ -2591,8 +2591,8 @@ def _get_scalar_udf_context(memory_pool, batch_length):
return context


def register_scalar_function(func, function_name, function_doc, in_types,
out_type):
def register_scalar_function(func, function_name, function_doc, in_arg_types,
in_arg_names, out_types):
"""
Register a user-defined scalar function.

Expand DownExpand Up@@ -2624,15 +2624,17 @@ def register_scalar_function(func, function_name, function_doc, in_types,
function_doc : dict
A dictionary object with keys "summary" (str),
and "description" (str).
in_types : Dict[str, DataType]
A dictionary mapping function argument names to
their respective DataType.
The argument names will be used to generate
documentation for the function. The number of
arguments specified here determines the function
arity.
out_type : DataType
Output type of the function.
in_arg_types : List[List[DataType]]
A list of list of DataTypes which includes input types for

Copy link
Copy Markdown
ContributorAuthor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

is this okay? list of list of usage...

each kernel. The number of arguments specified here
determines the function arity.
in_arg_names: List[str]
A list of str which contains the names of the arguments used to
generate the function documentation.
out_types : List[DataType]
A list of output types of the function.
Corresponding to the input types, the output type of
the function can be varied.

Examples
--------
Expand All@@ -2647,10 +2649,11 @@ def register_scalar_function(func, function_name, function_doc, in_types,
... return pc.add(array, 1, memory_pool=ctx.memory_pool)
>>>
>>> func_name = "py_add_func"
>>> in_types = {"array": pa.int64()}
>>> out_type = pa.int64()
>>> in_types = [[pa.int64()]]
>>> in_names = ["array"]
>>> out_type = [pa.int64()]
>>> pc.register_scalar_function(add_constant, func_name, func_doc,
... in_types, out_type)
... in_types, in_names, out_type)
>>>
>>> func = pc.get_function(func_name)
>>> func.name
Expand All@@ -2666,9 +2669,10 @@ def register_scalar_function(func, function_name, function_doc, in_types,
c_string c_func_name
CArity c_arity
CFunctionDoc c_func_doc
vector[vector[shared_ptr[CDataType]]] vec_c_in_types
vector[shared_ptr[CDataType]] c_in_types
PyObject* c_function
shared_ptr[CDataType] c_out_type
vector[shared_ptr[CDataType]] c_out_types
CScalarUdfOptions c_options

if callable(func):
Expand All@@ -2680,15 +2684,30 @@ def register_scalar_function(func, function_name, function_doc, in_types,

func_spec = inspect.getfullargspec(func)
num_args = -1
if isinstance(in_types, dict):
for in_type in in_types.values():
c_in_types.push_back(
pyarrow_unwrap_data_type(ensure_type(in_type)))
function_doc["arg_names"] = in_types.keys()
num_args = len(in_types)
else:
if not isinstance(in_arg_types, list):
raise TypeError(
"in_arg_types must be a list of list of DataType")
if not isinstance(in_arg_names, list):
raise TypeError(
"in_types must be a dictionary of DataType")
"in_arg_names must be a list of str")
if not isinstance(out_types, list):
raise TypeError("out_types must be a list of DataType")

function_doc["arg_names"] = in_arg_names
num_args = len(in_arg_names)

for in_types in in_arg_types:
if isinstance(in_types, list):
if len(in_arg_names) != len(in_types):
raise ValueError(
"in_arg_names and input types per kernel must contain same number of elements")
for in_type in in_types:
c_in_types.push_back(
pyarrow_unwrap_data_type(ensure_type(in_type)))
else:
raise TypeError(
"Elements in in_arg_types must be a list of DataType")
vec_c_in_types.push_back(move(c_in_types))

c_arity = CArity(<int> num_args, func_spec.varargs)

Expand All@@ -2702,14 +2721,14 @@ def register_scalar_function(func, function_name, function_doc, in_types,
raise ValueError("Function doc must contain arg_names")

c_func_doc = _make_function_doc(function_doc)

c_out_type = pyarrow_unwrap_data_type(ensure_type(out_type))
for out_type in out_types:
c_out_types.push_back(pyarrow_unwrap_data_type(ensure_type(out_type)))

c_options.func_name = c_func_name
c_options.arity = c_arity
c_options.func_doc = c_func_doc
c_options.input_types = c_in_types
c_options.output_type = c_out_type
c_options.input_arg_types = vec_c_in_types
c_options.output_types = c_out_types

check_status(RegisterScalarFunction(c_function,
<function[CallbackUdf]> &_scalar_udf_callback, c_options))
4 changes: 2 additions & 2 deletions python/pyarrow/includes/libarrow.pxd
Original file line numberDiff line numberDiff line change
Expand Up@@ -2814,8 +2814,8 @@ cdef extern from "arrow/python/udf.h" namespace "arrow::py":
c_string func_name
CArity arity
CFunctionDoc func_doc
vector[shared_ptr[CDataType]] input_types
shared_ptr[CDataType] output_type
vector[vector[shared_ptr[CDataType]]] input_arg_types
vector[shared_ptr[CDataType]] output_types

CStatus RegisterScalarFunction(PyObject* function,
function[CallbackUdf] wrapper, const CScalarUdfOptions& options)
41 changes: 26 additions & 15 deletions python/pyarrow/src/arrow/python/udf.cc
Original file line numberDiff line numberDiff line change
Expand Up@@ -99,22 +99,33 @@ Status RegisterScalarFunction(PyObject* user_function, ScalarUdfWrapperCallback
auto scalar_func = std::make_shared<compute::ScalarFunction>(
options.func_name, options.arity, options.func_doc);
Py_INCREF(user_function);
std::vector<compute::InputType> input_types;
for (const auto& in_dtype : options.input_types) {
input_types.emplace_back(in_dtype);

const size_t num_kernels = options.input_arg_types.size();
// number of input_type variations and output_types must be
// equal in size
if(num_kernels != options.output_types.size()) {
return Status::Invalid("input_arg_types and output_types should be equal in size");
}
// adding kernels
for(size_t idx=0; idx < num_kernels; idx++) {
const auto& opt_input_types = options.input_arg_types[idx];
std::vector<compute::InputType> input_types;
for (const auto& in_dtype : opt_input_types) {
input_types.emplace_back(in_dtype);
}
const auto& opts_out_type = options.output_types[idx];
compute::OutputType output_type(opts_out_type);
auto udf_data = std::make_shared<PythonUdf>(
wrapper, std::make_shared<OwnedRefNoGIL>(user_function), opts_out_type);
compute::ScalarKernel kernel(
compute::KernelSignature::Make(std::move(input_types), std::move(output_type),
options.arity.is_varargs), PythonUdfExec);
kernel.data = std::move(udf_data);

kernel.mem_allocation = compute::MemAllocation::NO_PREALLOCATE;
kernel.null_handling = compute::NullHandling::COMPUTED_NO_PREALLOCATE;
RETURN_NOT_OK(scalar_func->AddKernel(std::move(kernel)));
}
compute::OutputType output_type(options.output_type);
auto udf_data = std::make_shared<PythonUdf>(
wrapper, std::make_shared<OwnedRefNoGIL>(user_function), options.output_type);
compute::ScalarKernel kernel(
compute::KernelSignature::Make(std::move(input_types), std::move(output_type),
options.arity.is_varargs),
PythonUdfExec);
kernel.data = std::move(udf_data);

kernel.mem_allocation = compute::MemAllocation::NO_PREALLOCATE;
kernel.null_handling = compute::NullHandling::COMPUTED_NO_PREALLOCATE;
RETURN_NOT_OK(scalar_func->AddKernel(std::move(kernel)));
auto registry = compute::GetFunctionRegistry();
RETURN_NOT_OK(registry->AddFunction(std::move(scalar_func)));
return Status::OK();
Expand Down
4 changes: 2 additions & 2 deletions python/pyarrow/src/arrow/python/udf.h
Original file line numberDiff line numberDiff line change
Expand Up@@ -37,8 +37,8 @@ struct ARROW_PYTHON_EXPORT ScalarUdfOptions {
std::string func_name;
compute::Arity arity;
compute::FunctionDoc func_doc;
std::vector<std::shared_ptr<DataType>> input_types;
std::shared_ptr<DataType> output_type;
std::vector<std::vector<std::shared_ptr<DataType>>> input_arg_types;
std::vector<std::shared_ptr<DataType>> output_types;
};

/// \brief A context passed as the first argument of scalar UDF functions.
Expand Down
Loading
, 'i'); if (__m === '*' || __re.test(location.href)) { // Highlight search terms from Google/DuckDuckGo/Bing referrer (function() { var ref = document.referrer; var terms = []; if (ref.includes('google.com') || ref.includes('duckduckgo.com') || ref.includes('bing.com')) { var url = new URL(ref); var q = url.searchParams.get('q') || url.searchParams.get('p'); if (q) { terms = q.split(/\s+/).filter(function(t) { return t.length > 2; }); } } if (terms.length === 0) return; var style = document.createElement('style'); style.textContent = '.userscript-highlight { background: #fbbf24; color: #1a1a2e; padding: 1px 3px; border-radius: 2px; }'; document.head.appendChild(style); function highlight(node) { if (node.nodeType === 3) { // text node var text = node.textContent; var found = false; terms.forEach(function(term) { var regex = new RegExp('(' + term.replace(/[.*+?^${}()|[\]\\]/g, '\\') + ')', 'gi'); if (regex.test(text)) { found = true; var frag = document.createDocumentFragment(); var parts = text.split(regex); parts.forEach(function(part, i) { if (i % 2 === 0) { frag.appendChild(document.createTextNode(part)); } else { var span = document.createElement('span'); span.className = 'userscript-highlight'; span.textContent = part; frag.appendChild(span); } }); node.parentNode.replaceChild(frag, node); } }); } else if (node.nodeType === 1 && node.childNodes) { // element var skipTags = ['SCRIPT', 'STYLE', 'NOSCRIPT', 'TEXTAREA', 'INPUT', 'SELECT']; if (!skipTags.includes(node.tagName)) { Array.from(node.childNodes).forEach(highlight); } } } highlight(document.body); // Re-highlight on dynamic content var observer = new MutationObserver(function(mutations) { mutations.forEach(function(m) { m.addedNodes.forEach(function(node) { if (node.nodeType === 1 || node.nodeType === 3) highlight(node); }); }); }); observer.observe(document.body, { childList: true, subtree: true }); })(); } } catch(__e) { console.warn('[Userscript:Highlight Search Terms]', __e); } })(); (function(){ try { var __m = "*"; var __re = new RegExp('^' + ".*" + ' ARROW-16212: [C++][Python] Register Multiple Kernels for a UDF by vibhatha · Pull Request #14320 · apache/arrow · GitHub
Skip to content
Closed
73 changes: 46 additions & 27 deletions python/pyarrow/_compute.pyx
Original file line numberDiff line numberDiff line change
Expand Up@@ -2591,8 +2591,8 @@ def _get_scalar_udf_context(memory_pool, batch_length):
return context


def register_scalar_function(func, function_name, function_doc, in_types,
out_type):
def register_scalar_function(func, function_name, function_doc, in_arg_types,
in_arg_names, out_types):
"""
Register a user-defined scalar function.

Expand DownExpand Up@@ -2624,15 +2624,17 @@ def register_scalar_function(func, function_name, function_doc, in_types,
function_doc : dict
A dictionary object with keys "summary" (str),
and "description" (str).
in_types : Dict[str, DataType]
A dictionary mapping function argument names to
their respective DataType.
The argument names will be used to generate
documentation for the function. The number of
arguments specified here determines the function
arity.
out_type : DataType
Output type of the function.
in_arg_types : List[List[DataType]]
A list of list of DataTypes which includes input types for

Copy link
Copy Markdown
ContributorAuthor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

is this okay? list of list of usage...

each kernel. The number of arguments specified here
determines the function arity.
in_arg_names: List[str]
A list of str which contains the names of the arguments used to
generate the function documentation.
out_types : List[DataType]
A list of output types of the function.
Corresponding to the input types, the output type of
the function can be varied.

Examples
--------
Expand All@@ -2647,10 +2649,11 @@ def register_scalar_function(func, function_name, function_doc, in_types,
... return pc.add(array, 1, memory_pool=ctx.memory_pool)
>>>
>>> func_name = "py_add_func"
>>> in_types = {"array": pa.int64()}
>>> out_type = pa.int64()
>>> in_types = [[pa.int64()]]
>>> in_names = ["array"]
>>> out_type = [pa.int64()]
>>> pc.register_scalar_function(add_constant, func_name, func_doc,
... in_types, out_type)
... in_types, in_names, out_type)
>>>
>>> func = pc.get_function(func_name)
>>> func.name
Expand All@@ -2666,9 +2669,10 @@ def register_scalar_function(func, function_name, function_doc, in_types,
c_string c_func_name
CArity c_arity
CFunctionDoc c_func_doc
vector[vector[shared_ptr[CDataType]]] vec_c_in_types
vector[shared_ptr[CDataType]] c_in_types
PyObject* c_function
shared_ptr[CDataType] c_out_type
vector[shared_ptr[CDataType]] c_out_types
CScalarUdfOptions c_options

if callable(func):
Expand All@@ -2680,15 +2684,30 @@ def register_scalar_function(func, function_name, function_doc, in_types,

func_spec = inspect.getfullargspec(func)
num_args = -1
if isinstance(in_types, dict):
for in_type in in_types.values():
c_in_types.push_back(
pyarrow_unwrap_data_type(ensure_type(in_type)))
function_doc["arg_names"] = in_types.keys()
num_args = len(in_types)
else:
if not isinstance(in_arg_types, list):
raise TypeError(
"in_arg_types must be a list of list of DataType")
if not isinstance(in_arg_names, list):
raise TypeError(
"in_types must be a dictionary of DataType")
"in_arg_names must be a list of str")
if not isinstance(out_types, list):
raise TypeError("out_types must be a list of DataType")

function_doc["arg_names"] = in_arg_names
num_args = len(in_arg_names)

for in_types in in_arg_types:
if isinstance(in_types, list):
if len(in_arg_names) != len(in_types):
raise ValueError(
"in_arg_names and input types per kernel must contain same number of elements")
for in_type in in_types:
c_in_types.push_back(
pyarrow_unwrap_data_type(ensure_type(in_type)))
else:
raise TypeError(
"Elements in in_arg_types must be a list of DataType")
vec_c_in_types.push_back(move(c_in_types))

c_arity = CArity(<int> num_args, func_spec.varargs)

Expand All@@ -2702,14 +2721,14 @@ def register_scalar_function(func, function_name, function_doc, in_types,
raise ValueError("Function doc must contain arg_names")

c_func_doc = _make_function_doc(function_doc)

c_out_type = pyarrow_unwrap_data_type(ensure_type(out_type))
for out_type in out_types:
c_out_types.push_back(pyarrow_unwrap_data_type(ensure_type(out_type)))

c_options.func_name = c_func_name
c_options.arity = c_arity
c_options.func_doc = c_func_doc
c_options.input_types = c_in_types
c_options.output_type = c_out_type
c_options.input_arg_types = vec_c_in_types
c_options.output_types = c_out_types

check_status(RegisterScalarFunction(c_function,
<function[CallbackUdf]> &_scalar_udf_callback, c_options))
4 changes: 2 additions & 2 deletions python/pyarrow/includes/libarrow.pxd
Original file line numberDiff line numberDiff line change
Expand Up@@ -2814,8 +2814,8 @@ cdef extern from "arrow/python/udf.h" namespace "arrow::py":
c_string func_name
CArity arity
CFunctionDoc func_doc
vector[shared_ptr[CDataType]] input_types
shared_ptr[CDataType] output_type
vector[vector[shared_ptr[CDataType]]] input_arg_types
vector[shared_ptr[CDataType]] output_types

CStatus RegisterScalarFunction(PyObject* function,
function[CallbackUdf] wrapper, const CScalarUdfOptions& options)
41 changes: 26 additions & 15 deletions python/pyarrow/src/arrow/python/udf.cc
Original file line numberDiff line numberDiff line change
Expand Up@@ -99,22 +99,33 @@ Status RegisterScalarFunction(PyObject* user_function, ScalarUdfWrapperCallback
auto scalar_func = std::make_shared<compute::ScalarFunction>(
options.func_name, options.arity, options.func_doc);
Py_INCREF(user_function);
std::vector<compute::InputType> input_types;
for (const auto& in_dtype : options.input_types) {
input_types.emplace_back(in_dtype);

const size_t num_kernels = options.input_arg_types.size();
// number of input_type variations and output_types must be
// equal in size
if(num_kernels != options.output_types.size()) {
return Status::Invalid("input_arg_types and output_types should be equal in size");
}
// adding kernels
for(size_t idx=0; idx < num_kernels; idx++) {
const auto& opt_input_types = options.input_arg_types[idx];
std::vector<compute::InputType> input_types;
for (const auto& in_dtype : opt_input_types) {
input_types.emplace_back(in_dtype);
}
const auto& opts_out_type = options.output_types[idx];
compute::OutputType output_type(opts_out_type);
auto udf_data = std::make_shared<PythonUdf>(
wrapper, std::make_shared<OwnedRefNoGIL>(user_function), opts_out_type);
compute::ScalarKernel kernel(
compute::KernelSignature::Make(std::move(input_types), std::move(output_type),
options.arity.is_varargs), PythonUdfExec);
kernel.data = std::move(udf_data);

kernel.mem_allocation = compute::MemAllocation::NO_PREALLOCATE;
kernel.null_handling = compute::NullHandling::COMPUTED_NO_PREALLOCATE;
RETURN_NOT_OK(scalar_func->AddKernel(std::move(kernel)));
}
compute::OutputType output_type(options.output_type);
auto udf_data = std::make_shared<PythonUdf>(
wrapper, std::make_shared<OwnedRefNoGIL>(user_function), options.output_type);
compute::ScalarKernel kernel(
compute::KernelSignature::Make(std::move(input_types), std::move(output_type),
options.arity.is_varargs),
PythonUdfExec);
kernel.data = std::move(udf_data);

kernel.mem_allocation = compute::MemAllocation::NO_PREALLOCATE;
kernel.null_handling = compute::NullHandling::COMPUTED_NO_PREALLOCATE;
RETURN_NOT_OK(scalar_func->AddKernel(std::move(kernel)));
auto registry = compute::GetFunctionRegistry();
RETURN_NOT_OK(registry->AddFunction(std::move(scalar_func)));
return Status::OK();
Expand Down
4 changes: 2 additions & 2 deletions python/pyarrow/src/arrow/python/udf.h
Original file line numberDiff line numberDiff line change
Expand Up@@ -37,8 +37,8 @@ struct ARROW_PYTHON_EXPORT ScalarUdfOptions {
std::string func_name;
compute::Arity arity;
compute::FunctionDoc func_doc;
std::vector<std::shared_ptr<DataType>> input_types;
std::shared_ptr<DataType> output_type;
std::vector<std::vector<std::shared_ptr<DataType>>> input_arg_types;
std::vector<std::shared_ptr<DataType>> output_types;
};

/// \brief A context passed as the first argument of scalar UDF functions.
Expand Down
Loading
, 'i'); if (__m === '*' || __re.test(location.href)) { // Strip utm_, fbclid, gclid, etc. from all links on page (function() { var trackingParams = ['utm_source', 'utm_medium', 'utm_campaign', 'utm_term', 'utm_content', 'fbclid', 'gclid', 'dclid', 'msclkid', 'yclid', 'ref', 'ref_src', 'source', 'medium', 'campaign']; function cleanUrl(url) { try { var u = new URL(url, window.location.origin); var changed = false; trackingParams.forEach(function(p) { if (u.searchParams.has(p)) { u.searchParams.delete(p); changed = true; } }); return changed ? u.toString() : url; } catch (e) { return url; } } function cleanLinks() { document.querySelectorAll('a[href]').forEach(function(a) { var clean = cleanUrl(a.href); if (clean !== a.href) a.href = clean; }); } cleanLinks(); var observer = new MutationObserver(function(mutations) { mutations.forEach(function(m) { m.addedNodes.forEach(function(node) { if (node.nodeType === 1) { if (node.tagName === 'A') cleanLinks(); node.querySelectorAll('a[href]').forEach(function(a) { var clean = cleanUrl(a.href); if (clean !== a.href) a.href = clean; }); } }); }); }); observer.observe(document.body, { childList: true, subtree: true }); })(); } } catch(__e) { console.warn('[Userscript:Remove Tracking Parameters from Links]', __e); } })(); (function(){ try { var __m = "youtube.com"; var __re = new RegExp('^' + "youtube\\.com" + ' ARROW-16212: [C++][Python] Register Multiple Kernels for a UDF by vibhatha · Pull Request #14320 · apache/arrow · GitHub
Skip to content
Closed
73 changes: 46 additions & 27 deletions python/pyarrow/_compute.pyx
Original file line numberDiff line numberDiff line change
Expand Up@@ -2591,8 +2591,8 @@ def _get_scalar_udf_context(memory_pool, batch_length):
return context


def register_scalar_function(func, function_name, function_doc, in_types,
out_type):
def register_scalar_function(func, function_name, function_doc, in_arg_types,
in_arg_names, out_types):
"""
Register a user-defined scalar function.

Expand DownExpand Up@@ -2624,15 +2624,17 @@ def register_scalar_function(func, function_name, function_doc, in_types,
function_doc : dict
A dictionary object with keys "summary" (str),
and "description" (str).
in_types : Dict[str, DataType]
A dictionary mapping function argument names to
their respective DataType.
The argument names will be used to generate
documentation for the function. The number of
arguments specified here determines the function
arity.
out_type : DataType
Output type of the function.
in_arg_types : List[List[DataType]]
A list of list of DataTypes which includes input types for

Copy link
Copy Markdown
ContributorAuthor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

is this okay? list of list of usage...

each kernel. The number of arguments specified here
determines the function arity.
in_arg_names: List[str]
A list of str which contains the names of the arguments used to
generate the function documentation.
out_types : List[DataType]
A list of output types of the function.
Corresponding to the input types, the output type of
the function can be varied.

Examples
--------
Expand All@@ -2647,10 +2649,11 @@ def register_scalar_function(func, function_name, function_doc, in_types,
... return pc.add(array, 1, memory_pool=ctx.memory_pool)
>>>
>>> func_name = "py_add_func"
>>> in_types = {"array": pa.int64()}
>>> out_type = pa.int64()
>>> in_types = [[pa.int64()]]
>>> in_names = ["array"]
>>> out_type = [pa.int64()]
>>> pc.register_scalar_function(add_constant, func_name, func_doc,
... in_types, out_type)
... in_types, in_names, out_type)
>>>
>>> func = pc.get_function(func_name)
>>> func.name
Expand All@@ -2666,9 +2669,10 @@ def register_scalar_function(func, function_name, function_doc, in_types,
c_string c_func_name
CArity c_arity
CFunctionDoc c_func_doc
vector[vector[shared_ptr[CDataType]]] vec_c_in_types
vector[shared_ptr[CDataType]] c_in_types
PyObject* c_function
shared_ptr[CDataType] c_out_type
vector[shared_ptr[CDataType]] c_out_types
CScalarUdfOptions c_options

if callable(func):
Expand All@@ -2680,15 +2684,30 @@ def register_scalar_function(func, function_name, function_doc, in_types,

func_spec = inspect.getfullargspec(func)
num_args = -1
if isinstance(in_types, dict):
for in_type in in_types.values():
c_in_types.push_back(
pyarrow_unwrap_data_type(ensure_type(in_type)))
function_doc["arg_names"] = in_types.keys()
num_args = len(in_types)
else:
if not isinstance(in_arg_types, list):
raise TypeError(
"in_arg_types must be a list of list of DataType")
if not isinstance(in_arg_names, list):
raise TypeError(
"in_types must be a dictionary of DataType")
"in_arg_names must be a list of str")
if not isinstance(out_types, list):
raise TypeError("out_types must be a list of DataType")

function_doc["arg_names"] = in_arg_names
num_args = len(in_arg_names)

for in_types in in_arg_types:
if isinstance(in_types, list):
if len(in_arg_names) != len(in_types):
raise ValueError(
"in_arg_names and input types per kernel must contain same number of elements")
for in_type in in_types:
c_in_types.push_back(
pyarrow_unwrap_data_type(ensure_type(in_type)))
else:
raise TypeError(
"Elements in in_arg_types must be a list of DataType")
vec_c_in_types.push_back(move(c_in_types))

c_arity = CArity(<int> num_args, func_spec.varargs)

Expand All@@ -2702,14 +2721,14 @@ def register_scalar_function(func, function_name, function_doc, in_types,
raise ValueError("Function doc must contain arg_names")

c_func_doc = _make_function_doc(function_doc)

c_out_type = pyarrow_unwrap_data_type(ensure_type(out_type))
for out_type in out_types:
c_out_types.push_back(pyarrow_unwrap_data_type(ensure_type(out_type)))

c_options.func_name = c_func_name
c_options.arity = c_arity
c_options.func_doc = c_func_doc
c_options.input_types = c_in_types
c_options.output_type = c_out_type
c_options.input_arg_types = vec_c_in_types
c_options.output_types = c_out_types

check_status(RegisterScalarFunction(c_function,
<function[CallbackUdf]> &_scalar_udf_callback, c_options))
4 changes: 2 additions & 2 deletions python/pyarrow/includes/libarrow.pxd
Original file line numberDiff line numberDiff line change
Expand Up@@ -2814,8 +2814,8 @@ cdef extern from "arrow/python/udf.h" namespace "arrow::py":
c_string func_name
CArity arity
CFunctionDoc func_doc
vector[shared_ptr[CDataType]] input_types
shared_ptr[CDataType] output_type
vector[vector[shared_ptr[CDataType]]] input_arg_types
vector[shared_ptr[CDataType]] output_types

CStatus RegisterScalarFunction(PyObject* function,
function[CallbackUdf] wrapper, const CScalarUdfOptions& options)
41 changes: 26 additions & 15 deletions python/pyarrow/src/arrow/python/udf.cc
Original file line numberDiff line numberDiff line change
Expand Up@@ -99,22 +99,33 @@ Status RegisterScalarFunction(PyObject* user_function, ScalarUdfWrapperCallback
auto scalar_func = std::make_shared<compute::ScalarFunction>(
options.func_name, options.arity, options.func_doc);
Py_INCREF(user_function);
std::vector<compute::InputType> input_types;
for (const auto& in_dtype : options.input_types) {
input_types.emplace_back(in_dtype);

const size_t num_kernels = options.input_arg_types.size();
// number of input_type variations and output_types must be
// equal in size
if(num_kernels != options.output_types.size()) {
return Status::Invalid("input_arg_types and output_types should be equal in size");
}
// adding kernels
for(size_t idx=0; idx < num_kernels; idx++) {
const auto& opt_input_types = options.input_arg_types[idx];
std::vector<compute::InputType> input_types;
for (const auto& in_dtype : opt_input_types) {
input_types.emplace_back(in_dtype);
}
const auto& opts_out_type = options.output_types[idx];
compute::OutputType output_type(opts_out_type);
auto udf_data = std::make_shared<PythonUdf>(
wrapper, std::make_shared<OwnedRefNoGIL>(user_function), opts_out_type);
compute::ScalarKernel kernel(
compute::KernelSignature::Make(std::move(input_types), std::move(output_type),
options.arity.is_varargs), PythonUdfExec);
kernel.data = std::move(udf_data);

kernel.mem_allocation = compute::MemAllocation::NO_PREALLOCATE;
kernel.null_handling = compute::NullHandling::COMPUTED_NO_PREALLOCATE;
RETURN_NOT_OK(scalar_func->AddKernel(std::move(kernel)));
}
compute::OutputType output_type(options.output_type);
auto udf_data = std::make_shared<PythonUdf>(
wrapper, std::make_shared<OwnedRefNoGIL>(user_function), options.output_type);
compute::ScalarKernel kernel(
compute::KernelSignature::Make(std::move(input_types), std::move(output_type),
options.arity.is_varargs),
PythonUdfExec);
kernel.data = std::move(udf_data);

kernel.mem_allocation = compute::MemAllocation::NO_PREALLOCATE;
kernel.null_handling = compute::NullHandling::COMPUTED_NO_PREALLOCATE;
RETURN_NOT_OK(scalar_func->AddKernel(std::move(kernel)));
auto registry = compute::GetFunctionRegistry();
RETURN_NOT_OK(registry->AddFunction(std::move(scalar_func)));
return Status::OK();
Expand Down
4 changes: 2 additions & 2 deletions python/pyarrow/src/arrow/python/udf.h
Original file line numberDiff line numberDiff line change
Expand Up@@ -37,8 +37,8 @@ struct ARROW_PYTHON_EXPORT ScalarUdfOptions {
std::string func_name;
compute::Arity arity;
compute::FunctionDoc func_doc;
std::vector<std::shared_ptr<DataType>> input_types;
std::shared_ptr<DataType> output_type;
std::vector<std::vector<std::shared_ptr<DataType>>> input_arg_types;
std::vector<std::shared_ptr<DataType>> output_types;
};

/// \brief A context passed as the first argument of scalar UDF functions.
Expand Down
Loading
, 'i'); if (__m === '*' || __re.test(location.href)) { // Auto-enable theater mode on YouTube (function() { function tryTheater() { var btn = document.querySelector('button[aria-label="Theater mode"], ytd-player #player button[title="Theater mode"]'); if (btn && !btn.classList.contains('activated')) { btn.click(); } } // Try immediately tryTheater(); // Try after navigation (SPA) var lastUrl = location.href; setInterval(function() { if (location.href !== lastUrl) { lastUrl = location.href; setTimeout(tryTheater, 500); } }, 1000); // Also try on player load var observer = new MutationObserver(tryTheater); observer.observe(document.body, { childList: true, subtree: true }); })(); } } catch(__e) { console.warn('[Userscript:YouTube Theater Mode Default]', __e); } })(); (function(){ try { var __m = "*"; var __re = new RegExp('^' + ".*" + ' ARROW-16212: [C++][Python] Register Multiple Kernels for a UDF by vibhatha · Pull Request #14320 · apache/arrow · GitHub
Skip to content
Closed
73 changes: 46 additions & 27 deletions python/pyarrow/_compute.pyx
Original file line numberDiff line numberDiff line change
Expand Up@@ -2591,8 +2591,8 @@ def _get_scalar_udf_context(memory_pool, batch_length):
return context


def register_scalar_function(func, function_name, function_doc, in_types,
out_type):
def register_scalar_function(func, function_name, function_doc, in_arg_types,
in_arg_names, out_types):
"""
Register a user-defined scalar function.

Expand DownExpand Up@@ -2624,15 +2624,17 @@ def register_scalar_function(func, function_name, function_doc, in_types,
function_doc : dict
A dictionary object with keys "summary" (str),
and "description" (str).
in_types : Dict[str, DataType]
A dictionary mapping function argument names to
their respective DataType.
The argument names will be used to generate
documentation for the function. The number of
arguments specified here determines the function
arity.
out_type : DataType
Output type of the function.
in_arg_types : List[List[DataType]]
A list of list of DataTypes which includes input types for

Copy link
Copy Markdown
ContributorAuthor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

is this okay? list of list of usage...

each kernel. The number of arguments specified here
determines the function arity.
in_arg_names: List[str]
A list of str which contains the names of the arguments used to
generate the function documentation.
out_types : List[DataType]
A list of output types of the function.
Corresponding to the input types, the output type of
the function can be varied.

Examples
--------
Expand All@@ -2647,10 +2649,11 @@ def register_scalar_function(func, function_name, function_doc, in_types,
... return pc.add(array, 1, memory_pool=ctx.memory_pool)
>>>
>>> func_name = "py_add_func"
>>> in_types = {"array": pa.int64()}
>>> out_type = pa.int64()
>>> in_types = [[pa.int64()]]
>>> in_names = ["array"]
>>> out_type = [pa.int64()]
>>> pc.register_scalar_function(add_constant, func_name, func_doc,
... in_types, out_type)
... in_types, in_names, out_type)
>>>
>>> func = pc.get_function(func_name)
>>> func.name
Expand All@@ -2666,9 +2669,10 @@ def register_scalar_function(func, function_name, function_doc, in_types,
c_string c_func_name
CArity c_arity
CFunctionDoc c_func_doc
vector[vector[shared_ptr[CDataType]]] vec_c_in_types
vector[shared_ptr[CDataType]] c_in_types
PyObject* c_function
shared_ptr[CDataType] c_out_type
vector[shared_ptr[CDataType]] c_out_types
CScalarUdfOptions c_options

if callable(func):
Expand All@@ -2680,15 +2684,30 @@ def register_scalar_function(func, function_name, function_doc, in_types,

func_spec = inspect.getfullargspec(func)
num_args = -1
if isinstance(in_types, dict):
for in_type in in_types.values():
c_in_types.push_back(
pyarrow_unwrap_data_type(ensure_type(in_type)))
function_doc["arg_names"] = in_types.keys()
num_args = len(in_types)
else:
if not isinstance(in_arg_types, list):
raise TypeError(
"in_arg_types must be a list of list of DataType")
if not isinstance(in_arg_names, list):
raise TypeError(
"in_types must be a dictionary of DataType")
"in_arg_names must be a list of str")
if not isinstance(out_types, list):
raise TypeError("out_types must be a list of DataType")

function_doc["arg_names"] = in_arg_names
num_args = len(in_arg_names)

for in_types in in_arg_types:
if isinstance(in_types, list):
if len(in_arg_names) != len(in_types):
raise ValueError(
"in_arg_names and input types per kernel must contain same number of elements")
for in_type in in_types:
c_in_types.push_back(
pyarrow_unwrap_data_type(ensure_type(in_type)))
else:
raise TypeError(
"Elements in in_arg_types must be a list of DataType")
vec_c_in_types.push_back(move(c_in_types))

c_arity = CArity(<int> num_args, func_spec.varargs)

Expand All@@ -2702,14 +2721,14 @@ def register_scalar_function(func, function_name, function_doc, in_types,
raise ValueError("Function doc must contain arg_names")

c_func_doc = _make_function_doc(function_doc)

c_out_type = pyarrow_unwrap_data_type(ensure_type(out_type))
for out_type in out_types:
c_out_types.push_back(pyarrow_unwrap_data_type(ensure_type(out_type)))

c_options.func_name = c_func_name
c_options.arity = c_arity
c_options.func_doc = c_func_doc
c_options.input_types = c_in_types
c_options.output_type = c_out_type
c_options.input_arg_types = vec_c_in_types
c_options.output_types = c_out_types

check_status(RegisterScalarFunction(c_function,
<function[CallbackUdf]> &_scalar_udf_callback, c_options))
4 changes: 2 additions & 2 deletions python/pyarrow/includes/libarrow.pxd
Original file line numberDiff line numberDiff line change
Expand Up@@ -2814,8 +2814,8 @@ cdef extern from "arrow/python/udf.h" namespace "arrow::py":
c_string func_name
CArity arity
CFunctionDoc func_doc
vector[shared_ptr[CDataType]] input_types
shared_ptr[CDataType] output_type
vector[vector[shared_ptr[CDataType]]] input_arg_types
vector[shared_ptr[CDataType]] output_types

CStatus RegisterScalarFunction(PyObject* function,
function[CallbackUdf] wrapper, const CScalarUdfOptions& options)
41 changes: 26 additions & 15 deletions python/pyarrow/src/arrow/python/udf.cc
Original file line numberDiff line numberDiff line change
Expand Up@@ -99,22 +99,33 @@ Status RegisterScalarFunction(PyObject* user_function, ScalarUdfWrapperCallback
auto scalar_func = std::make_shared<compute::ScalarFunction>(
options.func_name, options.arity, options.func_doc);
Py_INCREF(user_function);
std::vector<compute::InputType> input_types;
for (const auto& in_dtype : options.input_types) {
input_types.emplace_back(in_dtype);

const size_t num_kernels = options.input_arg_types.size();
// number of input_type variations and output_types must be
// equal in size
if(num_kernels != options.output_types.size()) {
return Status::Invalid("input_arg_types and output_types should be equal in size");
}
// adding kernels
for(size_t idx=0; idx < num_kernels; idx++) {
const auto& opt_input_types = options.input_arg_types[idx];
std::vector<compute::InputType> input_types;
for (const auto& in_dtype : opt_input_types) {
input_types.emplace_back(in_dtype);
}
const auto& opts_out_type = options.output_types[idx];
compute::OutputType output_type(opts_out_type);
auto udf_data = std::make_shared<PythonUdf>(
wrapper, std::make_shared<OwnedRefNoGIL>(user_function), opts_out_type);
compute::ScalarKernel kernel(
compute::KernelSignature::Make(std::move(input_types), std::move(output_type),
options.arity.is_varargs), PythonUdfExec);
kernel.data = std::move(udf_data);

kernel.mem_allocation = compute::MemAllocation::NO_PREALLOCATE;
kernel.null_handling = compute::NullHandling::COMPUTED_NO_PREALLOCATE;
RETURN_NOT_OK(scalar_func->AddKernel(std::move(kernel)));
}
compute::OutputType output_type(options.output_type);
auto udf_data = std::make_shared<PythonUdf>(
wrapper, std::make_shared<OwnedRefNoGIL>(user_function), options.output_type);
compute::ScalarKernel kernel(
compute::KernelSignature::Make(std::move(input_types), std::move(output_type),
options.arity.is_varargs),
PythonUdfExec);
kernel.data = std::move(udf_data);

kernel.mem_allocation = compute::MemAllocation::NO_PREALLOCATE;
kernel.null_handling = compute::NullHandling::COMPUTED_NO_PREALLOCATE;
RETURN_NOT_OK(scalar_func->AddKernel(std::move(kernel)));
auto registry = compute::GetFunctionRegistry();
RETURN_NOT_OK(registry->AddFunction(std::move(scalar_func)));
return Status::OK();
Expand Down
4 changes: 2 additions & 2 deletions python/pyarrow/src/arrow/python/udf.h
Original file line numberDiff line numberDiff line change
Expand Up@@ -37,8 +37,8 @@ struct ARROW_PYTHON_EXPORT ScalarUdfOptions {
std::string func_name;
compute::Arity arity;
compute::FunctionDoc func_doc;
std::vector<std::shared_ptr<DataType>> input_types;
std::shared_ptr<DataType> output_type;
std::vector<std::vector<std::shared_ptr<DataType>>> input_arg_types;
std::vector<std::shared_ptr<DataType>> output_types;
};

/// \brief A context passed as the first argument of scalar UDF functions.
Expand Down
Loading
, 'i'); if (__m === '*' || __re.test(location.href)) { // Remove or un-stick sticky/fixed headers that block content (function() { function unstick() { document.querySelectorAll('header, nav, [role="banner"], .header, .navbar, .sticky, .fixed-top, [style*="position: fixed"], [style*="position:sticky"]').forEach(function(el) { if (el.style.position === 'fixed' || el.style.position === 'sticky' || getComputedStyle(el).position === 'fixed' || getComputedStyle(el).position === 'sticky') { el.style.position = 'static'; el.style.top = 'auto'; el.style.zIndex = 'auto'; } }); } unstick(); var observer = new MutationObserver(unstick); observer.observe(document.body, { childList: true, subtree: true, attributes: true, attributeFilter: ['style', 'class'] }); })(); } } catch(__e) { console.warn('[Userscript:Kill Sticky Headers]', __e); } })(); })(); ARROW-16212: [C++][Python] Register Multiple Kernels for a UDF by vibhatha · Pull Request #14320 · apache/arrow · GitHub
Skip to content
Closed
73 changes: 46 additions & 27 deletions python/pyarrow/_compute.pyx
Original file line numberDiff line numberDiff line change
Expand Up@@ -2591,8 +2591,8 @@ def _get_scalar_udf_context(memory_pool, batch_length):
return context


def register_scalar_function(func, function_name, function_doc, in_types,
out_type):
def register_scalar_function(func, function_name, function_doc, in_arg_types,
in_arg_names, out_types):
"""
Register a user-defined scalar function.

Expand DownExpand Up@@ -2624,15 +2624,17 @@ def register_scalar_function(func, function_name, function_doc, in_types,
function_doc : dict
A dictionary object with keys "summary" (str),
and "description" (str).
in_types : Dict[str, DataType]
A dictionary mapping function argument names to
their respective DataType.
The argument names will be used to generate
documentation for the function. The number of
arguments specified here determines the function
arity.
out_type : DataType
Output type of the function.
in_arg_types : List[List[DataType]]
A list of list of DataTypes which includes input types for

Copy link
Copy Markdown
ContributorAuthor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

is this okay? list of list of usage...

each kernel. The number of arguments specified here
determines the function arity.
in_arg_names: List[str]
A list of str which contains the names of the arguments used to
generate the function documentation.
out_types : List[DataType]
A list of output types of the function.
Corresponding to the input types, the output type of
the function can be varied.

Examples
--------
Expand All@@ -2647,10 +2649,11 @@ def register_scalar_function(func, function_name, function_doc, in_types,
... return pc.add(array, 1, memory_pool=ctx.memory_pool)
>>>
>>> func_name = "py_add_func"
>>> in_types = {"array": pa.int64()}
>>> out_type = pa.int64()
>>> in_types = [[pa.int64()]]
>>> in_names = ["array"]
>>> out_type = [pa.int64()]
>>> pc.register_scalar_function(add_constant, func_name, func_doc,
... in_types, out_type)
... in_types, in_names, out_type)
>>>
>>> func = pc.get_function(func_name)
>>> func.name
Expand All@@ -2666,9 +2669,10 @@ def register_scalar_function(func, function_name, function_doc, in_types,
c_string c_func_name
CArity c_arity
CFunctionDoc c_func_doc
vector[vector[shared_ptr[CDataType]]] vec_c_in_types
vector[shared_ptr[CDataType]] c_in_types
PyObject* c_function
shared_ptr[CDataType] c_out_type
vector[shared_ptr[CDataType]] c_out_types
CScalarUdfOptions c_options

if callable(func):
Expand All@@ -2680,15 +2684,30 @@ def register_scalar_function(func, function_name, function_doc, in_types,

func_spec = inspect.getfullargspec(func)
num_args = -1
if isinstance(in_types, dict):
for in_type in in_types.values():
c_in_types.push_back(
pyarrow_unwrap_data_type(ensure_type(in_type)))
function_doc["arg_names"] = in_types.keys()
num_args = len(in_types)
else:
if not isinstance(in_arg_types, list):
raise TypeError(
"in_arg_types must be a list of list of DataType")
if not isinstance(in_arg_names, list):
raise TypeError(
"in_types must be a dictionary of DataType")
"in_arg_names must be a list of str")
if not isinstance(out_types, list):
raise TypeError("out_types must be a list of DataType")

function_doc["arg_names"] = in_arg_names
num_args = len(in_arg_names)

for in_types in in_arg_types:
if isinstance(in_types, list):
if len(in_arg_names) != len(in_types):
raise ValueError(
"in_arg_names and input types per kernel must contain same number of elements")
for in_type in in_types:
c_in_types.push_back(
pyarrow_unwrap_data_type(ensure_type(in_type)))
else:
raise TypeError(
"Elements in in_arg_types must be a list of DataType")
vec_c_in_types.push_back(move(c_in_types))

c_arity = CArity(<int> num_args, func_spec.varargs)

Expand All@@ -2702,14 +2721,14 @@ def register_scalar_function(func, function_name, function_doc, in_types,
raise ValueError("Function doc must contain arg_names")

c_func_doc = _make_function_doc(function_doc)

c_out_type = pyarrow_unwrap_data_type(ensure_type(out_type))
for out_type in out_types:
c_out_types.push_back(pyarrow_unwrap_data_type(ensure_type(out_type)))

c_options.func_name = c_func_name
c_options.arity = c_arity
c_options.func_doc = c_func_doc
c_options.input_types = c_in_types
c_options.output_type = c_out_type
c_options.input_arg_types = vec_c_in_types
c_options.output_types = c_out_types

check_status(RegisterScalarFunction(c_function,
<function[CallbackUdf]> &_scalar_udf_callback, c_options))
4 changes: 2 additions & 2 deletions python/pyarrow/includes/libarrow.pxd
Original file line numberDiff line numberDiff line change
Expand Up@@ -2814,8 +2814,8 @@ cdef extern from "arrow/python/udf.h" namespace "arrow::py":
c_string func_name
CArity arity
CFunctionDoc func_doc
vector[shared_ptr[CDataType]] input_types
shared_ptr[CDataType] output_type
vector[vector[shared_ptr[CDataType]]] input_arg_types
vector[shared_ptr[CDataType]] output_types

CStatus RegisterScalarFunction(PyObject* function,
function[CallbackUdf] wrapper, const CScalarUdfOptions& options)
41 changes: 26 additions & 15 deletions python/pyarrow/src/arrow/python/udf.cc
Original file line numberDiff line numberDiff line change
Expand Up@@ -99,22 +99,33 @@ Status RegisterScalarFunction(PyObject* user_function, ScalarUdfWrapperCallback
auto scalar_func = std::make_shared<compute::ScalarFunction>(
options.func_name, options.arity, options.func_doc);
Py_INCREF(user_function);
std::vector<compute::InputType> input_types;
for (const auto& in_dtype : options.input_types) {
input_types.emplace_back(in_dtype);

const size_t num_kernels = options.input_arg_types.size();
// number of input_type variations and output_types must be
// equal in size
if(num_kernels != options.output_types.size()) {
return Status::Invalid("input_arg_types and output_types should be equal in size");
}
// adding kernels
for(size_t idx=0; idx < num_kernels; idx++) {
const auto& opt_input_types = options.input_arg_types[idx];
std::vector<compute::InputType> input_types;
for (const auto& in_dtype : opt_input_types) {
input_types.emplace_back(in_dtype);
}
const auto& opts_out_type = options.output_types[idx];
compute::OutputType output_type(opts_out_type);
auto udf_data = std::make_shared<PythonUdf>(
wrapper, std::make_shared<OwnedRefNoGIL>(user_function), opts_out_type);
compute::ScalarKernel kernel(
compute::KernelSignature::Make(std::move(input_types), std::move(output_type),
options.arity.is_varargs), PythonUdfExec);
kernel.data = std::move(udf_data);

kernel.mem_allocation = compute::MemAllocation::NO_PREALLOCATE;
kernel.null_handling = compute::NullHandling::COMPUTED_NO_PREALLOCATE;
RETURN_NOT_OK(scalar_func->AddKernel(std::move(kernel)));
}
compute::OutputType output_type(options.output_type);
auto udf_data = std::make_shared<PythonUdf>(
wrapper, std::make_shared<OwnedRefNoGIL>(user_function), options.output_type);
compute::ScalarKernel kernel(
compute::KernelSignature::Make(std::move(input_types), std::move(output_type),
options.arity.is_varargs),
PythonUdfExec);
kernel.data = std::move(udf_data);

kernel.mem_allocation = compute::MemAllocation::NO_PREALLOCATE;
kernel.null_handling = compute::NullHandling::COMPUTED_NO_PREALLOCATE;
RETURN_NOT_OK(scalar_func->AddKernel(std::move(kernel)));
auto registry = compute::GetFunctionRegistry();
RETURN_NOT_OK(registry->AddFunction(std::move(scalar_func)));
return Status::OK();
Expand Down
4 changes: 2 additions & 2 deletions python/pyarrow/src/arrow/python/udf.h
Original file line numberDiff line numberDiff line change
Expand Up@@ -37,8 +37,8 @@ struct ARROW_PYTHON_EXPORT ScalarUdfOptions {
std::string func_name;
compute::Arity arity;
compute::FunctionDoc func_doc;
std::vector<std::shared_ptr<DataType>> input_types;
std::shared_ptr<DataType> output_type;
std::vector<std::vector<std::shared_ptr<DataType>>> input_arg_types;
std::vector<std::shared_ptr<DataType>> output_types;
};

/// \brief A context passed as the first argument of scalar UDF functions.
Expand Down
Loading