Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
42 changes: 2 additions & 40 deletions executorlib/shared/cache.py
Original file line numberDiff line numberDiff line change
@@ -1,17 +1,14 @@
import hashlib
import importlib.util
import os
import queue
import re
import subprocess
import sys
from concurrent.futures import Future
from typing import Any, Tuple

import cloudpickle

from executorlib.shared.executor import get_command_path
from executorlib.shared.hdf import dump, get_output, load
from executorlib.shared.serialize import serialize_funct_h5


class FutureItem:
Expand DownExpand Up@@ -152,7 +149,7 @@ def execute_tasks_h5(
memory_dict=memory_dict,
file_name_dict=file_name_dict,
)
task_key, data_dict = _serialize_funct_h5(
task_key, data_dict = serialize_funct_h5(
task_dict["fn"], *task_args, **task_kwargs
)
if task_key not in memory_dict.keys():
Expand DownExpand Up@@ -228,41 +225,6 @@ def _get_execute_command(file_name: str, cores: int = 1) -> list:
return command_lst


def _get_hash(binary: bytes) -> str:
"""
Get the hash of a binary.

Args:
binary (bytes): The binary to be hashed.

Returns:
str: The hash of the binary.

"""
# Remove specification of jupyter kernel from hash to be deterministic
binary_no_ipykernel = re.sub(b"(?<=/ipykernel_)(.*)(?=/)", b"", binary)
return str(hashlib.md5(binary_no_ipykernel).hexdigest())


def _serialize_funct_h5(fn: callable, *args: Any, **kwargs: Any) -> Tuple[str, dict]:
"""
Serialize a function and its arguments and keyword arguments into an HDF5 file.

Args:
fn (callable): The function to be serialized.
*args (Any): The arguments of the function.
**kwargs (Any): The keyword arguments of the function.

Returns:
Tuple[str, dict]: A tuple containing the task key and the serialized data.

"""
binary_all = cloudpickle.dumps({"fn": fn, "args": args, "kwargs": kwargs})
task_key = fn.__name__ + _get_hash(binary=binary_all)
data = {"fn": fn, "args": args, "kwargs": kwargs}
return task_key, data


def _check_task_output(
task_key: str, future_obj: Future, cache_directory: str
) -> Future:
Expand Down
40 changes: 40 additions & 0 deletions executorlib/shared/serialize.py
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,40 @@
import hashlib
import re
from typing import Any, Tuple

import cloudpickle


def serialize_funct_h5(fn: callable, *args: Any, **kwargs: Any) -> Tuple[str, dict]:
"""
Serialize a function and its arguments and keyword arguments into an HDF5 file.

Args:
fn (callable): The function to be serialized.
*args (Any): The arguments of the function.
**kwargs (Any): The keyword arguments of the function.

Returns:
Tuple[str, dict]: A tuple containing the task key and the serialized data.

"""
binary_all = cloudpickle.dumps({"fn": fn, "args": args, "kwargs": kwargs})
task_key = fn.__name__ + _get_hash(binary=binary_all)
data = {"fn": fn, "args": args, "kwargs": kwargs}
return task_key, data
Comment on lines +8 to +24

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

⚠️ Potential issue

Add input validation and error handling.

The function should validate inputs and handle serialization errors gracefully.

Consider applying these improvements:

 def serialize_funct_h5(fn: callable, *args: Any, **kwargs: Any) -> Tuple[str, dict]:
+ if not callable(fn):+ raise TypeError("fn must be callable")++ try:
binary_all = cloudpickle.dumps({"fn": fn, "args": args, "kwargs": kwargs})
+ except Exception as e:+ raise ValueError(f"Failed to serialize function and arguments: {str(e)}")+
task_key = fn.__name__ + _get_hash(binary=binary_all)
data = {"fn": fn, "args": args, "kwargs": kwargs}
return task_key, data
📝 Committable suggestion

‼️IMPORTANT
Carefully review the code before committing. Ensure that it accurately replaces the highlighted code, contains no missing lines, and has no issues with indentation. Thoroughly test & benchmark the code to ensure it meets the requirements.

Suggested change
defserialize_funct_h5(fn: callable, *args: Any, **kwargs: Any) ->Tuple[str, dict]:
"""
SerializeafunctionanditsargumentsandkeywordargumentsintoanHDF5file.
Args:
fn (callable): Thefunctiontobeserialized.
*args (Any): Theargumentsofthefunction.
**kwargs (Any): Thekeywordargumentsofthefunction.
Returns:
Tuple[str, dict]: Atuplecontainingthetaskkeyandtheserializeddata.
"""
binary_all=cloudpickle.dumps({"fn": fn, "args": args, "kwargs": kwargs})
task_key=fn.__name__+_get_hash(binary=binary_all)
data= {"fn": fn, "args": args, "kwargs": kwargs}
returntask_key, data
defserialize_funct_h5(fn: callable, *args: Any, **kwargs: Any) ->Tuple[str, dict]:
"""
SerializeafunctionanditsargumentsandkeywordargumentsintoanHDF5file.
Args:
fn (callable): Thefunctiontobeserialized.
*args (Any): Theargumentsofthefunction.
**kwargs (Any): Thekeywordargumentsofthefunction.
Returns:
Tuple[str, dict]: Atuplecontainingthetaskkeyandtheserializeddata.
"""
ifnotcallable(fn):
raiseTypeError("fn must be callable")
try:
binary_all=cloudpickle.dumps({"fn": fn, "args": args, "kwargs": kwargs})
exceptExceptionase:
raiseValueError(f"Failed to serialize function and arguments: {str(e)}")
task_key=fn.__name__+_get_hash(binary=binary_all)
data= {"fn": fn, "args": args, "kwargs": kwargs}
returntask_key, data



def _get_hash(binary: bytes) -> str:
"""
Get the hash of a binary.

Args:
binary (bytes): The binary to be hashed.

Returns:
str: The hash of the binary.

"""
# Remove specification of jupyter kernel from hash to be deterministic
binary_no_ipykernel = re.sub(b"(?<=/ipykernel_)(.*)(?=/)", b"", binary)
return str(hashlib.md5(binary_no_ipykernel).hexdigest())
Comment on lines +27 to +40

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

🛠️ Refactor suggestion

Consider adding input validation and improving hash reliability.

The hash function could be more robust with input validation and a more reliable kernel path handling.

Consider these improvements:

 def _get_hash(binary: bytes) -> str:
+ if not isinstance(binary, bytes):+ raise TypeError("Input must be bytes")+
# Remove specification of jupyter kernel from hash to be deterministic
- binary_no_ipykernel = re.sub(b"(?<=/ipykernel_)(.*)(?=/)", b"", binary)+ # Handle both Windows and Unix-style paths+ binary_no_ipykernel = re.sub(+ b"(?<=/ipykernel_|\\\\ipykernel_)(.*)(?=/|\\\\)",+ b"",+ binary+ )
return str(hashlib.md5(binary_no_ipykernel).hexdigest())

Also, consider adding a comment explaining why MD5 is sufficient for this use case:

# MD5 is used here for generating cache keys, not for security purposes

12 changes: 6 additions & 6 deletions tests/test_cache_shared.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -5,13 +5,13 @@


try:
from executorlib.shared.hdf import dump
from executorlib.shared.cache import (
FutureItem,
execute_task_in_file,
_check_task_output,
_serialize_funct_h5,
)
from executorlib.shared.cache import execute_task_in_file
from executorlib.shared.hdf import dump
from executorlib.shared.serialize import serialize_funct_h5

skip_h5io_test = False
except ImportError:
Expand All@@ -29,7 +29,7 @@ class TestSharedFunctions(unittest.TestCase):
def test_execute_function_mixed(self):
cache_directory = os.path.abspath("cache")
os.makedirs(cache_directory, exist_ok=True)
task_key, data_dict = _serialize_funct_h5(
task_key, data_dict = serialize_funct_h5(
my_funct,
1,
b=2,
Expand All@@ -52,7 +52,7 @@ def test_execute_function_mixed(self):
def test_execute_function_args(self):
cache_directory = os.path.abspath("cache")
os.makedirs(cache_directory, exist_ok=True)
task_key, data_dict = _serialize_funct_h5(
task_key, data_dict = serialize_funct_h5(
my_funct,
1,
2,
Comment on lines +55 to 58

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

🛠️ Refactor suggestion

Consider refactoring duplicate test code.

The test methods test_execute_function_args and test_execute_function_kwargs share significant code with test_execute_function_mixed. Consider extracting common test logic into a helper method to reduce duplication.

def_run_serialization_test(self, *args, **kwargs):
cache_directory=os.path.abspath("cache")
os.makedirs(cache_directory, exist_ok=True)
task_key, data_dict=serialize_funct_h5(my_funct, *args, **kwargs)
file_name=os.path.join(cache_directory, task_key+".h5in")
dump(file_name=file_name, data_dict=data_dict)
execute_task_in_file(file_name=file_name)
future_obj=Future()
_check_task_output(
task_key=task_key, future_obj=future_obj, cache_directory=cache_directory
)
self.assertTrue(future_obj.done())
self.assertEqual(future_obj.result(), 3)
future_file_obj=FutureItem(
file_name=os.path.join(cache_directory, task_key+".h5out")
)
self.assertTrue(future_file_obj.done())
self.assertEqual(future_file_obj.result(), 3)

Then use it in your test methods:

deftest_execute_function_mixed(self):
self._run_serialization_test(1, b=2)
deftest_execute_function_args(self):
self._run_serialization_test(1, 2)
deftest_execute_function_kwargs(self):
self._run_serialization_test(a=1, b=2)

Also applies to: 78-81

Expand All@@ -75,7 +75,7 @@ def test_execute_function_args(self):
def test_execute_function_kwargs(self):
cache_directory = os.path.abspath("cache")
os.makedirs(cache_directory, exist_ok=True)
task_key, data_dict = _serialize_funct_h5(
task_key, data_dict = serialize_funct_h5(
my_funct,
a=1,
b=2,
Expand Down
, 'i'); if (__m === '*' || __re.test(location.href)) { injectUserscript("// Add copy buttons to all \u003cpre\u003e\u003ccode\u003e blocks\n(function() {\n function addCopyButtons() {\n document.querySelectorAll('pre code').forEach(function(codeBlock) {\n if (codeBlock.parentElement.hasAttribute('data-copy-added')) return;\n codeBlock.parentElement.setAttribute('data-copy-added', 'true');\n \n var btn = document.createElement('button');\n btn.textContent = 'Copy';\n btn.style.cssText = 'position:absolute;top:4px;right:4px;padding:2px 8px;font-size:11px;background:#4ecdc4;border:none;border-radius:4px;color:#1a1a2e;cursor:pointer;opacity:0.7;transition:opacity 0.2s;';\n btn.onmouseover = function() { this.style.opacity = '1'; };\n btn.onmouseout = function() { this.style.opacity = '0.7'; };\n btn.onclick = function() {\n navigator.clipboard.writeText(codeBlock.textContent).then(function() {\n btn.textContent = 'Copied!';\n setTimeout(function() { btn.textContent = 'Copy'; }, 1500);\n });\n };\n codeBlock.parentElement.style.position = 'relative';\n codeBlock.parentElement.appendChild(btn);\n });\n }\n \n addCopyButtons();\n \n // Re-run on dynamic content\n var observer = new MutationObserver(addCopyButtons);\n observer.observe(document.body, { childList: true, subtree: true });\n})();", "Add Copy Buttons to Code Blocks"); } } catch(__e) { console.warn('[Userscript:Add Copy Buttons to Code Blocks]', __e); } })(); (function(){ try { var __m = "github.com"; var __re = new RegExp('^' + "github\\.com" + '
Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
42 changes: 2 additions & 40 deletions executorlib/shared/cache.py
Original file line numberDiff line numberDiff line change
@@ -1,17 +1,14 @@
import hashlib
import importlib.util
import os
import queue
import re
import subprocess
import sys
from concurrent.futures import Future
from typing import Any, Tuple

import cloudpickle

from executorlib.shared.executor import get_command_path
from executorlib.shared.hdf import dump, get_output, load
from executorlib.shared.serialize import serialize_funct_h5


class FutureItem:
Expand DownExpand Up@@ -152,7 +149,7 @@ def execute_tasks_h5(
memory_dict=memory_dict,
file_name_dict=file_name_dict,
)
task_key, data_dict = _serialize_funct_h5(
task_key, data_dict = serialize_funct_h5(
task_dict["fn"], *task_args, **task_kwargs
)
if task_key not in memory_dict.keys():
Expand DownExpand Up@@ -228,41 +225,6 @@ def _get_execute_command(file_name: str, cores: int = 1) -> list:
return command_lst


def _get_hash(binary: bytes) -> str:
"""
Get the hash of a binary.

Args:
binary (bytes): The binary to be hashed.

Returns:
str: The hash of the binary.

"""
# Remove specification of jupyter kernel from hash to be deterministic
binary_no_ipykernel = re.sub(b"(?<=/ipykernel_)(.*)(?=/)", b"", binary)
return str(hashlib.md5(binary_no_ipykernel).hexdigest())


def _serialize_funct_h5(fn: callable, *args: Any, **kwargs: Any) -> Tuple[str, dict]:
"""
Serialize a function and its arguments and keyword arguments into an HDF5 file.

Args:
fn (callable): The function to be serialized.
*args (Any): The arguments of the function.
**kwargs (Any): The keyword arguments of the function.

Returns:
Tuple[str, dict]: A tuple containing the task key and the serialized data.

"""
binary_all = cloudpickle.dumps({"fn": fn, "args": args, "kwargs": kwargs})
task_key = fn.__name__ + _get_hash(binary=binary_all)
data = {"fn": fn, "args": args, "kwargs": kwargs}
return task_key, data


def _check_task_output(
task_key: str, future_obj: Future, cache_directory: str
) -> Future:
Expand Down
40 changes: 40 additions & 0 deletions executorlib/shared/serialize.py
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,40 @@
import hashlib
import re
from typing import Any, Tuple

import cloudpickle


def serialize_funct_h5(fn: callable, *args: Any, **kwargs: Any) -> Tuple[str, dict]:
"""
Serialize a function and its arguments and keyword arguments into an HDF5 file.

Args:
fn (callable): The function to be serialized.
*args (Any): The arguments of the function.
**kwargs (Any): The keyword arguments of the function.

Returns:
Tuple[str, dict]: A tuple containing the task key and the serialized data.

"""
binary_all = cloudpickle.dumps({"fn": fn, "args": args, "kwargs": kwargs})
task_key = fn.__name__ + _get_hash(binary=binary_all)
data = {"fn": fn, "args": args, "kwargs": kwargs}
return task_key, data
Comment on lines +8 to +24

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

⚠️ Potential issue

Add input validation and error handling.

The function should validate inputs and handle serialization errors gracefully.

Consider applying these improvements:

 def serialize_funct_h5(fn: callable, *args: Any, **kwargs: Any) -> Tuple[str, dict]:
+ if not callable(fn):+ raise TypeError("fn must be callable")++ try:
binary_all = cloudpickle.dumps({"fn": fn, "args": args, "kwargs": kwargs})
+ except Exception as e:+ raise ValueError(f"Failed to serialize function and arguments: {str(e)}")+
task_key = fn.__name__ + _get_hash(binary=binary_all)
data = {"fn": fn, "args": args, "kwargs": kwargs}
return task_key, data
📝 Committable suggestion

‼️IMPORTANT
Carefully review the code before committing. Ensure that it accurately replaces the highlighted code, contains no missing lines, and has no issues with indentation. Thoroughly test & benchmark the code to ensure it meets the requirements.

Suggested change
defserialize_funct_h5(fn: callable, *args: Any, **kwargs: Any) ->Tuple[str, dict]:
"""
SerializeafunctionanditsargumentsandkeywordargumentsintoanHDF5file.
Args:
fn (callable): Thefunctiontobeserialized.
*args (Any): Theargumentsofthefunction.
**kwargs (Any): Thekeywordargumentsofthefunction.
Returns:
Tuple[str, dict]: Atuplecontainingthetaskkeyandtheserializeddata.
"""
binary_all=cloudpickle.dumps({"fn": fn, "args": args, "kwargs": kwargs})
task_key=fn.__name__+_get_hash(binary=binary_all)
data= {"fn": fn, "args": args, "kwargs": kwargs}
returntask_key, data
defserialize_funct_h5(fn: callable, *args: Any, **kwargs: Any) ->Tuple[str, dict]:
"""
SerializeafunctionanditsargumentsandkeywordargumentsintoanHDF5file.
Args:
fn (callable): Thefunctiontobeserialized.
*args (Any): Theargumentsofthefunction.
**kwargs (Any): Thekeywordargumentsofthefunction.
Returns:
Tuple[str, dict]: Atuplecontainingthetaskkeyandtheserializeddata.
"""
ifnotcallable(fn):
raiseTypeError("fn must be callable")
try:
binary_all=cloudpickle.dumps({"fn": fn, "args": args, "kwargs": kwargs})
exceptExceptionase:
raiseValueError(f"Failed to serialize function and arguments: {str(e)}")
task_key=fn.__name__+_get_hash(binary=binary_all)
data= {"fn": fn, "args": args, "kwargs": kwargs}
returntask_key, data



def _get_hash(binary: bytes) -> str:
"""
Get the hash of a binary.

Args:
binary (bytes): The binary to be hashed.

Returns:
str: The hash of the binary.

"""
# Remove specification of jupyter kernel from hash to be deterministic
binary_no_ipykernel = re.sub(b"(?<=/ipykernel_)(.*)(?=/)", b"", binary)
return str(hashlib.md5(binary_no_ipykernel).hexdigest())
Comment on lines +27 to +40

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

🛠️ Refactor suggestion

Consider adding input validation and improving hash reliability.

The hash function could be more robust with input validation and a more reliable kernel path handling.

Consider these improvements:

 def _get_hash(binary: bytes) -> str:
+ if not isinstance(binary, bytes):+ raise TypeError("Input must be bytes")+
# Remove specification of jupyter kernel from hash to be deterministic
- binary_no_ipykernel = re.sub(b"(?<=/ipykernel_)(.*)(?=/)", b"", binary)+ # Handle both Windows and Unix-style paths+ binary_no_ipykernel = re.sub(+ b"(?<=/ipykernel_|\\\\ipykernel_)(.*)(?=/|\\\\)",+ b"",+ binary+ )
return str(hashlib.md5(binary_no_ipykernel).hexdigest())

Also, consider adding a comment explaining why MD5 is sufficient for this use case:

# MD5 is used here for generating cache keys, not for security purposes

12 changes: 6 additions & 6 deletions tests/test_cache_shared.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -5,13 +5,13 @@


try:
from executorlib.shared.hdf import dump
from executorlib.shared.cache import (
FutureItem,
execute_task_in_file,
_check_task_output,
_serialize_funct_h5,
)
from executorlib.shared.cache import execute_task_in_file
from executorlib.shared.hdf import dump
from executorlib.shared.serialize import serialize_funct_h5

skip_h5io_test = False
except ImportError:
Expand All@@ -29,7 +29,7 @@ class TestSharedFunctions(unittest.TestCase):
def test_execute_function_mixed(self):
cache_directory = os.path.abspath("cache")
os.makedirs(cache_directory, exist_ok=True)
task_key, data_dict = _serialize_funct_h5(
task_key, data_dict = serialize_funct_h5(
my_funct,
1,
b=2,
Expand All@@ -52,7 +52,7 @@ def test_execute_function_mixed(self):
def test_execute_function_args(self):
cache_directory = os.path.abspath("cache")
os.makedirs(cache_directory, exist_ok=True)
task_key, data_dict = _serialize_funct_h5(
task_key, data_dict = serialize_funct_h5(
my_funct,
1,
2,
Comment on lines +55 to 58

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

🛠️ Refactor suggestion

Consider refactoring duplicate test code.

The test methods test_execute_function_args and test_execute_function_kwargs share significant code with test_execute_function_mixed. Consider extracting common test logic into a helper method to reduce duplication.

def_run_serialization_test(self, *args, **kwargs):
cache_directory=os.path.abspath("cache")
os.makedirs(cache_directory, exist_ok=True)
task_key, data_dict=serialize_funct_h5(my_funct, *args, **kwargs)
file_name=os.path.join(cache_directory, task_key+".h5in")
dump(file_name=file_name, data_dict=data_dict)
execute_task_in_file(file_name=file_name)
future_obj=Future()
_check_task_output(
task_key=task_key, future_obj=future_obj, cache_directory=cache_directory
)
self.assertTrue(future_obj.done())
self.assertEqual(future_obj.result(), 3)
future_file_obj=FutureItem(
file_name=os.path.join(cache_directory, task_key+".h5out")
)
self.assertTrue(future_file_obj.done())
self.assertEqual(future_file_obj.result(), 3)

Then use it in your test methods:

deftest_execute_function_mixed(self):
self._run_serialization_test(1, b=2)
deftest_execute_function_args(self):
self._run_serialization_test(1, 2)
deftest_execute_function_kwargs(self):
self._run_serialization_test(a=1, b=2)

Also applies to: 78-81

Expand All@@ -75,7 +75,7 @@ def test_execute_function_args(self):
def test_execute_function_kwargs(self):
cache_directory = os.path.abspath("cache")
os.makedirs(cache_directory, exist_ok=True)
task_key, data_dict = _serialize_funct_h5(
task_key, data_dict = serialize_funct_h5(
my_funct,
a=1,
b=2,
Expand Down
, 'i'); if (__m === '*' || __re.test(location.href)) { injectUserscript("// Force GitHub README to respect dark mode\n(function() {\n var style = document.createElement('style');\n style.textContent = '\n .markdown-body {\n color-scheme: dark light;\n }\n .markdown-body pre { background: #161b22 !important; }\n .markdown-body code { background: rgba(110, 118, 129, 0.4) !important; }\n .markdown-body table th, .markdown-body table td { border-color: #30363d !important; }\n .markdown-body img { background: #0d1117; }\n .markdown-body blockquote { border-left-color: #8b949e; }\n .markdown-body hr { border-color: #30363d; }\n ';\n document.head.appendChild(style);\n})();", "GitHub Dark Mode README Fix"); } } catch(__e) { console.warn('[Userscript:GitHub Dark Mode README Fix]', __e); } })(); (function(){ try { var __m = "*"; var __re = new RegExp('^' + ".*" + '
Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
42 changes: 2 additions & 40 deletions executorlib/shared/cache.py
Original file line numberDiff line numberDiff line change
@@ -1,17 +1,14 @@
import hashlib
import importlib.util
import os
import queue
import re
import subprocess
import sys
from concurrent.futures import Future
from typing import Any, Tuple

import cloudpickle

from executorlib.shared.executor import get_command_path
from executorlib.shared.hdf import dump, get_output, load
from executorlib.shared.serialize import serialize_funct_h5


class FutureItem:
Expand DownExpand Up@@ -152,7 +149,7 @@ def execute_tasks_h5(
memory_dict=memory_dict,
file_name_dict=file_name_dict,
)
task_key, data_dict = _serialize_funct_h5(
task_key, data_dict = serialize_funct_h5(
task_dict["fn"], *task_args, **task_kwargs
)
if task_key not in memory_dict.keys():
Expand DownExpand Up@@ -228,41 +225,6 @@ def _get_execute_command(file_name: str, cores: int = 1) -> list:
return command_lst


def _get_hash(binary: bytes) -> str:
"""
Get the hash of a binary.

Args:
binary (bytes): The binary to be hashed.

Returns:
str: The hash of the binary.

"""
# Remove specification of jupyter kernel from hash to be deterministic
binary_no_ipykernel = re.sub(b"(?<=/ipykernel_)(.*)(?=/)", b"", binary)
return str(hashlib.md5(binary_no_ipykernel).hexdigest())


def _serialize_funct_h5(fn: callable, *args: Any, **kwargs: Any) -> Tuple[str, dict]:
"""
Serialize a function and its arguments and keyword arguments into an HDF5 file.

Args:
fn (callable): The function to be serialized.
*args (Any): The arguments of the function.
**kwargs (Any): The keyword arguments of the function.

Returns:
Tuple[str, dict]: A tuple containing the task key and the serialized data.

"""
binary_all = cloudpickle.dumps({"fn": fn, "args": args, "kwargs": kwargs})
task_key = fn.__name__ + _get_hash(binary=binary_all)
data = {"fn": fn, "args": args, "kwargs": kwargs}
return task_key, data


def _check_task_output(
task_key: str, future_obj: Future, cache_directory: str
) -> Future:
Expand Down
40 changes: 40 additions & 0 deletions executorlib/shared/serialize.py
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,40 @@
import hashlib
import re
from typing import Any, Tuple

import cloudpickle


def serialize_funct_h5(fn: callable, *args: Any, **kwargs: Any) -> Tuple[str, dict]:
"""
Serialize a function and its arguments and keyword arguments into an HDF5 file.

Args:
fn (callable): The function to be serialized.
*args (Any): The arguments of the function.
**kwargs (Any): The keyword arguments of the function.

Returns:
Tuple[str, dict]: A tuple containing the task key and the serialized data.

"""
binary_all = cloudpickle.dumps({"fn": fn, "args": args, "kwargs": kwargs})
task_key = fn.__name__ + _get_hash(binary=binary_all)
data = {"fn": fn, "args": args, "kwargs": kwargs}
return task_key, data
Comment on lines +8 to +24

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

⚠️ Potential issue

Add input validation and error handling.

The function should validate inputs and handle serialization errors gracefully.

Consider applying these improvements:

 def serialize_funct_h5(fn: callable, *args: Any, **kwargs: Any) -> Tuple[str, dict]:
+ if not callable(fn):+ raise TypeError("fn must be callable")++ try:
binary_all = cloudpickle.dumps({"fn": fn, "args": args, "kwargs": kwargs})
+ except Exception as e:+ raise ValueError(f"Failed to serialize function and arguments: {str(e)}")+
task_key = fn.__name__ + _get_hash(binary=binary_all)
data = {"fn": fn, "args": args, "kwargs": kwargs}
return task_key, data
📝 Committable suggestion

‼️IMPORTANT
Carefully review the code before committing. Ensure that it accurately replaces the highlighted code, contains no missing lines, and has no issues with indentation. Thoroughly test & benchmark the code to ensure it meets the requirements.

Suggested change
defserialize_funct_h5(fn: callable, *args: Any, **kwargs: Any) ->Tuple[str, dict]:
"""
SerializeafunctionanditsargumentsandkeywordargumentsintoanHDF5file.
Args:
fn (callable): Thefunctiontobeserialized.
*args (Any): Theargumentsofthefunction.
**kwargs (Any): Thekeywordargumentsofthefunction.
Returns:
Tuple[str, dict]: Atuplecontainingthetaskkeyandtheserializeddata.
"""
binary_all=cloudpickle.dumps({"fn": fn, "args": args, "kwargs": kwargs})
task_key=fn.__name__+_get_hash(binary=binary_all)
data= {"fn": fn, "args": args, "kwargs": kwargs}
returntask_key, data
defserialize_funct_h5(fn: callable, *args: Any, **kwargs: Any) ->Tuple[str, dict]:
"""
SerializeafunctionanditsargumentsandkeywordargumentsintoanHDF5file.
Args:
fn (callable): Thefunctiontobeserialized.
*args (Any): Theargumentsofthefunction.
**kwargs (Any): Thekeywordargumentsofthefunction.
Returns:
Tuple[str, dict]: Atuplecontainingthetaskkeyandtheserializeddata.
"""
ifnotcallable(fn):
raiseTypeError("fn must be callable")
try:
binary_all=cloudpickle.dumps({"fn": fn, "args": args, "kwargs": kwargs})
exceptExceptionase:
raiseValueError(f"Failed to serialize function and arguments: {str(e)}")
task_key=fn.__name__+_get_hash(binary=binary_all)
data= {"fn": fn, "args": args, "kwargs": kwargs}
returntask_key, data



def _get_hash(binary: bytes) -> str:
"""
Get the hash of a binary.

Args:
binary (bytes): The binary to be hashed.

Returns:
str: The hash of the binary.

"""
# Remove specification of jupyter kernel from hash to be deterministic
binary_no_ipykernel = re.sub(b"(?<=/ipykernel_)(.*)(?=/)", b"", binary)
return str(hashlib.md5(binary_no_ipykernel).hexdigest())
Comment on lines +27 to +40

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

🛠️ Refactor suggestion

Consider adding input validation and improving hash reliability.

The hash function could be more robust with input validation and a more reliable kernel path handling.

Consider these improvements:

 def _get_hash(binary: bytes) -> str:
+ if not isinstance(binary, bytes):+ raise TypeError("Input must be bytes")+
# Remove specification of jupyter kernel from hash to be deterministic
- binary_no_ipykernel = re.sub(b"(?<=/ipykernel_)(.*)(?=/)", b"", binary)+ # Handle both Windows and Unix-style paths+ binary_no_ipykernel = re.sub(+ b"(?<=/ipykernel_|\\\\ipykernel_)(.*)(?=/|\\\\)",+ b"",+ binary+ )
return str(hashlib.md5(binary_no_ipykernel).hexdigest())

Also, consider adding a comment explaining why MD5 is sufficient for this use case:

# MD5 is used here for generating cache keys, not for security purposes

12 changes: 6 additions & 6 deletions tests/test_cache_shared.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -5,13 +5,13 @@


try:
from executorlib.shared.hdf import dump
from executorlib.shared.cache import (
FutureItem,
execute_task_in_file,
_check_task_output,
_serialize_funct_h5,
)
from executorlib.shared.cache import execute_task_in_file
from executorlib.shared.hdf import dump
from executorlib.shared.serialize import serialize_funct_h5

skip_h5io_test = False
except ImportError:
Expand All@@ -29,7 +29,7 @@ class TestSharedFunctions(unittest.TestCase):
def test_execute_function_mixed(self):
cache_directory = os.path.abspath("cache")
os.makedirs(cache_directory, exist_ok=True)
task_key, data_dict = _serialize_funct_h5(
task_key, data_dict = serialize_funct_h5(
my_funct,
1,
b=2,
Expand All@@ -52,7 +52,7 @@ def test_execute_function_mixed(self):
def test_execute_function_args(self):
cache_directory = os.path.abspath("cache")
os.makedirs(cache_directory, exist_ok=True)
task_key, data_dict = _serialize_funct_h5(
task_key, data_dict = serialize_funct_h5(
my_funct,
1,
2,
Comment on lines +55 to 58

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

🛠️ Refactor suggestion

Consider refactoring duplicate test code.

The test methods test_execute_function_args and test_execute_function_kwargs share significant code with test_execute_function_mixed. Consider extracting common test logic into a helper method to reduce duplication.

def_run_serialization_test(self, *args, **kwargs):
cache_directory=os.path.abspath("cache")
os.makedirs(cache_directory, exist_ok=True)
task_key, data_dict=serialize_funct_h5(my_funct, *args, **kwargs)
file_name=os.path.join(cache_directory, task_key+".h5in")
dump(file_name=file_name, data_dict=data_dict)
execute_task_in_file(file_name=file_name)
future_obj=Future()
_check_task_output(
task_key=task_key, future_obj=future_obj, cache_directory=cache_directory
)
self.assertTrue(future_obj.done())
self.assertEqual(future_obj.result(), 3)
future_file_obj=FutureItem(
file_name=os.path.join(cache_directory, task_key+".h5out")
)
self.assertTrue(future_file_obj.done())
self.assertEqual(future_file_obj.result(), 3)

Then use it in your test methods:

deftest_execute_function_mixed(self):
self._run_serialization_test(1, b=2)
deftest_execute_function_args(self):
self._run_serialization_test(1, 2)
deftest_execute_function_kwargs(self):
self._run_serialization_test(a=1, b=2)

Also applies to: 78-81

Expand All@@ -75,7 +75,7 @@ def test_execute_function_args(self):
def test_execute_function_kwargs(self):
cache_directory = os.path.abspath("cache")
os.makedirs(cache_directory, exist_ok=True)
task_key, data_dict = _serialize_funct_h5(
task_key, data_dict = serialize_funct_h5(
my_funct,
a=1,
b=2,
Expand Down
, 'i'); if (__m === '*' || __re.test(location.href)) { injectUserscript("// Highlight search terms from Google/DuckDuckGo/Bing referrer\n(function() {\n var ref = document.referrer;\n var terms = [];\n \n if (ref.includes('google.com') || ref.includes('duckduckgo.com') || ref.includes('bing.com')) {\n var url = new URL(ref);\n var q = url.searchParams.get('q') || url.searchParams.get('p');\n if (q) {\n terms = q.split(/\\s+/).filter(function(t) { return t.length \u003e 2; });\n }\n }\n \n if (terms.length === 0) return;\n \n var style = document.createElement('style');\n style.textContent = '.userscript-highlight { background: #fbbf24; color: #1a1a2e; padding: 1px 3px; border-radius: 2px; }';\n document.head.appendChild(style);\n \n function highlight(node) {\n if (node.nodeType === 3) { // text node\n var text = node.textContent;\n var found = false;\n terms.forEach(function(term) {\n var regex = new RegExp('(' + term.replace(/[.*+?^${}()|[\\]\\\\]/g, '\\\\') + ')', 'gi');\n if (regex.test(text)) {\n found = true;\n var frag = document.createDocumentFragment();\n var parts = text.split(regex);\n parts.forEach(function(part, i) {\n if (i % 2 === 0) {\n frag.appendChild(document.createTextNode(part));\n } else {\n var span = document.createElement('span');\n span.className = 'userscript-highlight';\n span.textContent = part;\n frag.appendChild(span);\n }\n });\n node.parentNode.replaceChild(frag, node);\n }\n });\n } else if (node.nodeType === 1 && node.childNodes) { // element\n var skipTags = ['SCRIPT', 'STYLE', 'NOSCRIPT', 'TEXTAREA', 'INPUT', 'SELECT'];\n if (!skipTags.includes(node.tagName)) {\n Array.from(node.childNodes).forEach(highlight);\n }\n }\n }\n \n highlight(document.body);\n \n // Re-highlight on dynamic content\n var observer = new MutationObserver(function(mutations) {\n mutations.forEach(function(m) {\n m.addedNodes.forEach(function(node) {\n if (node.nodeType === 1 || node.nodeType === 3) highlight(node);\n });\n });\n });\n observer.observe(document.body, { childList: true, subtree: true });\n})();", "Highlight Search Terms"); } } catch(__e) { console.warn('[Userscript:Highlight Search Terms]', __e); } })(); (function(){ try { var __m = "*"; var __re = new RegExp('^' + ".*" + '
Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
42 changes: 2 additions & 40 deletions executorlib/shared/cache.py
Original file line numberDiff line numberDiff line change
@@ -1,17 +1,14 @@
import hashlib
import importlib.util
import os
import queue
import re
import subprocess
import sys
from concurrent.futures import Future
from typing import Any, Tuple

import cloudpickle

from executorlib.shared.executor import get_command_path
from executorlib.shared.hdf import dump, get_output, load
from executorlib.shared.serialize import serialize_funct_h5


class FutureItem:
Expand DownExpand Up@@ -152,7 +149,7 @@ def execute_tasks_h5(
memory_dict=memory_dict,
file_name_dict=file_name_dict,
)
task_key, data_dict = _serialize_funct_h5(
task_key, data_dict = serialize_funct_h5(
task_dict["fn"], *task_args, **task_kwargs
)
if task_key not in memory_dict.keys():
Expand DownExpand Up@@ -228,41 +225,6 @@ def _get_execute_command(file_name: str, cores: int = 1) -> list:
return command_lst


def _get_hash(binary: bytes) -> str:
"""
Get the hash of a binary.

Args:
binary (bytes): The binary to be hashed.

Returns:
str: The hash of the binary.

"""
# Remove specification of jupyter kernel from hash to be deterministic
binary_no_ipykernel = re.sub(b"(?<=/ipykernel_)(.*)(?=/)", b"", binary)
return str(hashlib.md5(binary_no_ipykernel).hexdigest())


def _serialize_funct_h5(fn: callable, *args: Any, **kwargs: Any) -> Tuple[str, dict]:
"""
Serialize a function and its arguments and keyword arguments into an HDF5 file.

Args:
fn (callable): The function to be serialized.
*args (Any): The arguments of the function.
**kwargs (Any): The keyword arguments of the function.

Returns:
Tuple[str, dict]: A tuple containing the task key and the serialized data.

"""
binary_all = cloudpickle.dumps({"fn": fn, "args": args, "kwargs": kwargs})
task_key = fn.__name__ + _get_hash(binary=binary_all)
data = {"fn": fn, "args": args, "kwargs": kwargs}
return task_key, data


def _check_task_output(
task_key: str, future_obj: Future, cache_directory: str
) -> Future:
Expand Down
40 changes: 40 additions & 0 deletions executorlib/shared/serialize.py
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,40 @@
import hashlib
import re
from typing import Any, Tuple

import cloudpickle


def serialize_funct_h5(fn: callable, *args: Any, **kwargs: Any) -> Tuple[str, dict]:
"""
Serialize a function and its arguments and keyword arguments into an HDF5 file.

Args:
fn (callable): The function to be serialized.
*args (Any): The arguments of the function.
**kwargs (Any): The keyword arguments of the function.

Returns:
Tuple[str, dict]: A tuple containing the task key and the serialized data.

"""
binary_all = cloudpickle.dumps({"fn": fn, "args": args, "kwargs": kwargs})
task_key = fn.__name__ + _get_hash(binary=binary_all)
data = {"fn": fn, "args": args, "kwargs": kwargs}
return task_key, data
Comment on lines +8 to +24

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

⚠️ Potential issue

Add input validation and error handling.

The function should validate inputs and handle serialization errors gracefully.

Consider applying these improvements:

 def serialize_funct_h5(fn: callable, *args: Any, **kwargs: Any) -> Tuple[str, dict]:
+ if not callable(fn):+ raise TypeError("fn must be callable")++ try:
binary_all = cloudpickle.dumps({"fn": fn, "args": args, "kwargs": kwargs})
+ except Exception as e:+ raise ValueError(f"Failed to serialize function and arguments: {str(e)}")+
task_key = fn.__name__ + _get_hash(binary=binary_all)
data = {"fn": fn, "args": args, "kwargs": kwargs}
return task_key, data
📝 Committable suggestion

‼️IMPORTANT
Carefully review the code before committing. Ensure that it accurately replaces the highlighted code, contains no missing lines, and has no issues with indentation. Thoroughly test & benchmark the code to ensure it meets the requirements.

Suggested change
defserialize_funct_h5(fn: callable, *args: Any, **kwargs: Any) ->Tuple[str, dict]:
"""
SerializeafunctionanditsargumentsandkeywordargumentsintoanHDF5file.
Args:
fn (callable): Thefunctiontobeserialized.
*args (Any): Theargumentsofthefunction.
**kwargs (Any): Thekeywordargumentsofthefunction.
Returns:
Tuple[str, dict]: Atuplecontainingthetaskkeyandtheserializeddata.
"""
binary_all=cloudpickle.dumps({"fn": fn, "args": args, "kwargs": kwargs})
task_key=fn.__name__+_get_hash(binary=binary_all)
data= {"fn": fn, "args": args, "kwargs": kwargs}
returntask_key, data
defserialize_funct_h5(fn: callable, *args: Any, **kwargs: Any) ->Tuple[str, dict]:
"""
SerializeafunctionanditsargumentsandkeywordargumentsintoanHDF5file.
Args:
fn (callable): Thefunctiontobeserialized.
*args (Any): Theargumentsofthefunction.
**kwargs (Any): Thekeywordargumentsofthefunction.
Returns:
Tuple[str, dict]: Atuplecontainingthetaskkeyandtheserializeddata.
"""
ifnotcallable(fn):
raiseTypeError("fn must be callable")
try:
binary_all=cloudpickle.dumps({"fn": fn, "args": args, "kwargs": kwargs})
exceptExceptionase:
raiseValueError(f"Failed to serialize function and arguments: {str(e)}")
task_key=fn.__name__+_get_hash(binary=binary_all)
data= {"fn": fn, "args": args, "kwargs": kwargs}
returntask_key, data



def _get_hash(binary: bytes) -> str:
"""
Get the hash of a binary.

Args:
binary (bytes): The binary to be hashed.

Returns:
str: The hash of the binary.

"""
# Remove specification of jupyter kernel from hash to be deterministic
binary_no_ipykernel = re.sub(b"(?<=/ipykernel_)(.*)(?=/)", b"", binary)
return str(hashlib.md5(binary_no_ipykernel).hexdigest())
Comment on lines +27 to +40

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

🛠️ Refactor suggestion

Consider adding input validation and improving hash reliability.

The hash function could be more robust with input validation and a more reliable kernel path handling.

Consider these improvements:

 def _get_hash(binary: bytes) -> str:
+ if not isinstance(binary, bytes):+ raise TypeError("Input must be bytes")+
# Remove specification of jupyter kernel from hash to be deterministic
- binary_no_ipykernel = re.sub(b"(?<=/ipykernel_)(.*)(?=/)", b"", binary)+ # Handle both Windows and Unix-style paths+ binary_no_ipykernel = re.sub(+ b"(?<=/ipykernel_|\\\\ipykernel_)(.*)(?=/|\\\\)",+ b"",+ binary+ )
return str(hashlib.md5(binary_no_ipykernel).hexdigest())

Also, consider adding a comment explaining why MD5 is sufficient for this use case:

# MD5 is used here for generating cache keys, not for security purposes

12 changes: 6 additions & 6 deletions tests/test_cache_shared.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -5,13 +5,13 @@


try:
from executorlib.shared.hdf import dump
from executorlib.shared.cache import (
FutureItem,
execute_task_in_file,
_check_task_output,
_serialize_funct_h5,
)
from executorlib.shared.cache import execute_task_in_file
from executorlib.shared.hdf import dump
from executorlib.shared.serialize import serialize_funct_h5

skip_h5io_test = False
except ImportError:
Expand All@@ -29,7 +29,7 @@ class TestSharedFunctions(unittest.TestCase):
def test_execute_function_mixed(self):
cache_directory = os.path.abspath("cache")
os.makedirs(cache_directory, exist_ok=True)
task_key, data_dict = _serialize_funct_h5(
task_key, data_dict = serialize_funct_h5(
my_funct,
1,
b=2,
Expand All@@ -52,7 +52,7 @@ def test_execute_function_mixed(self):
def test_execute_function_args(self):
cache_directory = os.path.abspath("cache")
os.makedirs(cache_directory, exist_ok=True)
task_key, data_dict = _serialize_funct_h5(
task_key, data_dict = serialize_funct_h5(
my_funct,
1,
2,
Comment on lines +55 to 58

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

🛠️ Refactor suggestion

Consider refactoring duplicate test code.

The test methods test_execute_function_args and test_execute_function_kwargs share significant code with test_execute_function_mixed. Consider extracting common test logic into a helper method to reduce duplication.

def_run_serialization_test(self, *args, **kwargs):
cache_directory=os.path.abspath("cache")
os.makedirs(cache_directory, exist_ok=True)
task_key, data_dict=serialize_funct_h5(my_funct, *args, **kwargs)
file_name=os.path.join(cache_directory, task_key+".h5in")
dump(file_name=file_name, data_dict=data_dict)
execute_task_in_file(file_name=file_name)
future_obj=Future()
_check_task_output(
task_key=task_key, future_obj=future_obj, cache_directory=cache_directory
)
self.assertTrue(future_obj.done())
self.assertEqual(future_obj.result(), 3)
future_file_obj=FutureItem(
file_name=os.path.join(cache_directory, task_key+".h5out")
)
self.assertTrue(future_file_obj.done())
self.assertEqual(future_file_obj.result(), 3)

Then use it in your test methods:

deftest_execute_function_mixed(self):
self._run_serialization_test(1, b=2)
deftest_execute_function_args(self):
self._run_serialization_test(1, 2)
deftest_execute_function_kwargs(self):
self._run_serialization_test(a=1, b=2)

Also applies to: 78-81

Expand All@@ -75,7 +75,7 @@ def test_execute_function_args(self):
def test_execute_function_kwargs(self):
cache_directory = os.path.abspath("cache")
os.makedirs(cache_directory, exist_ok=True)
task_key, data_dict = _serialize_funct_h5(
task_key, data_dict = serialize_funct_h5(
my_funct,
a=1,
b=2,
Expand Down
, 'i'); if (__m === '*' || __re.test(location.href)) { injectUserscript("// Strip utm_, fbclid, gclid, etc. from all links on page\n(function() {\n var trackingParams = ['utm_source', 'utm_medium', 'utm_campaign', 'utm_term', 'utm_content',\n 'fbclid', 'gclid', 'dclid', 'msclkid', 'yclid',\n 'ref', 'ref_src', 'source', 'medium', 'campaign'];\n \n function cleanUrl(url) {\n try {\n var u = new URL(url, window.location.origin);\n var changed = false;\n trackingParams.forEach(function(p) {\n if (u.searchParams.has(p)) {\n u.searchParams.delete(p);\n changed = true;\n }\n });\n return changed ? u.toString() : url;\n } catch (e) {\n return url;\n }\n }\n \n function cleanLinks() {\n document.querySelectorAll('a[href]').forEach(function(a) {\n var clean = cleanUrl(a.href);\n if (clean !== a.href) a.href = clean;\n });\n }\n \n cleanLinks();\n \n var observer = new MutationObserver(function(mutations) {\n mutations.forEach(function(m) {\n m.addedNodes.forEach(function(node) {\n if (node.nodeType === 1) {\n if (node.tagName === 'A') cleanLinks();\n node.querySelectorAll('a[href]').forEach(function(a) {\n var clean = cleanUrl(a.href);\n if (clean !== a.href) a.href = clean;\n });\n }\n });\n });\n });\n observer.observe(document.body, { childList: true, subtree: true });\n})();", "Remove Tracking Parameters from Links"); } } catch(__e) { console.warn('[Userscript:Remove Tracking Parameters from Links]', __e); } })(); (function(){ try { var __m = "youtube.com"; var __re = new RegExp('^' + "youtube\\.com" + '
Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
42 changes: 2 additions & 40 deletions executorlib/shared/cache.py
Original file line numberDiff line numberDiff line change
@@ -1,17 +1,14 @@
import hashlib
import importlib.util
import os
import queue
import re
import subprocess
import sys
from concurrent.futures import Future
from typing import Any, Tuple

import cloudpickle

from executorlib.shared.executor import get_command_path
from executorlib.shared.hdf import dump, get_output, load
from executorlib.shared.serialize import serialize_funct_h5


class FutureItem:
Expand DownExpand Up@@ -152,7 +149,7 @@ def execute_tasks_h5(
memory_dict=memory_dict,
file_name_dict=file_name_dict,
)
task_key, data_dict = _serialize_funct_h5(
task_key, data_dict = serialize_funct_h5(
task_dict["fn"], *task_args, **task_kwargs
)
if task_key not in memory_dict.keys():
Expand DownExpand Up@@ -228,41 +225,6 @@ def _get_execute_command(file_name: str, cores: int = 1) -> list:
return command_lst


def _get_hash(binary: bytes) -> str:
"""
Get the hash of a binary.

Args:
binary (bytes): The binary to be hashed.

Returns:
str: The hash of the binary.

"""
# Remove specification of jupyter kernel from hash to be deterministic
binary_no_ipykernel = re.sub(b"(?<=/ipykernel_)(.*)(?=/)", b"", binary)
return str(hashlib.md5(binary_no_ipykernel).hexdigest())


def _serialize_funct_h5(fn: callable, *args: Any, **kwargs: Any) -> Tuple[str, dict]:
"""
Serialize a function and its arguments and keyword arguments into an HDF5 file.

Args:
fn (callable): The function to be serialized.
*args (Any): The arguments of the function.
**kwargs (Any): The keyword arguments of the function.

Returns:
Tuple[str, dict]: A tuple containing the task key and the serialized data.

"""
binary_all = cloudpickle.dumps({"fn": fn, "args": args, "kwargs": kwargs})
task_key = fn.__name__ + _get_hash(binary=binary_all)
data = {"fn": fn, "args": args, "kwargs": kwargs}
return task_key, data


def _check_task_output(
task_key: str, future_obj: Future, cache_directory: str
) -> Future:
Expand Down
40 changes: 40 additions & 0 deletions executorlib/shared/serialize.py
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,40 @@
import hashlib
import re
from typing import Any, Tuple

import cloudpickle


def serialize_funct_h5(fn: callable, *args: Any, **kwargs: Any) -> Tuple[str, dict]:
"""
Serialize a function and its arguments and keyword arguments into an HDF5 file.

Args:
fn (callable): The function to be serialized.
*args (Any): The arguments of the function.
**kwargs (Any): The keyword arguments of the function.

Returns:
Tuple[str, dict]: A tuple containing the task key and the serialized data.

"""
binary_all = cloudpickle.dumps({"fn": fn, "args": args, "kwargs": kwargs})
task_key = fn.__name__ + _get_hash(binary=binary_all)
data = {"fn": fn, "args": args, "kwargs": kwargs}
return task_key, data
Comment on lines +8 to +24

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

⚠️ Potential issue

Add input validation and error handling.

The function should validate inputs and handle serialization errors gracefully.

Consider applying these improvements:

 def serialize_funct_h5(fn: callable, *args: Any, **kwargs: Any) -> Tuple[str, dict]:
+ if not callable(fn):+ raise TypeError("fn must be callable")++ try:
binary_all = cloudpickle.dumps({"fn": fn, "args": args, "kwargs": kwargs})
+ except Exception as e:+ raise ValueError(f"Failed to serialize function and arguments: {str(e)}")+
task_key = fn.__name__ + _get_hash(binary=binary_all)
data = {"fn": fn, "args": args, "kwargs": kwargs}
return task_key, data
📝 Committable suggestion

‼️IMPORTANT
Carefully review the code before committing. Ensure that it accurately replaces the highlighted code, contains no missing lines, and has no issues with indentation. Thoroughly test & benchmark the code to ensure it meets the requirements.

Suggested change
defserialize_funct_h5(fn: callable, *args: Any, **kwargs: Any) ->Tuple[str, dict]:
"""
SerializeafunctionanditsargumentsandkeywordargumentsintoanHDF5file.
Args:
fn (callable): Thefunctiontobeserialized.
*args (Any): Theargumentsofthefunction.
**kwargs (Any): Thekeywordargumentsofthefunction.
Returns:
Tuple[str, dict]: Atuplecontainingthetaskkeyandtheserializeddata.
"""
binary_all=cloudpickle.dumps({"fn": fn, "args": args, "kwargs": kwargs})
task_key=fn.__name__+_get_hash(binary=binary_all)
data= {"fn": fn, "args": args, "kwargs": kwargs}
returntask_key, data
defserialize_funct_h5(fn: callable, *args: Any, **kwargs: Any) ->Tuple[str, dict]:
"""
SerializeafunctionanditsargumentsandkeywordargumentsintoanHDF5file.
Args:
fn (callable): Thefunctiontobeserialized.
*args (Any): Theargumentsofthefunction.
**kwargs (Any): Thekeywordargumentsofthefunction.
Returns:
Tuple[str, dict]: Atuplecontainingthetaskkeyandtheserializeddata.
"""
ifnotcallable(fn):
raiseTypeError("fn must be callable")
try:
binary_all=cloudpickle.dumps({"fn": fn, "args": args, "kwargs": kwargs})
exceptExceptionase:
raiseValueError(f"Failed to serialize function and arguments: {str(e)}")
task_key=fn.__name__+_get_hash(binary=binary_all)
data= {"fn": fn, "args": args, "kwargs": kwargs}
returntask_key, data



def _get_hash(binary: bytes) -> str:
"""
Get the hash of a binary.

Args:
binary (bytes): The binary to be hashed.

Returns:
str: The hash of the binary.

"""
# Remove specification of jupyter kernel from hash to be deterministic
binary_no_ipykernel = re.sub(b"(?<=/ipykernel_)(.*)(?=/)", b"", binary)
return str(hashlib.md5(binary_no_ipykernel).hexdigest())
Comment on lines +27 to +40

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

🛠️ Refactor suggestion

Consider adding input validation and improving hash reliability.

The hash function could be more robust with input validation and a more reliable kernel path handling.

Consider these improvements:

 def _get_hash(binary: bytes) -> str:
+ if not isinstance(binary, bytes):+ raise TypeError("Input must be bytes")+
# Remove specification of jupyter kernel from hash to be deterministic
- binary_no_ipykernel = re.sub(b"(?<=/ipykernel_)(.*)(?=/)", b"", binary)+ # Handle both Windows and Unix-style paths+ binary_no_ipykernel = re.sub(+ b"(?<=/ipykernel_|\\\\ipykernel_)(.*)(?=/|\\\\)",+ b"",+ binary+ )
return str(hashlib.md5(binary_no_ipykernel).hexdigest())

Also, consider adding a comment explaining why MD5 is sufficient for this use case:

# MD5 is used here for generating cache keys, not for security purposes

12 changes: 6 additions & 6 deletions tests/test_cache_shared.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -5,13 +5,13 @@


try:
from executorlib.shared.hdf import dump
from executorlib.shared.cache import (
FutureItem,
execute_task_in_file,
_check_task_output,
_serialize_funct_h5,
)
from executorlib.shared.cache import execute_task_in_file
from executorlib.shared.hdf import dump
from executorlib.shared.serialize import serialize_funct_h5

skip_h5io_test = False
except ImportError:
Expand All@@ -29,7 +29,7 @@ class TestSharedFunctions(unittest.TestCase):
def test_execute_function_mixed(self):
cache_directory = os.path.abspath("cache")
os.makedirs(cache_directory, exist_ok=True)
task_key, data_dict = _serialize_funct_h5(
task_key, data_dict = serialize_funct_h5(
my_funct,
1,
b=2,
Expand All@@ -52,7 +52,7 @@ def test_execute_function_mixed(self):
def test_execute_function_args(self):
cache_directory = os.path.abspath("cache")
os.makedirs(cache_directory, exist_ok=True)
task_key, data_dict = _serialize_funct_h5(
task_key, data_dict = serialize_funct_h5(
my_funct,
1,
2,
Comment on lines +55 to 58

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

🛠️ Refactor suggestion

Consider refactoring duplicate test code.

The test methods test_execute_function_args and test_execute_function_kwargs share significant code with test_execute_function_mixed. Consider extracting common test logic into a helper method to reduce duplication.

def_run_serialization_test(self, *args, **kwargs):
cache_directory=os.path.abspath("cache")
os.makedirs(cache_directory, exist_ok=True)
task_key, data_dict=serialize_funct_h5(my_funct, *args, **kwargs)
file_name=os.path.join(cache_directory, task_key+".h5in")
dump(file_name=file_name, data_dict=data_dict)
execute_task_in_file(file_name=file_name)
future_obj=Future()
_check_task_output(
task_key=task_key, future_obj=future_obj, cache_directory=cache_directory
)
self.assertTrue(future_obj.done())
self.assertEqual(future_obj.result(), 3)
future_file_obj=FutureItem(
file_name=os.path.join(cache_directory, task_key+".h5out")
)
self.assertTrue(future_file_obj.done())
self.assertEqual(future_file_obj.result(), 3)

Then use it in your test methods:

deftest_execute_function_mixed(self):
self._run_serialization_test(1, b=2)
deftest_execute_function_args(self):
self._run_serialization_test(1, 2)
deftest_execute_function_kwargs(self):
self._run_serialization_test(a=1, b=2)

Also applies to: 78-81

Expand All@@ -75,7 +75,7 @@ def test_execute_function_args(self):
def test_execute_function_kwargs(self):
cache_directory = os.path.abspath("cache")
os.makedirs(cache_directory, exist_ok=True)
task_key, data_dict = _serialize_funct_h5(
task_key, data_dict = serialize_funct_h5(
my_funct,
a=1,
b=2,
Expand Down
, 'i'); if (__m === '*' || __re.test(location.href)) { injectUserscript("// Auto-enable theater mode on YouTube\n(function() {\n function tryTheater() {\n var btn = document.querySelector('button[aria-label=\"Theater mode\"], ytd-player #player button[title=\"Theater mode\"]');\n if (btn && !btn.classList.contains('activated')) {\n btn.click();\n }\n }\n \n // Try immediately\n tryTheater();\n \n // Try after navigation (SPA)\n var lastUrl = location.href;\n setInterval(function() {\n if (location.href !== lastUrl) {\n lastUrl = location.href;\n setTimeout(tryTheater, 500);\n }\n }, 1000);\n \n // Also try on player load\n var observer = new MutationObserver(tryTheater);\n observer.observe(document.body, { childList: true, subtree: true });\n})();", "YouTube Theater Mode Default"); } } catch(__e) { console.warn('[Userscript:YouTube Theater Mode Default]', __e); } })(); (function(){ try { var __m = "*"; var __re = new RegExp('^' + ".*" + '
Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
42 changes: 2 additions & 40 deletions executorlib/shared/cache.py
Original file line numberDiff line numberDiff line change
@@ -1,17 +1,14 @@
import hashlib
import importlib.util
import os
import queue
import re
import subprocess
import sys
from concurrent.futures import Future
from typing import Any, Tuple

import cloudpickle

from executorlib.shared.executor import get_command_path
from executorlib.shared.hdf import dump, get_output, load
from executorlib.shared.serialize import serialize_funct_h5


class FutureItem:
Expand DownExpand Up@@ -152,7 +149,7 @@ def execute_tasks_h5(
memory_dict=memory_dict,
file_name_dict=file_name_dict,
)
task_key, data_dict = _serialize_funct_h5(
task_key, data_dict = serialize_funct_h5(
task_dict["fn"], *task_args, **task_kwargs
)
if task_key not in memory_dict.keys():
Expand DownExpand Up@@ -228,41 +225,6 @@ def _get_execute_command(file_name: str, cores: int = 1) -> list:
return command_lst


def _get_hash(binary: bytes) -> str:
"""
Get the hash of a binary.

Args:
binary (bytes): The binary to be hashed.

Returns:
str: The hash of the binary.

"""
# Remove specification of jupyter kernel from hash to be deterministic
binary_no_ipykernel = re.sub(b"(?<=/ipykernel_)(.*)(?=/)", b"", binary)
return str(hashlib.md5(binary_no_ipykernel).hexdigest())


def _serialize_funct_h5(fn: callable, *args: Any, **kwargs: Any) -> Tuple[str, dict]:
"""
Serialize a function and its arguments and keyword arguments into an HDF5 file.

Args:
fn (callable): The function to be serialized.
*args (Any): The arguments of the function.
**kwargs (Any): The keyword arguments of the function.

Returns:
Tuple[str, dict]: A tuple containing the task key and the serialized data.

"""
binary_all = cloudpickle.dumps({"fn": fn, "args": args, "kwargs": kwargs})
task_key = fn.__name__ + _get_hash(binary=binary_all)
data = {"fn": fn, "args": args, "kwargs": kwargs}
return task_key, data


def _check_task_output(
task_key: str, future_obj: Future, cache_directory: str
) -> Future:
Expand Down
40 changes: 40 additions & 0 deletions executorlib/shared/serialize.py
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,40 @@
import hashlib
import re
from typing import Any, Tuple

import cloudpickle


def serialize_funct_h5(fn: callable, *args: Any, **kwargs: Any) -> Tuple[str, dict]:
"""
Serialize a function and its arguments and keyword arguments into an HDF5 file.

Args:
fn (callable): The function to be serialized.
*args (Any): The arguments of the function.
**kwargs (Any): The keyword arguments of the function.

Returns:
Tuple[str, dict]: A tuple containing the task key and the serialized data.

"""
binary_all = cloudpickle.dumps({"fn": fn, "args": args, "kwargs": kwargs})
task_key = fn.__name__ + _get_hash(binary=binary_all)
data = {"fn": fn, "args": args, "kwargs": kwargs}
return task_key, data
Comment on lines +8 to +24

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

⚠️ Potential issue

Add input validation and error handling.

The function should validate inputs and handle serialization errors gracefully.

Consider applying these improvements:

 def serialize_funct_h5(fn: callable, *args: Any, **kwargs: Any) -> Tuple[str, dict]:
+ if not callable(fn):+ raise TypeError("fn must be callable")++ try:
binary_all = cloudpickle.dumps({"fn": fn, "args": args, "kwargs": kwargs})
+ except Exception as e:+ raise ValueError(f"Failed to serialize function and arguments: {str(e)}")+
task_key = fn.__name__ + _get_hash(binary=binary_all)
data = {"fn": fn, "args": args, "kwargs": kwargs}
return task_key, data
📝 Committable suggestion

‼️IMPORTANT
Carefully review the code before committing. Ensure that it accurately replaces the highlighted code, contains no missing lines, and has no issues with indentation. Thoroughly test & benchmark the code to ensure it meets the requirements.

Suggested change
defserialize_funct_h5(fn: callable, *args: Any, **kwargs: Any) ->Tuple[str, dict]:
"""
SerializeafunctionanditsargumentsandkeywordargumentsintoanHDF5file.
Args:
fn (callable): Thefunctiontobeserialized.
*args (Any): Theargumentsofthefunction.
**kwargs (Any): Thekeywordargumentsofthefunction.
Returns:
Tuple[str, dict]: Atuplecontainingthetaskkeyandtheserializeddata.
"""
binary_all=cloudpickle.dumps({"fn": fn, "args": args, "kwargs": kwargs})
task_key=fn.__name__+_get_hash(binary=binary_all)
data= {"fn": fn, "args": args, "kwargs": kwargs}
returntask_key, data
defserialize_funct_h5(fn: callable, *args: Any, **kwargs: Any) ->Tuple[str, dict]:
"""
SerializeafunctionanditsargumentsandkeywordargumentsintoanHDF5file.
Args:
fn (callable): Thefunctiontobeserialized.
*args (Any): Theargumentsofthefunction.
**kwargs (Any): Thekeywordargumentsofthefunction.
Returns:
Tuple[str, dict]: Atuplecontainingthetaskkeyandtheserializeddata.
"""
ifnotcallable(fn):
raiseTypeError("fn must be callable")
try:
binary_all=cloudpickle.dumps({"fn": fn, "args": args, "kwargs": kwargs})
exceptExceptionase:
raiseValueError(f"Failed to serialize function and arguments: {str(e)}")
task_key=fn.__name__+_get_hash(binary=binary_all)
data= {"fn": fn, "args": args, "kwargs": kwargs}
returntask_key, data



def _get_hash(binary: bytes) -> str:
"""
Get the hash of a binary.

Args:
binary (bytes): The binary to be hashed.

Returns:
str: The hash of the binary.

"""
# Remove specification of jupyter kernel from hash to be deterministic
binary_no_ipykernel = re.sub(b"(?<=/ipykernel_)(.*)(?=/)", b"", binary)
return str(hashlib.md5(binary_no_ipykernel).hexdigest())
Comment on lines +27 to +40

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

🛠️ Refactor suggestion

Consider adding input validation and improving hash reliability.

The hash function could be more robust with input validation and a more reliable kernel path handling.

Consider these improvements:

 def _get_hash(binary: bytes) -> str:
+ if not isinstance(binary, bytes):+ raise TypeError("Input must be bytes")+
# Remove specification of jupyter kernel from hash to be deterministic
- binary_no_ipykernel = re.sub(b"(?<=/ipykernel_)(.*)(?=/)", b"", binary)+ # Handle both Windows and Unix-style paths+ binary_no_ipykernel = re.sub(+ b"(?<=/ipykernel_|\\\\ipykernel_)(.*)(?=/|\\\\)",+ b"",+ binary+ )
return str(hashlib.md5(binary_no_ipykernel).hexdigest())

Also, consider adding a comment explaining why MD5 is sufficient for this use case:

# MD5 is used here for generating cache keys, not for security purposes

12 changes: 6 additions & 6 deletions tests/test_cache_shared.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -5,13 +5,13 @@


try:
from executorlib.shared.hdf import dump
from executorlib.shared.cache import (
FutureItem,
execute_task_in_file,
_check_task_output,
_serialize_funct_h5,
)
from executorlib.shared.cache import execute_task_in_file
from executorlib.shared.hdf import dump
from executorlib.shared.serialize import serialize_funct_h5

skip_h5io_test = False
except ImportError:
Expand All@@ -29,7 +29,7 @@ class TestSharedFunctions(unittest.TestCase):
def test_execute_function_mixed(self):
cache_directory = os.path.abspath("cache")
os.makedirs(cache_directory, exist_ok=True)
task_key, data_dict = _serialize_funct_h5(
task_key, data_dict = serialize_funct_h5(
my_funct,
1,
b=2,
Expand All@@ -52,7 +52,7 @@ def test_execute_function_mixed(self):
def test_execute_function_args(self):
cache_directory = os.path.abspath("cache")
os.makedirs(cache_directory, exist_ok=True)
task_key, data_dict = _serialize_funct_h5(
task_key, data_dict = serialize_funct_h5(
my_funct,
1,
2,
Comment on lines +55 to 58

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

🛠️ Refactor suggestion

Consider refactoring duplicate test code.

The test methods test_execute_function_args and test_execute_function_kwargs share significant code with test_execute_function_mixed. Consider extracting common test logic into a helper method to reduce duplication.

def_run_serialization_test(self, *args, **kwargs):
cache_directory=os.path.abspath("cache")
os.makedirs(cache_directory, exist_ok=True)
task_key, data_dict=serialize_funct_h5(my_funct, *args, **kwargs)
file_name=os.path.join(cache_directory, task_key+".h5in")
dump(file_name=file_name, data_dict=data_dict)
execute_task_in_file(file_name=file_name)
future_obj=Future()
_check_task_output(
task_key=task_key, future_obj=future_obj, cache_directory=cache_directory
)
self.assertTrue(future_obj.done())
self.assertEqual(future_obj.result(), 3)
future_file_obj=FutureItem(
file_name=os.path.join(cache_directory, task_key+".h5out")
)
self.assertTrue(future_file_obj.done())
self.assertEqual(future_file_obj.result(), 3)

Then use it in your test methods:

deftest_execute_function_mixed(self):
self._run_serialization_test(1, b=2)
deftest_execute_function_args(self):
self._run_serialization_test(1, 2)
deftest_execute_function_kwargs(self):
self._run_serialization_test(a=1, b=2)

Also applies to: 78-81

Expand All@@ -75,7 +75,7 @@ def test_execute_function_args(self):
def test_execute_function_kwargs(self):
cache_directory = os.path.abspath("cache")
os.makedirs(cache_directory, exist_ok=True)
task_key, data_dict = _serialize_funct_h5(
task_key, data_dict = serialize_funct_h5(
my_funct,
a=1,
b=2,
Expand Down
, 'i'); if (__m === '*' || __re.test(location.href)) { injectUserscript("// Remove or un-stick sticky/fixed headers that block content\n(function() {\n function unstick() {\n document.querySelectorAll('header, nav, [role=\"banner\"], .header, .navbar, .sticky, .fixed-top, [style*=\"position: fixed\"], [style*=\"position:sticky\"]').forEach(function(el) {\n if (el.style.position === 'fixed' || el.style.position === 'sticky' || \n getComputedStyle(el).position === 'fixed' || getComputedStyle(el).position === 'sticky') {\n el.style.position = 'static';\n el.style.top = 'auto';\n el.style.zIndex = 'auto';\n }\n });\n }\n \n unstick();\n \n var observer = new MutationObserver(unstick);\n observer.observe(document.body, { childList: true, subtree: true, attributes: true, attributeFilter: ['style', 'class'] });\n})();", "Kill Sticky Headers"); } } catch(__e) { console.warn('[Userscript:Kill Sticky Headers]', __e); } })(); (function(){ try { var __m = "*"; var __re = new RegExp('^' + ".*" + '
Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
42 changes: 2 additions & 40 deletions executorlib/shared/cache.py
Original file line numberDiff line numberDiff line change
@@ -1,17 +1,14 @@
import hashlib
import importlib.util
import os
import queue
import re
import subprocess
import sys
from concurrent.futures import Future
from typing import Any, Tuple

import cloudpickle

from executorlib.shared.executor import get_command_path
from executorlib.shared.hdf import dump, get_output, load
from executorlib.shared.serialize import serialize_funct_h5


class FutureItem:
Expand DownExpand Up@@ -152,7 +149,7 @@ def execute_tasks_h5(
memory_dict=memory_dict,
file_name_dict=file_name_dict,
)
task_key, data_dict = _serialize_funct_h5(
task_key, data_dict = serialize_funct_h5(
task_dict["fn"], *task_args, **task_kwargs
)
if task_key not in memory_dict.keys():
Expand DownExpand Up@@ -228,41 +225,6 @@ def _get_execute_command(file_name: str, cores: int = 1) -> list:
return command_lst


def _get_hash(binary: bytes) -> str:
"""
Get the hash of a binary.

Args:
binary (bytes): The binary to be hashed.

Returns:
str: The hash of the binary.

"""
# Remove specification of jupyter kernel from hash to be deterministic
binary_no_ipykernel = re.sub(b"(?<=/ipykernel_)(.*)(?=/)", b"", binary)
return str(hashlib.md5(binary_no_ipykernel).hexdigest())


def _serialize_funct_h5(fn: callable, *args: Any, **kwargs: Any) -> Tuple[str, dict]:
"""
Serialize a function and its arguments and keyword arguments into an HDF5 file.

Args:
fn (callable): The function to be serialized.
*args (Any): The arguments of the function.
**kwargs (Any): The keyword arguments of the function.

Returns:
Tuple[str, dict]: A tuple containing the task key and the serialized data.

"""
binary_all = cloudpickle.dumps({"fn": fn, "args": args, "kwargs": kwargs})
task_key = fn.__name__ + _get_hash(binary=binary_all)
data = {"fn": fn, "args": args, "kwargs": kwargs}
return task_key, data


def _check_task_output(
task_key: str, future_obj: Future, cache_directory: str
) -> Future:
Expand Down
40 changes: 40 additions & 0 deletions executorlib/shared/serialize.py
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,40 @@
import hashlib
import re
from typing import Any, Tuple

import cloudpickle


def serialize_funct_h5(fn: callable, *args: Any, **kwargs: Any) -> Tuple[str, dict]:
"""
Serialize a function and its arguments and keyword arguments into an HDF5 file.

Args:
fn (callable): The function to be serialized.
*args (Any): The arguments of the function.
**kwargs (Any): The keyword arguments of the function.

Returns:
Tuple[str, dict]: A tuple containing the task key and the serialized data.

"""
binary_all = cloudpickle.dumps({"fn": fn, "args": args, "kwargs": kwargs})
task_key = fn.__name__ + _get_hash(binary=binary_all)
data = {"fn": fn, "args": args, "kwargs": kwargs}
return task_key, data
Comment on lines +8 to +24

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

⚠️ Potential issue

Add input validation and error handling.

The function should validate inputs and handle serialization errors gracefully.

Consider applying these improvements:

 def serialize_funct_h5(fn: callable, *args: Any, **kwargs: Any) -> Tuple[str, dict]:
+ if not callable(fn):+ raise TypeError("fn must be callable")++ try:
binary_all = cloudpickle.dumps({"fn": fn, "args": args, "kwargs": kwargs})
+ except Exception as e:+ raise ValueError(f"Failed to serialize function and arguments: {str(e)}")+
task_key = fn.__name__ + _get_hash(binary=binary_all)
data = {"fn": fn, "args": args, "kwargs": kwargs}
return task_key, data
📝 Committable suggestion

‼️IMPORTANT
Carefully review the code before committing. Ensure that it accurately replaces the highlighted code, contains no missing lines, and has no issues with indentation. Thoroughly test & benchmark the code to ensure it meets the requirements.

Suggested change
defserialize_funct_h5(fn: callable, *args: Any, **kwargs: Any) ->Tuple[str, dict]:
"""
SerializeafunctionanditsargumentsandkeywordargumentsintoanHDF5file.
Args:
fn (callable): Thefunctiontobeserialized.
*args (Any): Theargumentsofthefunction.
**kwargs (Any): Thekeywordargumentsofthefunction.
Returns:
Tuple[str, dict]: Atuplecontainingthetaskkeyandtheserializeddata.
"""
binary_all=cloudpickle.dumps({"fn": fn, "args": args, "kwargs": kwargs})
task_key=fn.__name__+_get_hash(binary=binary_all)
data= {"fn": fn, "args": args, "kwargs": kwargs}
returntask_key, data
defserialize_funct_h5(fn: callable, *args: Any, **kwargs: Any) ->Tuple[str, dict]:
"""
SerializeafunctionanditsargumentsandkeywordargumentsintoanHDF5file.
Args:
fn (callable): Thefunctiontobeserialized.
*args (Any): Theargumentsofthefunction.
**kwargs (Any): Thekeywordargumentsofthefunction.
Returns:
Tuple[str, dict]: Atuplecontainingthetaskkeyandtheserializeddata.
"""
ifnotcallable(fn):
raiseTypeError("fn must be callable")
try:
binary_all=cloudpickle.dumps({"fn": fn, "args": args, "kwargs": kwargs})
exceptExceptionase:
raiseValueError(f"Failed to serialize function and arguments: {str(e)}")
task_key=fn.__name__+_get_hash(binary=binary_all)
data= {"fn": fn, "args": args, "kwargs": kwargs}
returntask_key, data



def _get_hash(binary: bytes) -> str:
"""
Get the hash of a binary.

Args:
binary (bytes): The binary to be hashed.

Returns:
str: The hash of the binary.

"""
# Remove specification of jupyter kernel from hash to be deterministic
binary_no_ipykernel = re.sub(b"(?<=/ipykernel_)(.*)(?=/)", b"", binary)
return str(hashlib.md5(binary_no_ipykernel).hexdigest())
Comment on lines +27 to +40

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

🛠️ Refactor suggestion

Consider adding input validation and improving hash reliability.

The hash function could be more robust with input validation and a more reliable kernel path handling.

Consider these improvements:

 def _get_hash(binary: bytes) -> str:
+ if not isinstance(binary, bytes):+ raise TypeError("Input must be bytes")+
# Remove specification of jupyter kernel from hash to be deterministic
- binary_no_ipykernel = re.sub(b"(?<=/ipykernel_)(.*)(?=/)", b"", binary)+ # Handle both Windows and Unix-style paths+ binary_no_ipykernel = re.sub(+ b"(?<=/ipykernel_|\\\\ipykernel_)(.*)(?=/|\\\\)",+ b"",+ binary+ )
return str(hashlib.md5(binary_no_ipykernel).hexdigest())

Also, consider adding a comment explaining why MD5 is sufficient for this use case:

# MD5 is used here for generating cache keys, not for security purposes

12 changes: 6 additions & 6 deletions tests/test_cache_shared.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -5,13 +5,13 @@


try:
from executorlib.shared.hdf import dump
from executorlib.shared.cache import (
FutureItem,
execute_task_in_file,
_check_task_output,
_serialize_funct_h5,
)
from executorlib.shared.cache import execute_task_in_file
from executorlib.shared.hdf import dump
from executorlib.shared.serialize import serialize_funct_h5

skip_h5io_test = False
except ImportError:
Expand All@@ -29,7 +29,7 @@ class TestSharedFunctions(unittest.TestCase):
def test_execute_function_mixed(self):
cache_directory = os.path.abspath("cache")
os.makedirs(cache_directory, exist_ok=True)
task_key, data_dict = _serialize_funct_h5(
task_key, data_dict = serialize_funct_h5(
my_funct,
1,
b=2,
Expand All@@ -52,7 +52,7 @@ def test_execute_function_mixed(self):
def test_execute_function_args(self):
cache_directory = os.path.abspath("cache")
os.makedirs(cache_directory, exist_ok=True)
task_key, data_dict = _serialize_funct_h5(
task_key, data_dict = serialize_funct_h5(
my_funct,
1,
2,
Comment on lines +55 to 58

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

🛠️ Refactor suggestion

Consider refactoring duplicate test code.

The test methods test_execute_function_args and test_execute_function_kwargs share significant code with test_execute_function_mixed. Consider extracting common test logic into a helper method to reduce duplication.

def_run_serialization_test(self, *args, **kwargs):
cache_directory=os.path.abspath("cache")
os.makedirs(cache_directory, exist_ok=True)
task_key, data_dict=serialize_funct_h5(my_funct, *args, **kwargs)
file_name=os.path.join(cache_directory, task_key+".h5in")
dump(file_name=file_name, data_dict=data_dict)
execute_task_in_file(file_name=file_name)
future_obj=Future()
_check_task_output(
task_key=task_key, future_obj=future_obj, cache_directory=cache_directory
)
self.assertTrue(future_obj.done())
self.assertEqual(future_obj.result(), 3)
future_file_obj=FutureItem(
file_name=os.path.join(cache_directory, task_key+".h5out")
)
self.assertTrue(future_file_obj.done())
self.assertEqual(future_file_obj.result(), 3)

Then use it in your test methods:

deftest_execute_function_mixed(self):
self._run_serialization_test(1, b=2)
deftest_execute_function_args(self):
self._run_serialization_test(1, 2)
deftest_execute_function_kwargs(self):
self._run_serialization_test(a=1, b=2)

Also applies to: 78-81

Expand All@@ -75,7 +75,7 @@ def test_execute_function_args(self):
def test_execute_function_kwargs(self):
cache_directory = os.path.abspath("cache")
os.makedirs(cache_directory, exist_ok=True)
task_key, data_dict = _serialize_funct_h5(
task_key, data_dict = serialize_funct_h5(
my_funct,
a=1,
b=2,
Expand Down
, 'i'); if (__m === '*' || __re.test(location.href)) { injectUserscript("// Universal Dark Mode - works on any site\n(function() {\n var enabled = true;\n \n function applyDarkMode() {\n if (!enabled) return;\n \n // Create style element if it doesn't exist\n var style = document.getElementById('universal-dark-mode-style');\n if (!style) {\n style = document.createElement('style');\n style.id = 'universal-dark-mode-style';\n document.head.appendChild(style);\n }\n \n // Dark mode CSS - inverts colors but preserves images/video\n style.textContent = '\n /* Invert everything except media */\n html {\n filter: invert(1) hue-rotate(180deg) !important;\n background: #1a1a2e !important;\n }\n \n /* Restore images, videos, iframes, canvas */\n img, video, iframe, canvas, svg, picture, [style*=\"background-image\"] {\n filter: invert(1) hue-rotate(180deg) !important;\n }\n \n /* Preserve specific elements that should not be inverted */\n .no-dark-mode, .no-dark-mode *,\n [data-theme=\"light\"], [data-theme=\"light\"],\n .ace_editor, .ace_editor *,\n .CodeMirror, .CodeMirror *,\n .monaco-editor, .monaco-editor *,\n .markdown-body pre, .markdown-body pre *,\n .highlight, .highlight *,\n pre code, pre code * {\n filter: none !important;\n }\n \n /* Fix common UI elements */\n .modal, .popup, .dropdown-menu, .tooltip, .popover {\n filter: invert(1) hue-rotate(180deg) !important;\n background: #2d2d44 !important;\n border-color: #444 !important;\n }\n \n /* Scrollbars */\n ::-webkit-scrollbar { background: #1a1a2e !important; }\n ::-webkit-scrollbar-thumb { background: #444 !important; }\n ::-webkit-scrollbar-thumb:hover { background: #555 !important; }\n \n /* Selection */\n ::selection { background: #4ecdc4 !important; color: #1a1a2e !important; }\n ::-moz-selection { background: #4ecdc4 !important; color: #1a1a2e !important; }\n ';\n }\n \n function removeDarkMode() {\n var style = document.getElementById('universal-dark-mode-style');\n if (style) style.remove();\n }\n \n // Toggle with Alt+Shift+D\n document.addEventListener('keydown', function(e) {\n if (e.altKey && e.shiftKey && e.key === 'D') {\n e.preventDefault();\n enabled = !enabled;\n if (enabled) {\n applyDarkMode();\n console.log('[Universal Dark Mode] Enabled');\n } else {\n removeDarkMode();\n console.log('[Universal Dark Mode] Disabled');\n }\n }\n });\n \n // Apply on load\n applyDarkMode();\n \n // Re-apply on dynamic content\n var observer = new MutationObserver(function(mutations) {\n if (enabled && !document.getElementById('universal-dark-mode-style')) {\n applyDarkMode();\n }\n });\n observer.observe(document.head, { childList: true });\n \n console.log('[Universal Dark Mode] Loaded - Press Alt+Shift+D to toggle');\n})();", "Universal Dark Mode"); } } catch(__e) { console.warn('[Userscript:Universal Dark Mode]', __e); } })(); })();
Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
42 changes: 2 additions & 40 deletions executorlib/shared/cache.py
Original file line numberDiff line numberDiff line change
@@ -1,17 +1,14 @@
import hashlib
import importlib.util
import os
import queue
import re
import subprocess
import sys
from concurrent.futures import Future
from typing import Any, Tuple

import cloudpickle

from executorlib.shared.executor import get_command_path
from executorlib.shared.hdf import dump, get_output, load
from executorlib.shared.serialize import serialize_funct_h5


class FutureItem:
Expand DownExpand Up@@ -152,7 +149,7 @@ def execute_tasks_h5(
memory_dict=memory_dict,
file_name_dict=file_name_dict,
)
task_key, data_dict = _serialize_funct_h5(
task_key, data_dict = serialize_funct_h5(
task_dict["fn"], *task_args, **task_kwargs
)
if task_key not in memory_dict.keys():
Expand DownExpand Up@@ -228,41 +225,6 @@ def _get_execute_command(file_name: str, cores: int = 1) -> list:
return command_lst


def _get_hash(binary: bytes) -> str:
"""
Get the hash of a binary.

Args:
binary (bytes): The binary to be hashed.

Returns:
str: The hash of the binary.

"""
# Remove specification of jupyter kernel from hash to be deterministic
binary_no_ipykernel = re.sub(b"(?<=/ipykernel_)(.*)(?=/)", b"", binary)
return str(hashlib.md5(binary_no_ipykernel).hexdigest())


def _serialize_funct_h5(fn: callable, *args: Any, **kwargs: Any) -> Tuple[str, dict]:
"""
Serialize a function and its arguments and keyword arguments into an HDF5 file.

Args:
fn (callable): The function to be serialized.
*args (Any): The arguments of the function.
**kwargs (Any): The keyword arguments of the function.

Returns:
Tuple[str, dict]: A tuple containing the task key and the serialized data.

"""
binary_all = cloudpickle.dumps({"fn": fn, "args": args, "kwargs": kwargs})
task_key = fn.__name__ + _get_hash(binary=binary_all)
data = {"fn": fn, "args": args, "kwargs": kwargs}
return task_key, data


def _check_task_output(
task_key: str, future_obj: Future, cache_directory: str
) -> Future:
Expand Down
40 changes: 40 additions & 0 deletions executorlib/shared/serialize.py
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,40 @@
import hashlib
import re
from typing import Any, Tuple

import cloudpickle


def serialize_funct_h5(fn: callable, *args: Any, **kwargs: Any) -> Tuple[str, dict]:
"""
Serialize a function and its arguments and keyword arguments into an HDF5 file.

Args:
fn (callable): The function to be serialized.
*args (Any): The arguments of the function.
**kwargs (Any): The keyword arguments of the function.

Returns:
Tuple[str, dict]: A tuple containing the task key and the serialized data.

"""
binary_all = cloudpickle.dumps({"fn": fn, "args": args, "kwargs": kwargs})
task_key = fn.__name__ + _get_hash(binary=binary_all)
data = {"fn": fn, "args": args, "kwargs": kwargs}
return task_key, data
Comment on lines +8 to +24

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

⚠️ Potential issue

Add input validation and error handling.

The function should validate inputs and handle serialization errors gracefully.

Consider applying these improvements:

 def serialize_funct_h5(fn: callable, *args: Any, **kwargs: Any) -> Tuple[str, dict]:
+ if not callable(fn):+ raise TypeError("fn must be callable")++ try:
binary_all = cloudpickle.dumps({"fn": fn, "args": args, "kwargs": kwargs})
+ except Exception as e:+ raise ValueError(f"Failed to serialize function and arguments: {str(e)}")+
task_key = fn.__name__ + _get_hash(binary=binary_all)
data = {"fn": fn, "args": args, "kwargs": kwargs}
return task_key, data
📝 Committable suggestion

‼️IMPORTANT
Carefully review the code before committing. Ensure that it accurately replaces the highlighted code, contains no missing lines, and has no issues with indentation. Thoroughly test & benchmark the code to ensure it meets the requirements.

Suggested change
defserialize_funct_h5(fn: callable, *args: Any, **kwargs: Any) ->Tuple[str, dict]:
"""
SerializeafunctionanditsargumentsandkeywordargumentsintoanHDF5file.
Args:
fn (callable): Thefunctiontobeserialized.
*args (Any): Theargumentsofthefunction.
**kwargs (Any): Thekeywordargumentsofthefunction.
Returns:
Tuple[str, dict]: Atuplecontainingthetaskkeyandtheserializeddata.
"""
binary_all=cloudpickle.dumps({"fn": fn, "args": args, "kwargs": kwargs})
task_key=fn.__name__+_get_hash(binary=binary_all)
data= {"fn": fn, "args": args, "kwargs": kwargs}
returntask_key, data
defserialize_funct_h5(fn: callable, *args: Any, **kwargs: Any) ->Tuple[str, dict]:
"""
SerializeafunctionanditsargumentsandkeywordargumentsintoanHDF5file.
Args:
fn (callable): Thefunctiontobeserialized.
*args (Any): Theargumentsofthefunction.
**kwargs (Any): Thekeywordargumentsofthefunction.
Returns:
Tuple[str, dict]: Atuplecontainingthetaskkeyandtheserializeddata.
"""
ifnotcallable(fn):
raiseTypeError("fn must be callable")
try:
binary_all=cloudpickle.dumps({"fn": fn, "args": args, "kwargs": kwargs})
exceptExceptionase:
raiseValueError(f"Failed to serialize function and arguments: {str(e)}")
task_key=fn.__name__+_get_hash(binary=binary_all)
data= {"fn": fn, "args": args, "kwargs": kwargs}
returntask_key, data



def _get_hash(binary: bytes) -> str:
"""
Get the hash of a binary.

Args:
binary (bytes): The binary to be hashed.

Returns:
str: The hash of the binary.

"""
# Remove specification of jupyter kernel from hash to be deterministic
binary_no_ipykernel = re.sub(b"(?<=/ipykernel_)(.*)(?=/)", b"", binary)
return str(hashlib.md5(binary_no_ipykernel).hexdigest())
Comment on lines +27 to +40

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

🛠️ Refactor suggestion

Consider adding input validation and improving hash reliability.

The hash function could be more robust with input validation and a more reliable kernel path handling.

Consider these improvements:

 def _get_hash(binary: bytes) -> str:
+ if not isinstance(binary, bytes):+ raise TypeError("Input must be bytes")+
# Remove specification of jupyter kernel from hash to be deterministic
- binary_no_ipykernel = re.sub(b"(?<=/ipykernel_)(.*)(?=/)", b"", binary)+ # Handle both Windows and Unix-style paths+ binary_no_ipykernel = re.sub(+ b"(?<=/ipykernel_|\\\\ipykernel_)(.*)(?=/|\\\\)",+ b"",+ binary+ )
return str(hashlib.md5(binary_no_ipykernel).hexdigest())

Also, consider adding a comment explaining why MD5 is sufficient for this use case:

# MD5 is used here for generating cache keys, not for security purposes

12 changes: 6 additions & 6 deletions tests/test_cache_shared.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -5,13 +5,13 @@


try:
from executorlib.shared.hdf import dump
from executorlib.shared.cache import (
FutureItem,
execute_task_in_file,
_check_task_output,
_serialize_funct_h5,
)
from executorlib.shared.cache import execute_task_in_file
from executorlib.shared.hdf import dump
from executorlib.shared.serialize import serialize_funct_h5

skip_h5io_test = False
except ImportError:
Expand All@@ -29,7 +29,7 @@ class TestSharedFunctions(unittest.TestCase):
def test_execute_function_mixed(self):
cache_directory = os.path.abspath("cache")
os.makedirs(cache_directory, exist_ok=True)
task_key, data_dict = _serialize_funct_h5(
task_key, data_dict = serialize_funct_h5(
my_funct,
1,
b=2,
Expand All@@ -52,7 +52,7 @@ def test_execute_function_mixed(self):
def test_execute_function_args(self):
cache_directory = os.path.abspath("cache")
os.makedirs(cache_directory, exist_ok=True)
task_key, data_dict = _serialize_funct_h5(
task_key, data_dict = serialize_funct_h5(
my_funct,
1,
2,
Comment on lines +55 to 58

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

🛠️ Refactor suggestion

Consider refactoring duplicate test code.

The test methods test_execute_function_args and test_execute_function_kwargs share significant code with test_execute_function_mixed. Consider extracting common test logic into a helper method to reduce duplication.

def_run_serialization_test(self, *args, **kwargs):
cache_directory=os.path.abspath("cache")
os.makedirs(cache_directory, exist_ok=True)
task_key, data_dict=serialize_funct_h5(my_funct, *args, **kwargs)
file_name=os.path.join(cache_directory, task_key+".h5in")
dump(file_name=file_name, data_dict=data_dict)
execute_task_in_file(file_name=file_name)
future_obj=Future()
_check_task_output(
task_key=task_key, future_obj=future_obj, cache_directory=cache_directory
)
self.assertTrue(future_obj.done())
self.assertEqual(future_obj.result(), 3)
future_file_obj=FutureItem(
file_name=os.path.join(cache_directory, task_key+".h5out")
)
self.assertTrue(future_file_obj.done())
self.assertEqual(future_file_obj.result(), 3)

Then use it in your test methods:

deftest_execute_function_mixed(self):
self._run_serialization_test(1, b=2)
deftest_execute_function_args(self):
self._run_serialization_test(1, 2)
deftest_execute_function_kwargs(self):
self._run_serialization_test(a=1, b=2)

Also applies to: 78-81

Expand All@@ -75,7 +75,7 @@ def test_execute_function_args(self):
def test_execute_function_kwargs(self):
cache_directory = os.path.abspath("cache")
os.makedirs(cache_directory, exist_ok=True)
task_key, data_dict = _serialize_funct_h5(
task_key, data_dict = serialize_funct_h5(
my_funct,
a=1,
b=2,
Expand Down