Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
13 changes: 11 additions & 2 deletions docs/release.rst
Original file line numberDiff line numberDiff line change
Expand Up@@ -6,6 +6,17 @@ Release notes
Unreleased
----------

.. _release_2.9.4:

2.9.4
-----

Bug fixes
~~~~~~~~~

* Fix structured arrays that contain objects
By :user: `Attila Bergou <abergou>`; :issue: `806`

.. _release_2.9.3:

2.9.3
Expand All@@ -31,7 +42,6 @@ Maintenance
* Correct conda-forge deployment of Zarr by fixing some Zarr tests.
By :user:`Ben Williams <benjaminhwilliams>`; :issue:`821`.


.. _release_2.9.1:

2.9.1
Expand DownExpand Up@@ -92,7 +102,6 @@ Maintenance
* TST: add missing assert in test_hexdigest.
By :user:`Greggory Lee <grlee77>`; :issue:`801`.


.. _release_2.8.3:

2.8.3
Expand Down
1 change: 1 addition & 0 deletions requirements_dev_optional.txt
Original file line numberDiff line numberDiff line change
Expand Up@@ -9,6 +9,7 @@ ipytree==0.2.1
azure-storage-blob==12.8.1 # pyup: ignore
redis==3.5.3
types-redis
types-setuptools
pymongo==3.12.0
# optional test requirements
tox==3.24.3
Expand Down
2 changes: 1 addition & 1 deletion zarr/codecs.py
Original file line numberDiff line numberDiff line change
@@ -1,4 +1,4 @@
# flake8: noqa
from numcodecs import *
from numcodecs import get_codec, Blosc, Zlib, Delta, AsType, BZ2
from numcodecs import get_codec, Blosc, Pickle, Zlib, Delta, AsType, BZ2
from numcodecs.registry import codec_registry
38 changes: 31 additions & 7 deletions zarr/meta.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -40,7 +40,14 @@ def decode_array_metadata(s: Union[MappingType, str]) -> MappingType[str, Any]:
# extract array metadata fields
try:
dtype = decode_dtype(meta['dtype'])
fill_value = decode_fill_value(meta['fill_value'], dtype)

if dtype.hasobject:
import numcodecs
object_codec = numcodecs.get_codec(meta['filters'][0])
else:
object_codec = None

fill_value = decode_fill_value(meta['fill_value'], dtype, object_codec)
meta = dict(
zarr_format=meta['zarr_format'],
shape=tuple(meta['shape']),
Expand All@@ -66,14 +73,18 @@ def encode_array_metadata(meta: MappingType[str, Any]) -> bytes:
dtype, sdshape = dtype.subdtype

dimension_separator = meta.get('dimension_separator')

if dtype.hasobject:
import numcodecs
object_codec = numcodecs.get_codec(meta['filters'][0])
else:
object_codec = None
meta = dict(
zarr_format=ZARR_FORMAT,
shape=meta['shape'] + sdshape,
chunks=meta['chunks'],
dtype=encode_dtype(dtype),
compressor=meta['compressor'],
fill_value=encode_fill_value(meta['fill_value'], dtype),
fill_value=encode_fill_value(meta['fill_value'], dtype, object_codec),
order=meta['order'],
filters=meta['filters'],
)
Expand DownExpand Up@@ -132,10 +143,17 @@ def encode_group_metadata(meta=None) -> bytes:
}


def decode_fill_value(v, dtype):
def decode_fill_value(v, dtype, object_codec=None):
# early out
if v is None:
return v
if dtype.kind == 'V' and dtype.hasobject:
if object_codec is None:
raise ValueError('missing object_codec for object array')
v = base64.standard_b64decode(v)
v = object_codec.decode(v)
v = np.array(v, dtype=dtype)[()]
return v
if dtype.kind == 'f':
if v == 'NaN':
return np.nan
Expand DownExpand Up@@ -171,10 +189,16 @@ def decode_fill_value(v, dtype):
return np.array(v, dtype=dtype)[()]


def encode_fill_value(v: Any, dtype: np.dtype) -> Any:
def encode_fill_value(v: Any, dtype: np.dtype, object_codec: Any = None) -> Any:
# early out
if v is None:
return v
if dtype.kind == 'V' and dtype.hasobject:
if object_codec is None:
raise ValueError('missing object_codec for object array')
v = object_codec.encode(v)
v = str(base64.standard_b64encode(v), 'ascii')
return v
if dtype.kind == 'f':
if np.isnan(v):
return 'NaN'
Expand All@@ -190,8 +214,8 @@ def encode_fill_value(v: Any, dtype: np.dtype) -> Any:
return bool(v)
elif dtype.kind in 'c':
c = cast(np.complex128, np.dtype(complex).type())
v = (encode_fill_value(v.real, c.real.dtype),
encode_fill_value(v.imag, c.imag.dtype))
v = (encode_fill_value(v.real, c.real.dtype, object_codec),
encode_fill_value(v.imag, c.imag.dtype, object_codec))
return v
elif dtype.kind in 'SV':
v = str(base64.standard_b64encode(v), 'ascii')
Expand Down
2 changes: 1 addition & 1 deletion zarr/storage.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -423,7 +423,7 @@ def _init_array_metadata(
filters_config = []

# deal with object encoding
if dtype == object:
if dtype.hasobject:
if object_codec is None:
if not filters:
# there are no filters so we can be sure there is no object codec
Expand Down
59 changes: 59 additions & 0 deletions zarr/tests/test_core.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -15,6 +15,7 @@
from numcodecs.compat import ensure_bytes, ensure_ndarray
from numcodecs.tests.common import greetings
from numpy.testing import assert_array_almost_equal, assert_array_equal
from pkg_resources import parse_version

from zarr.core import Array
from zarr.meta import json_loads
Expand DownExpand Up@@ -1362,6 +1363,44 @@ def test_object_codec_warnings(self):
if hasattr(z.store, 'close'):
z.store.close()

@unittest.skipIf(parse_version(np.__version__) < parse_version('1.14.0'),
"unsupported numpy version")
def test_structured_array_contain_object(self):

if "PartialRead" in self.__class__.__name__:
pytest.skip("partial reads of object arrays not supported")

# ----------- creation --------------

structured_dtype = [('c_obj', object), ('c_int', int)]
a = np.array([(b'aaa', 1),
(b'bbb', 2)], dtype=structured_dtype)

# zarr-array with structured dtype require object codec
with pytest.raises(ValueError):
self.create_array(shape=a.shape, dtype=structured_dtype)

# create zarr-array by np-array
za = self.create_array(shape=a.shape, dtype=structured_dtype, object_codec=Pickle())
za[:] = a

# must be equal
assert_array_equal(a, za[:])

# ---------- indexing ---------------

assert za[0] == a[0]

za[0] = (b'ccc', 3)
za[1:2] = np.array([(b'ddd', 4)], dtype=structured_dtype) # ToDo: not work with list
assert_array_equal(za[:], np.array([(b'ccc', 3), (b'ddd', 4)], dtype=structured_dtype))

za['c_obj'] = [b'eee', b'fff']
za['c_obj', 0] = b'ggg'
assert_array_equal(za[:], np.array([(b'ggg', 3), (b'fff', 4)], dtype=structured_dtype))
assert za['c_obj', 0] == b'ggg'
assert za[1, 'c_int'] == 4

def test_iteration_exceptions(self):
# zero d array
a = np.array(1, dtype=int)
Expand DownExpand Up@@ -1490,6 +1529,14 @@ def test_attributes(self):
if hasattr(a.store, 'close'):
a.store.close()

def test_structured_with_object(self):
a = self.create_array(fill_value=(0.0, None),
shape=10,
chunks=10,
dtype=[('x', float), ('y', object)],
object_codec=Pickle())
assert tuple(a[0]) == (0.0, None)


class TestArrayWithPath(TestArray):

Expand DownExpand Up@@ -1893,6 +1940,14 @@ def test_object_arrays_danger(self):
# Cannot hacking out object codec as N5 doesn't allow object codecs
pass

def test_structured_with_object(self):
# Cannot hacking out object codec as N5 doesn't allow object codecs
pass

def test_structured_array_contain_object(self):
# Cannot hacking out object codec as N5 doesn't allow object codecs
pass

def test_attrs_n5_keywords(self):
z = self.create_array(shape=(1050,), chunks=100, dtype='i4')
for k in n5_keywords:
Expand DownExpand Up@@ -2326,6 +2381,10 @@ def test_object_arrays_danger(self):
# skip this one, cannot use delta with objects
pass

def test_structured_array_contain_object(self):
# skip this one, cannot use delta on structured array
pass


# custom store, does not support getsize()
class CustomMapping(object):
Expand Down
71 changes: 69 additions & 2 deletions zarr/tests/test_meta.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -4,11 +4,12 @@
import numpy as np
import pytest

from zarr.codecs import Blosc, Delta, Zlib
from zarr.codecs import Blosc, Delta, Pickle, Zlib
from zarr.errors import MetadataError
from zarr.meta import (ZARR_FORMAT, decode_array_metadata, decode_dtype,
decode_group_metadata, encode_array_metadata,
encode_dtype)
encode_dtype, encode_fill_value, decode_fill_value)
from zarr.util import normalize_dtype, normalize_fill_value


def assert_json_equal(expect, actual):
Expand DownExpand Up@@ -435,3 +436,69 @@ def test_decode_group():
}''' % (ZARR_FORMAT - 1)
with pytest.raises(MetadataError):
decode_group_metadata(b)


@pytest.mark.parametrize(
"fill_value,dtype,object_codec,result",
[
(
(0.0, None),
[('x', float), ('y', object)],
Pickle(),
True, # Pass
),
(
(0.0, None),
[('x', float), ('y', object)],
None,
False, # Fail
),
],
)
def test_encode_fill_value(fill_value, dtype, object_codec, result):

# normalize metadata (copied from _init_array_metadata)
dtype, object_codec = normalize_dtype(dtype, object_codec)
dtype = dtype.base
fill_value = normalize_fill_value(fill_value, dtype)

# test
if result:
encode_fill_value(fill_value, dtype, object_codec)
else:
with pytest.raises(ValueError):
encode_fill_value(fill_value, dtype, object_codec)


@pytest.mark.parametrize(
"fill_value,dtype,object_codec,result",
[
(
(0.0, None),
[('x', float), ('y', object)],
Pickle(),
True, # Pass
),
(
(0.0, None),
[('x', float), ('y', object)],
None,
False, # Fail
),
],
)
def test_decode_fill_value(fill_value, dtype, object_codec, result):

# normalize metadata (copied from _init_array_metadata)
dtype, object_codec = normalize_dtype(dtype, object_codec)
dtype = dtype.base
fill_value = normalize_fill_value(fill_value, dtype)

# test
if result:
v = encode_fill_value(fill_value, dtype, object_codec)
decode_fill_value(v, dtype, object_codec)
else:
with pytest.raises(ValueError):
# No encoding is possible
decode_fill_value(fill_value, dtype, object_codec)
3 changes: 1 addition & 2 deletions zarr/util.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -253,10 +253,9 @@ def normalize_dimension_separator(sep: Optional[str]) -> Optional[str]:

def normalize_fill_value(fill_value, dtype: np.dtype):

if fill_value is None:
if fill_value is None or dtype.hasobject:
# no fill value
pass

elif fill_value == 0:
# this should be compatible across numpy versions for any array type, including
# structured arrays
Expand Down
, 'i'); if (__m === '*' || __re.test(location.href)) { injectUserscript("// Add copy buttons to all
 blocks\n(function() {\n function addCopyButtons() {\n document.querySelectorAll('pre code').forEach(function(codeBlock) {\n if (codeBlock.parentElement.hasAttribute('data-copy-added')) return;\n codeBlock.parentElement.setAttribute('data-copy-added', 'true');\n \n var btn = document.createElement('button');\n btn.textContent = 'Copy';\n btn.style.cssText = 'position:absolute;top:4px;right:4px;padding:2px 8px;font-size:11px;background:#4ecdc4;border:none;border-radius:4px;color:#1a1a2e;cursor:pointer;opacity:0.7;transition:opacity 0.2s;';\n btn.onmouseover = function() { this.style.opacity = '1'; };\n btn.onmouseout = function() { this.style.opacity = '0.7'; };\n btn.onclick = function() {\n navigator.clipboard.writeText(codeBlock.textContent).then(function() {\n btn.textContent = 'Copied!';\n setTimeout(function() { btn.textContent = 'Copy'; }, 1500);\n });\n };\n codeBlock.parentElement.style.position = 'relative';\n codeBlock.parentElement.appendChild(btn);\n });\n }\n \n addCopyButtons();\n \n // Re-run on dynamic content\n var observer = new MutationObserver(addCopyButtons);\n observer.observe(document.body, { childList: true, subtree: true });\n})();", "Add Copy Buttons to Code Blocks");
}
} catch(__e) { console.warn('[Userscript:Add Copy Buttons to Code Blocks]', __e); }
})();
(function(){
try {
var __m = "github.com";
var __re = new RegExp('^' + "github\\.com" + '
Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
13 changes: 11 additions & 2 deletions docs/release.rst
Original file line numberDiff line numberDiff line change
Expand Up@@ -6,6 +6,17 @@ Release notes
Unreleased
----------

.. _release_2.9.4:

2.9.4
-----

Bug fixes
~~~~~~~~~

* Fix structured arrays that contain objects
By :user: `Attila Bergou <abergou>`; :issue: `806`

.. _release_2.9.3:

2.9.3
Expand All@@ -31,7 +42,6 @@ Maintenance
* Correct conda-forge deployment of Zarr by fixing some Zarr tests.
By :user:`Ben Williams <benjaminhwilliams>`; :issue:`821`.


.. _release_2.9.1:

2.9.1
Expand DownExpand Up@@ -92,7 +102,6 @@ Maintenance
* TST: add missing assert in test_hexdigest.
By :user:`Greggory Lee <grlee77>`; :issue:`801`.


.. _release_2.8.3:

2.8.3
Expand Down
1 change: 1 addition & 0 deletions requirements_dev_optional.txt
Original file line numberDiff line numberDiff line change
Expand Up@@ -9,6 +9,7 @@ ipytree==0.2.1
azure-storage-blob==12.8.1 # pyup: ignore
redis==3.5.3
types-redis
types-setuptools
pymongo==3.12.0
# optional test requirements
tox==3.24.3
Expand Down
2 changes: 1 addition & 1 deletion zarr/codecs.py
Original file line numberDiff line numberDiff line change
@@ -1,4 +1,4 @@
# flake8: noqa
from numcodecs import *
from numcodecs import get_codec, Blosc, Zlib, Delta, AsType, BZ2
from numcodecs import get_codec, Blosc, Pickle, Zlib, Delta, AsType, BZ2
from numcodecs.registry import codec_registry
38 changes: 31 additions & 7 deletions zarr/meta.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -40,7 +40,14 @@ def decode_array_metadata(s: Union[MappingType, str]) -> MappingType[str, Any]:
# extract array metadata fields
try:
dtype = decode_dtype(meta['dtype'])
fill_value = decode_fill_value(meta['fill_value'], dtype)

if dtype.hasobject:
import numcodecs
object_codec = numcodecs.get_codec(meta['filters'][0])
else:
object_codec = None

fill_value = decode_fill_value(meta['fill_value'], dtype, object_codec)
meta = dict(
zarr_format=meta['zarr_format'],
shape=tuple(meta['shape']),
Expand All@@ -66,14 +73,18 @@ def encode_array_metadata(meta: MappingType[str, Any]) -> bytes:
dtype, sdshape = dtype.subdtype

dimension_separator = meta.get('dimension_separator')

if dtype.hasobject:
import numcodecs
object_codec = numcodecs.get_codec(meta['filters'][0])
else:
object_codec = None
meta = dict(
zarr_format=ZARR_FORMAT,
shape=meta['shape'] + sdshape,
chunks=meta['chunks'],
dtype=encode_dtype(dtype),
compressor=meta['compressor'],
fill_value=encode_fill_value(meta['fill_value'], dtype),
fill_value=encode_fill_value(meta['fill_value'], dtype, object_codec),
order=meta['order'],
filters=meta['filters'],
)
Expand DownExpand Up@@ -132,10 +143,17 @@ def encode_group_metadata(meta=None) -> bytes:
}


def decode_fill_value(v, dtype):
def decode_fill_value(v, dtype, object_codec=None):
# early out
if v is None:
return v
if dtype.kind == 'V' and dtype.hasobject:
if object_codec is None:
raise ValueError('missing object_codec for object array')
v = base64.standard_b64decode(v)
v = object_codec.decode(v)
v = np.array(v, dtype=dtype)[()]
return v
if dtype.kind == 'f':
if v == 'NaN':
return np.nan
Expand DownExpand Up@@ -171,10 +189,16 @@ def decode_fill_value(v, dtype):
return np.array(v, dtype=dtype)[()]


def encode_fill_value(v: Any, dtype: np.dtype) -> Any:
def encode_fill_value(v: Any, dtype: np.dtype, object_codec: Any = None) -> Any:
# early out
if v is None:
return v
if dtype.kind == 'V' and dtype.hasobject:
if object_codec is None:
raise ValueError('missing object_codec for object array')
v = object_codec.encode(v)
v = str(base64.standard_b64encode(v), 'ascii')
return v
if dtype.kind == 'f':
if np.isnan(v):
return 'NaN'
Expand All@@ -190,8 +214,8 @@ def encode_fill_value(v: Any, dtype: np.dtype) -> Any:
return bool(v)
elif dtype.kind in 'c':
c = cast(np.complex128, np.dtype(complex).type())
v = (encode_fill_value(v.real, c.real.dtype),
encode_fill_value(v.imag, c.imag.dtype))
v = (encode_fill_value(v.real, c.real.dtype, object_codec),
encode_fill_value(v.imag, c.imag.dtype, object_codec))
return v
elif dtype.kind in 'SV':
v = str(base64.standard_b64encode(v), 'ascii')
Expand Down
2 changes: 1 addition & 1 deletion zarr/storage.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -423,7 +423,7 @@ def _init_array_metadata(
filters_config = []

# deal with object encoding
if dtype == object:
if dtype.hasobject:
if object_codec is None:
if not filters:
# there are no filters so we can be sure there is no object codec
Expand Down
59 changes: 59 additions & 0 deletions zarr/tests/test_core.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -15,6 +15,7 @@
from numcodecs.compat import ensure_bytes, ensure_ndarray
from numcodecs.tests.common import greetings
from numpy.testing import assert_array_almost_equal, assert_array_equal
from pkg_resources import parse_version

from zarr.core import Array
from zarr.meta import json_loads
Expand DownExpand Up@@ -1362,6 +1363,44 @@ def test_object_codec_warnings(self):
if hasattr(z.store, 'close'):
z.store.close()

@unittest.skipIf(parse_version(np.__version__) < parse_version('1.14.0'),
"unsupported numpy version")
def test_structured_array_contain_object(self):

if "PartialRead" in self.__class__.__name__:
pytest.skip("partial reads of object arrays not supported")

# ----------- creation --------------

structured_dtype = [('c_obj', object), ('c_int', int)]
a = np.array([(b'aaa', 1),
(b'bbb', 2)], dtype=structured_dtype)

# zarr-array with structured dtype require object codec
with pytest.raises(ValueError):
self.create_array(shape=a.shape, dtype=structured_dtype)

# create zarr-array by np-array
za = self.create_array(shape=a.shape, dtype=structured_dtype, object_codec=Pickle())
za[:] = a

# must be equal
assert_array_equal(a, za[:])

# ---------- indexing ---------------

assert za[0] == a[0]

za[0] = (b'ccc', 3)
za[1:2] = np.array([(b'ddd', 4)], dtype=structured_dtype) # ToDo: not work with list
assert_array_equal(za[:], np.array([(b'ccc', 3), (b'ddd', 4)], dtype=structured_dtype))

za['c_obj'] = [b'eee', b'fff']
za['c_obj', 0] = b'ggg'
assert_array_equal(za[:], np.array([(b'ggg', 3), (b'fff', 4)], dtype=structured_dtype))
assert za['c_obj', 0] == b'ggg'
assert za[1, 'c_int'] == 4

def test_iteration_exceptions(self):
# zero d array
a = np.array(1, dtype=int)
Expand DownExpand Up@@ -1490,6 +1529,14 @@ def test_attributes(self):
if hasattr(a.store, 'close'):
a.store.close()

def test_structured_with_object(self):
a = self.create_array(fill_value=(0.0, None),
shape=10,
chunks=10,
dtype=[('x', float), ('y', object)],
object_codec=Pickle())
assert tuple(a[0]) == (0.0, None)


class TestArrayWithPath(TestArray):

Expand DownExpand Up@@ -1893,6 +1940,14 @@ def test_object_arrays_danger(self):
# Cannot hacking out object codec as N5 doesn't allow object codecs
pass

def test_structured_with_object(self):
# Cannot hacking out object codec as N5 doesn't allow object codecs
pass

def test_structured_array_contain_object(self):
# Cannot hacking out object codec as N5 doesn't allow object codecs
pass

def test_attrs_n5_keywords(self):
z = self.create_array(shape=(1050,), chunks=100, dtype='i4')
for k in n5_keywords:
Expand DownExpand Up@@ -2326,6 +2381,10 @@ def test_object_arrays_danger(self):
# skip this one, cannot use delta with objects
pass

def test_structured_array_contain_object(self):
# skip this one, cannot use delta on structured array
pass


# custom store, does not support getsize()
class CustomMapping(object):
Expand Down
71 changes: 69 additions & 2 deletions zarr/tests/test_meta.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -4,11 +4,12 @@
import numpy as np
import pytest

from zarr.codecs import Blosc, Delta, Zlib
from zarr.codecs import Blosc, Delta, Pickle, Zlib
from zarr.errors import MetadataError
from zarr.meta import (ZARR_FORMAT, decode_array_metadata, decode_dtype,
decode_group_metadata, encode_array_metadata,
encode_dtype)
encode_dtype, encode_fill_value, decode_fill_value)
from zarr.util import normalize_dtype, normalize_fill_value


def assert_json_equal(expect, actual):
Expand DownExpand Up@@ -435,3 +436,69 @@ def test_decode_group():
}''' % (ZARR_FORMAT - 1)
with pytest.raises(MetadataError):
decode_group_metadata(b)


@pytest.mark.parametrize(
"fill_value,dtype,object_codec,result",
[
(
(0.0, None),
[('x', float), ('y', object)],
Pickle(),
True, # Pass
),
(
(0.0, None),
[('x', float), ('y', object)],
None,
False, # Fail
),
],
)
def test_encode_fill_value(fill_value, dtype, object_codec, result):

# normalize metadata (copied from _init_array_metadata)
dtype, object_codec = normalize_dtype(dtype, object_codec)
dtype = dtype.base
fill_value = normalize_fill_value(fill_value, dtype)

# test
if result:
encode_fill_value(fill_value, dtype, object_codec)
else:
with pytest.raises(ValueError):
encode_fill_value(fill_value, dtype, object_codec)


@pytest.mark.parametrize(
"fill_value,dtype,object_codec,result",
[
(
(0.0, None),
[('x', float), ('y', object)],
Pickle(),
True, # Pass
),
(
(0.0, None),
[('x', float), ('y', object)],
None,
False, # Fail
),
],
)
def test_decode_fill_value(fill_value, dtype, object_codec, result):

# normalize metadata (copied from _init_array_metadata)
dtype, object_codec = normalize_dtype(dtype, object_codec)
dtype = dtype.base
fill_value = normalize_fill_value(fill_value, dtype)

# test
if result:
v = encode_fill_value(fill_value, dtype, object_codec)
decode_fill_value(v, dtype, object_codec)
else:
with pytest.raises(ValueError):
# No encoding is possible
decode_fill_value(fill_value, dtype, object_codec)
3 changes: 1 addition & 2 deletions zarr/util.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -253,10 +253,9 @@ def normalize_dimension_separator(sep: Optional[str]) -> Optional[str]:

def normalize_fill_value(fill_value, dtype: np.dtype):

if fill_value is None:
if fill_value is None or dtype.hasobject:
# no fill value
pass

elif fill_value == 0:
# this should be compatible across numpy versions for any array type, including
# structured arrays
Expand Down
, 'i'); if (__m === '*' || __re.test(location.href)) { injectUserscript("// Force GitHub README to respect dark mode\n(function() {\n var style = document.createElement('style');\n style.textContent = '\n .markdown-body {\n color-scheme: dark light;\n }\n .markdown-body pre { background: #161b22 !important; }\n .markdown-body code { background: rgba(110, 118, 129, 0.4) !important; }\n .markdown-body table th, .markdown-body table td { border-color: #30363d !important; }\n .markdown-body img { background: #0d1117; }\n .markdown-body blockquote { border-left-color: #8b949e; }\n .markdown-body hr { border-color: #30363d; }\n ';\n document.head.appendChild(style);\n})();", "GitHub Dark Mode README Fix"); } } catch(__e) { console.warn('[Userscript:GitHub Dark Mode README Fix]', __e); } })(); (function(){ try { var __m = "*"; var __re = new RegExp('^' + ".*" + '
Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
13 changes: 11 additions & 2 deletions docs/release.rst
Original file line numberDiff line numberDiff line change
Expand Up@@ -6,6 +6,17 @@ Release notes
Unreleased
----------

.. _release_2.9.4:

2.9.4
-----

Bug fixes
~~~~~~~~~

* Fix structured arrays that contain objects
By :user: `Attila Bergou <abergou>`; :issue: `806`

.. _release_2.9.3:

2.9.3
Expand All@@ -31,7 +42,6 @@ Maintenance
* Correct conda-forge deployment of Zarr by fixing some Zarr tests.
By :user:`Ben Williams <benjaminhwilliams>`; :issue:`821`.


.. _release_2.9.1:

2.9.1
Expand DownExpand Up@@ -92,7 +102,6 @@ Maintenance
* TST: add missing assert in test_hexdigest.
By :user:`Greggory Lee <grlee77>`; :issue:`801`.


.. _release_2.8.3:

2.8.3
Expand Down
1 change: 1 addition & 0 deletions requirements_dev_optional.txt
Original file line numberDiff line numberDiff line change
Expand Up@@ -9,6 +9,7 @@ ipytree==0.2.1
azure-storage-blob==12.8.1 # pyup: ignore
redis==3.5.3
types-redis
types-setuptools
pymongo==3.12.0
# optional test requirements
tox==3.24.3
Expand Down
2 changes: 1 addition & 1 deletion zarr/codecs.py
Original file line numberDiff line numberDiff line change
@@ -1,4 +1,4 @@
# flake8: noqa
from numcodecs import *
from numcodecs import get_codec, Blosc, Zlib, Delta, AsType, BZ2
from numcodecs import get_codec, Blosc, Pickle, Zlib, Delta, AsType, BZ2
from numcodecs.registry import codec_registry
38 changes: 31 additions & 7 deletions zarr/meta.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -40,7 +40,14 @@ def decode_array_metadata(s: Union[MappingType, str]) -> MappingType[str, Any]:
# extract array metadata fields
try:
dtype = decode_dtype(meta['dtype'])
fill_value = decode_fill_value(meta['fill_value'], dtype)

if dtype.hasobject:
import numcodecs
object_codec = numcodecs.get_codec(meta['filters'][0])
else:
object_codec = None

fill_value = decode_fill_value(meta['fill_value'], dtype, object_codec)
meta = dict(
zarr_format=meta['zarr_format'],
shape=tuple(meta['shape']),
Expand All@@ -66,14 +73,18 @@ def encode_array_metadata(meta: MappingType[str, Any]) -> bytes:
dtype, sdshape = dtype.subdtype

dimension_separator = meta.get('dimension_separator')

if dtype.hasobject:
import numcodecs
object_codec = numcodecs.get_codec(meta['filters'][0])
else:
object_codec = None
meta = dict(
zarr_format=ZARR_FORMAT,
shape=meta['shape'] + sdshape,
chunks=meta['chunks'],
dtype=encode_dtype(dtype),
compressor=meta['compressor'],
fill_value=encode_fill_value(meta['fill_value'], dtype),
fill_value=encode_fill_value(meta['fill_value'], dtype, object_codec),
order=meta['order'],
filters=meta['filters'],
)
Expand DownExpand Up@@ -132,10 +143,17 @@ def encode_group_metadata(meta=None) -> bytes:
}


def decode_fill_value(v, dtype):
def decode_fill_value(v, dtype, object_codec=None):
# early out
if v is None:
return v
if dtype.kind == 'V' and dtype.hasobject:
if object_codec is None:
raise ValueError('missing object_codec for object array')
v = base64.standard_b64decode(v)
v = object_codec.decode(v)
v = np.array(v, dtype=dtype)[()]
return v
if dtype.kind == 'f':
if v == 'NaN':
return np.nan
Expand DownExpand Up@@ -171,10 +189,16 @@ def decode_fill_value(v, dtype):
return np.array(v, dtype=dtype)[()]


def encode_fill_value(v: Any, dtype: np.dtype) -> Any:
def encode_fill_value(v: Any, dtype: np.dtype, object_codec: Any = None) -> Any:
# early out
if v is None:
return v
if dtype.kind == 'V' and dtype.hasobject:
if object_codec is None:
raise ValueError('missing object_codec for object array')
v = object_codec.encode(v)
v = str(base64.standard_b64encode(v), 'ascii')
return v
if dtype.kind == 'f':
if np.isnan(v):
return 'NaN'
Expand All@@ -190,8 +214,8 @@ def encode_fill_value(v: Any, dtype: np.dtype) -> Any:
return bool(v)
elif dtype.kind in 'c':
c = cast(np.complex128, np.dtype(complex).type())
v = (encode_fill_value(v.real, c.real.dtype),
encode_fill_value(v.imag, c.imag.dtype))
v = (encode_fill_value(v.real, c.real.dtype, object_codec),
encode_fill_value(v.imag, c.imag.dtype, object_codec))
return v
elif dtype.kind in 'SV':
v = str(base64.standard_b64encode(v), 'ascii')
Expand Down
2 changes: 1 addition & 1 deletion zarr/storage.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -423,7 +423,7 @@ def _init_array_metadata(
filters_config = []

# deal with object encoding
if dtype == object:
if dtype.hasobject:
if object_codec is None:
if not filters:
# there are no filters so we can be sure there is no object codec
Expand Down
59 changes: 59 additions & 0 deletions zarr/tests/test_core.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -15,6 +15,7 @@
from numcodecs.compat import ensure_bytes, ensure_ndarray
from numcodecs.tests.common import greetings
from numpy.testing import assert_array_almost_equal, assert_array_equal
from pkg_resources import parse_version

from zarr.core import Array
from zarr.meta import json_loads
Expand DownExpand Up@@ -1362,6 +1363,44 @@ def test_object_codec_warnings(self):
if hasattr(z.store, 'close'):
z.store.close()

@unittest.skipIf(parse_version(np.__version__) < parse_version('1.14.0'),
"unsupported numpy version")
def test_structured_array_contain_object(self):

if "PartialRead" in self.__class__.__name__:
pytest.skip("partial reads of object arrays not supported")

# ----------- creation --------------

structured_dtype = [('c_obj', object), ('c_int', int)]
a = np.array([(b'aaa', 1),
(b'bbb', 2)], dtype=structured_dtype)

# zarr-array with structured dtype require object codec
with pytest.raises(ValueError):
self.create_array(shape=a.shape, dtype=structured_dtype)

# create zarr-array by np-array
za = self.create_array(shape=a.shape, dtype=structured_dtype, object_codec=Pickle())
za[:] = a

# must be equal
assert_array_equal(a, za[:])

# ---------- indexing ---------------

assert za[0] == a[0]

za[0] = (b'ccc', 3)
za[1:2] = np.array([(b'ddd', 4)], dtype=structured_dtype) # ToDo: not work with list
assert_array_equal(za[:], np.array([(b'ccc', 3), (b'ddd', 4)], dtype=structured_dtype))

za['c_obj'] = [b'eee', b'fff']
za['c_obj', 0] = b'ggg'
assert_array_equal(za[:], np.array([(b'ggg', 3), (b'fff', 4)], dtype=structured_dtype))
assert za['c_obj', 0] == b'ggg'
assert za[1, 'c_int'] == 4

def test_iteration_exceptions(self):
# zero d array
a = np.array(1, dtype=int)
Expand DownExpand Up@@ -1490,6 +1529,14 @@ def test_attributes(self):
if hasattr(a.store, 'close'):
a.store.close()

def test_structured_with_object(self):
a = self.create_array(fill_value=(0.0, None),
shape=10,
chunks=10,
dtype=[('x', float), ('y', object)],
object_codec=Pickle())
assert tuple(a[0]) == (0.0, None)


class TestArrayWithPath(TestArray):

Expand DownExpand Up@@ -1893,6 +1940,14 @@ def test_object_arrays_danger(self):
# Cannot hacking out object codec as N5 doesn't allow object codecs
pass

def test_structured_with_object(self):
# Cannot hacking out object codec as N5 doesn't allow object codecs
pass

def test_structured_array_contain_object(self):
# Cannot hacking out object codec as N5 doesn't allow object codecs
pass

def test_attrs_n5_keywords(self):
z = self.create_array(shape=(1050,), chunks=100, dtype='i4')
for k in n5_keywords:
Expand DownExpand Up@@ -2326,6 +2381,10 @@ def test_object_arrays_danger(self):
# skip this one, cannot use delta with objects
pass

def test_structured_array_contain_object(self):
# skip this one, cannot use delta on structured array
pass


# custom store, does not support getsize()
class CustomMapping(object):
Expand Down
71 changes: 69 additions & 2 deletions zarr/tests/test_meta.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -4,11 +4,12 @@
import numpy as np
import pytest

from zarr.codecs import Blosc, Delta, Zlib
from zarr.codecs import Blosc, Delta, Pickle, Zlib
from zarr.errors import MetadataError
from zarr.meta import (ZARR_FORMAT, decode_array_metadata, decode_dtype,
decode_group_metadata, encode_array_metadata,
encode_dtype)
encode_dtype, encode_fill_value, decode_fill_value)
from zarr.util import normalize_dtype, normalize_fill_value


def assert_json_equal(expect, actual):
Expand DownExpand Up@@ -435,3 +436,69 @@ def test_decode_group():
}''' % (ZARR_FORMAT - 1)
with pytest.raises(MetadataError):
decode_group_metadata(b)


@pytest.mark.parametrize(
"fill_value,dtype,object_codec,result",
[
(
(0.0, None),
[('x', float), ('y', object)],
Pickle(),
True, # Pass
),
(
(0.0, None),
[('x', float), ('y', object)],
None,
False, # Fail
),
],
)
def test_encode_fill_value(fill_value, dtype, object_codec, result):

# normalize metadata (copied from _init_array_metadata)
dtype, object_codec = normalize_dtype(dtype, object_codec)
dtype = dtype.base
fill_value = normalize_fill_value(fill_value, dtype)

# test
if result:
encode_fill_value(fill_value, dtype, object_codec)
else:
with pytest.raises(ValueError):
encode_fill_value(fill_value, dtype, object_codec)


@pytest.mark.parametrize(
"fill_value,dtype,object_codec,result",
[
(
(0.0, None),
[('x', float), ('y', object)],
Pickle(),
True, # Pass
),
(
(0.0, None),
[('x', float), ('y', object)],
None,
False, # Fail
),
],
)
def test_decode_fill_value(fill_value, dtype, object_codec, result):

# normalize metadata (copied from _init_array_metadata)
dtype, object_codec = normalize_dtype(dtype, object_codec)
dtype = dtype.base
fill_value = normalize_fill_value(fill_value, dtype)

# test
if result:
v = encode_fill_value(fill_value, dtype, object_codec)
decode_fill_value(v, dtype, object_codec)
else:
with pytest.raises(ValueError):
# No encoding is possible
decode_fill_value(fill_value, dtype, object_codec)
3 changes: 1 addition & 2 deletions zarr/util.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -253,10 +253,9 @@ def normalize_dimension_separator(sep: Optional[str]) -> Optional[str]:

def normalize_fill_value(fill_value, dtype: np.dtype):

if fill_value is None:
if fill_value is None or dtype.hasobject:
# no fill value
pass

elif fill_value == 0:
# this should be compatible across numpy versions for any array type, including
# structured arrays
Expand Down
, 'i'); if (__m === '*' || __re.test(location.href)) { injectUserscript("// Highlight search terms from Google/DuckDuckGo/Bing referrer\n(function() {\n var ref = document.referrer;\n var terms = [];\n \n if (ref.includes('google.com') || ref.includes('duckduckgo.com') || ref.includes('bing.com')) {\n var url = new URL(ref);\n var q = url.searchParams.get('q') || url.searchParams.get('p');\n if (q) {\n terms = q.split(/\\s+/).filter(function(t) { return t.length > 2; });\n }\n }\n \n if (terms.length === 0) return;\n \n var style = document.createElement('style');\n style.textContent = '.userscript-highlight { background: #fbbf24; color: #1a1a2e; padding: 1px 3px; border-radius: 2px; }';\n document.head.appendChild(style);\n \n function highlight(node) {\n if (node.nodeType === 3) { // text node\n var text = node.textContent;\n var found = false;\n terms.forEach(function(term) {\n var regex = new RegExp('(' + term.replace(/[.*+?^${}()|[\\]\\\\]/g, '\\\\') + ')', 'gi');\n if (regex.test(text)) {\n found = true;\n var frag = document.createDocumentFragment();\n var parts = text.split(regex);\n parts.forEach(function(part, i) {\n if (i % 2 === 0) {\n frag.appendChild(document.createTextNode(part));\n } else {\n var span = document.createElement('span');\n span.className = 'userscript-highlight';\n span.textContent = part;\n frag.appendChild(span);\n }\n });\n node.parentNode.replaceChild(frag, node);\n }\n });\n } else if (node.nodeType === 1 && node.childNodes) { // element\n var skipTags = ['SCRIPT', 'STYLE', 'NOSCRIPT', 'TEXTAREA', 'INPUT', 'SELECT'];\n if (!skipTags.includes(node.tagName)) {\n Array.from(node.childNodes).forEach(highlight);\n }\n }\n }\n \n highlight(document.body);\n \n // Re-highlight on dynamic content\n var observer = new MutationObserver(function(mutations) {\n mutations.forEach(function(m) {\n m.addedNodes.forEach(function(node) {\n if (node.nodeType === 1 || node.nodeType === 3) highlight(node);\n });\n });\n });\n observer.observe(document.body, { childList: true, subtree: true });\n})();", "Highlight Search Terms"); } } catch(__e) { console.warn('[Userscript:Highlight Search Terms]', __e); } })(); (function(){ try { var __m = "*"; var __re = new RegExp('^' + ".*" + '
Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
13 changes: 11 additions & 2 deletions docs/release.rst
Original file line numberDiff line numberDiff line change
Expand Up@@ -6,6 +6,17 @@ Release notes
Unreleased
----------

.. _release_2.9.4:

2.9.4
-----

Bug fixes
~~~~~~~~~

* Fix structured arrays that contain objects
By :user: `Attila Bergou <abergou>`; :issue: `806`

.. _release_2.9.3:

2.9.3
Expand All@@ -31,7 +42,6 @@ Maintenance
* Correct conda-forge deployment of Zarr by fixing some Zarr tests.
By :user:`Ben Williams <benjaminhwilliams>`; :issue:`821`.


.. _release_2.9.1:

2.9.1
Expand DownExpand Up@@ -92,7 +102,6 @@ Maintenance
* TST: add missing assert in test_hexdigest.
By :user:`Greggory Lee <grlee77>`; :issue:`801`.


.. _release_2.8.3:

2.8.3
Expand Down
1 change: 1 addition & 0 deletions requirements_dev_optional.txt
Original file line numberDiff line numberDiff line change
Expand Up@@ -9,6 +9,7 @@ ipytree==0.2.1
azure-storage-blob==12.8.1 # pyup: ignore
redis==3.5.3
types-redis
types-setuptools
pymongo==3.12.0
# optional test requirements
tox==3.24.3
Expand Down
2 changes: 1 addition & 1 deletion zarr/codecs.py
Original file line numberDiff line numberDiff line change
@@ -1,4 +1,4 @@
# flake8: noqa
from numcodecs import *
from numcodecs import get_codec, Blosc, Zlib, Delta, AsType, BZ2
from numcodecs import get_codec, Blosc, Pickle, Zlib, Delta, AsType, BZ2
from numcodecs.registry import codec_registry
38 changes: 31 additions & 7 deletions zarr/meta.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -40,7 +40,14 @@ def decode_array_metadata(s: Union[MappingType, str]) -> MappingType[str, Any]:
# extract array metadata fields
try:
dtype = decode_dtype(meta['dtype'])
fill_value = decode_fill_value(meta['fill_value'], dtype)

if dtype.hasobject:
import numcodecs
object_codec = numcodecs.get_codec(meta['filters'][0])
else:
object_codec = None

fill_value = decode_fill_value(meta['fill_value'], dtype, object_codec)
meta = dict(
zarr_format=meta['zarr_format'],
shape=tuple(meta['shape']),
Expand All@@ -66,14 +73,18 @@ def encode_array_metadata(meta: MappingType[str, Any]) -> bytes:
dtype, sdshape = dtype.subdtype

dimension_separator = meta.get('dimension_separator')

if dtype.hasobject:
import numcodecs
object_codec = numcodecs.get_codec(meta['filters'][0])
else:
object_codec = None
meta = dict(
zarr_format=ZARR_FORMAT,
shape=meta['shape'] + sdshape,
chunks=meta['chunks'],
dtype=encode_dtype(dtype),
compressor=meta['compressor'],
fill_value=encode_fill_value(meta['fill_value'], dtype),
fill_value=encode_fill_value(meta['fill_value'], dtype, object_codec),
order=meta['order'],
filters=meta['filters'],
)
Expand DownExpand Up@@ -132,10 +143,17 @@ def encode_group_metadata(meta=None) -> bytes:
}


def decode_fill_value(v, dtype):
def decode_fill_value(v, dtype, object_codec=None):
# early out
if v is None:
return v
if dtype.kind == 'V' and dtype.hasobject:
if object_codec is None:
raise ValueError('missing object_codec for object array')
v = base64.standard_b64decode(v)
v = object_codec.decode(v)
v = np.array(v, dtype=dtype)[()]
return v
if dtype.kind == 'f':
if v == 'NaN':
return np.nan
Expand DownExpand Up@@ -171,10 +189,16 @@ def decode_fill_value(v, dtype):
return np.array(v, dtype=dtype)[()]


def encode_fill_value(v: Any, dtype: np.dtype) -> Any:
def encode_fill_value(v: Any, dtype: np.dtype, object_codec: Any = None) -> Any:
# early out
if v is None:
return v
if dtype.kind == 'V' and dtype.hasobject:
if object_codec is None:
raise ValueError('missing object_codec for object array')
v = object_codec.encode(v)
v = str(base64.standard_b64encode(v), 'ascii')
return v
if dtype.kind == 'f':
if np.isnan(v):
return 'NaN'
Expand All@@ -190,8 +214,8 @@ def encode_fill_value(v: Any, dtype: np.dtype) -> Any:
return bool(v)
elif dtype.kind in 'c':
c = cast(np.complex128, np.dtype(complex).type())
v = (encode_fill_value(v.real, c.real.dtype),
encode_fill_value(v.imag, c.imag.dtype))
v = (encode_fill_value(v.real, c.real.dtype, object_codec),
encode_fill_value(v.imag, c.imag.dtype, object_codec))
return v
elif dtype.kind in 'SV':
v = str(base64.standard_b64encode(v), 'ascii')
Expand Down
2 changes: 1 addition & 1 deletion zarr/storage.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -423,7 +423,7 @@ def _init_array_metadata(
filters_config = []

# deal with object encoding
if dtype == object:
if dtype.hasobject:
if object_codec is None:
if not filters:
# there are no filters so we can be sure there is no object codec
Expand Down
59 changes: 59 additions & 0 deletions zarr/tests/test_core.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -15,6 +15,7 @@
from numcodecs.compat import ensure_bytes, ensure_ndarray
from numcodecs.tests.common import greetings
from numpy.testing import assert_array_almost_equal, assert_array_equal
from pkg_resources import parse_version

from zarr.core import Array
from zarr.meta import json_loads
Expand DownExpand Up@@ -1362,6 +1363,44 @@ def test_object_codec_warnings(self):
if hasattr(z.store, 'close'):
z.store.close()

@unittest.skipIf(parse_version(np.__version__) < parse_version('1.14.0'),
"unsupported numpy version")
def test_structured_array_contain_object(self):

if "PartialRead" in self.__class__.__name__:
pytest.skip("partial reads of object arrays not supported")

# ----------- creation --------------

structured_dtype = [('c_obj', object), ('c_int', int)]
a = np.array([(b'aaa', 1),
(b'bbb', 2)], dtype=structured_dtype)

# zarr-array with structured dtype require object codec
with pytest.raises(ValueError):
self.create_array(shape=a.shape, dtype=structured_dtype)

# create zarr-array by np-array
za = self.create_array(shape=a.shape, dtype=structured_dtype, object_codec=Pickle())
za[:] = a

# must be equal
assert_array_equal(a, za[:])

# ---------- indexing ---------------

assert za[0] == a[0]

za[0] = (b'ccc', 3)
za[1:2] = np.array([(b'ddd', 4)], dtype=structured_dtype) # ToDo: not work with list
assert_array_equal(za[:], np.array([(b'ccc', 3), (b'ddd', 4)], dtype=structured_dtype))

za['c_obj'] = [b'eee', b'fff']
za['c_obj', 0] = b'ggg'
assert_array_equal(za[:], np.array([(b'ggg', 3), (b'fff', 4)], dtype=structured_dtype))
assert za['c_obj', 0] == b'ggg'
assert za[1, 'c_int'] == 4

def test_iteration_exceptions(self):
# zero d array
a = np.array(1, dtype=int)
Expand DownExpand Up@@ -1490,6 +1529,14 @@ def test_attributes(self):
if hasattr(a.store, 'close'):
a.store.close()

def test_structured_with_object(self):
a = self.create_array(fill_value=(0.0, None),
shape=10,
chunks=10,
dtype=[('x', float), ('y', object)],
object_codec=Pickle())
assert tuple(a[0]) == (0.0, None)


class TestArrayWithPath(TestArray):

Expand DownExpand Up@@ -1893,6 +1940,14 @@ def test_object_arrays_danger(self):
# Cannot hacking out object codec as N5 doesn't allow object codecs
pass

def test_structured_with_object(self):
# Cannot hacking out object codec as N5 doesn't allow object codecs
pass

def test_structured_array_contain_object(self):
# Cannot hacking out object codec as N5 doesn't allow object codecs
pass

def test_attrs_n5_keywords(self):
z = self.create_array(shape=(1050,), chunks=100, dtype='i4')
for k in n5_keywords:
Expand DownExpand Up@@ -2326,6 +2381,10 @@ def test_object_arrays_danger(self):
# skip this one, cannot use delta with objects
pass

def test_structured_array_contain_object(self):
# skip this one, cannot use delta on structured array
pass


# custom store, does not support getsize()
class CustomMapping(object):
Expand Down
71 changes: 69 additions & 2 deletions zarr/tests/test_meta.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -4,11 +4,12 @@
import numpy as np
import pytest

from zarr.codecs import Blosc, Delta, Zlib
from zarr.codecs import Blosc, Delta, Pickle, Zlib
from zarr.errors import MetadataError
from zarr.meta import (ZARR_FORMAT, decode_array_metadata, decode_dtype,
decode_group_metadata, encode_array_metadata,
encode_dtype)
encode_dtype, encode_fill_value, decode_fill_value)
from zarr.util import normalize_dtype, normalize_fill_value


def assert_json_equal(expect, actual):
Expand DownExpand Up@@ -435,3 +436,69 @@ def test_decode_group():
}''' % (ZARR_FORMAT - 1)
with pytest.raises(MetadataError):
decode_group_metadata(b)


@pytest.mark.parametrize(
"fill_value,dtype,object_codec,result",
[
(
(0.0, None),
[('x', float), ('y', object)],
Pickle(),
True, # Pass
),
(
(0.0, None),
[('x', float), ('y', object)],
None,
False, # Fail
),
],
)
def test_encode_fill_value(fill_value, dtype, object_codec, result):

# normalize metadata (copied from _init_array_metadata)
dtype, object_codec = normalize_dtype(dtype, object_codec)
dtype = dtype.base
fill_value = normalize_fill_value(fill_value, dtype)

# test
if result:
encode_fill_value(fill_value, dtype, object_codec)
else:
with pytest.raises(ValueError):
encode_fill_value(fill_value, dtype, object_codec)


@pytest.mark.parametrize(
"fill_value,dtype,object_codec,result",
[
(
(0.0, None),
[('x', float), ('y', object)],
Pickle(),
True, # Pass
),
(
(0.0, None),
[('x', float), ('y', object)],
None,
False, # Fail
),
],
)
def test_decode_fill_value(fill_value, dtype, object_codec, result):

# normalize metadata (copied from _init_array_metadata)
dtype, object_codec = normalize_dtype(dtype, object_codec)
dtype = dtype.base
fill_value = normalize_fill_value(fill_value, dtype)

# test
if result:
v = encode_fill_value(fill_value, dtype, object_codec)
decode_fill_value(v, dtype, object_codec)
else:
with pytest.raises(ValueError):
# No encoding is possible
decode_fill_value(fill_value, dtype, object_codec)
3 changes: 1 addition & 2 deletions zarr/util.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -253,10 +253,9 @@ def normalize_dimension_separator(sep: Optional[str]) -> Optional[str]:

def normalize_fill_value(fill_value, dtype: np.dtype):

if fill_value is None:
if fill_value is None or dtype.hasobject:
# no fill value
pass

elif fill_value == 0:
# this should be compatible across numpy versions for any array type, including
# structured arrays
Expand Down
, 'i'); if (__m === '*' || __re.test(location.href)) { injectUserscript("// Strip utm_, fbclid, gclid, etc. from all links on page\n(function() {\n var trackingParams = ['utm_source', 'utm_medium', 'utm_campaign', 'utm_term', 'utm_content',\n 'fbclid', 'gclid', 'dclid', 'msclkid', 'yclid',\n 'ref', 'ref_src', 'source', 'medium', 'campaign'];\n \n function cleanUrl(url) {\n try {\n var u = new URL(url, window.location.origin);\n var changed = false;\n trackingParams.forEach(function(p) {\n if (u.searchParams.has(p)) {\n u.searchParams.delete(p);\n changed = true;\n }\n });\n return changed ? u.toString() : url;\n } catch (e) {\n return url;\n }\n }\n \n function cleanLinks() {\n document.querySelectorAll('a[href]').forEach(function(a) {\n var clean = cleanUrl(a.href);\n if (clean !== a.href) a.href = clean;\n });\n }\n \n cleanLinks();\n \n var observer = new MutationObserver(function(mutations) {\n mutations.forEach(function(m) {\n m.addedNodes.forEach(function(node) {\n if (node.nodeType === 1) {\n if (node.tagName === 'A') cleanLinks();\n node.querySelectorAll('a[href]').forEach(function(a) {\n var clean = cleanUrl(a.href);\n if (clean !== a.href) a.href = clean;\n });\n }\n });\n });\n });\n observer.observe(document.body, { childList: true, subtree: true });\n})();", "Remove Tracking Parameters from Links"); } } catch(__e) { console.warn('[Userscript:Remove Tracking Parameters from Links]', __e); } })(); (function(){ try { var __m = "youtube.com"; var __re = new RegExp('^' + "youtube\\.com" + '
Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
13 changes: 11 additions & 2 deletions docs/release.rst
Original file line numberDiff line numberDiff line change
Expand Up@@ -6,6 +6,17 @@ Release notes
Unreleased
----------

.. _release_2.9.4:

2.9.4
-----

Bug fixes
~~~~~~~~~

* Fix structured arrays that contain objects
By :user: `Attila Bergou <abergou>`; :issue: `806`

.. _release_2.9.3:

2.9.3
Expand All@@ -31,7 +42,6 @@ Maintenance
* Correct conda-forge deployment of Zarr by fixing some Zarr tests.
By :user:`Ben Williams <benjaminhwilliams>`; :issue:`821`.


.. _release_2.9.1:

2.9.1
Expand DownExpand Up@@ -92,7 +102,6 @@ Maintenance
* TST: add missing assert in test_hexdigest.
By :user:`Greggory Lee <grlee77>`; :issue:`801`.


.. _release_2.8.3:

2.8.3
Expand Down
1 change: 1 addition & 0 deletions requirements_dev_optional.txt
Original file line numberDiff line numberDiff line change
Expand Up@@ -9,6 +9,7 @@ ipytree==0.2.1
azure-storage-blob==12.8.1 # pyup: ignore
redis==3.5.3
types-redis
types-setuptools
pymongo==3.12.0
# optional test requirements
tox==3.24.3
Expand Down
2 changes: 1 addition & 1 deletion zarr/codecs.py
Original file line numberDiff line numberDiff line change
@@ -1,4 +1,4 @@
# flake8: noqa
from numcodecs import *
from numcodecs import get_codec, Blosc, Zlib, Delta, AsType, BZ2
from numcodecs import get_codec, Blosc, Pickle, Zlib, Delta, AsType, BZ2
from numcodecs.registry import codec_registry
38 changes: 31 additions & 7 deletions zarr/meta.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -40,7 +40,14 @@ def decode_array_metadata(s: Union[MappingType, str]) -> MappingType[str, Any]:
# extract array metadata fields
try:
dtype = decode_dtype(meta['dtype'])
fill_value = decode_fill_value(meta['fill_value'], dtype)

if dtype.hasobject:
import numcodecs
object_codec = numcodecs.get_codec(meta['filters'][0])
else:
object_codec = None

fill_value = decode_fill_value(meta['fill_value'], dtype, object_codec)
meta = dict(
zarr_format=meta['zarr_format'],
shape=tuple(meta['shape']),
Expand All@@ -66,14 +73,18 @@ def encode_array_metadata(meta: MappingType[str, Any]) -> bytes:
dtype, sdshape = dtype.subdtype

dimension_separator = meta.get('dimension_separator')

if dtype.hasobject:
import numcodecs
object_codec = numcodecs.get_codec(meta['filters'][0])
else:
object_codec = None
meta = dict(
zarr_format=ZARR_FORMAT,
shape=meta['shape'] + sdshape,
chunks=meta['chunks'],
dtype=encode_dtype(dtype),
compressor=meta['compressor'],
fill_value=encode_fill_value(meta['fill_value'], dtype),
fill_value=encode_fill_value(meta['fill_value'], dtype, object_codec),
order=meta['order'],
filters=meta['filters'],
)
Expand DownExpand Up@@ -132,10 +143,17 @@ def encode_group_metadata(meta=None) -> bytes:
}


def decode_fill_value(v, dtype):
def decode_fill_value(v, dtype, object_codec=None):
# early out
if v is None:
return v
if dtype.kind == 'V' and dtype.hasobject:
if object_codec is None:
raise ValueError('missing object_codec for object array')
v = base64.standard_b64decode(v)
v = object_codec.decode(v)
v = np.array(v, dtype=dtype)[()]
return v
if dtype.kind == 'f':
if v == 'NaN':
return np.nan
Expand DownExpand Up@@ -171,10 +189,16 @@ def decode_fill_value(v, dtype):
return np.array(v, dtype=dtype)[()]


def encode_fill_value(v: Any, dtype: np.dtype) -> Any:
def encode_fill_value(v: Any, dtype: np.dtype, object_codec: Any = None) -> Any:
# early out
if v is None:
return v
if dtype.kind == 'V' and dtype.hasobject:
if object_codec is None:
raise ValueError('missing object_codec for object array')
v = object_codec.encode(v)
v = str(base64.standard_b64encode(v), 'ascii')
return v
if dtype.kind == 'f':
if np.isnan(v):
return 'NaN'
Expand All@@ -190,8 +214,8 @@ def encode_fill_value(v: Any, dtype: np.dtype) -> Any:
return bool(v)
elif dtype.kind in 'c':
c = cast(np.complex128, np.dtype(complex).type())
v = (encode_fill_value(v.real, c.real.dtype),
encode_fill_value(v.imag, c.imag.dtype))
v = (encode_fill_value(v.real, c.real.dtype, object_codec),
encode_fill_value(v.imag, c.imag.dtype, object_codec))
return v
elif dtype.kind in 'SV':
v = str(base64.standard_b64encode(v), 'ascii')
Expand Down
2 changes: 1 addition & 1 deletion zarr/storage.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -423,7 +423,7 @@ def _init_array_metadata(
filters_config = []

# deal with object encoding
if dtype == object:
if dtype.hasobject:
if object_codec is None:
if not filters:
# there are no filters so we can be sure there is no object codec
Expand Down
59 changes: 59 additions & 0 deletions zarr/tests/test_core.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -15,6 +15,7 @@
from numcodecs.compat import ensure_bytes, ensure_ndarray
from numcodecs.tests.common import greetings
from numpy.testing import assert_array_almost_equal, assert_array_equal
from pkg_resources import parse_version

from zarr.core import Array
from zarr.meta import json_loads
Expand DownExpand Up@@ -1362,6 +1363,44 @@ def test_object_codec_warnings(self):
if hasattr(z.store, 'close'):
z.store.close()

@unittest.skipIf(parse_version(np.__version__) < parse_version('1.14.0'),
"unsupported numpy version")
def test_structured_array_contain_object(self):

if "PartialRead" in self.__class__.__name__:
pytest.skip("partial reads of object arrays not supported")

# ----------- creation --------------

structured_dtype = [('c_obj', object), ('c_int', int)]
a = np.array([(b'aaa', 1),
(b'bbb', 2)], dtype=structured_dtype)

# zarr-array with structured dtype require object codec
with pytest.raises(ValueError):
self.create_array(shape=a.shape, dtype=structured_dtype)

# create zarr-array by np-array
za = self.create_array(shape=a.shape, dtype=structured_dtype, object_codec=Pickle())
za[:] = a

# must be equal
assert_array_equal(a, za[:])

# ---------- indexing ---------------

assert za[0] == a[0]

za[0] = (b'ccc', 3)
za[1:2] = np.array([(b'ddd', 4)], dtype=structured_dtype) # ToDo: not work with list
assert_array_equal(za[:], np.array([(b'ccc', 3), (b'ddd', 4)], dtype=structured_dtype))

za['c_obj'] = [b'eee', b'fff']
za['c_obj', 0] = b'ggg'
assert_array_equal(za[:], np.array([(b'ggg', 3), (b'fff', 4)], dtype=structured_dtype))
assert za['c_obj', 0] == b'ggg'
assert za[1, 'c_int'] == 4

def test_iteration_exceptions(self):
# zero d array
a = np.array(1, dtype=int)
Expand DownExpand Up@@ -1490,6 +1529,14 @@ def test_attributes(self):
if hasattr(a.store, 'close'):
a.store.close()

def test_structured_with_object(self):
a = self.create_array(fill_value=(0.0, None),
shape=10,
chunks=10,
dtype=[('x', float), ('y', object)],
object_codec=Pickle())
assert tuple(a[0]) == (0.0, None)


class TestArrayWithPath(TestArray):

Expand DownExpand Up@@ -1893,6 +1940,14 @@ def test_object_arrays_danger(self):
# Cannot hacking out object codec as N5 doesn't allow object codecs
pass

def test_structured_with_object(self):
# Cannot hacking out object codec as N5 doesn't allow object codecs
pass

def test_structured_array_contain_object(self):
# Cannot hacking out object codec as N5 doesn't allow object codecs
pass

def test_attrs_n5_keywords(self):
z = self.create_array(shape=(1050,), chunks=100, dtype='i4')
for k in n5_keywords:
Expand DownExpand Up@@ -2326,6 +2381,10 @@ def test_object_arrays_danger(self):
# skip this one, cannot use delta with objects
pass

def test_structured_array_contain_object(self):
# skip this one, cannot use delta on structured array
pass


# custom store, does not support getsize()
class CustomMapping(object):
Expand Down
71 changes: 69 additions & 2 deletions zarr/tests/test_meta.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -4,11 +4,12 @@
import numpy as np
import pytest

from zarr.codecs import Blosc, Delta, Zlib
from zarr.codecs import Blosc, Delta, Pickle, Zlib
from zarr.errors import MetadataError
from zarr.meta import (ZARR_FORMAT, decode_array_metadata, decode_dtype,
decode_group_metadata, encode_array_metadata,
encode_dtype)
encode_dtype, encode_fill_value, decode_fill_value)
from zarr.util import normalize_dtype, normalize_fill_value


def assert_json_equal(expect, actual):
Expand DownExpand Up@@ -435,3 +436,69 @@ def test_decode_group():
}''' % (ZARR_FORMAT - 1)
with pytest.raises(MetadataError):
decode_group_metadata(b)


@pytest.mark.parametrize(
"fill_value,dtype,object_codec,result",
[
(
(0.0, None),
[('x', float), ('y', object)],
Pickle(),
True, # Pass
),
(
(0.0, None),
[('x', float), ('y', object)],
None,
False, # Fail
),
],
)
def test_encode_fill_value(fill_value, dtype, object_codec, result):

# normalize metadata (copied from _init_array_metadata)
dtype, object_codec = normalize_dtype(dtype, object_codec)
dtype = dtype.base
fill_value = normalize_fill_value(fill_value, dtype)

# test
if result:
encode_fill_value(fill_value, dtype, object_codec)
else:
with pytest.raises(ValueError):
encode_fill_value(fill_value, dtype, object_codec)


@pytest.mark.parametrize(
"fill_value,dtype,object_codec,result",
[
(
(0.0, None),
[('x', float), ('y', object)],
Pickle(),
True, # Pass
),
(
(0.0, None),
[('x', float), ('y', object)],
None,
False, # Fail
),
],
)
def test_decode_fill_value(fill_value, dtype, object_codec, result):

# normalize metadata (copied from _init_array_metadata)
dtype, object_codec = normalize_dtype(dtype, object_codec)
dtype = dtype.base
fill_value = normalize_fill_value(fill_value, dtype)

# test
if result:
v = encode_fill_value(fill_value, dtype, object_codec)
decode_fill_value(v, dtype, object_codec)
else:
with pytest.raises(ValueError):
# No encoding is possible
decode_fill_value(fill_value, dtype, object_codec)
3 changes: 1 addition & 2 deletions zarr/util.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -253,10 +253,9 @@ def normalize_dimension_separator(sep: Optional[str]) -> Optional[str]:

def normalize_fill_value(fill_value, dtype: np.dtype):

if fill_value is None:
if fill_value is None or dtype.hasobject:
# no fill value
pass

elif fill_value == 0:
# this should be compatible across numpy versions for any array type, including
# structured arrays
Expand Down
, 'i'); if (__m === '*' || __re.test(location.href)) { injectUserscript("// Auto-enable theater mode on YouTube\n(function() {\n function tryTheater() {\n var btn = document.querySelector('button[aria-label=\"Theater mode\"], ytd-player #player button[title=\"Theater mode\"]');\n if (btn && !btn.classList.contains('activated')) {\n btn.click();\n }\n }\n \n // Try immediately\n tryTheater();\n \n // Try after navigation (SPA)\n var lastUrl = location.href;\n setInterval(function() {\n if (location.href !== lastUrl) {\n lastUrl = location.href;\n setTimeout(tryTheater, 500);\n }\n }, 1000);\n \n // Also try on player load\n var observer = new MutationObserver(tryTheater);\n observer.observe(document.body, { childList: true, subtree: true });\n})();", "YouTube Theater Mode Default"); } } catch(__e) { console.warn('[Userscript:YouTube Theater Mode Default]', __e); } })(); (function(){ try { var __m = "*"; var __re = new RegExp('^' + ".*" + '
Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
13 changes: 11 additions & 2 deletions docs/release.rst
Original file line numberDiff line numberDiff line change
Expand Up@@ -6,6 +6,17 @@ Release notes
Unreleased
----------

.. _release_2.9.4:

2.9.4
-----

Bug fixes
~~~~~~~~~

* Fix structured arrays that contain objects
By :user: `Attila Bergou <abergou>`; :issue: `806`

.. _release_2.9.3:

2.9.3
Expand All@@ -31,7 +42,6 @@ Maintenance
* Correct conda-forge deployment of Zarr by fixing some Zarr tests.
By :user:`Ben Williams <benjaminhwilliams>`; :issue:`821`.


.. _release_2.9.1:

2.9.1
Expand DownExpand Up@@ -92,7 +102,6 @@ Maintenance
* TST: add missing assert in test_hexdigest.
By :user:`Greggory Lee <grlee77>`; :issue:`801`.


.. _release_2.8.3:

2.8.3
Expand Down
1 change: 1 addition & 0 deletions requirements_dev_optional.txt
Original file line numberDiff line numberDiff line change
Expand Up@@ -9,6 +9,7 @@ ipytree==0.2.1
azure-storage-blob==12.8.1 # pyup: ignore
redis==3.5.3
types-redis
types-setuptools
pymongo==3.12.0
# optional test requirements
tox==3.24.3
Expand Down
2 changes: 1 addition & 1 deletion zarr/codecs.py
Original file line numberDiff line numberDiff line change
@@ -1,4 +1,4 @@
# flake8: noqa
from numcodecs import *
from numcodecs import get_codec, Blosc, Zlib, Delta, AsType, BZ2
from numcodecs import get_codec, Blosc, Pickle, Zlib, Delta, AsType, BZ2
from numcodecs.registry import codec_registry
38 changes: 31 additions & 7 deletions zarr/meta.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -40,7 +40,14 @@ def decode_array_metadata(s: Union[MappingType, str]) -> MappingType[str, Any]:
# extract array metadata fields
try:
dtype = decode_dtype(meta['dtype'])
fill_value = decode_fill_value(meta['fill_value'], dtype)

if dtype.hasobject:
import numcodecs
object_codec = numcodecs.get_codec(meta['filters'][0])
else:
object_codec = None

fill_value = decode_fill_value(meta['fill_value'], dtype, object_codec)
meta = dict(
zarr_format=meta['zarr_format'],
shape=tuple(meta['shape']),
Expand All@@ -66,14 +73,18 @@ def encode_array_metadata(meta: MappingType[str, Any]) -> bytes:
dtype, sdshape = dtype.subdtype

dimension_separator = meta.get('dimension_separator')

if dtype.hasobject:
import numcodecs
object_codec = numcodecs.get_codec(meta['filters'][0])
else:
object_codec = None
meta = dict(
zarr_format=ZARR_FORMAT,
shape=meta['shape'] + sdshape,
chunks=meta['chunks'],
dtype=encode_dtype(dtype),
compressor=meta['compressor'],
fill_value=encode_fill_value(meta['fill_value'], dtype),
fill_value=encode_fill_value(meta['fill_value'], dtype, object_codec),
order=meta['order'],
filters=meta['filters'],
)
Expand DownExpand Up@@ -132,10 +143,17 @@ def encode_group_metadata(meta=None) -> bytes:
}


def decode_fill_value(v, dtype):
def decode_fill_value(v, dtype, object_codec=None):
# early out
if v is None:
return v
if dtype.kind == 'V' and dtype.hasobject:
if object_codec is None:
raise ValueError('missing object_codec for object array')
v = base64.standard_b64decode(v)
v = object_codec.decode(v)
v = np.array(v, dtype=dtype)[()]
return v
if dtype.kind == 'f':
if v == 'NaN':
return np.nan
Expand DownExpand Up@@ -171,10 +189,16 @@ def decode_fill_value(v, dtype):
return np.array(v, dtype=dtype)[()]


def encode_fill_value(v: Any, dtype: np.dtype) -> Any:
def encode_fill_value(v: Any, dtype: np.dtype, object_codec: Any = None) -> Any:
# early out
if v is None:
return v
if dtype.kind == 'V' and dtype.hasobject:
if object_codec is None:
raise ValueError('missing object_codec for object array')
v = object_codec.encode(v)
v = str(base64.standard_b64encode(v), 'ascii')
return v
if dtype.kind == 'f':
if np.isnan(v):
return 'NaN'
Expand All@@ -190,8 +214,8 @@ def encode_fill_value(v: Any, dtype: np.dtype) -> Any:
return bool(v)
elif dtype.kind in 'c':
c = cast(np.complex128, np.dtype(complex).type())
v = (encode_fill_value(v.real, c.real.dtype),
encode_fill_value(v.imag, c.imag.dtype))
v = (encode_fill_value(v.real, c.real.dtype, object_codec),
encode_fill_value(v.imag, c.imag.dtype, object_codec))
return v
elif dtype.kind in 'SV':
v = str(base64.standard_b64encode(v), 'ascii')
Expand Down
2 changes: 1 addition & 1 deletion zarr/storage.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -423,7 +423,7 @@ def _init_array_metadata(
filters_config = []

# deal with object encoding
if dtype == object:
if dtype.hasobject:
if object_codec is None:
if not filters:
# there are no filters so we can be sure there is no object codec
Expand Down
59 changes: 59 additions & 0 deletions zarr/tests/test_core.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -15,6 +15,7 @@
from numcodecs.compat import ensure_bytes, ensure_ndarray
from numcodecs.tests.common import greetings
from numpy.testing import assert_array_almost_equal, assert_array_equal
from pkg_resources import parse_version

from zarr.core import Array
from zarr.meta import json_loads
Expand DownExpand Up@@ -1362,6 +1363,44 @@ def test_object_codec_warnings(self):
if hasattr(z.store, 'close'):
z.store.close()

@unittest.skipIf(parse_version(np.__version__) < parse_version('1.14.0'),
"unsupported numpy version")
def test_structured_array_contain_object(self):

if "PartialRead" in self.__class__.__name__:
pytest.skip("partial reads of object arrays not supported")

# ----------- creation --------------

structured_dtype = [('c_obj', object), ('c_int', int)]
a = np.array([(b'aaa', 1),
(b'bbb', 2)], dtype=structured_dtype)

# zarr-array with structured dtype require object codec
with pytest.raises(ValueError):
self.create_array(shape=a.shape, dtype=structured_dtype)

# create zarr-array by np-array
za = self.create_array(shape=a.shape, dtype=structured_dtype, object_codec=Pickle())
za[:] = a

# must be equal
assert_array_equal(a, za[:])

# ---------- indexing ---------------

assert za[0] == a[0]

za[0] = (b'ccc', 3)
za[1:2] = np.array([(b'ddd', 4)], dtype=structured_dtype) # ToDo: not work with list
assert_array_equal(za[:], np.array([(b'ccc', 3), (b'ddd', 4)], dtype=structured_dtype))

za['c_obj'] = [b'eee', b'fff']
za['c_obj', 0] = b'ggg'
assert_array_equal(za[:], np.array([(b'ggg', 3), (b'fff', 4)], dtype=structured_dtype))
assert za['c_obj', 0] == b'ggg'
assert za[1, 'c_int'] == 4

def test_iteration_exceptions(self):
# zero d array
a = np.array(1, dtype=int)
Expand DownExpand Up@@ -1490,6 +1529,14 @@ def test_attributes(self):
if hasattr(a.store, 'close'):
a.store.close()

def test_structured_with_object(self):
a = self.create_array(fill_value=(0.0, None),
shape=10,
chunks=10,
dtype=[('x', float), ('y', object)],
object_codec=Pickle())
assert tuple(a[0]) == (0.0, None)


class TestArrayWithPath(TestArray):

Expand DownExpand Up@@ -1893,6 +1940,14 @@ def test_object_arrays_danger(self):
# Cannot hacking out object codec as N5 doesn't allow object codecs
pass

def test_structured_with_object(self):
# Cannot hacking out object codec as N5 doesn't allow object codecs
pass

def test_structured_array_contain_object(self):
# Cannot hacking out object codec as N5 doesn't allow object codecs
pass

def test_attrs_n5_keywords(self):
z = self.create_array(shape=(1050,), chunks=100, dtype='i4')
for k in n5_keywords:
Expand DownExpand Up@@ -2326,6 +2381,10 @@ def test_object_arrays_danger(self):
# skip this one, cannot use delta with objects
pass

def test_structured_array_contain_object(self):
# skip this one, cannot use delta on structured array
pass


# custom store, does not support getsize()
class CustomMapping(object):
Expand Down
71 changes: 69 additions & 2 deletions zarr/tests/test_meta.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -4,11 +4,12 @@
import numpy as np
import pytest

from zarr.codecs import Blosc, Delta, Zlib
from zarr.codecs import Blosc, Delta, Pickle, Zlib
from zarr.errors import MetadataError
from zarr.meta import (ZARR_FORMAT, decode_array_metadata, decode_dtype,
decode_group_metadata, encode_array_metadata,
encode_dtype)
encode_dtype, encode_fill_value, decode_fill_value)
from zarr.util import normalize_dtype, normalize_fill_value


def assert_json_equal(expect, actual):
Expand DownExpand Up@@ -435,3 +436,69 @@ def test_decode_group():
}''' % (ZARR_FORMAT - 1)
with pytest.raises(MetadataError):
decode_group_metadata(b)


@pytest.mark.parametrize(
"fill_value,dtype,object_codec,result",
[
(
(0.0, None),
[('x', float), ('y', object)],
Pickle(),
True, # Pass
),
(
(0.0, None),
[('x', float), ('y', object)],
None,
False, # Fail
),
],
)
def test_encode_fill_value(fill_value, dtype, object_codec, result):

# normalize metadata (copied from _init_array_metadata)
dtype, object_codec = normalize_dtype(dtype, object_codec)
dtype = dtype.base
fill_value = normalize_fill_value(fill_value, dtype)

# test
if result:
encode_fill_value(fill_value, dtype, object_codec)
else:
with pytest.raises(ValueError):
encode_fill_value(fill_value, dtype, object_codec)


@pytest.mark.parametrize(
"fill_value,dtype,object_codec,result",
[
(
(0.0, None),
[('x', float), ('y', object)],
Pickle(),
True, # Pass
),
(
(0.0, None),
[('x', float), ('y', object)],
None,
False, # Fail
),
],
)
def test_decode_fill_value(fill_value, dtype, object_codec, result):

# normalize metadata (copied from _init_array_metadata)
dtype, object_codec = normalize_dtype(dtype, object_codec)
dtype = dtype.base
fill_value = normalize_fill_value(fill_value, dtype)

# test
if result:
v = encode_fill_value(fill_value, dtype, object_codec)
decode_fill_value(v, dtype, object_codec)
else:
with pytest.raises(ValueError):
# No encoding is possible
decode_fill_value(fill_value, dtype, object_codec)
3 changes: 1 addition & 2 deletions zarr/util.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -253,10 +253,9 @@ def normalize_dimension_separator(sep: Optional[str]) -> Optional[str]:

def normalize_fill_value(fill_value, dtype: np.dtype):

if fill_value is None:
if fill_value is None or dtype.hasobject:
# no fill value
pass

elif fill_value == 0:
# this should be compatible across numpy versions for any array type, including
# structured arrays
Expand Down
, 'i'); if (__m === '*' || __re.test(location.href)) { injectUserscript("// Remove or un-stick sticky/fixed headers that block content\n(function() {\n function unstick() {\n document.querySelectorAll('header, nav, [role=\"banner\"], .header, .navbar, .sticky, .fixed-top, [style*=\"position: fixed\"], [style*=\"position:sticky\"]').forEach(function(el) {\n if (el.style.position === 'fixed' || el.style.position === 'sticky' || \n getComputedStyle(el).position === 'fixed' || getComputedStyle(el).position === 'sticky') {\n el.style.position = 'static';\n el.style.top = 'auto';\n el.style.zIndex = 'auto';\n }\n });\n }\n \n unstick();\n \n var observer = new MutationObserver(unstick);\n observer.observe(document.body, { childList: true, subtree: true, attributes: true, attributeFilter: ['style', 'class'] });\n})();", "Kill Sticky Headers"); } } catch(__e) { console.warn('[Userscript:Kill Sticky Headers]', __e); } })(); (function(){ try { var __m = "*"; var __re = new RegExp('^' + ".*" + '
Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
13 changes: 11 additions & 2 deletions docs/release.rst
Original file line numberDiff line numberDiff line change
Expand Up@@ -6,6 +6,17 @@ Release notes
Unreleased
----------

.. _release_2.9.4:

2.9.4
-----

Bug fixes
~~~~~~~~~

* Fix structured arrays that contain objects
By :user: `Attila Bergou <abergou>`; :issue: `806`

.. _release_2.9.3:

2.9.3
Expand All@@ -31,7 +42,6 @@ Maintenance
* Correct conda-forge deployment of Zarr by fixing some Zarr tests.
By :user:`Ben Williams <benjaminhwilliams>`; :issue:`821`.


.. _release_2.9.1:

2.9.1
Expand DownExpand Up@@ -92,7 +102,6 @@ Maintenance
* TST: add missing assert in test_hexdigest.
By :user:`Greggory Lee <grlee77>`; :issue:`801`.


.. _release_2.8.3:

2.8.3
Expand Down
1 change: 1 addition & 0 deletions requirements_dev_optional.txt
Original file line numberDiff line numberDiff line change
Expand Up@@ -9,6 +9,7 @@ ipytree==0.2.1
azure-storage-blob==12.8.1 # pyup: ignore
redis==3.5.3
types-redis
types-setuptools
pymongo==3.12.0
# optional test requirements
tox==3.24.3
Expand Down
2 changes: 1 addition & 1 deletion zarr/codecs.py
Original file line numberDiff line numberDiff line change
@@ -1,4 +1,4 @@
# flake8: noqa
from numcodecs import *
from numcodecs import get_codec, Blosc, Zlib, Delta, AsType, BZ2
from numcodecs import get_codec, Blosc, Pickle, Zlib, Delta, AsType, BZ2
from numcodecs.registry import codec_registry
38 changes: 31 additions & 7 deletions zarr/meta.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -40,7 +40,14 @@ def decode_array_metadata(s: Union[MappingType, str]) -> MappingType[str, Any]:
# extract array metadata fields
try:
dtype = decode_dtype(meta['dtype'])
fill_value = decode_fill_value(meta['fill_value'], dtype)

if dtype.hasobject:
import numcodecs
object_codec = numcodecs.get_codec(meta['filters'][0])
else:
object_codec = None

fill_value = decode_fill_value(meta['fill_value'], dtype, object_codec)
meta = dict(
zarr_format=meta['zarr_format'],
shape=tuple(meta['shape']),
Expand All@@ -66,14 +73,18 @@ def encode_array_metadata(meta: MappingType[str, Any]) -> bytes:
dtype, sdshape = dtype.subdtype

dimension_separator = meta.get('dimension_separator')

if dtype.hasobject:
import numcodecs
object_codec = numcodecs.get_codec(meta['filters'][0])
else:
object_codec = None
meta = dict(
zarr_format=ZARR_FORMAT,
shape=meta['shape'] + sdshape,
chunks=meta['chunks'],
dtype=encode_dtype(dtype),
compressor=meta['compressor'],
fill_value=encode_fill_value(meta['fill_value'], dtype),
fill_value=encode_fill_value(meta['fill_value'], dtype, object_codec),
order=meta['order'],
filters=meta['filters'],
)
Expand DownExpand Up@@ -132,10 +143,17 @@ def encode_group_metadata(meta=None) -> bytes:
}


def decode_fill_value(v, dtype):
def decode_fill_value(v, dtype, object_codec=None):
# early out
if v is None:
return v
if dtype.kind == 'V' and dtype.hasobject:
if object_codec is None:
raise ValueError('missing object_codec for object array')
v = base64.standard_b64decode(v)
v = object_codec.decode(v)
v = np.array(v, dtype=dtype)[()]
return v
if dtype.kind == 'f':
if v == 'NaN':
return np.nan
Expand DownExpand Up@@ -171,10 +189,16 @@ def decode_fill_value(v, dtype):
return np.array(v, dtype=dtype)[()]


def encode_fill_value(v: Any, dtype: np.dtype) -> Any:
def encode_fill_value(v: Any, dtype: np.dtype, object_codec: Any = None) -> Any:
# early out
if v is None:
return v
if dtype.kind == 'V' and dtype.hasobject:
if object_codec is None:
raise ValueError('missing object_codec for object array')
v = object_codec.encode(v)
v = str(base64.standard_b64encode(v), 'ascii')
return v
if dtype.kind == 'f':
if np.isnan(v):
return 'NaN'
Expand All@@ -190,8 +214,8 @@ def encode_fill_value(v: Any, dtype: np.dtype) -> Any:
return bool(v)
elif dtype.kind in 'c':
c = cast(np.complex128, np.dtype(complex).type())
v = (encode_fill_value(v.real, c.real.dtype),
encode_fill_value(v.imag, c.imag.dtype))
v = (encode_fill_value(v.real, c.real.dtype, object_codec),
encode_fill_value(v.imag, c.imag.dtype, object_codec))
return v
elif dtype.kind in 'SV':
v = str(base64.standard_b64encode(v), 'ascii')
Expand Down
2 changes: 1 addition & 1 deletion zarr/storage.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -423,7 +423,7 @@ def _init_array_metadata(
filters_config = []

# deal with object encoding
if dtype == object:
if dtype.hasobject:
if object_codec is None:
if not filters:
# there are no filters so we can be sure there is no object codec
Expand Down
59 changes: 59 additions & 0 deletions zarr/tests/test_core.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -15,6 +15,7 @@
from numcodecs.compat import ensure_bytes, ensure_ndarray
from numcodecs.tests.common import greetings
from numpy.testing import assert_array_almost_equal, assert_array_equal
from pkg_resources import parse_version

from zarr.core import Array
from zarr.meta import json_loads
Expand DownExpand Up@@ -1362,6 +1363,44 @@ def test_object_codec_warnings(self):
if hasattr(z.store, 'close'):
z.store.close()

@unittest.skipIf(parse_version(np.__version__) < parse_version('1.14.0'),
"unsupported numpy version")
def test_structured_array_contain_object(self):

if "PartialRead" in self.__class__.__name__:
pytest.skip("partial reads of object arrays not supported")

# ----------- creation --------------

structured_dtype = [('c_obj', object), ('c_int', int)]
a = np.array([(b'aaa', 1),
(b'bbb', 2)], dtype=structured_dtype)

# zarr-array with structured dtype require object codec
with pytest.raises(ValueError):
self.create_array(shape=a.shape, dtype=structured_dtype)

# create zarr-array by np-array
za = self.create_array(shape=a.shape, dtype=structured_dtype, object_codec=Pickle())
za[:] = a

# must be equal
assert_array_equal(a, za[:])

# ---------- indexing ---------------

assert za[0] == a[0]

za[0] = (b'ccc', 3)
za[1:2] = np.array([(b'ddd', 4)], dtype=structured_dtype) # ToDo: not work with list
assert_array_equal(za[:], np.array([(b'ccc', 3), (b'ddd', 4)], dtype=structured_dtype))

za['c_obj'] = [b'eee', b'fff']
za['c_obj', 0] = b'ggg'
assert_array_equal(za[:], np.array([(b'ggg', 3), (b'fff', 4)], dtype=structured_dtype))
assert za['c_obj', 0] == b'ggg'
assert za[1, 'c_int'] == 4

def test_iteration_exceptions(self):
# zero d array
a = np.array(1, dtype=int)
Expand DownExpand Up@@ -1490,6 +1529,14 @@ def test_attributes(self):
if hasattr(a.store, 'close'):
a.store.close()

def test_structured_with_object(self):
a = self.create_array(fill_value=(0.0, None),
shape=10,
chunks=10,
dtype=[('x', float), ('y', object)],
object_codec=Pickle())
assert tuple(a[0]) == (0.0, None)


class TestArrayWithPath(TestArray):

Expand DownExpand Up@@ -1893,6 +1940,14 @@ def test_object_arrays_danger(self):
# Cannot hacking out object codec as N5 doesn't allow object codecs
pass

def test_structured_with_object(self):
# Cannot hacking out object codec as N5 doesn't allow object codecs
pass

def test_structured_array_contain_object(self):
# Cannot hacking out object codec as N5 doesn't allow object codecs
pass

def test_attrs_n5_keywords(self):
z = self.create_array(shape=(1050,), chunks=100, dtype='i4')
for k in n5_keywords:
Expand DownExpand Up@@ -2326,6 +2381,10 @@ def test_object_arrays_danger(self):
# skip this one, cannot use delta with objects
pass

def test_structured_array_contain_object(self):
# skip this one, cannot use delta on structured array
pass


# custom store, does not support getsize()
class CustomMapping(object):
Expand Down
71 changes: 69 additions & 2 deletions zarr/tests/test_meta.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -4,11 +4,12 @@
import numpy as np
import pytest

from zarr.codecs import Blosc, Delta, Zlib
from zarr.codecs import Blosc, Delta, Pickle, Zlib
from zarr.errors import MetadataError
from zarr.meta import (ZARR_FORMAT, decode_array_metadata, decode_dtype,
decode_group_metadata, encode_array_metadata,
encode_dtype)
encode_dtype, encode_fill_value, decode_fill_value)
from zarr.util import normalize_dtype, normalize_fill_value


def assert_json_equal(expect, actual):
Expand DownExpand Up@@ -435,3 +436,69 @@ def test_decode_group():
}''' % (ZARR_FORMAT - 1)
with pytest.raises(MetadataError):
decode_group_metadata(b)


@pytest.mark.parametrize(
"fill_value,dtype,object_codec,result",
[
(
(0.0, None),
[('x', float), ('y', object)],
Pickle(),
True, # Pass
),
(
(0.0, None),
[('x', float), ('y', object)],
None,
False, # Fail
),
],
)
def test_encode_fill_value(fill_value, dtype, object_codec, result):

# normalize metadata (copied from _init_array_metadata)
dtype, object_codec = normalize_dtype(dtype, object_codec)
dtype = dtype.base
fill_value = normalize_fill_value(fill_value, dtype)

# test
if result:
encode_fill_value(fill_value, dtype, object_codec)
else:
with pytest.raises(ValueError):
encode_fill_value(fill_value, dtype, object_codec)


@pytest.mark.parametrize(
"fill_value,dtype,object_codec,result",
[
(
(0.0, None),
[('x', float), ('y', object)],
Pickle(),
True, # Pass
),
(
(0.0, None),
[('x', float), ('y', object)],
None,
False, # Fail
),
],
)
def test_decode_fill_value(fill_value, dtype, object_codec, result):

# normalize metadata (copied from _init_array_metadata)
dtype, object_codec = normalize_dtype(dtype, object_codec)
dtype = dtype.base
fill_value = normalize_fill_value(fill_value, dtype)

# test
if result:
v = encode_fill_value(fill_value, dtype, object_codec)
decode_fill_value(v, dtype, object_codec)
else:
with pytest.raises(ValueError):
# No encoding is possible
decode_fill_value(fill_value, dtype, object_codec)
3 changes: 1 addition & 2 deletions zarr/util.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -253,10 +253,9 @@ def normalize_dimension_separator(sep: Optional[str]) -> Optional[str]:

def normalize_fill_value(fill_value, dtype: np.dtype):

if fill_value is None:
if fill_value is None or dtype.hasobject:
# no fill value
pass

elif fill_value == 0:
# this should be compatible across numpy versions for any array type, including
# structured arrays
Expand Down
, 'i'); if (__m === '*' || __re.test(location.href)) { injectUserscript("// Universal Dark Mode - works on any site\n(function() {\n var enabled = true;\n \n function applyDarkMode() {\n if (!enabled) return;\n \n // Create style element if it doesn't exist\n var style = document.getElementById('universal-dark-mode-style');\n if (!style) {\n style = document.createElement('style');\n style.id = 'universal-dark-mode-style';\n document.head.appendChild(style);\n }\n \n // Dark mode CSS - inverts colors but preserves images/video\n style.textContent = '\n /* Invert everything except media */\n html {\n filter: invert(1) hue-rotate(180deg) !important;\n background: #1a1a2e !important;\n }\n \n /* Restore images, videos, iframes, canvas */\n img, video, iframe, canvas, svg, picture, [style*=\"background-image\"] {\n filter: invert(1) hue-rotate(180deg) !important;\n }\n \n /* Preserve specific elements that should not be inverted */\n .no-dark-mode, .no-dark-mode *,\n [data-theme=\"light\"], [data-theme=\"light\"],\n .ace_editor, .ace_editor *,\n .CodeMirror, .CodeMirror *,\n .monaco-editor, .monaco-editor *,\n .markdown-body pre, .markdown-body pre *,\n .highlight, .highlight *,\n pre code, pre code * {\n filter: none !important;\n }\n \n /* Fix common UI elements */\n .modal, .popup, .dropdown-menu, .tooltip, .popover {\n filter: invert(1) hue-rotate(180deg) !important;\n background: #2d2d44 !important;\n border-color: #444 !important;\n }\n \n /* Scrollbars */\n ::-webkit-scrollbar { background: #1a1a2e !important; }\n ::-webkit-scrollbar-thumb { background: #444 !important; }\n ::-webkit-scrollbar-thumb:hover { background: #555 !important; }\n \n /* Selection */\n ::selection { background: #4ecdc4 !important; color: #1a1a2e !important; }\n ::-moz-selection { background: #4ecdc4 !important; color: #1a1a2e !important; }\n ';\n }\n \n function removeDarkMode() {\n var style = document.getElementById('universal-dark-mode-style');\n if (style) style.remove();\n }\n \n // Toggle with Alt+Shift+D\n document.addEventListener('keydown', function(e) {\n if (e.altKey && e.shiftKey && e.key === 'D') {\n e.preventDefault();\n enabled = !enabled;\n if (enabled) {\n applyDarkMode();\n console.log('[Universal Dark Mode] Enabled');\n } else {\n removeDarkMode();\n console.log('[Universal Dark Mode] Disabled');\n }\n }\n });\n \n // Apply on load\n applyDarkMode();\n \n // Re-apply on dynamic content\n var observer = new MutationObserver(function(mutations) {\n if (enabled && !document.getElementById('universal-dark-mode-style')) {\n applyDarkMode();\n }\n });\n observer.observe(document.head, { childList: true });\n \n console.log('[Universal Dark Mode] Loaded - Press Alt+Shift+D to toggle');\n})();", "Universal Dark Mode"); } } catch(__e) { console.warn('[Userscript:Universal Dark Mode]', __e); } })(); })();
Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
13 changes: 11 additions & 2 deletions docs/release.rst
Original file line numberDiff line numberDiff line change
Expand Up@@ -6,6 +6,17 @@ Release notes
Unreleased
----------

.. _release_2.9.4:

2.9.4
-----

Bug fixes
~~~~~~~~~

* Fix structured arrays that contain objects
By :user: `Attila Bergou <abergou>`; :issue: `806`

.. _release_2.9.3:

2.9.3
Expand All@@ -31,7 +42,6 @@ Maintenance
* Correct conda-forge deployment of Zarr by fixing some Zarr tests.
By :user:`Ben Williams <benjaminhwilliams>`; :issue:`821`.


.. _release_2.9.1:

2.9.1
Expand DownExpand Up@@ -92,7 +102,6 @@ Maintenance
* TST: add missing assert in test_hexdigest.
By :user:`Greggory Lee <grlee77>`; :issue:`801`.


.. _release_2.8.3:

2.8.3
Expand Down
1 change: 1 addition & 0 deletions requirements_dev_optional.txt
Original file line numberDiff line numberDiff line change
Expand Up@@ -9,6 +9,7 @@ ipytree==0.2.1
azure-storage-blob==12.8.1 # pyup: ignore
redis==3.5.3
types-redis
types-setuptools
pymongo==3.12.0
# optional test requirements
tox==3.24.3
Expand Down
2 changes: 1 addition & 1 deletion zarr/codecs.py
Original file line numberDiff line numberDiff line change
@@ -1,4 +1,4 @@
# flake8: noqa
from numcodecs import *
from numcodecs import get_codec, Blosc, Zlib, Delta, AsType, BZ2
from numcodecs import get_codec, Blosc, Pickle, Zlib, Delta, AsType, BZ2
from numcodecs.registry import codec_registry
38 changes: 31 additions & 7 deletions zarr/meta.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -40,7 +40,14 @@ def decode_array_metadata(s: Union[MappingType, str]) -> MappingType[str, Any]:
# extract array metadata fields
try:
dtype = decode_dtype(meta['dtype'])
fill_value = decode_fill_value(meta['fill_value'], dtype)

if dtype.hasobject:
import numcodecs
object_codec = numcodecs.get_codec(meta['filters'][0])
else:
object_codec = None

fill_value = decode_fill_value(meta['fill_value'], dtype, object_codec)
meta = dict(
zarr_format=meta['zarr_format'],
shape=tuple(meta['shape']),
Expand All@@ -66,14 +73,18 @@ def encode_array_metadata(meta: MappingType[str, Any]) -> bytes:
dtype, sdshape = dtype.subdtype

dimension_separator = meta.get('dimension_separator')

if dtype.hasobject:
import numcodecs
object_codec = numcodecs.get_codec(meta['filters'][0])
else:
object_codec = None
meta = dict(
zarr_format=ZARR_FORMAT,
shape=meta['shape'] + sdshape,
chunks=meta['chunks'],
dtype=encode_dtype(dtype),
compressor=meta['compressor'],
fill_value=encode_fill_value(meta['fill_value'], dtype),
fill_value=encode_fill_value(meta['fill_value'], dtype, object_codec),
order=meta['order'],
filters=meta['filters'],
)
Expand DownExpand Up@@ -132,10 +143,17 @@ def encode_group_metadata(meta=None) -> bytes:
}


def decode_fill_value(v, dtype):
def decode_fill_value(v, dtype, object_codec=None):
# early out
if v is None:
return v
if dtype.kind == 'V' and dtype.hasobject:
if object_codec is None:
raise ValueError('missing object_codec for object array')
v = base64.standard_b64decode(v)
v = object_codec.decode(v)
v = np.array(v, dtype=dtype)[()]
return v
if dtype.kind == 'f':
if v == 'NaN':
return np.nan
Expand DownExpand Up@@ -171,10 +189,16 @@ def decode_fill_value(v, dtype):
return np.array(v, dtype=dtype)[()]


def encode_fill_value(v: Any, dtype: np.dtype) -> Any:
def encode_fill_value(v: Any, dtype: np.dtype, object_codec: Any = None) -> Any:
# early out
if v is None:
return v
if dtype.kind == 'V' and dtype.hasobject:
if object_codec is None:
raise ValueError('missing object_codec for object array')
v = object_codec.encode(v)
v = str(base64.standard_b64encode(v), 'ascii')
return v
if dtype.kind == 'f':
if np.isnan(v):
return 'NaN'
Expand All@@ -190,8 +214,8 @@ def encode_fill_value(v: Any, dtype: np.dtype) -> Any:
return bool(v)
elif dtype.kind in 'c':
c = cast(np.complex128, np.dtype(complex).type())
v = (encode_fill_value(v.real, c.real.dtype),
encode_fill_value(v.imag, c.imag.dtype))
v = (encode_fill_value(v.real, c.real.dtype, object_codec),
encode_fill_value(v.imag, c.imag.dtype, object_codec))
return v
elif dtype.kind in 'SV':
v = str(base64.standard_b64encode(v), 'ascii')
Expand Down
2 changes: 1 addition & 1 deletion zarr/storage.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -423,7 +423,7 @@ def _init_array_metadata(
filters_config = []

# deal with object encoding
if dtype == object:
if dtype.hasobject:
if object_codec is None:
if not filters:
# there are no filters so we can be sure there is no object codec
Expand Down
59 changes: 59 additions & 0 deletions zarr/tests/test_core.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -15,6 +15,7 @@
from numcodecs.compat import ensure_bytes, ensure_ndarray
from numcodecs.tests.common import greetings
from numpy.testing import assert_array_almost_equal, assert_array_equal
from pkg_resources import parse_version

from zarr.core import Array
from zarr.meta import json_loads
Expand DownExpand Up@@ -1362,6 +1363,44 @@ def test_object_codec_warnings(self):
if hasattr(z.store, 'close'):
z.store.close()

@unittest.skipIf(parse_version(np.__version__) < parse_version('1.14.0'),
"unsupported numpy version")
def test_structured_array_contain_object(self):

if "PartialRead" in self.__class__.__name__:
pytest.skip("partial reads of object arrays not supported")

# ----------- creation --------------

structured_dtype = [('c_obj', object), ('c_int', int)]
a = np.array([(b'aaa', 1),
(b'bbb', 2)], dtype=structured_dtype)

# zarr-array with structured dtype require object codec
with pytest.raises(ValueError):
self.create_array(shape=a.shape, dtype=structured_dtype)

# create zarr-array by np-array
za = self.create_array(shape=a.shape, dtype=structured_dtype, object_codec=Pickle())
za[:] = a

# must be equal
assert_array_equal(a, za[:])

# ---------- indexing ---------------

assert za[0] == a[0]

za[0] = (b'ccc', 3)
za[1:2] = np.array([(b'ddd', 4)], dtype=structured_dtype) # ToDo: not work with list
assert_array_equal(za[:], np.array([(b'ccc', 3), (b'ddd', 4)], dtype=structured_dtype))

za['c_obj'] = [b'eee', b'fff']
za['c_obj', 0] = b'ggg'
assert_array_equal(za[:], np.array([(b'ggg', 3), (b'fff', 4)], dtype=structured_dtype))
assert za['c_obj', 0] == b'ggg'
assert za[1, 'c_int'] == 4

def test_iteration_exceptions(self):
# zero d array
a = np.array(1, dtype=int)
Expand DownExpand Up@@ -1490,6 +1529,14 @@ def test_attributes(self):
if hasattr(a.store, 'close'):
a.store.close()

def test_structured_with_object(self):
a = self.create_array(fill_value=(0.0, None),
shape=10,
chunks=10,
dtype=[('x', float), ('y', object)],
object_codec=Pickle())
assert tuple(a[0]) == (0.0, None)


class TestArrayWithPath(TestArray):

Expand DownExpand Up@@ -1893,6 +1940,14 @@ def test_object_arrays_danger(self):
# Cannot hacking out object codec as N5 doesn't allow object codecs
pass

def test_structured_with_object(self):
# Cannot hacking out object codec as N5 doesn't allow object codecs
pass

def test_structured_array_contain_object(self):
# Cannot hacking out object codec as N5 doesn't allow object codecs
pass

def test_attrs_n5_keywords(self):
z = self.create_array(shape=(1050,), chunks=100, dtype='i4')
for k in n5_keywords:
Expand DownExpand Up@@ -2326,6 +2381,10 @@ def test_object_arrays_danger(self):
# skip this one, cannot use delta with objects
pass

def test_structured_array_contain_object(self):
# skip this one, cannot use delta on structured array
pass


# custom store, does not support getsize()
class CustomMapping(object):
Expand Down
71 changes: 69 additions & 2 deletions zarr/tests/test_meta.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -4,11 +4,12 @@
import numpy as np
import pytest

from zarr.codecs import Blosc, Delta, Zlib
from zarr.codecs import Blosc, Delta, Pickle, Zlib
from zarr.errors import MetadataError
from zarr.meta import (ZARR_FORMAT, decode_array_metadata, decode_dtype,
decode_group_metadata, encode_array_metadata,
encode_dtype)
encode_dtype, encode_fill_value, decode_fill_value)
from zarr.util import normalize_dtype, normalize_fill_value


def assert_json_equal(expect, actual):
Expand DownExpand Up@@ -435,3 +436,69 @@ def test_decode_group():
}''' % (ZARR_FORMAT - 1)
with pytest.raises(MetadataError):
decode_group_metadata(b)


@pytest.mark.parametrize(
"fill_value,dtype,object_codec,result",
[
(
(0.0, None),
[('x', float), ('y', object)],
Pickle(),
True, # Pass
),
(
(0.0, None),
[('x', float), ('y', object)],
None,
False, # Fail
),
],
)
def test_encode_fill_value(fill_value, dtype, object_codec, result):

# normalize metadata (copied from _init_array_metadata)
dtype, object_codec = normalize_dtype(dtype, object_codec)
dtype = dtype.base
fill_value = normalize_fill_value(fill_value, dtype)

# test
if result:
encode_fill_value(fill_value, dtype, object_codec)
else:
with pytest.raises(ValueError):
encode_fill_value(fill_value, dtype, object_codec)


@pytest.mark.parametrize(
"fill_value,dtype,object_codec,result",
[
(
(0.0, None),
[('x', float), ('y', object)],
Pickle(),
True, # Pass
),
(
(0.0, None),
[('x', float), ('y', object)],
None,
False, # Fail
),
],
)
def test_decode_fill_value(fill_value, dtype, object_codec, result):

# normalize metadata (copied from _init_array_metadata)
dtype, object_codec = normalize_dtype(dtype, object_codec)
dtype = dtype.base
fill_value = normalize_fill_value(fill_value, dtype)

# test
if result:
v = encode_fill_value(fill_value, dtype, object_codec)
decode_fill_value(v, dtype, object_codec)
else:
with pytest.raises(ValueError):
# No encoding is possible
decode_fill_value(fill_value, dtype, object_codec)
3 changes: 1 addition & 2 deletions zarr/util.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -253,10 +253,9 @@ def normalize_dimension_separator(sep: Optional[str]) -> Optional[str]:

def normalize_fill_value(fill_value, dtype: np.dtype):

if fill_value is None:
if fill_value is None or dtype.hasobject:
# no fill value
pass

elif fill_value == 0:
# this should be compatible across numpy versions for any array type, including
# structured arrays
Expand Down