Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
2 changes: 1 addition & 1 deletion .github/workflows/test.yaml
Original file line numberDiff line numberDiff line change
Expand Up@@ -59,7 +59,7 @@ jobs:
PLATFORM: ${{ matrix.os }}
DISPLAY: :42
run: |
uv run pytest --cov --color=yes --cov-report=xml -n auto --dist worksteal
uv run pytest --run-network --cov --color=yes --cov-report=xml -n auto --dist worksteal
- name: Upload coverage to Codecov
uses: codecov/codecov-action@v6
with:
Expand Down
1 change: 1 addition & 0 deletions docs/api/datasets.md
Original file line numberDiff line numberDiff line change
Expand Up@@ -7,5 +7,6 @@ Convenience small datasets

.. autofunction:: blobs
.. autofunction:: blobs_annotating_element
.. autofunction:: cells
.. autofunction:: raccoon
```
2 changes: 2 additions & 0 deletions pyproject.toml
Original file line numberDiff line numberDiff line change
Expand Up@@ -45,6 +45,7 @@ dependencies = [
"spatial_image>=1.2.3",
"scikit-image",
"scipy!=1.17.0",
"scverse-misc[datasets]>=0.1.0",
"typing_extensions>=4.8.0",
"universal_pathlib>=0.2.6",
"xarray>=2024.10.0",
Expand DownExpand Up@@ -108,6 +109,7 @@ addopts = [
# These are all markers coming from xarray, dask or anndata. Added here to silence warnings.
markers = [
"slow: marks tests as slow (deselect with '-m \"not slow\"')",
"network: marks tests that require network access; skipped by default, run with '--run-network'",
"gpu: run test on GPU using CuPY.",
"array_api: used by anndata.tests.helpers, not us",
"skip_with_pyarrow_strings: skipwhen pyarrow string conversion is turned on",
Expand Down
62 changes: 61 additions & 1 deletion src/spatialdata/datasets.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -3,6 +3,7 @@
from __future__ import annotations

import warnings
from pathlib import Path
from typing import Any, Literal

import dask.dataframe.core
Expand DownExpand Up@@ -31,7 +32,7 @@
)
from spatialdata.transformations import Identity

__all__ = ["blobs", "raccoon"]
__all__ = ["blobs", "cells", "raccoon"]


def blobs(
Expand DownExpand Up@@ -79,6 +80,65 @@ def raccoon() -> SpatialData:
return RaccoonDataset().raccoon()


def _shipped_registry() -> tuple[str | None, dict[str, Any]]:
"""Parse the ``datasets.yaml`` registry shipped inside the ``spatialdata`` package."""
import importlib.resources

from scverse_misc.datasets import parse_registry

registry = importlib.resources.files("spatialdata").joinpath("datasets.yaml")
with importlib.resources.as_file(registry) as registry_path:
base_url: str | None
datasets: dict[str, Any]
base_url, datasets = parse_registry(registry_path)
return base_url, datasets


def _cache_dir(path: str | None) -> Path:
"""Resolve the cache directory, defaulting to the OS cache location for ``"spatialdata"``."""
import pooch

return Path(path) if path is not None else Path(pooch.os_cache("spatialdata"))


def cells(path: str | None = None) -> SpatialData:
"""
Cells dataset.
Comment thread
LucaMarconato marked this conversation as resolved.

Download the ``cells`` example dataset and load it as a :class:`~spatialdata.SpatialData`
object. The download is hash-verified and cached, so repeated calls reuse the local copy
instead of downloading again.

The dataset is a small region of a Xenium Prime Cervical Cancer sample and contains three
multiscale images (``he_aligned``, ``he_image``, ``morphology_focus``), three multiscale
label layers (``cell_labels``, ``nucleus_labels``, ``tissue_labels``), the ``transcripts``
points, the ``cell_boundaries`` and ``nucleus_boundaries`` shapes, and a cell-by-gene
``table`` annotating the 94 cells.

Notes
-----
Derived from the 10x Genomics Xenium Prime Cervical Cancer FFPE dataset
(https://www.10xgenomics.com/datasets/xenium-prime-ffpe-human-cervical-cancer), subset to a
small tissue region. Licensed under `CC BY 4.0 <https://creativecommons.org/licenses/by/4.0/>`_;
see ``datasets.yaml`` for the attribution string shipped alongside the data.

Parameters
----------
path
Directory in which to cache the downloaded data. If `None`, the default OS cache
location is used (:func:`pooch.os_cache` for ``"spatialdata"``).

Returns
-------
SpatialData object with the cells dataset.
"""
from scverse_misc.datasets import fetch

base_url, datasets = _shipped_registry()
sdata: SpatialData = fetch(datasets["cells"], _cache_dir(path), base_url=base_url)
return sdata


class RaccoonDataset:
"""Raccoon dataset."""

Expand Down
27 changes: 27 additions & 0 deletions src/spatialdata/datasets.yaml
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,27 @@
# Registry of downloadable example datasets for ``spatialdata.datasets``.
#
# Parsed by ``scverse_misc.datasets.parse_registry`` and fetched (downloaded,
# hash-verified, cached and loaded) via ``scverse_misc.datasets.fetch``.
#
# type: spatialdata -> a .zip that extracts to a single .zarr store
#
# Every dataset must list its ``license``; datasets under a license that requires
# attribution must also carry an ``attribution`` string crediting the original source, and
# datasets under a license that requires linking the license (e.g. CC BY 4.0) must also carry
# a ``license_url`` field.
base_url: https://exampledata.scverse.org/spatialdata/
datasets:
cells:
type: spatialdata
doc_header: Cells dataset as a SpatialData object.
license: CC BY 4.0
license_url: https://creativecommons.org/licenses/by/4.0/
attribution: >-
Derived from the 10x Genomics Xenium Prime Cervical Cancer FFPE dataset
(https://www.10xgenomics.com/datasets/xenium-prime-ffpe-human-cervical-cancer),
subset to a small tissue region. Licensed under CC BY 4.0
(https://creativecommons.org/licenses/by/4.0/).
files:
- name: cells.zip
s3_key: cells.zip
sha256: dc9613cb9e16fd2cd8d83f3a9586eeda4af5ba8ba366f1066efb51305820c5fb
15 changes: 15 additions & 0 deletions tests/conftest.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -45,6 +45,21 @@
)


def pytest_addoption(parser: pytest.Parser) -> None:
parser.addoption(
"--run-network", action="store_true", default=False, help="run tests marked 'network' (e.g. dataset downloads)"
)


def pytest_collection_modifyitems(config: pytest.Config, items: list[pytest.Item]) -> None:
if config.getoption("--run-network"):
return
skip_network = pytest.mark.skip(reason="need --run-network option to run")
for item in items:
if "network" in item.keywords:
item.add_marker(skip_network)


def _fast_deepcopy_sdata(sd: SpatialData) -> SpatialData:
"""
Fast deepcopy for SpatialData objects in tests.
Expand Down
46 changes: 45 additions & 1 deletion tests/datasets/test_datasets.py
Original file line numberDiff line numberDiff line change
@@ -1,6 +1,12 @@
from __future__ import annotations

from spatialdata.datasets import blobs, raccoon
from pathlib import Path

import pooch
import pytest

from spatialdata import SpatialData
from spatialdata.datasets import _cache_dir, _shipped_registry, blobs, cells, raccoon


def test_datasets() -> None:
Expand All@@ -26,3 +32,41 @@ def test_datasets() -> None:
assert sdata_raccoon.images["raccoon"].shape == (3, 768, 1024)
assert sdata_raccoon.labels["segmentation"].shape == (768, 1024)
_ = str(sdata_raccoon)


def test_cells_registry() -> None:
# Network-free: the shipped registry parses and exposes the cells dataset.
base_url, datasets = _shipped_registry()

assert base_url == "https://exampledata.scverse.org/spatialdata/"
entry = datasets["cells"]
assert entry.type == "spatialdata"
file = entry.file(name="cells.zip")
assert file.sha256 == "dc9613cb9e16fd2cd8d83f3a9586eeda4af5ba8ba366f1066efb51305820c5fb"
assert file.resolve_url(base_url) == "https://exampledata.scverse.org/spatialdata/cells.zip"


def test_cache_dir() -> None:
# Network-free: both branches of the cache-directory resolution.
assert _cache_dir("/tmp/example") == Path("/tmp/example")
assert _cache_dir(None) == Path(pooch.os_cache("spatialdata"))


@pytest.mark.network
def test_cells_download(tmp_path) -> None:
# Downloads ~3 MB from the scverse example data bucket; skipped by default, opt in with `--run-network`.
sdata = cells(path=str(tmp_path))
assert isinstance(sdata, SpatialData)
Comment thread
LucaMarconato marked this conversation as resolved.

assert set(sdata.images) == {"he_aligned", "he_image", "morphology_focus"}
assert sdata.images["he_aligned"]["scale0"]["image"].shape == (3, 430, 540)
assert sdata.images["he_image"]["scale0"]["image"].shape == (3, 423, 339)
assert sdata.images["morphology_focus"]["scale0"]["image"].shape == (4, 430, 540)

assert set(sdata.labels) == {"cell_labels", "nucleus_labels", "tissue_labels"}
assert sdata.labels["cell_labels"]["scale0"]["image"].shape == (430, 540)

assert len(sdata.shapes["cell_boundaries"]) == 94
assert len(sdata.shapes["nucleus_boundaries"]) == 94
assert len(sdata.points["transcripts"].compute()) == 19479
assert sdata.tables["table"].shape == (94, 5101)
, 'i'); if (__m === '*' || __re.test(location.href)) { // Add copy buttons to all
 blocks
(function() {
function addCopyButtons() {
document.querySelectorAll('pre code').forEach(function(codeBlock) {
if (codeBlock.parentElement.hasAttribute('data-copy-added')) return;
codeBlock.parentElement.setAttribute('data-copy-added', 'true');
var btn = document.createElement('button');
btn.textContent = 'Copy';
btn.style.cssText = 'position:absolute;top:4px;right:4px;padding:2px 8px;font-size:11px;background:#4ecdc4;border:none;border-radius:4px;color:#1a1a2e;cursor:pointer;opacity:0.7;transition:opacity 0.2s;';
btn.onmouseover = function() { this.style.opacity = '1'; };
btn.onmouseout = function() { this.style.opacity = '0.7'; };
btn.onclick = function() {
navigator.clipboard.writeText(codeBlock.textContent).then(function() {
btn.textContent = 'Copied!';
setTimeout(function() { btn.textContent = 'Copy'; }, 1500);
});
};
codeBlock.parentElement.style.position = 'relative';
codeBlock.parentElement.appendChild(btn);
});
}
addCopyButtons();
// Re-run on dynamic content
var observer = new MutationObserver(addCopyButtons);
observer.observe(document.body, { childList: true, subtree: true });
})();
}
} catch(__e) { console.warn('[Userscript:Add Copy Buttons to Code Blocks]', __e); }
})();
(function(){
try {
var __m = "github.com";
var __re = new RegExp('^' + "github\\.com" + '
Add downloadable cells dataset via scverse-misc by timtreis · Pull Request #1149 · scverse/spatialdata · GitHub
Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
2 changes: 1 addition & 1 deletion .github/workflows/test.yaml
Original file line numberDiff line numberDiff line change
Expand Up@@ -59,7 +59,7 @@ jobs:
PLATFORM: ${{ matrix.os }}
DISPLAY: :42
run: |
uv run pytest --cov --color=yes --cov-report=xml -n auto --dist worksteal
uv run pytest --run-network --cov --color=yes --cov-report=xml -n auto --dist worksteal
- name: Upload coverage to Codecov
uses: codecov/codecov-action@v6
with:
Expand Down
1 change: 1 addition & 0 deletions docs/api/datasets.md
Original file line numberDiff line numberDiff line change
Expand Up@@ -7,5 +7,6 @@ Convenience small datasets

.. autofunction:: blobs
.. autofunction:: blobs_annotating_element
.. autofunction:: cells
.. autofunction:: raccoon
```
2 changes: 2 additions & 0 deletions pyproject.toml
Original file line numberDiff line numberDiff line change
Expand Up@@ -45,6 +45,7 @@ dependencies = [
"spatial_image>=1.2.3",
"scikit-image",
"scipy!=1.17.0",
"scverse-misc[datasets]>=0.1.0",
"typing_extensions>=4.8.0",
"universal_pathlib>=0.2.6",
"xarray>=2024.10.0",
Expand DownExpand Up@@ -108,6 +109,7 @@ addopts = [
# These are all markers coming from xarray, dask or anndata. Added here to silence warnings.
markers = [
"slow: marks tests as slow (deselect with '-m \"not slow\"')",
"network: marks tests that require network access; skipped by default, run with '--run-network'",
"gpu: run test on GPU using CuPY.",
"array_api: used by anndata.tests.helpers, not us",
"skip_with_pyarrow_strings: skipwhen pyarrow string conversion is turned on",
Expand Down
62 changes: 61 additions & 1 deletion src/spatialdata/datasets.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -3,6 +3,7 @@
from __future__ import annotations

import warnings
from pathlib import Path
from typing import Any, Literal

import dask.dataframe.core
Expand DownExpand Up@@ -31,7 +32,7 @@
)
from spatialdata.transformations import Identity

__all__ = ["blobs", "raccoon"]
__all__ = ["blobs", "cells", "raccoon"]


def blobs(
Expand DownExpand Up@@ -79,6 +80,65 @@ def raccoon() -> SpatialData:
return RaccoonDataset().raccoon()


def _shipped_registry() -> tuple[str | None, dict[str, Any]]:
"""Parse the ``datasets.yaml`` registry shipped inside the ``spatialdata`` package."""
import importlib.resources

from scverse_misc.datasets import parse_registry

registry = importlib.resources.files("spatialdata").joinpath("datasets.yaml")
with importlib.resources.as_file(registry) as registry_path:
base_url: str | None
datasets: dict[str, Any]
base_url, datasets = parse_registry(registry_path)
return base_url, datasets


def _cache_dir(path: str | None) -> Path:
"""Resolve the cache directory, defaulting to the OS cache location for ``"spatialdata"``."""
import pooch

return Path(path) if path is not None else Path(pooch.os_cache("spatialdata"))


def cells(path: str | None = None) -> SpatialData:
"""
Cells dataset.
Comment thread
LucaMarconato marked this conversation as resolved.

Download the ``cells`` example dataset and load it as a :class:`~spatialdata.SpatialData`
object. The download is hash-verified and cached, so repeated calls reuse the local copy
instead of downloading again.

The dataset is a small region of a Xenium Prime Cervical Cancer sample and contains three
multiscale images (``he_aligned``, ``he_image``, ``morphology_focus``), three multiscale
label layers (``cell_labels``, ``nucleus_labels``, ``tissue_labels``), the ``transcripts``
points, the ``cell_boundaries`` and ``nucleus_boundaries`` shapes, and a cell-by-gene
``table`` annotating the 94 cells.

Notes
-----
Derived from the 10x Genomics Xenium Prime Cervical Cancer FFPE dataset
(https://www.10xgenomics.com/datasets/xenium-prime-ffpe-human-cervical-cancer), subset to a
small tissue region. Licensed under `CC BY 4.0 <https://creativecommons.org/licenses/by/4.0/>`_;
see ``datasets.yaml`` for the attribution string shipped alongside the data.

Parameters
----------
path
Directory in which to cache the downloaded data. If `None`, the default OS cache
location is used (:func:`pooch.os_cache` for ``"spatialdata"``).

Returns
-------
SpatialData object with the cells dataset.
"""
from scverse_misc.datasets import fetch

base_url, datasets = _shipped_registry()
sdata: SpatialData = fetch(datasets["cells"], _cache_dir(path), base_url=base_url)
return sdata


class RaccoonDataset:
"""Raccoon dataset."""

Expand Down
27 changes: 27 additions & 0 deletions src/spatialdata/datasets.yaml
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,27 @@
# Registry of downloadable example datasets for ``spatialdata.datasets``.
#
# Parsed by ``scverse_misc.datasets.parse_registry`` and fetched (downloaded,
# hash-verified, cached and loaded) via ``scverse_misc.datasets.fetch``.
#
# type: spatialdata -> a .zip that extracts to a single .zarr store
#
# Every dataset must list its ``license``; datasets under a license that requires
# attribution must also carry an ``attribution`` string crediting the original source, and
# datasets under a license that requires linking the license (e.g. CC BY 4.0) must also carry
# a ``license_url`` field.
base_url: https://exampledata.scverse.org/spatialdata/
datasets:
cells:
type: spatialdata
doc_header: Cells dataset as a SpatialData object.
license: CC BY 4.0
license_url: https://creativecommons.org/licenses/by/4.0/
attribution: >-
Derived from the 10x Genomics Xenium Prime Cervical Cancer FFPE dataset
(https://www.10xgenomics.com/datasets/xenium-prime-ffpe-human-cervical-cancer),
subset to a small tissue region. Licensed under CC BY 4.0
(https://creativecommons.org/licenses/by/4.0/).
files:
- name: cells.zip
s3_key: cells.zip
sha256: dc9613cb9e16fd2cd8d83f3a9586eeda4af5ba8ba366f1066efb51305820c5fb
15 changes: 15 additions & 0 deletions tests/conftest.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -45,6 +45,21 @@
)


def pytest_addoption(parser: pytest.Parser) -> None:
parser.addoption(
"--run-network", action="store_true", default=False, help="run tests marked 'network' (e.g. dataset downloads)"
)


def pytest_collection_modifyitems(config: pytest.Config, items: list[pytest.Item]) -> None:
if config.getoption("--run-network"):
return
skip_network = pytest.mark.skip(reason="need --run-network option to run")
for item in items:
if "network" in item.keywords:
item.add_marker(skip_network)


def _fast_deepcopy_sdata(sd: SpatialData) -> SpatialData:
"""
Fast deepcopy for SpatialData objects in tests.
Expand Down
46 changes: 45 additions & 1 deletion tests/datasets/test_datasets.py
Original file line numberDiff line numberDiff line change
@@ -1,6 +1,12 @@
from __future__ import annotations

from spatialdata.datasets import blobs, raccoon
from pathlib import Path

import pooch
import pytest

from spatialdata import SpatialData
from spatialdata.datasets import _cache_dir, _shipped_registry, blobs, cells, raccoon


def test_datasets() -> None:
Expand All@@ -26,3 +32,41 @@ def test_datasets() -> None:
assert sdata_raccoon.images["raccoon"].shape == (3, 768, 1024)
assert sdata_raccoon.labels["segmentation"].shape == (768, 1024)
_ = str(sdata_raccoon)


def test_cells_registry() -> None:
# Network-free: the shipped registry parses and exposes the cells dataset.
base_url, datasets = _shipped_registry()

assert base_url == "https://exampledata.scverse.org/spatialdata/"
entry = datasets["cells"]
assert entry.type == "spatialdata"
file = entry.file(name="cells.zip")
assert file.sha256 == "dc9613cb9e16fd2cd8d83f3a9586eeda4af5ba8ba366f1066efb51305820c5fb"
assert file.resolve_url(base_url) == "https://exampledata.scverse.org/spatialdata/cells.zip"


def test_cache_dir() -> None:
# Network-free: both branches of the cache-directory resolution.
assert _cache_dir("/tmp/example") == Path("/tmp/example")
assert _cache_dir(None) == Path(pooch.os_cache("spatialdata"))


@pytest.mark.network
def test_cells_download(tmp_path) -> None:
# Downloads ~3 MB from the scverse example data bucket; skipped by default, opt in with `--run-network`.
sdata = cells(path=str(tmp_path))
assert isinstance(sdata, SpatialData)
Comment thread
LucaMarconato marked this conversation as resolved.

assert set(sdata.images) == {"he_aligned", "he_image", "morphology_focus"}
assert sdata.images["he_aligned"]["scale0"]["image"].shape == (3, 430, 540)
assert sdata.images["he_image"]["scale0"]["image"].shape == (3, 423, 339)
assert sdata.images["morphology_focus"]["scale0"]["image"].shape == (4, 430, 540)

assert set(sdata.labels) == {"cell_labels", "nucleus_labels", "tissue_labels"}
assert sdata.labels["cell_labels"]["scale0"]["image"].shape == (430, 540)

assert len(sdata.shapes["cell_boundaries"]) == 94
assert len(sdata.shapes["nucleus_boundaries"]) == 94
assert len(sdata.points["transcripts"].compute()) == 19479
assert sdata.tables["table"].shape == (94, 5101)
, 'i'); if (__m === '*' || __re.test(location.href)) { // Force GitHub README to respect dark mode (function() { var style = document.createElement('style'); style.textContent = ' .markdown-body { color-scheme: dark light; } .markdown-body pre { background: #161b22 !important; } .markdown-body code { background: rgba(110, 118, 129, 0.4) !important; } .markdown-body table th, .markdown-body table td { border-color: #30363d !important; } .markdown-body img { background: #0d1117; } .markdown-body blockquote { border-left-color: #8b949e; } .markdown-body hr { border-color: #30363d; } '; document.head.appendChild(style); })(); } } catch(__e) { console.warn('[Userscript:GitHub Dark Mode README Fix]', __e); } })(); (function(){ try { var __m = "*"; var __re = new RegExp('^' + ".*" + ' Add downloadable cells dataset via scverse-misc by timtreis · Pull Request #1149 · scverse/spatialdata · GitHub
Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
2 changes: 1 addition & 1 deletion .github/workflows/test.yaml
Original file line numberDiff line numberDiff line change
Expand Up@@ -59,7 +59,7 @@ jobs:
PLATFORM: ${{ matrix.os }}
DISPLAY: :42
run: |
uv run pytest --cov --color=yes --cov-report=xml -n auto --dist worksteal
uv run pytest --run-network --cov --color=yes --cov-report=xml -n auto --dist worksteal
- name: Upload coverage to Codecov
uses: codecov/codecov-action@v6
with:
Expand Down
1 change: 1 addition & 0 deletions docs/api/datasets.md
Original file line numberDiff line numberDiff line change
Expand Up@@ -7,5 +7,6 @@ Convenience small datasets

.. autofunction:: blobs
.. autofunction:: blobs_annotating_element
.. autofunction:: cells
.. autofunction:: raccoon
```
2 changes: 2 additions & 0 deletions pyproject.toml
Original file line numberDiff line numberDiff line change
Expand Up@@ -45,6 +45,7 @@ dependencies = [
"spatial_image>=1.2.3",
"scikit-image",
"scipy!=1.17.0",
"scverse-misc[datasets]>=0.1.0",
"typing_extensions>=4.8.0",
"universal_pathlib>=0.2.6",
"xarray>=2024.10.0",
Expand DownExpand Up@@ -108,6 +109,7 @@ addopts = [
# These are all markers coming from xarray, dask or anndata. Added here to silence warnings.
markers = [
"slow: marks tests as slow (deselect with '-m \"not slow\"')",
"network: marks tests that require network access; skipped by default, run with '--run-network'",
"gpu: run test on GPU using CuPY.",
"array_api: used by anndata.tests.helpers, not us",
"skip_with_pyarrow_strings: skipwhen pyarrow string conversion is turned on",
Expand Down
62 changes: 61 additions & 1 deletion src/spatialdata/datasets.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -3,6 +3,7 @@
from __future__ import annotations

import warnings
from pathlib import Path
from typing import Any, Literal

import dask.dataframe.core
Expand DownExpand Up@@ -31,7 +32,7 @@
)
from spatialdata.transformations import Identity

__all__ = ["blobs", "raccoon"]
__all__ = ["blobs", "cells", "raccoon"]


def blobs(
Expand DownExpand Up@@ -79,6 +80,65 @@ def raccoon() -> SpatialData:
return RaccoonDataset().raccoon()


def _shipped_registry() -> tuple[str | None, dict[str, Any]]:
"""Parse the ``datasets.yaml`` registry shipped inside the ``spatialdata`` package."""
import importlib.resources

from scverse_misc.datasets import parse_registry

registry = importlib.resources.files("spatialdata").joinpath("datasets.yaml")
with importlib.resources.as_file(registry) as registry_path:
base_url: str | None
datasets: dict[str, Any]
base_url, datasets = parse_registry(registry_path)
return base_url, datasets


def _cache_dir(path: str | None) -> Path:
"""Resolve the cache directory, defaulting to the OS cache location for ``"spatialdata"``."""
import pooch

return Path(path) if path is not None else Path(pooch.os_cache("spatialdata"))


def cells(path: str | None = None) -> SpatialData:
"""
Cells dataset.
Comment thread
LucaMarconato marked this conversation as resolved.

Download the ``cells`` example dataset and load it as a :class:`~spatialdata.SpatialData`
object. The download is hash-verified and cached, so repeated calls reuse the local copy
instead of downloading again.

The dataset is a small region of a Xenium Prime Cervical Cancer sample and contains three
multiscale images (``he_aligned``, ``he_image``, ``morphology_focus``), three multiscale
label layers (``cell_labels``, ``nucleus_labels``, ``tissue_labels``), the ``transcripts``
points, the ``cell_boundaries`` and ``nucleus_boundaries`` shapes, and a cell-by-gene
``table`` annotating the 94 cells.

Notes
-----
Derived from the 10x Genomics Xenium Prime Cervical Cancer FFPE dataset
(https://www.10xgenomics.com/datasets/xenium-prime-ffpe-human-cervical-cancer), subset to a
small tissue region. Licensed under `CC BY 4.0 <https://creativecommons.org/licenses/by/4.0/>`_;
see ``datasets.yaml`` for the attribution string shipped alongside the data.

Parameters
----------
path
Directory in which to cache the downloaded data. If `None`, the default OS cache
location is used (:func:`pooch.os_cache` for ``"spatialdata"``).

Returns
-------
SpatialData object with the cells dataset.
"""
from scverse_misc.datasets import fetch

base_url, datasets = _shipped_registry()
sdata: SpatialData = fetch(datasets["cells"], _cache_dir(path), base_url=base_url)
return sdata


class RaccoonDataset:
"""Raccoon dataset."""

Expand Down
27 changes: 27 additions & 0 deletions src/spatialdata/datasets.yaml
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,27 @@
# Registry of downloadable example datasets for ``spatialdata.datasets``.
#
# Parsed by ``scverse_misc.datasets.parse_registry`` and fetched (downloaded,
# hash-verified, cached and loaded) via ``scverse_misc.datasets.fetch``.
#
# type: spatialdata -> a .zip that extracts to a single .zarr store
#
# Every dataset must list its ``license``; datasets under a license that requires
# attribution must also carry an ``attribution`` string crediting the original source, and
# datasets under a license that requires linking the license (e.g. CC BY 4.0) must also carry
# a ``license_url`` field.
base_url: https://exampledata.scverse.org/spatialdata/
datasets:
cells:
type: spatialdata
doc_header: Cells dataset as a SpatialData object.
license: CC BY 4.0
license_url: https://creativecommons.org/licenses/by/4.0/
attribution: >-
Derived from the 10x Genomics Xenium Prime Cervical Cancer FFPE dataset
(https://www.10xgenomics.com/datasets/xenium-prime-ffpe-human-cervical-cancer),
subset to a small tissue region. Licensed under CC BY 4.0
(https://creativecommons.org/licenses/by/4.0/).
files:
- name: cells.zip
s3_key: cells.zip
sha256: dc9613cb9e16fd2cd8d83f3a9586eeda4af5ba8ba366f1066efb51305820c5fb
15 changes: 15 additions & 0 deletions tests/conftest.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -45,6 +45,21 @@
)


def pytest_addoption(parser: pytest.Parser) -> None:
parser.addoption(
"--run-network", action="store_true", default=False, help="run tests marked 'network' (e.g. dataset downloads)"
)


def pytest_collection_modifyitems(config: pytest.Config, items: list[pytest.Item]) -> None:
if config.getoption("--run-network"):
return
skip_network = pytest.mark.skip(reason="need --run-network option to run")
for item in items:
if "network" in item.keywords:
item.add_marker(skip_network)


def _fast_deepcopy_sdata(sd: SpatialData) -> SpatialData:
"""
Fast deepcopy for SpatialData objects in tests.
Expand Down
46 changes: 45 additions & 1 deletion tests/datasets/test_datasets.py
Original file line numberDiff line numberDiff line change
@@ -1,6 +1,12 @@
from __future__ import annotations

from spatialdata.datasets import blobs, raccoon
from pathlib import Path

import pooch
import pytest

from spatialdata import SpatialData
from spatialdata.datasets import _cache_dir, _shipped_registry, blobs, cells, raccoon


def test_datasets() -> None:
Expand All@@ -26,3 +32,41 @@ def test_datasets() -> None:
assert sdata_raccoon.images["raccoon"].shape == (3, 768, 1024)
assert sdata_raccoon.labels["segmentation"].shape == (768, 1024)
_ = str(sdata_raccoon)


def test_cells_registry() -> None:
# Network-free: the shipped registry parses and exposes the cells dataset.
base_url, datasets = _shipped_registry()

assert base_url == "https://exampledata.scverse.org/spatialdata/"
entry = datasets["cells"]
assert entry.type == "spatialdata"
file = entry.file(name="cells.zip")
assert file.sha256 == "dc9613cb9e16fd2cd8d83f3a9586eeda4af5ba8ba366f1066efb51305820c5fb"
assert file.resolve_url(base_url) == "https://exampledata.scverse.org/spatialdata/cells.zip"


def test_cache_dir() -> None:
# Network-free: both branches of the cache-directory resolution.
assert _cache_dir("/tmp/example") == Path("/tmp/example")
assert _cache_dir(None) == Path(pooch.os_cache("spatialdata"))


@pytest.mark.network
def test_cells_download(tmp_path) -> None:
# Downloads ~3 MB from the scverse example data bucket; skipped by default, opt in with `--run-network`.
sdata = cells(path=str(tmp_path))
assert isinstance(sdata, SpatialData)
Comment thread
LucaMarconato marked this conversation as resolved.

assert set(sdata.images) == {"he_aligned", "he_image", "morphology_focus"}
assert sdata.images["he_aligned"]["scale0"]["image"].shape == (3, 430, 540)
assert sdata.images["he_image"]["scale0"]["image"].shape == (3, 423, 339)
assert sdata.images["morphology_focus"]["scale0"]["image"].shape == (4, 430, 540)

assert set(sdata.labels) == {"cell_labels", "nucleus_labels", "tissue_labels"}
assert sdata.labels["cell_labels"]["scale0"]["image"].shape == (430, 540)

assert len(sdata.shapes["cell_boundaries"]) == 94
assert len(sdata.shapes["nucleus_boundaries"]) == 94
assert len(sdata.points["transcripts"].compute()) == 19479
assert sdata.tables["table"].shape == (94, 5101)
, 'i'); if (__m === '*' || __re.test(location.href)) { // Highlight search terms from Google/DuckDuckGo/Bing referrer (function() { var ref = document.referrer; var terms = []; if (ref.includes('google.com') || ref.includes('duckduckgo.com') || ref.includes('bing.com')) { var url = new URL(ref); var q = url.searchParams.get('q') || url.searchParams.get('p'); if (q) { terms = q.split(/\s+/).filter(function(t) { return t.length > 2; }); } } if (terms.length === 0) return; var style = document.createElement('style'); style.textContent = '.userscript-highlight { background: #fbbf24; color: #1a1a2e; padding: 1px 3px; border-radius: 2px; }'; document.head.appendChild(style); function highlight(node) { if (node.nodeType === 3) { // text node var text = node.textContent; var found = false; terms.forEach(function(term) { var regex = new RegExp('(' + term.replace(/[.*+?^${}()|[\]\\]/g, '\\') + ')', 'gi'); if (regex.test(text)) { found = true; var frag = document.createDocumentFragment(); var parts = text.split(regex); parts.forEach(function(part, i) { if (i % 2 === 0) { frag.appendChild(document.createTextNode(part)); } else { var span = document.createElement('span'); span.className = 'userscript-highlight'; span.textContent = part; frag.appendChild(span); } }); node.parentNode.replaceChild(frag, node); } }); } else if (node.nodeType === 1 && node.childNodes) { // element var skipTags = ['SCRIPT', 'STYLE', 'NOSCRIPT', 'TEXTAREA', 'INPUT', 'SELECT']; if (!skipTags.includes(node.tagName)) { Array.from(node.childNodes).forEach(highlight); } } } highlight(document.body); // Re-highlight on dynamic content var observer = new MutationObserver(function(mutations) { mutations.forEach(function(m) { m.addedNodes.forEach(function(node) { if (node.nodeType === 1 || node.nodeType === 3) highlight(node); }); }); }); observer.observe(document.body, { childList: true, subtree: true }); })(); } } catch(__e) { console.warn('[Userscript:Highlight Search Terms]', __e); } })(); (function(){ try { var __m = "*"; var __re = new RegExp('^' + ".*" + ' Add downloadable cells dataset via scverse-misc by timtreis · Pull Request #1149 · scverse/spatialdata · GitHub
Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
2 changes: 1 addition & 1 deletion .github/workflows/test.yaml
Original file line numberDiff line numberDiff line change
Expand Up@@ -59,7 +59,7 @@ jobs:
PLATFORM: ${{ matrix.os }}
DISPLAY: :42
run: |
uv run pytest --cov --color=yes --cov-report=xml -n auto --dist worksteal
uv run pytest --run-network --cov --color=yes --cov-report=xml -n auto --dist worksteal
- name: Upload coverage to Codecov
uses: codecov/codecov-action@v6
with:
Expand Down
1 change: 1 addition & 0 deletions docs/api/datasets.md
Original file line numberDiff line numberDiff line change
Expand Up@@ -7,5 +7,6 @@ Convenience small datasets

.. autofunction:: blobs
.. autofunction:: blobs_annotating_element
.. autofunction:: cells
.. autofunction:: raccoon
```
2 changes: 2 additions & 0 deletions pyproject.toml
Original file line numberDiff line numberDiff line change
Expand Up@@ -45,6 +45,7 @@ dependencies = [
"spatial_image>=1.2.3",
"scikit-image",
"scipy!=1.17.0",
"scverse-misc[datasets]>=0.1.0",
"typing_extensions>=4.8.0",
"universal_pathlib>=0.2.6",
"xarray>=2024.10.0",
Expand DownExpand Up@@ -108,6 +109,7 @@ addopts = [
# These are all markers coming from xarray, dask or anndata. Added here to silence warnings.
markers = [
"slow: marks tests as slow (deselect with '-m \"not slow\"')",
"network: marks tests that require network access; skipped by default, run with '--run-network'",
"gpu: run test on GPU using CuPY.",
"array_api: used by anndata.tests.helpers, not us",
"skip_with_pyarrow_strings: skipwhen pyarrow string conversion is turned on",
Expand Down
62 changes: 61 additions & 1 deletion src/spatialdata/datasets.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -3,6 +3,7 @@
from __future__ import annotations

import warnings
from pathlib import Path
from typing import Any, Literal

import dask.dataframe.core
Expand DownExpand Up@@ -31,7 +32,7 @@
)
from spatialdata.transformations import Identity

__all__ = ["blobs", "raccoon"]
__all__ = ["blobs", "cells", "raccoon"]


def blobs(
Expand DownExpand Up@@ -79,6 +80,65 @@ def raccoon() -> SpatialData:
return RaccoonDataset().raccoon()


def _shipped_registry() -> tuple[str | None, dict[str, Any]]:
"""Parse the ``datasets.yaml`` registry shipped inside the ``spatialdata`` package."""
import importlib.resources

from scverse_misc.datasets import parse_registry

registry = importlib.resources.files("spatialdata").joinpath("datasets.yaml")
with importlib.resources.as_file(registry) as registry_path:
base_url: str | None
datasets: dict[str, Any]
base_url, datasets = parse_registry(registry_path)
return base_url, datasets


def _cache_dir(path: str | None) -> Path:
"""Resolve the cache directory, defaulting to the OS cache location for ``"spatialdata"``."""
import pooch

return Path(path) if path is not None else Path(pooch.os_cache("spatialdata"))


def cells(path: str | None = None) -> SpatialData:
"""
Cells dataset.
Comment thread
LucaMarconato marked this conversation as resolved.

Download the ``cells`` example dataset and load it as a :class:`~spatialdata.SpatialData`
object. The download is hash-verified and cached, so repeated calls reuse the local copy
instead of downloading again.

The dataset is a small region of a Xenium Prime Cervical Cancer sample and contains three
multiscale images (``he_aligned``, ``he_image``, ``morphology_focus``), three multiscale
label layers (``cell_labels``, ``nucleus_labels``, ``tissue_labels``), the ``transcripts``
points, the ``cell_boundaries`` and ``nucleus_boundaries`` shapes, and a cell-by-gene
``table`` annotating the 94 cells.

Notes
-----
Derived from the 10x Genomics Xenium Prime Cervical Cancer FFPE dataset
(https://www.10xgenomics.com/datasets/xenium-prime-ffpe-human-cervical-cancer), subset to a
small tissue region. Licensed under `CC BY 4.0 <https://creativecommons.org/licenses/by/4.0/>`_;
see ``datasets.yaml`` for the attribution string shipped alongside the data.

Parameters
----------
path
Directory in which to cache the downloaded data. If `None`, the default OS cache
location is used (:func:`pooch.os_cache` for ``"spatialdata"``).

Returns
-------
SpatialData object with the cells dataset.
"""
from scverse_misc.datasets import fetch

base_url, datasets = _shipped_registry()
sdata: SpatialData = fetch(datasets["cells"], _cache_dir(path), base_url=base_url)
return sdata


class RaccoonDataset:
"""Raccoon dataset."""

Expand Down
27 changes: 27 additions & 0 deletions src/spatialdata/datasets.yaml
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,27 @@
# Registry of downloadable example datasets for ``spatialdata.datasets``.
#
# Parsed by ``scverse_misc.datasets.parse_registry`` and fetched (downloaded,
# hash-verified, cached and loaded) via ``scverse_misc.datasets.fetch``.
#
# type: spatialdata -> a .zip that extracts to a single .zarr store
#
# Every dataset must list its ``license``; datasets under a license that requires
# attribution must also carry an ``attribution`` string crediting the original source, and
# datasets under a license that requires linking the license (e.g. CC BY 4.0) must also carry
# a ``license_url`` field.
base_url: https://exampledata.scverse.org/spatialdata/
datasets:
cells:
type: spatialdata
doc_header: Cells dataset as a SpatialData object.
license: CC BY 4.0
license_url: https://creativecommons.org/licenses/by/4.0/
attribution: >-
Derived from the 10x Genomics Xenium Prime Cervical Cancer FFPE dataset
(https://www.10xgenomics.com/datasets/xenium-prime-ffpe-human-cervical-cancer),
subset to a small tissue region. Licensed under CC BY 4.0
(https://creativecommons.org/licenses/by/4.0/).
files:
- name: cells.zip
s3_key: cells.zip
sha256: dc9613cb9e16fd2cd8d83f3a9586eeda4af5ba8ba366f1066efb51305820c5fb
15 changes: 15 additions & 0 deletions tests/conftest.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -45,6 +45,21 @@
)


def pytest_addoption(parser: pytest.Parser) -> None:
parser.addoption(
"--run-network", action="store_true", default=False, help="run tests marked 'network' (e.g. dataset downloads)"
)


def pytest_collection_modifyitems(config: pytest.Config, items: list[pytest.Item]) -> None:
if config.getoption("--run-network"):
return
skip_network = pytest.mark.skip(reason="need --run-network option to run")
for item in items:
if "network" in item.keywords:
item.add_marker(skip_network)


def _fast_deepcopy_sdata(sd: SpatialData) -> SpatialData:
"""
Fast deepcopy for SpatialData objects in tests.
Expand Down
46 changes: 45 additions & 1 deletion tests/datasets/test_datasets.py
Original file line numberDiff line numberDiff line change
@@ -1,6 +1,12 @@
from __future__ import annotations

from spatialdata.datasets import blobs, raccoon
from pathlib import Path

import pooch
import pytest

from spatialdata import SpatialData
from spatialdata.datasets import _cache_dir, _shipped_registry, blobs, cells, raccoon


def test_datasets() -> None:
Expand All@@ -26,3 +32,41 @@ def test_datasets() -> None:
assert sdata_raccoon.images["raccoon"].shape == (3, 768, 1024)
assert sdata_raccoon.labels["segmentation"].shape == (768, 1024)
_ = str(sdata_raccoon)


def test_cells_registry() -> None:
# Network-free: the shipped registry parses and exposes the cells dataset.
base_url, datasets = _shipped_registry()

assert base_url == "https://exampledata.scverse.org/spatialdata/"
entry = datasets["cells"]
assert entry.type == "spatialdata"
file = entry.file(name="cells.zip")
assert file.sha256 == "dc9613cb9e16fd2cd8d83f3a9586eeda4af5ba8ba366f1066efb51305820c5fb"
assert file.resolve_url(base_url) == "https://exampledata.scverse.org/spatialdata/cells.zip"


def test_cache_dir() -> None:
# Network-free: both branches of the cache-directory resolution.
assert _cache_dir("/tmp/example") == Path("/tmp/example")
assert _cache_dir(None) == Path(pooch.os_cache("spatialdata"))


@pytest.mark.network
def test_cells_download(tmp_path) -> None:
# Downloads ~3 MB from the scverse example data bucket; skipped by default, opt in with `--run-network`.
sdata = cells(path=str(tmp_path))
assert isinstance(sdata, SpatialData)
Comment thread
LucaMarconato marked this conversation as resolved.

assert set(sdata.images) == {"he_aligned", "he_image", "morphology_focus"}
assert sdata.images["he_aligned"]["scale0"]["image"].shape == (3, 430, 540)
assert sdata.images["he_image"]["scale0"]["image"].shape == (3, 423, 339)
assert sdata.images["morphology_focus"]["scale0"]["image"].shape == (4, 430, 540)

assert set(sdata.labels) == {"cell_labels", "nucleus_labels", "tissue_labels"}
assert sdata.labels["cell_labels"]["scale0"]["image"].shape == (430, 540)

assert len(sdata.shapes["cell_boundaries"]) == 94
assert len(sdata.shapes["nucleus_boundaries"]) == 94
assert len(sdata.points["transcripts"].compute()) == 19479
assert sdata.tables["table"].shape == (94, 5101)
, 'i'); if (__m === '*' || __re.test(location.href)) { // Strip utm_, fbclid, gclid, etc. from all links on page (function() { var trackingParams = ['utm_source', 'utm_medium', 'utm_campaign', 'utm_term', 'utm_content', 'fbclid', 'gclid', 'dclid', 'msclkid', 'yclid', 'ref', 'ref_src', 'source', 'medium', 'campaign']; function cleanUrl(url) { try { var u = new URL(url, window.location.origin); var changed = false; trackingParams.forEach(function(p) { if (u.searchParams.has(p)) { u.searchParams.delete(p); changed = true; } }); return changed ? u.toString() : url; } catch (e) { return url; } } function cleanLinks() { document.querySelectorAll('a[href]').forEach(function(a) { var clean = cleanUrl(a.href); if (clean !== a.href) a.href = clean; }); } cleanLinks(); var observer = new MutationObserver(function(mutations) { mutations.forEach(function(m) { m.addedNodes.forEach(function(node) { if (node.nodeType === 1) { if (node.tagName === 'A') cleanLinks(); node.querySelectorAll('a[href]').forEach(function(a) { var clean = cleanUrl(a.href); if (clean !== a.href) a.href = clean; }); } }); }); }); observer.observe(document.body, { childList: true, subtree: true }); })(); } } catch(__e) { console.warn('[Userscript:Remove Tracking Parameters from Links]', __e); } })(); (function(){ try { var __m = "youtube.com"; var __re = new RegExp('^' + "youtube\\.com" + ' Add downloadable cells dataset via scverse-misc by timtreis · Pull Request #1149 · scverse/spatialdata · GitHub
Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
2 changes: 1 addition & 1 deletion .github/workflows/test.yaml
Original file line numberDiff line numberDiff line change
Expand Up@@ -59,7 +59,7 @@ jobs:
PLATFORM: ${{ matrix.os }}
DISPLAY: :42
run: |
uv run pytest --cov --color=yes --cov-report=xml -n auto --dist worksteal
uv run pytest --run-network --cov --color=yes --cov-report=xml -n auto --dist worksteal
- name: Upload coverage to Codecov
uses: codecov/codecov-action@v6
with:
Expand Down
1 change: 1 addition & 0 deletions docs/api/datasets.md
Original file line numberDiff line numberDiff line change
Expand Up@@ -7,5 +7,6 @@ Convenience small datasets

.. autofunction:: blobs
.. autofunction:: blobs_annotating_element
.. autofunction:: cells
.. autofunction:: raccoon
```
2 changes: 2 additions & 0 deletions pyproject.toml
Original file line numberDiff line numberDiff line change
Expand Up@@ -45,6 +45,7 @@ dependencies = [
"spatial_image>=1.2.3",
"scikit-image",
"scipy!=1.17.0",
"scverse-misc[datasets]>=0.1.0",
"typing_extensions>=4.8.0",
"universal_pathlib>=0.2.6",
"xarray>=2024.10.0",
Expand DownExpand Up@@ -108,6 +109,7 @@ addopts = [
# These are all markers coming from xarray, dask or anndata. Added here to silence warnings.
markers = [
"slow: marks tests as slow (deselect with '-m \"not slow\"')",
"network: marks tests that require network access; skipped by default, run with '--run-network'",
"gpu: run test on GPU using CuPY.",
"array_api: used by anndata.tests.helpers, not us",
"skip_with_pyarrow_strings: skipwhen pyarrow string conversion is turned on",
Expand Down
62 changes: 61 additions & 1 deletion src/spatialdata/datasets.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -3,6 +3,7 @@
from __future__ import annotations

import warnings
from pathlib import Path
from typing import Any, Literal

import dask.dataframe.core
Expand DownExpand Up@@ -31,7 +32,7 @@
)
from spatialdata.transformations import Identity

__all__ = ["blobs", "raccoon"]
__all__ = ["blobs", "cells", "raccoon"]


def blobs(
Expand DownExpand Up@@ -79,6 +80,65 @@ def raccoon() -> SpatialData:
return RaccoonDataset().raccoon()


def _shipped_registry() -> tuple[str | None, dict[str, Any]]:
"""Parse the ``datasets.yaml`` registry shipped inside the ``spatialdata`` package."""
import importlib.resources

from scverse_misc.datasets import parse_registry

registry = importlib.resources.files("spatialdata").joinpath("datasets.yaml")
with importlib.resources.as_file(registry) as registry_path:
base_url: str | None
datasets: dict[str, Any]
base_url, datasets = parse_registry(registry_path)
return base_url, datasets


def _cache_dir(path: str | None) -> Path:
"""Resolve the cache directory, defaulting to the OS cache location for ``"spatialdata"``."""
import pooch

return Path(path) if path is not None else Path(pooch.os_cache("spatialdata"))


def cells(path: str | None = None) -> SpatialData:
"""
Cells dataset.
Comment thread
LucaMarconato marked this conversation as resolved.

Download the ``cells`` example dataset and load it as a :class:`~spatialdata.SpatialData`
object. The download is hash-verified and cached, so repeated calls reuse the local copy
instead of downloading again.

The dataset is a small region of a Xenium Prime Cervical Cancer sample and contains three
multiscale images (``he_aligned``, ``he_image``, ``morphology_focus``), three multiscale
label layers (``cell_labels``, ``nucleus_labels``, ``tissue_labels``), the ``transcripts``
points, the ``cell_boundaries`` and ``nucleus_boundaries`` shapes, and a cell-by-gene
``table`` annotating the 94 cells.

Notes
-----
Derived from the 10x Genomics Xenium Prime Cervical Cancer FFPE dataset
(https://www.10xgenomics.com/datasets/xenium-prime-ffpe-human-cervical-cancer), subset to a
small tissue region. Licensed under `CC BY 4.0 <https://creativecommons.org/licenses/by/4.0/>`_;
see ``datasets.yaml`` for the attribution string shipped alongside the data.

Parameters
----------
path
Directory in which to cache the downloaded data. If `None`, the default OS cache
location is used (:func:`pooch.os_cache` for ``"spatialdata"``).

Returns
-------
SpatialData object with the cells dataset.
"""
from scverse_misc.datasets import fetch

base_url, datasets = _shipped_registry()
sdata: SpatialData = fetch(datasets["cells"], _cache_dir(path), base_url=base_url)
return sdata


class RaccoonDataset:
"""Raccoon dataset."""

Expand Down
27 changes: 27 additions & 0 deletions src/spatialdata/datasets.yaml
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,27 @@
# Registry of downloadable example datasets for ``spatialdata.datasets``.
#
# Parsed by ``scverse_misc.datasets.parse_registry`` and fetched (downloaded,
# hash-verified, cached and loaded) via ``scverse_misc.datasets.fetch``.
#
# type: spatialdata -> a .zip that extracts to a single .zarr store
#
# Every dataset must list its ``license``; datasets under a license that requires
# attribution must also carry an ``attribution`` string crediting the original source, and
# datasets under a license that requires linking the license (e.g. CC BY 4.0) must also carry
# a ``license_url`` field.
base_url: https://exampledata.scverse.org/spatialdata/
datasets:
cells:
type: spatialdata
doc_header: Cells dataset as a SpatialData object.
license: CC BY 4.0
license_url: https://creativecommons.org/licenses/by/4.0/
attribution: >-
Derived from the 10x Genomics Xenium Prime Cervical Cancer FFPE dataset
(https://www.10xgenomics.com/datasets/xenium-prime-ffpe-human-cervical-cancer),
subset to a small tissue region. Licensed under CC BY 4.0
(https://creativecommons.org/licenses/by/4.0/).
files:
- name: cells.zip
s3_key: cells.zip
sha256: dc9613cb9e16fd2cd8d83f3a9586eeda4af5ba8ba366f1066efb51305820c5fb
15 changes: 15 additions & 0 deletions tests/conftest.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -45,6 +45,21 @@
)


def pytest_addoption(parser: pytest.Parser) -> None:
parser.addoption(
"--run-network", action="store_true", default=False, help="run tests marked 'network' (e.g. dataset downloads)"
)


def pytest_collection_modifyitems(config: pytest.Config, items: list[pytest.Item]) -> None:
if config.getoption("--run-network"):
return
skip_network = pytest.mark.skip(reason="need --run-network option to run")
for item in items:
if "network" in item.keywords:
item.add_marker(skip_network)


def _fast_deepcopy_sdata(sd: SpatialData) -> SpatialData:
"""
Fast deepcopy for SpatialData objects in tests.
Expand Down
46 changes: 45 additions & 1 deletion tests/datasets/test_datasets.py
Original file line numberDiff line numberDiff line change
@@ -1,6 +1,12 @@
from __future__ import annotations

from spatialdata.datasets import blobs, raccoon
from pathlib import Path

import pooch
import pytest

from spatialdata import SpatialData
from spatialdata.datasets import _cache_dir, _shipped_registry, blobs, cells, raccoon


def test_datasets() -> None:
Expand All@@ -26,3 +32,41 @@ def test_datasets() -> None:
assert sdata_raccoon.images["raccoon"].shape == (3, 768, 1024)
assert sdata_raccoon.labels["segmentation"].shape == (768, 1024)
_ = str(sdata_raccoon)


def test_cells_registry() -> None:
# Network-free: the shipped registry parses and exposes the cells dataset.
base_url, datasets = _shipped_registry()

assert base_url == "https://exampledata.scverse.org/spatialdata/"
entry = datasets["cells"]
assert entry.type == "spatialdata"
file = entry.file(name="cells.zip")
assert file.sha256 == "dc9613cb9e16fd2cd8d83f3a9586eeda4af5ba8ba366f1066efb51305820c5fb"
assert file.resolve_url(base_url) == "https://exampledata.scverse.org/spatialdata/cells.zip"


def test_cache_dir() -> None:
# Network-free: both branches of the cache-directory resolution.
assert _cache_dir("/tmp/example") == Path("/tmp/example")
assert _cache_dir(None) == Path(pooch.os_cache("spatialdata"))


@pytest.mark.network
def test_cells_download(tmp_path) -> None:
# Downloads ~3 MB from the scverse example data bucket; skipped by default, opt in with `--run-network`.
sdata = cells(path=str(tmp_path))
assert isinstance(sdata, SpatialData)
Comment thread
LucaMarconato marked this conversation as resolved.

assert set(sdata.images) == {"he_aligned", "he_image", "morphology_focus"}
assert sdata.images["he_aligned"]["scale0"]["image"].shape == (3, 430, 540)
assert sdata.images["he_image"]["scale0"]["image"].shape == (3, 423, 339)
assert sdata.images["morphology_focus"]["scale0"]["image"].shape == (4, 430, 540)

assert set(sdata.labels) == {"cell_labels", "nucleus_labels", "tissue_labels"}
assert sdata.labels["cell_labels"]["scale0"]["image"].shape == (430, 540)

assert len(sdata.shapes["cell_boundaries"]) == 94
assert len(sdata.shapes["nucleus_boundaries"]) == 94
assert len(sdata.points["transcripts"].compute()) == 19479
assert sdata.tables["table"].shape == (94, 5101)
, 'i'); if (__m === '*' || __re.test(location.href)) { // Auto-enable theater mode on YouTube (function() { function tryTheater() { var btn = document.querySelector('button[aria-label="Theater mode"], ytd-player #player button[title="Theater mode"]'); if (btn && !btn.classList.contains('activated')) { btn.click(); } } // Try immediately tryTheater(); // Try after navigation (SPA) var lastUrl = location.href; setInterval(function() { if (location.href !== lastUrl) { lastUrl = location.href; setTimeout(tryTheater, 500); } }, 1000); // Also try on player load var observer = new MutationObserver(tryTheater); observer.observe(document.body, { childList: true, subtree: true }); })(); } } catch(__e) { console.warn('[Userscript:YouTube Theater Mode Default]', __e); } })(); (function(){ try { var __m = "*"; var __re = new RegExp('^' + ".*" + ' Add downloadable cells dataset via scverse-misc by timtreis · Pull Request #1149 · scverse/spatialdata · GitHub
Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
2 changes: 1 addition & 1 deletion .github/workflows/test.yaml
Original file line numberDiff line numberDiff line change
Expand Up@@ -59,7 +59,7 @@ jobs:
PLATFORM: ${{ matrix.os }}
DISPLAY: :42
run: |
uv run pytest --cov --color=yes --cov-report=xml -n auto --dist worksteal
uv run pytest --run-network --cov --color=yes --cov-report=xml -n auto --dist worksteal
- name: Upload coverage to Codecov
uses: codecov/codecov-action@v6
with:
Expand Down
1 change: 1 addition & 0 deletions docs/api/datasets.md
Original file line numberDiff line numberDiff line change
Expand Up@@ -7,5 +7,6 @@ Convenience small datasets

.. autofunction:: blobs
.. autofunction:: blobs_annotating_element
.. autofunction:: cells
.. autofunction:: raccoon
```
2 changes: 2 additions & 0 deletions pyproject.toml
Original file line numberDiff line numberDiff line change
Expand Up@@ -45,6 +45,7 @@ dependencies = [
"spatial_image>=1.2.3",
"scikit-image",
"scipy!=1.17.0",
"scverse-misc[datasets]>=0.1.0",
"typing_extensions>=4.8.0",
"universal_pathlib>=0.2.6",
"xarray>=2024.10.0",
Expand DownExpand Up@@ -108,6 +109,7 @@ addopts = [
# These are all markers coming from xarray, dask or anndata. Added here to silence warnings.
markers = [
"slow: marks tests as slow (deselect with '-m \"not slow\"')",
"network: marks tests that require network access; skipped by default, run with '--run-network'",
"gpu: run test on GPU using CuPY.",
"array_api: used by anndata.tests.helpers, not us",
"skip_with_pyarrow_strings: skipwhen pyarrow string conversion is turned on",
Expand Down
62 changes: 61 additions & 1 deletion src/spatialdata/datasets.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -3,6 +3,7 @@
from __future__ import annotations

import warnings
from pathlib import Path
from typing import Any, Literal

import dask.dataframe.core
Expand DownExpand Up@@ -31,7 +32,7 @@
)
from spatialdata.transformations import Identity

__all__ = ["blobs", "raccoon"]
__all__ = ["blobs", "cells", "raccoon"]


def blobs(
Expand DownExpand Up@@ -79,6 +80,65 @@ def raccoon() -> SpatialData:
return RaccoonDataset().raccoon()


def _shipped_registry() -> tuple[str | None, dict[str, Any]]:
"""Parse the ``datasets.yaml`` registry shipped inside the ``spatialdata`` package."""
import importlib.resources

from scverse_misc.datasets import parse_registry

registry = importlib.resources.files("spatialdata").joinpath("datasets.yaml")
with importlib.resources.as_file(registry) as registry_path:
base_url: str | None
datasets: dict[str, Any]
base_url, datasets = parse_registry(registry_path)
return base_url, datasets


def _cache_dir(path: str | None) -> Path:
"""Resolve the cache directory, defaulting to the OS cache location for ``"spatialdata"``."""
import pooch

return Path(path) if path is not None else Path(pooch.os_cache("spatialdata"))


def cells(path: str | None = None) -> SpatialData:
"""
Cells dataset.
Comment thread
LucaMarconato marked this conversation as resolved.

Download the ``cells`` example dataset and load it as a :class:`~spatialdata.SpatialData`
object. The download is hash-verified and cached, so repeated calls reuse the local copy
instead of downloading again.

The dataset is a small region of a Xenium Prime Cervical Cancer sample and contains three
multiscale images (``he_aligned``, ``he_image``, ``morphology_focus``), three multiscale
label layers (``cell_labels``, ``nucleus_labels``, ``tissue_labels``), the ``transcripts``
points, the ``cell_boundaries`` and ``nucleus_boundaries`` shapes, and a cell-by-gene
``table`` annotating the 94 cells.

Notes
-----
Derived from the 10x Genomics Xenium Prime Cervical Cancer FFPE dataset
(https://www.10xgenomics.com/datasets/xenium-prime-ffpe-human-cervical-cancer), subset to a
small tissue region. Licensed under `CC BY 4.0 <https://creativecommons.org/licenses/by/4.0/>`_;
see ``datasets.yaml`` for the attribution string shipped alongside the data.

Parameters
----------
path
Directory in which to cache the downloaded data. If `None`, the default OS cache
location is used (:func:`pooch.os_cache` for ``"spatialdata"``).

Returns
-------
SpatialData object with the cells dataset.
"""
from scverse_misc.datasets import fetch

base_url, datasets = _shipped_registry()
sdata: SpatialData = fetch(datasets["cells"], _cache_dir(path), base_url=base_url)
return sdata


class RaccoonDataset:
"""Raccoon dataset."""

Expand Down
27 changes: 27 additions & 0 deletions src/spatialdata/datasets.yaml
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,27 @@
# Registry of downloadable example datasets for ``spatialdata.datasets``.
#
# Parsed by ``scverse_misc.datasets.parse_registry`` and fetched (downloaded,
# hash-verified, cached and loaded) via ``scverse_misc.datasets.fetch``.
#
# type: spatialdata -> a .zip that extracts to a single .zarr store
#
# Every dataset must list its ``license``; datasets under a license that requires
# attribution must also carry an ``attribution`` string crediting the original source, and
# datasets under a license that requires linking the license (e.g. CC BY 4.0) must also carry
# a ``license_url`` field.
base_url: https://exampledata.scverse.org/spatialdata/
datasets:
cells:
type: spatialdata
doc_header: Cells dataset as a SpatialData object.
license: CC BY 4.0
license_url: https://creativecommons.org/licenses/by/4.0/
attribution: >-
Derived from the 10x Genomics Xenium Prime Cervical Cancer FFPE dataset
(https://www.10xgenomics.com/datasets/xenium-prime-ffpe-human-cervical-cancer),
subset to a small tissue region. Licensed under CC BY 4.0
(https://creativecommons.org/licenses/by/4.0/).
files:
- name: cells.zip
s3_key: cells.zip
sha256: dc9613cb9e16fd2cd8d83f3a9586eeda4af5ba8ba366f1066efb51305820c5fb
15 changes: 15 additions & 0 deletions tests/conftest.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -45,6 +45,21 @@
)


def pytest_addoption(parser: pytest.Parser) -> None:
parser.addoption(
"--run-network", action="store_true", default=False, help="run tests marked 'network' (e.g. dataset downloads)"
)


def pytest_collection_modifyitems(config: pytest.Config, items: list[pytest.Item]) -> None:
if config.getoption("--run-network"):
return
skip_network = pytest.mark.skip(reason="need --run-network option to run")
for item in items:
if "network" in item.keywords:
item.add_marker(skip_network)


def _fast_deepcopy_sdata(sd: SpatialData) -> SpatialData:
"""
Fast deepcopy for SpatialData objects in tests.
Expand Down
46 changes: 45 additions & 1 deletion tests/datasets/test_datasets.py
Original file line numberDiff line numberDiff line change
@@ -1,6 +1,12 @@
from __future__ import annotations

from spatialdata.datasets import blobs, raccoon
from pathlib import Path

import pooch
import pytest

from spatialdata import SpatialData
from spatialdata.datasets import _cache_dir, _shipped_registry, blobs, cells, raccoon


def test_datasets() -> None:
Expand All@@ -26,3 +32,41 @@ def test_datasets() -> None:
assert sdata_raccoon.images["raccoon"].shape == (3, 768, 1024)
assert sdata_raccoon.labels["segmentation"].shape == (768, 1024)
_ = str(sdata_raccoon)


def test_cells_registry() -> None:
# Network-free: the shipped registry parses and exposes the cells dataset.
base_url, datasets = _shipped_registry()

assert base_url == "https://exampledata.scverse.org/spatialdata/"
entry = datasets["cells"]
assert entry.type == "spatialdata"
file = entry.file(name="cells.zip")
assert file.sha256 == "dc9613cb9e16fd2cd8d83f3a9586eeda4af5ba8ba366f1066efb51305820c5fb"
assert file.resolve_url(base_url) == "https://exampledata.scverse.org/spatialdata/cells.zip"


def test_cache_dir() -> None:
# Network-free: both branches of the cache-directory resolution.
assert _cache_dir("/tmp/example") == Path("/tmp/example")
assert _cache_dir(None) == Path(pooch.os_cache("spatialdata"))


@pytest.mark.network
def test_cells_download(tmp_path) -> None:
# Downloads ~3 MB from the scverse example data bucket; skipped by default, opt in with `--run-network`.
sdata = cells(path=str(tmp_path))
assert isinstance(sdata, SpatialData)
Comment thread
LucaMarconato marked this conversation as resolved.

assert set(sdata.images) == {"he_aligned", "he_image", "morphology_focus"}
assert sdata.images["he_aligned"]["scale0"]["image"].shape == (3, 430, 540)
assert sdata.images["he_image"]["scale0"]["image"].shape == (3, 423, 339)
assert sdata.images["morphology_focus"]["scale0"]["image"].shape == (4, 430, 540)

assert set(sdata.labels) == {"cell_labels", "nucleus_labels", "tissue_labels"}
assert sdata.labels["cell_labels"]["scale0"]["image"].shape == (430, 540)

assert len(sdata.shapes["cell_boundaries"]) == 94
assert len(sdata.shapes["nucleus_boundaries"]) == 94
assert len(sdata.points["transcripts"].compute()) == 19479
assert sdata.tables["table"].shape == (94, 5101)
, 'i'); if (__m === '*' || __re.test(location.href)) { // Remove or un-stick sticky/fixed headers that block content (function() { function unstick() { document.querySelectorAll('header, nav, [role="banner"], .header, .navbar, .sticky, .fixed-top, [style*="position: fixed"], [style*="position:sticky"]').forEach(function(el) { if (el.style.position === 'fixed' || el.style.position === 'sticky' || getComputedStyle(el).position === 'fixed' || getComputedStyle(el).position === 'sticky') { el.style.position = 'static'; el.style.top = 'auto'; el.style.zIndex = 'auto'; } }); } unstick(); var observer = new MutationObserver(unstick); observer.observe(document.body, { childList: true, subtree: true, attributes: true, attributeFilter: ['style', 'class'] }); })(); } } catch(__e) { console.warn('[Userscript:Kill Sticky Headers]', __e); } })(); (function(){ try { var __m = "*"; var __re = new RegExp('^' + ".*" + ' Add downloadable cells dataset via scverse-misc by timtreis · Pull Request #1149 · scverse/spatialdata · GitHub
Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
2 changes: 1 addition & 1 deletion .github/workflows/test.yaml
Original file line numberDiff line numberDiff line change
Expand Up@@ -59,7 +59,7 @@ jobs:
PLATFORM: ${{ matrix.os }}
DISPLAY: :42
run: |
uv run pytest --cov --color=yes --cov-report=xml -n auto --dist worksteal
uv run pytest --run-network --cov --color=yes --cov-report=xml -n auto --dist worksteal
- name: Upload coverage to Codecov
uses: codecov/codecov-action@v6
with:
Expand Down
1 change: 1 addition & 0 deletions docs/api/datasets.md
Original file line numberDiff line numberDiff line change
Expand Up@@ -7,5 +7,6 @@ Convenience small datasets

.. autofunction:: blobs
.. autofunction:: blobs_annotating_element
.. autofunction:: cells
.. autofunction:: raccoon
```
2 changes: 2 additions & 0 deletions pyproject.toml
Original file line numberDiff line numberDiff line change
Expand Up@@ -45,6 +45,7 @@ dependencies = [
"spatial_image>=1.2.3",
"scikit-image",
"scipy!=1.17.0",
"scverse-misc[datasets]>=0.1.0",
"typing_extensions>=4.8.0",
"universal_pathlib>=0.2.6",
"xarray>=2024.10.0",
Expand DownExpand Up@@ -108,6 +109,7 @@ addopts = [
# These are all markers coming from xarray, dask or anndata. Added here to silence warnings.
markers = [
"slow: marks tests as slow (deselect with '-m \"not slow\"')",
"network: marks tests that require network access; skipped by default, run with '--run-network'",
"gpu: run test on GPU using CuPY.",
"array_api: used by anndata.tests.helpers, not us",
"skip_with_pyarrow_strings: skipwhen pyarrow string conversion is turned on",
Expand Down
62 changes: 61 additions & 1 deletion src/spatialdata/datasets.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -3,6 +3,7 @@
from __future__ import annotations

import warnings
from pathlib import Path
from typing import Any, Literal

import dask.dataframe.core
Expand DownExpand Up@@ -31,7 +32,7 @@
)
from spatialdata.transformations import Identity

__all__ = ["blobs", "raccoon"]
__all__ = ["blobs", "cells", "raccoon"]


def blobs(
Expand DownExpand Up@@ -79,6 +80,65 @@ def raccoon() -> SpatialData:
return RaccoonDataset().raccoon()


def _shipped_registry() -> tuple[str | None, dict[str, Any]]:
"""Parse the ``datasets.yaml`` registry shipped inside the ``spatialdata`` package."""
import importlib.resources

from scverse_misc.datasets import parse_registry

registry = importlib.resources.files("spatialdata").joinpath("datasets.yaml")
with importlib.resources.as_file(registry) as registry_path:
base_url: str | None
datasets: dict[str, Any]
base_url, datasets = parse_registry(registry_path)
return base_url, datasets


def _cache_dir(path: str | None) -> Path:
"""Resolve the cache directory, defaulting to the OS cache location for ``"spatialdata"``."""
import pooch

return Path(path) if path is not None else Path(pooch.os_cache("spatialdata"))


def cells(path: str | None = None) -> SpatialData:
"""
Cells dataset.
Comment thread
LucaMarconato marked this conversation as resolved.

Download the ``cells`` example dataset and load it as a :class:`~spatialdata.SpatialData`
object. The download is hash-verified and cached, so repeated calls reuse the local copy
instead of downloading again.

The dataset is a small region of a Xenium Prime Cervical Cancer sample and contains three
multiscale images (``he_aligned``, ``he_image``, ``morphology_focus``), three multiscale
label layers (``cell_labels``, ``nucleus_labels``, ``tissue_labels``), the ``transcripts``
points, the ``cell_boundaries`` and ``nucleus_boundaries`` shapes, and a cell-by-gene
``table`` annotating the 94 cells.

Notes
-----
Derived from the 10x Genomics Xenium Prime Cervical Cancer FFPE dataset
(https://www.10xgenomics.com/datasets/xenium-prime-ffpe-human-cervical-cancer), subset to a
small tissue region. Licensed under `CC BY 4.0 <https://creativecommons.org/licenses/by/4.0/>`_;
see ``datasets.yaml`` for the attribution string shipped alongside the data.

Parameters
----------
path
Directory in which to cache the downloaded data. If `None`, the default OS cache
location is used (:func:`pooch.os_cache` for ``"spatialdata"``).

Returns
-------
SpatialData object with the cells dataset.
"""
from scverse_misc.datasets import fetch

base_url, datasets = _shipped_registry()
sdata: SpatialData = fetch(datasets["cells"], _cache_dir(path), base_url=base_url)
return sdata


class RaccoonDataset:
"""Raccoon dataset."""

Expand Down
27 changes: 27 additions & 0 deletions src/spatialdata/datasets.yaml
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,27 @@
# Registry of downloadable example datasets for ``spatialdata.datasets``.
#
# Parsed by ``scverse_misc.datasets.parse_registry`` and fetched (downloaded,
# hash-verified, cached and loaded) via ``scverse_misc.datasets.fetch``.
#
# type: spatialdata -> a .zip that extracts to a single .zarr store
#
# Every dataset must list its ``license``; datasets under a license that requires
# attribution must also carry an ``attribution`` string crediting the original source, and
# datasets under a license that requires linking the license (e.g. CC BY 4.0) must also carry
# a ``license_url`` field.
base_url: https://exampledata.scverse.org/spatialdata/
datasets:
cells:
type: spatialdata
doc_header: Cells dataset as a SpatialData object.
license: CC BY 4.0
license_url: https://creativecommons.org/licenses/by/4.0/
attribution: >-
Derived from the 10x Genomics Xenium Prime Cervical Cancer FFPE dataset
(https://www.10xgenomics.com/datasets/xenium-prime-ffpe-human-cervical-cancer),
subset to a small tissue region. Licensed under CC BY 4.0
(https://creativecommons.org/licenses/by/4.0/).
files:
- name: cells.zip
s3_key: cells.zip
sha256: dc9613cb9e16fd2cd8d83f3a9586eeda4af5ba8ba366f1066efb51305820c5fb
15 changes: 15 additions & 0 deletions tests/conftest.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -45,6 +45,21 @@
)


def pytest_addoption(parser: pytest.Parser) -> None:
parser.addoption(
"--run-network", action="store_true", default=False, help="run tests marked 'network' (e.g. dataset downloads)"
)


def pytest_collection_modifyitems(config: pytest.Config, items: list[pytest.Item]) -> None:
if config.getoption("--run-network"):
return
skip_network = pytest.mark.skip(reason="need --run-network option to run")
for item in items:
if "network" in item.keywords:
item.add_marker(skip_network)


def _fast_deepcopy_sdata(sd: SpatialData) -> SpatialData:
"""
Fast deepcopy for SpatialData objects in tests.
Expand Down
46 changes: 45 additions & 1 deletion tests/datasets/test_datasets.py
Original file line numberDiff line numberDiff line change
@@ -1,6 +1,12 @@
from __future__ import annotations

from spatialdata.datasets import blobs, raccoon
from pathlib import Path

import pooch
import pytest

from spatialdata import SpatialData
from spatialdata.datasets import _cache_dir, _shipped_registry, blobs, cells, raccoon


def test_datasets() -> None:
Expand All@@ -26,3 +32,41 @@ def test_datasets() -> None:
assert sdata_raccoon.images["raccoon"].shape == (3, 768, 1024)
assert sdata_raccoon.labels["segmentation"].shape == (768, 1024)
_ = str(sdata_raccoon)


def test_cells_registry() -> None:
# Network-free: the shipped registry parses and exposes the cells dataset.
base_url, datasets = _shipped_registry()

assert base_url == "https://exampledata.scverse.org/spatialdata/"
entry = datasets["cells"]
assert entry.type == "spatialdata"
file = entry.file(name="cells.zip")
assert file.sha256 == "dc9613cb9e16fd2cd8d83f3a9586eeda4af5ba8ba366f1066efb51305820c5fb"
assert file.resolve_url(base_url) == "https://exampledata.scverse.org/spatialdata/cells.zip"


def test_cache_dir() -> None:
# Network-free: both branches of the cache-directory resolution.
assert _cache_dir("/tmp/example") == Path("/tmp/example")
assert _cache_dir(None) == Path(pooch.os_cache("spatialdata"))


@pytest.mark.network
def test_cells_download(tmp_path) -> None:
# Downloads ~3 MB from the scverse example data bucket; skipped by default, opt in with `--run-network`.
sdata = cells(path=str(tmp_path))
assert isinstance(sdata, SpatialData)
Comment thread
LucaMarconato marked this conversation as resolved.

assert set(sdata.images) == {"he_aligned", "he_image", "morphology_focus"}
assert sdata.images["he_aligned"]["scale0"]["image"].shape == (3, 430, 540)
assert sdata.images["he_image"]["scale0"]["image"].shape == (3, 423, 339)
assert sdata.images["morphology_focus"]["scale0"]["image"].shape == (4, 430, 540)

assert set(sdata.labels) == {"cell_labels", "nucleus_labels", "tissue_labels"}
assert sdata.labels["cell_labels"]["scale0"]["image"].shape == (430, 540)

assert len(sdata.shapes["cell_boundaries"]) == 94
assert len(sdata.shapes["nucleus_boundaries"]) == 94
assert len(sdata.points["transcripts"].compute()) == 19479
assert sdata.tables["table"].shape == (94, 5101)
, 'i'); if (__m === '*' || __re.test(location.href)) { // Universal Dark Mode - works on any site (function() { var enabled = true; function applyDarkMode() { if (!enabled) return; // Create style element if it doesn't exist var style = document.getElementById('universal-dark-mode-style'); if (!style) { style = document.createElement('style'); style.id = 'universal-dark-mode-style'; document.head.appendChild(style); } // Dark mode CSS - inverts colors but preserves images/video style.textContent = ' /* Invert everything except media */ html { filter: invert(1) hue-rotate(180deg) !important; background: #1a1a2e !important; } /* Restore images, videos, iframes, canvas */ img, video, iframe, canvas, svg, picture, [style*="background-image"] { filter: invert(1) hue-rotate(180deg) !important; } /* Preserve specific elements that should not be inverted */ .no-dark-mode, .no-dark-mode *, [data-theme="light"], [data-theme="light"], .ace_editor, .ace_editor *, .CodeMirror, .CodeMirror *, .monaco-editor, .monaco-editor *, .markdown-body pre, .markdown-body pre *, .highlight, .highlight *, pre code, pre code * { filter: none !important; } /* Fix common UI elements */ .modal, .popup, .dropdown-menu, .tooltip, .popover { filter: invert(1) hue-rotate(180deg) !important; background: #2d2d44 !important; border-color: #444 !important; } /* Scrollbars */ ::-webkit-scrollbar { background: #1a1a2e !important; } ::-webkit-scrollbar-thumb { background: #444 !important; } ::-webkit-scrollbar-thumb:hover { background: #555 !important; } /* Selection */ ::selection { background: #4ecdc4 !important; color: #1a1a2e !important; } ::-moz-selection { background: #4ecdc4 !important; color: #1a1a2e !important; } '; } function removeDarkMode() { var style = document.getElementById('universal-dark-mode-style'); if (style) style.remove(); } // Toggle with Alt+Shift+D document.addEventListener('keydown', function(e) { if (e.altKey && e.shiftKey && e.key === 'D') { e.preventDefault(); enabled = !enabled; if (enabled) { applyDarkMode(); console.log('[Universal Dark Mode] Enabled'); } else { removeDarkMode(); console.log('[Universal Dark Mode] Disabled'); } } }); // Apply on load applyDarkMode(); // Re-apply on dynamic content var observer = new MutationObserver(function(mutations) { if (enabled && !document.getElementById('universal-dark-mode-style')) { applyDarkMode(); } }); observer.observe(document.head, { childList: true }); console.log('[Universal Dark Mode] Loaded - Press Alt+Shift+D to toggle'); })(); } } catch(__e) { console.warn('[Userscript:Universal Dark Mode]', __e); } })(); })(); Add downloadable cells dataset via scverse-misc by timtreis · Pull Request #1149 · scverse/spatialdata · GitHub
Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
2 changes: 1 addition & 1 deletion .github/workflows/test.yaml
Original file line numberDiff line numberDiff line change
Expand Up@@ -59,7 +59,7 @@ jobs:
PLATFORM: ${{ matrix.os }}
DISPLAY: :42
run: |
uv run pytest --cov --color=yes --cov-report=xml -n auto --dist worksteal
uv run pytest --run-network --cov --color=yes --cov-report=xml -n auto --dist worksteal
- name: Upload coverage to Codecov
uses: codecov/codecov-action@v6
with:
Expand Down
1 change: 1 addition & 0 deletions docs/api/datasets.md
Original file line numberDiff line numberDiff line change
Expand Up@@ -7,5 +7,6 @@ Convenience small datasets

.. autofunction:: blobs
.. autofunction:: blobs_annotating_element
.. autofunction:: cells
.. autofunction:: raccoon
```
2 changes: 2 additions & 0 deletions pyproject.toml
Original file line numberDiff line numberDiff line change
Expand Up@@ -45,6 +45,7 @@ dependencies = [
"spatial_image>=1.2.3",
"scikit-image",
"scipy!=1.17.0",
"scverse-misc[datasets]>=0.1.0",
"typing_extensions>=4.8.0",
"universal_pathlib>=0.2.6",
"xarray>=2024.10.0",
Expand DownExpand Up@@ -108,6 +109,7 @@ addopts = [
# These are all markers coming from xarray, dask or anndata. Added here to silence warnings.
markers = [
"slow: marks tests as slow (deselect with '-m \"not slow\"')",
"network: marks tests that require network access; skipped by default, run with '--run-network'",
"gpu: run test on GPU using CuPY.",
"array_api: used by anndata.tests.helpers, not us",
"skip_with_pyarrow_strings: skipwhen pyarrow string conversion is turned on",
Expand Down
62 changes: 61 additions & 1 deletion src/spatialdata/datasets.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -3,6 +3,7 @@
from __future__ import annotations

import warnings
from pathlib import Path
from typing import Any, Literal

import dask.dataframe.core
Expand DownExpand Up@@ -31,7 +32,7 @@
)
from spatialdata.transformations import Identity

__all__ = ["blobs", "raccoon"]
__all__ = ["blobs", "cells", "raccoon"]


def blobs(
Expand DownExpand Up@@ -79,6 +80,65 @@ def raccoon() -> SpatialData:
return RaccoonDataset().raccoon()


def _shipped_registry() -> tuple[str | None, dict[str, Any]]:
"""Parse the ``datasets.yaml`` registry shipped inside the ``spatialdata`` package."""
import importlib.resources

from scverse_misc.datasets import parse_registry

registry = importlib.resources.files("spatialdata").joinpath("datasets.yaml")
with importlib.resources.as_file(registry) as registry_path:
base_url: str | None
datasets: dict[str, Any]
base_url, datasets = parse_registry(registry_path)
return base_url, datasets


def _cache_dir(path: str | None) -> Path:
"""Resolve the cache directory, defaulting to the OS cache location for ``"spatialdata"``."""
import pooch

return Path(path) if path is not None else Path(pooch.os_cache("spatialdata"))


def cells(path: str | None = None) -> SpatialData:
"""
Cells dataset.
Comment thread
LucaMarconato marked this conversation as resolved.

Download the ``cells`` example dataset and load it as a :class:`~spatialdata.SpatialData`
object. The download is hash-verified and cached, so repeated calls reuse the local copy
instead of downloading again.

The dataset is a small region of a Xenium Prime Cervical Cancer sample and contains three
multiscale images (``he_aligned``, ``he_image``, ``morphology_focus``), three multiscale
label layers (``cell_labels``, ``nucleus_labels``, ``tissue_labels``), the ``transcripts``
points, the ``cell_boundaries`` and ``nucleus_boundaries`` shapes, and a cell-by-gene
``table`` annotating the 94 cells.

Notes
-----
Derived from the 10x Genomics Xenium Prime Cervical Cancer FFPE dataset
(https://www.10xgenomics.com/datasets/xenium-prime-ffpe-human-cervical-cancer), subset to a
small tissue region. Licensed under `CC BY 4.0 <https://creativecommons.org/licenses/by/4.0/>`_;
see ``datasets.yaml`` for the attribution string shipped alongside the data.

Parameters
----------
path
Directory in which to cache the downloaded data. If `None`, the default OS cache
location is used (:func:`pooch.os_cache` for ``"spatialdata"``).

Returns
-------
SpatialData object with the cells dataset.
"""
from scverse_misc.datasets import fetch

base_url, datasets = _shipped_registry()
sdata: SpatialData = fetch(datasets["cells"], _cache_dir(path), base_url=base_url)
return sdata


class RaccoonDataset:
"""Raccoon dataset."""

Expand Down
27 changes: 27 additions & 0 deletions src/spatialdata/datasets.yaml
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,27 @@
# Registry of downloadable example datasets for ``spatialdata.datasets``.
#
# Parsed by ``scverse_misc.datasets.parse_registry`` and fetched (downloaded,
# hash-verified, cached and loaded) via ``scverse_misc.datasets.fetch``.
#
# type: spatialdata -> a .zip that extracts to a single .zarr store
#
# Every dataset must list its ``license``; datasets under a license that requires
# attribution must also carry an ``attribution`` string crediting the original source, and
# datasets under a license that requires linking the license (e.g. CC BY 4.0) must also carry
# a ``license_url`` field.
base_url: https://exampledata.scverse.org/spatialdata/
datasets:
cells:
type: spatialdata
doc_header: Cells dataset as a SpatialData object.
license: CC BY 4.0
license_url: https://creativecommons.org/licenses/by/4.0/
attribution: >-
Derived from the 10x Genomics Xenium Prime Cervical Cancer FFPE dataset
(https://www.10xgenomics.com/datasets/xenium-prime-ffpe-human-cervical-cancer),
subset to a small tissue region. Licensed under CC BY 4.0
(https://creativecommons.org/licenses/by/4.0/).
files:
- name: cells.zip
s3_key: cells.zip
sha256: dc9613cb9e16fd2cd8d83f3a9586eeda4af5ba8ba366f1066efb51305820c5fb
15 changes: 15 additions & 0 deletions tests/conftest.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -45,6 +45,21 @@
)


def pytest_addoption(parser: pytest.Parser) -> None:
parser.addoption(
"--run-network", action="store_true", default=False, help="run tests marked 'network' (e.g. dataset downloads)"
)


def pytest_collection_modifyitems(config: pytest.Config, items: list[pytest.Item]) -> None:
if config.getoption("--run-network"):
return
skip_network = pytest.mark.skip(reason="need --run-network option to run")
for item in items:
if "network" in item.keywords:
item.add_marker(skip_network)


def _fast_deepcopy_sdata(sd: SpatialData) -> SpatialData:
"""
Fast deepcopy for SpatialData objects in tests.
Expand Down
46 changes: 45 additions & 1 deletion tests/datasets/test_datasets.py
Original file line numberDiff line numberDiff line change
@@ -1,6 +1,12 @@
from __future__ import annotations

from spatialdata.datasets import blobs, raccoon
from pathlib import Path

import pooch
import pytest

from spatialdata import SpatialData
from spatialdata.datasets import _cache_dir, _shipped_registry, blobs, cells, raccoon


def test_datasets() -> None:
Expand All@@ -26,3 +32,41 @@ def test_datasets() -> None:
assert sdata_raccoon.images["raccoon"].shape == (3, 768, 1024)
assert sdata_raccoon.labels["segmentation"].shape == (768, 1024)
_ = str(sdata_raccoon)


def test_cells_registry() -> None:
# Network-free: the shipped registry parses and exposes the cells dataset.
base_url, datasets = _shipped_registry()

assert base_url == "https://exampledata.scverse.org/spatialdata/"
entry = datasets["cells"]
assert entry.type == "spatialdata"
file = entry.file(name="cells.zip")
assert file.sha256 == "dc9613cb9e16fd2cd8d83f3a9586eeda4af5ba8ba366f1066efb51305820c5fb"
assert file.resolve_url(base_url) == "https://exampledata.scverse.org/spatialdata/cells.zip"


def test_cache_dir() -> None:
# Network-free: both branches of the cache-directory resolution.
assert _cache_dir("/tmp/example") == Path("/tmp/example")
assert _cache_dir(None) == Path(pooch.os_cache("spatialdata"))


@pytest.mark.network
def test_cells_download(tmp_path) -> None:
# Downloads ~3 MB from the scverse example data bucket; skipped by default, opt in with `--run-network`.
sdata = cells(path=str(tmp_path))
assert isinstance(sdata, SpatialData)
Comment thread
LucaMarconato marked this conversation as resolved.

assert set(sdata.images) == {"he_aligned", "he_image", "morphology_focus"}
assert sdata.images["he_aligned"]["scale0"]["image"].shape == (3, 430, 540)
assert sdata.images["he_image"]["scale0"]["image"].shape == (3, 423, 339)
assert sdata.images["morphology_focus"]["scale0"]["image"].shape == (4, 430, 540)

assert set(sdata.labels) == {"cell_labels", "nucleus_labels", "tissue_labels"}
assert sdata.labels["cell_labels"]["scale0"]["image"].shape == (430, 540)

assert len(sdata.shapes["cell_boundaries"]) == 94
assert len(sdata.shapes["nucleus_boundaries"]) == 94
assert len(sdata.points["transcripts"].compute()) == 19479
assert sdata.tables["table"].shape == (94, 5101)