Skip to content
Merged
Show file tree
Hide file tree
Changes from 12 commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
1 change: 1 addition & 0 deletions docs/api/datasets.md
Original file line number Diff line number Diff line change
Expand Up @@ -7,5 +7,6 @@ Convenience small datasets

.. autofunction:: blobs
.. autofunction:: blobs_annotating_element
.. autofunction:: cells
.. autofunction:: raccoon
```
1 change: 1 addition & 0 deletions pyproject.toml
Original file line number Diff line number Diff line change
Expand Up @@ -45,6 +45,7 @@ dependencies = [
"spatial_image>=1.2.3",
"scikit-image",
"scipy!=1.17.0",
"scverse-misc[datasets]>=0.1.0",
"typing_extensions>=4.8.0",
"universal_pathlib>=0.2.6",
"xarray>=2024.10.0",
Expand Down
55 changes: 54 additions & 1 deletion src/spatialdata/datasets.py
Original file line number Diff line number Diff line change
Expand Up @@ -3,6 +3,7 @@
from __future__ import annotations

import warnings
from pathlib import Path
from typing import Any, Literal

import dask.dataframe.core
Expand Down Expand Up @@ -31,7 +32,7 @@
)
from spatialdata.transformations import Identity

__all__ = ["blobs", "raccoon"]
__all__ = ["blobs", "cells", "raccoon"]


def blobs(
Expand Down Expand Up @@ -79,6 +80,58 @@ def raccoon() -> SpatialData:
return RaccoonDataset().raccoon()


def _shipped_registry() -> tuple[str | None, dict[str, Any]]:
"""Parse the ``datasets.yaml`` registry shipped inside the ``spatialdata`` package."""
import importlib.resources

from scverse_misc.datasets import parse_registry

registry = importlib.resources.files("spatialdata").joinpath("datasets.yaml")
with importlib.resources.as_file(registry) as registry_path:
base_url: str | None
datasets: dict[str, Any]
base_url, datasets = parse_registry(registry_path)
return base_url, datasets


def _cache_dir(path: str | None) -> Path:
"""Resolve the cache directory, defaulting to the OS cache location for ``"spatialdata"``."""
import pooch

return Path(path) if path is not None else Path(pooch.os_cache("spatialdata"))


def cells(path: str | None = None) -> SpatialData:
"""
Cells dataset.
Comment thread
LucaMarconato marked this conversation as resolved.

Download the ``cells`` example dataset and load it as a :class:`~spatialdata.SpatialData`
object. The download is hash-verified and cached, so repeated calls reuse the local copy
instead of downloading again.

The dataset is a small region of a Xenium Prime Cervical Cancer sample and contains three
multiscale images (``he_aligned``, ``he_image``, ``morphology_focus``), three multiscale
label layers (``cell_labels``, ``nucleus_labels``, ``tissue_labels``), the ``transcripts``
points, the ``cell_boundaries`` and ``nucleus_boundaries`` shapes, and a cell-by-gene
``table`` annotating the 94 cells.

Parameters
----------
path
Directory in which to cache the downloaded data. If `None`, the default OS cache
location is used (:func:`pooch.os_cache` for ``"spatialdata"``).

Returns
-------
SpatialData object with the cells dataset.
"""
from scverse_misc.datasets import fetch

base_url, datasets = _shipped_registry()
sdata: SpatialData = fetch(datasets["cells"], _cache_dir(path), base_url=base_url)
return sdata


class RaccoonDataset:
"""Raccoon dataset."""

Expand Down
23 changes: 23 additions & 0 deletions src/spatialdata/datasets.yaml
Original file line number Diff line number Diff line change
@@ -0,0 +1,23 @@
# Registry of downloadable example datasets for ``spatialdata.datasets``.
#
# Parsed by ``scverse_misc.datasets.parse_registry`` and fetched (downloaded,
# hash-verified, cached and loaded) via ``scverse_misc.datasets.fetch``.
#
# type: spatialdata -> a .zip that extracts to a single .zarr store
#
# Every dataset must list its ``license``; datasets under a license that requires
# attribution must also carry an ``attribution`` string crediting the original source.
base_url: https://exampledata.scverse.org/spatialdata/
datasets:
cells:
type: spatialdata
doc_header: Cells dataset as a SpatialData object.
license: CC BY 4.0
attribution: >-
Derived from the 10x Genomics Xenium Prime Cervical Cancer FFPE dataset
(https://www.10xgenomics.com/datasets/xenium-prime-ffpe-human-cervical-cancer),
subset to a small tissue region. Licensed under CC BY 4.0.
files:
- name: cells.zip
s3_key: cells.zip
sha256: dc9613cb9e16fd2cd8d83f3a9586eeda4af5ba8ba366f1066efb51305820c5fb
46 changes: 45 additions & 1 deletion tests/datasets/test_datasets.py
Original file line number Diff line number Diff line change
@@ -1,6 +1,12 @@
from __future__ import annotations

from spatialdata.datasets import blobs, raccoon
from pathlib import Path

import pooch
import pytest

from spatialdata import SpatialData
from spatialdata.datasets import _cache_dir, _shipped_registry, blobs, cells, raccoon


def test_datasets() -> None:
Expand All @@ -26,3 +32,41 @@ def test_datasets() -> None:
assert sdata_raccoon.images["raccoon"].shape == (3, 768, 1024)
assert sdata_raccoon.labels["segmentation"].shape == (768, 1024)
_ = str(sdata_raccoon)


def test_cells_registry() -> None:
# Network-free: the shipped registry parses and exposes the cells dataset.
base_url, datasets = _shipped_registry()

assert base_url == "https://exampledata.scverse.org/spatialdata/"
entry = datasets["cells"]
assert entry.type == "spatialdata"
file = entry.file(name="cells.zip")
assert file.sha256 == "dc9613cb9e16fd2cd8d83f3a9586eeda4af5ba8ba366f1066efb51305820c5fb"
assert file.resolve_url(base_url) == "https://exampledata.scverse.org/spatialdata/cells.zip"


def test_cache_dir() -> None:
# Network-free: both branches of the cache-directory resolution.
assert _cache_dir("/tmp/example") == Path("/tmp/example")
assert _cache_dir(None) == Path(pooch.os_cache("spatialdata"))


@pytest.mark.slow
def test_cells_download(tmp_path) -> None:
# Downloads ~3 MB from the scverse example data bucket; opt out with `-m "not slow"`.

Copy link
Copy Markdown
Member

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Rather than opt-out, I'll try turning this into opt-in and skip the test by default.

Copy link
Copy Markdown
Member

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Done, made opt-in and opting in in the GitHub CI.

It's a small dataset here and not a big deal, but later when we test cloud support it's good that we differentiate between tests require a network and tests that do not.

sdata = cells(path=str(tmp_path))
assert isinstance(sdata, SpatialData)
Comment thread
LucaMarconato marked this conversation as resolved.

assert set(sdata.images) == {"he_aligned", "he_image", "morphology_focus"}
assert sdata.images["he_aligned"]["scale0"]["image"].shape == (3, 430, 540)
assert sdata.images["he_image"]["scale0"]["image"].shape == (3, 423, 339)
assert sdata.images["morphology_focus"]["scale0"]["image"].shape == (4, 430, 540)

assert set(sdata.labels) == {"cell_labels", "nucleus_labels", "tissue_labels"}
assert sdata.labels["cell_labels"]["scale0"]["image"].shape == (430, 540)

assert len(sdata.shapes["cell_boundaries"]) == 94
assert len(sdata.shapes["nucleus_boundaries"]) == 94
assert len(sdata.points["transcripts"].compute()) == 19479
assert sdata.tables["table"].shape == (94, 5101)
Loading