torch_em.data.datasets.electron_microscopy.neuromast_connectomics

The NeuromastConnectomics dataset contains dense electron microscopy connectomic reconstructions of zebrafish lateral-line neuromasts, with proofread cell instance segmentation of hair cells and afferent/efferent neurons.

Ten of the paper's fourteen named neuromast volumes are included here; the remaining four were not found as separate public entries at the time of writing this module.

The data is available via the Hudspeth Lab's public WEBKNOSSOS instance (organization 'The Rockefeller University'). No license is specified by the data provider. The dataset was published in https://doi.org/10.7554/eLife.33988. Please cite this publication if you use the dataset in your research.

  1"""The NeuromastConnectomics dataset contains dense electron microscopy connectomic
  2reconstructions of zebrafish lateral-line neuromasts, with proofread cell instance
  3segmentation of hair cells and afferent/efferent neurons.
  4
  5Ten of the paper's fourteen named neuromast volumes are included here; the remaining four
  6were not found as separate public entries at the time of writing this module.
  7
  8The data is available via the Hudspeth Lab's public WEBKNOSSOS instance (organization
  9'The Rockefeller University'). No license is specified by the data provider.
 10The dataset was published in https://doi.org/10.7554/eLife.33988.
 11Please cite this publication if you use the dataset in your research.
 12"""
 13
 14import os
 15from typing import Dict, List, Literal, Tuple, Union
 16
 17import numpy as np
 18
 19from torch.utils.data import DataLoader, Dataset
 20
 21import torch_em
 22
 23from .. import util
 24
 25
 26DATASTORE_URL = "https://data-humerus.webknossos.org/data/zarr"
 27
 28SAMPLES: Dict[str, dict] = {
 29    "wt1": {"dataset_id": "5b3ce7bb1d0000340240b921", "shape_zyx": (1024, 9216, 9216), "resolution_nm": (30, 6, 6)},
 30    "wt2": {"dataset_id": "5b3ce7bb1d0000340240b923", "shape_zyx": (1024, 10240, 13312), "resolution_nm": (50, 5, 5)},
 31    "wt6": {"dataset_id": "5b3ce7bb1d0000340240b91d", "shape_zyx": (2048, 10240, 11264), "resolution_nm": (30, 5, 5)},
 32    "wt7": {"dataset_id": "5b3ce7bb1d0000340240b91c", "shape_zyx": (1024, 8192, 10240), "resolution_nm": (30, 6, 6)},
 33    "wt8": {"dataset_id": "5b3ce7bb1d0000340240b91f", "shape_zyx": (2048, 11264, 11264), "resolution_nm": (30, 6, 6)},
 34    "t1": {"dataset_id": "5b3ce7bb1d0000340240b91b", "shape_zyx": (2048, 6144, 6144), "resolution_nm": (30, 6, 6)},
 35    "t3": {"dataset_id": "5b4238a61d0000c42a1ab5e3", "shape_zyx": (2048, 7168, 8192), "resolution_nm": (30, 6, 6)},
 36    "t4": {"dataset_id": "5b4238a61d0000c42a1ab5e5", "shape_zyx": (2048, 10240, 9216), "resolution_nm": (30, 6, 6)},
 37    "n1": {"dataset_id": "5b4238a61d00004b1d1ab5e2", "shape_zyx": (1024, 9216, 11264), "resolution_nm": (30, 6, 6)},
 38    "n2": {"dataset_id": "5b4238a61d0000c42a1ab5e4", "shape_zyx": (1024, 9216, 9216), "resolution_nm": (30, 6, 6)},
 39}
 40
 41
 42def _bbox_to_str(bounding_box):
 43    import hashlib
 44    return hashlib.md5("_".join(str(v) for v in bounding_box).encode()).hexdigest()[:12]
 45
 46
 47def _open_remote_array(dataset_id, layer):
 48    import fsspec
 49    import zarr
 50    return zarr.open(fsspec.get_mapper(f"{DATASTORE_URL}/{dataset_id}/{layer}/1-1-1"), mode="r", zarr_format=2)
 51
 52
 53def get_neuromast_connectomics_data(
 54    path: Union[os.PathLike, str],
 55    bounding_box: Tuple[int, int, int, int, int, int],
 56    sample: Literal["wt1", "wt2", "wt6", "wt7", "wt8", "t1", "t3", "t4", "n1", "n2"] = "wt1",
 57    download: bool = False,
 58) -> str:
 59    """Stream a subvolume of the NeuromastConnectomics dataset and cache it as a zarr v3 store.
 60
 61    Args:
 62        path: Filepath to a folder where the cached zarr store will be saved.
 63        bounding_box: The region to fetch as (z_min, z_max, y_min, y_max, x_min, x_max) in voxel
 64            coordinates, at the dataset's native resolution (see `SAMPLES`).
 65        sample: Which neuromast to use. See `SAMPLES` for the available keys.
 66        download: Whether to stream and cache the data if it is not present.
 67
 68    Returns:
 69        The filepath to the cached zarr store.
 70    """
 71    import zarr
 72    from zarr.codecs import BloscCodec
 73
 74    if sample not in SAMPLES:
 75        raise ValueError(f"sample must be one of {list(SAMPLES)}, got {sample!r}")
 76
 77    os.makedirs(str(path), exist_ok=True)
 78    zarr_path = os.path.join(str(path), f"{sample}_{_bbox_to_str(bounding_box)}.zarr")
 79
 80    root = zarr.open_group(zarr_path, mode="a")
 81    if "raw" in root and "labels" in root:
 82        return zarr_path
 83
 84    if not download:
 85        raise RuntimeError(f"No cached data found at '{zarr_path}'. Set download=True to stream it.")
 86
 87    z_min, z_max, y_min, y_max, x_min, x_max = bounding_box
 88    info = SAMPLES[sample]
 89    shape_z, shape_y, shape_x = info["shape_zyx"]
 90    assert z_max <= shape_z and y_max <= shape_y and x_max <= shape_x, \
 91        f"Bounding box exceeds the {sample} volume {info['shape_zyx']}"
 92
 93    color = _open_remote_array(info["dataset_id"], "color")
 94    seg = _open_remote_array(info["dataset_id"], "segmentation")
 95
 96    # the remote arrays are indexed as (channel, x, y, z); torch_em volumes use (z, y, x).
 97    raw = np.transpose(color[0, x_min:x_max, y_min:y_max, z_min:z_max], (2, 1, 0))
 98    labels = np.transpose(seg[0, x_min:x_max, y_min:y_max, z_min:z_max], (2, 1, 0))
 99
100    def _make_array(name, data, shuffle):
101        array = root.create_array(
102            name, shape=data.shape, chunks=(32, 512, 512), dtype=data.dtype,
103            compressors=BloscCodec(cname="zstd", clevel=6, shuffle=shuffle),
104        )
105        array[:] = data
106
107    root.attrs["bounding_box"] = list(bounding_box)
108    root.attrs["sample"] = sample
109    root.attrs["resolution_nm"] = list(info["resolution_nm"])
110    root.attrs["labels_are_exhaustive"] = False
111
112    _make_array("raw", raw, shuffle="shuffle")
113    _make_array("labels", labels, shuffle="bitshuffle")
114
115    return zarr_path
116
117
118def get_neuromast_connectomics_paths(
119    path: Union[os.PathLike, str],
120    bounding_boxes: List[Tuple[int, int, int, int, int, int]],
121    sample: Literal["wt1", "wt2", "wt6", "wt7", "wt8", "t1", "t3", "t4", "n1", "n2"] = "wt1",
122    download: bool = False,
123) -> List[str]:
124    """Get paths to cached NeuromastConnectomics zarr stores.
125
126    Args:
127        path: Filepath to a folder where the cached zarr stores will be saved.
128        bounding_boxes: List of regions to fetch, each as (z_min, z_max, y_min, y_max, x_min, x_max).
129        sample: Which neuromast to use. See `SAMPLES` for the available keys.
130        download: Whether to stream and cache the data if it is not present.
131
132    Returns:
133        List of filepaths to the cached zarr stores.
134    """
135    return [get_neuromast_connectomics_data(path, bbox, sample, download) for bbox in bounding_boxes]
136
137
138def get_neuromast_connectomics_dataset(
139    path: Union[os.PathLike, str],
140    patch_shape: Tuple[int, int, int],
141    bounding_boxes: List[Tuple[int, int, int, int, int, int]],
142    sample: Literal["wt1", "wt2", "wt6", "wt7", "wt8", "t1", "t3", "t4", "n1", "n2"] = "wt1",
143    download: bool = False,
144    **kwargs,
145) -> Dataset:
146    """Get the NeuromastConnectomics dataset for cell instance segmentation.
147
148    Args:
149        path: Filepath to a folder where the cached zarr stores will be saved.
150        patch_shape: The patch shape (z, y, x) to use for training.
151        bounding_boxes: List of subvolumes to use, each as (z_min, z_max, y_min, y_max, x_min, x_max).
152        sample: Which neuromast to use. See `SAMPLES` for the available keys.
153        download: Whether to stream and cache data if not already present.
154        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`.
155
156    Returns:
157        The segmentation dataset.
158    """
159    assert len(patch_shape) == 3
160
161    paths = get_neuromast_connectomics_paths(path, bounding_boxes, sample, download)
162    kwargs = util.update_kwargs(kwargs, "is_seg_dataset", True)
163
164    return torch_em.default_segmentation_dataset(
165        raw_paths=paths,
166        raw_key="raw",
167        label_paths=paths,
168        label_key="labels",
169        patch_shape=patch_shape,
170        **kwargs,
171    )
172
173
174def get_neuromast_connectomics_loader(
175    path: Union[os.PathLike, str],
176    patch_shape: Tuple[int, int, int],
177    batch_size: int,
178    bounding_boxes: List[Tuple[int, int, int, int, int, int]],
179    sample: Literal["wt1", "wt2", "wt6", "wt7", "wt8", "t1", "t3", "t4", "n1", "n2"] = "wt1",
180    download: bool = False,
181    **kwargs,
182) -> DataLoader:
183    """Get the DataLoader for cell instance segmentation in the NeuromastConnectomics dataset.
184
185    Args:
186        path: Filepath to a folder where the cached zarr stores will be saved.
187        patch_shape: The patch shape (z, y, x) to use for training.
188        batch_size: The batch size for training.
189        bounding_boxes: List of subvolumes to use, each as (z_min, z_max, y_min, y_max, x_min, x_max).
190        sample: Which neuromast to use. See `SAMPLES` for the available keys.
191        download: Whether to stream and cache data if not already present.
192        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the
193            PyTorch DataLoader.
194
195    Returns:
196        The DataLoader.
197    """
198    ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs)
199    dataset = get_neuromast_connectomics_dataset(path, patch_shape, bounding_boxes, sample, download, **ds_kwargs)
200    return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)
DATASTORE_URL = 'https://data-humerus.webknossos.org/data/zarr'
SAMPLES: Dict[str, dict] = {'wt1': {'dataset_id': '5b3ce7bb1d0000340240b921', 'shape_zyx': (1024, 9216, 9216), 'resolution_nm': (30, 6, 6)}, 'wt2': {'dataset_id': '5b3ce7bb1d0000340240b923', 'shape_zyx': (1024, 10240, 13312), 'resolution_nm': (50, 5, 5)}, 'wt6': {'dataset_id': '5b3ce7bb1d0000340240b91d', 'shape_zyx': (2048, 10240, 11264), 'resolution_nm': (30, 5, 5)}, 'wt7': {'dataset_id': '5b3ce7bb1d0000340240b91c', 'shape_zyx': (1024, 8192, 10240), 'resolution_nm': (30, 6, 6)}, 'wt8': {'dataset_id': '5b3ce7bb1d0000340240b91f', 'shape_zyx': (2048, 11264, 11264), 'resolution_nm': (30, 6, 6)}, 't1': {'dataset_id': '5b3ce7bb1d0000340240b91b', 'shape_zyx': (2048, 6144, 6144), 'resolution_nm': (30, 6, 6)}, 't3': {'dataset_id': '5b4238a61d0000c42a1ab5e3', 'shape_zyx': (2048, 7168, 8192), 'resolution_nm': (30, 6, 6)}, 't4': {'dataset_id': '5b4238a61d0000c42a1ab5e5', 'shape_zyx': (2048, 10240, 9216), 'resolution_nm': (30, 6, 6)}, 'n1': {'dataset_id': '5b4238a61d00004b1d1ab5e2', 'shape_zyx': (1024, 9216, 11264), 'resolution_nm': (30, 6, 6)}, 'n2': {'dataset_id': '5b4238a61d0000c42a1ab5e4', 'shape_zyx': (1024, 9216, 9216), 'resolution_nm': (30, 6, 6)}}
def get_neuromast_connectomics_data( path: Union[os.PathLike, str], bounding_box: Tuple[int, int, int, int, int, int], sample: Literal['wt1', 'wt2', 'wt6', 'wt7', 'wt8', 't1', 't3', 't4', 'n1', 'n2'] = 'wt1', download: bool = False) -> str:
 54def get_neuromast_connectomics_data(
 55    path: Union[os.PathLike, str],
 56    bounding_box: Tuple[int, int, int, int, int, int],
 57    sample: Literal["wt1", "wt2", "wt6", "wt7", "wt8", "t1", "t3", "t4", "n1", "n2"] = "wt1",
 58    download: bool = False,
 59) -> str:
 60    """Stream a subvolume of the NeuromastConnectomics dataset and cache it as a zarr v3 store.
 61
 62    Args:
 63        path: Filepath to a folder where the cached zarr store will be saved.
 64        bounding_box: The region to fetch as (z_min, z_max, y_min, y_max, x_min, x_max) in voxel
 65            coordinates, at the dataset's native resolution (see `SAMPLES`).
 66        sample: Which neuromast to use. See `SAMPLES` for the available keys.
 67        download: Whether to stream and cache the data if it is not present.
 68
 69    Returns:
 70        The filepath to the cached zarr store.
 71    """
 72    import zarr
 73    from zarr.codecs import BloscCodec
 74
 75    if sample not in SAMPLES:
 76        raise ValueError(f"sample must be one of {list(SAMPLES)}, got {sample!r}")
 77
 78    os.makedirs(str(path), exist_ok=True)
 79    zarr_path = os.path.join(str(path), f"{sample}_{_bbox_to_str(bounding_box)}.zarr")
 80
 81    root = zarr.open_group(zarr_path, mode="a")
 82    if "raw" in root and "labels" in root:
 83        return zarr_path
 84
 85    if not download:
 86        raise RuntimeError(f"No cached data found at '{zarr_path}'. Set download=True to stream it.")
 87
 88    z_min, z_max, y_min, y_max, x_min, x_max = bounding_box
 89    info = SAMPLES[sample]
 90    shape_z, shape_y, shape_x = info["shape_zyx"]
 91    assert z_max <= shape_z and y_max <= shape_y and x_max <= shape_x, \
 92        f"Bounding box exceeds the {sample} volume {info['shape_zyx']}"
 93
 94    color = _open_remote_array(info["dataset_id"], "color")
 95    seg = _open_remote_array(info["dataset_id"], "segmentation")
 96
 97    # the remote arrays are indexed as (channel, x, y, z); torch_em volumes use (z, y, x).
 98    raw = np.transpose(color[0, x_min:x_max, y_min:y_max, z_min:z_max], (2, 1, 0))
 99    labels = np.transpose(seg[0, x_min:x_max, y_min:y_max, z_min:z_max], (2, 1, 0))
100
101    def _make_array(name, data, shuffle):
102        array = root.create_array(
103            name, shape=data.shape, chunks=(32, 512, 512), dtype=data.dtype,
104            compressors=BloscCodec(cname="zstd", clevel=6, shuffle=shuffle),
105        )
106        array[:] = data
107
108    root.attrs["bounding_box"] = list(bounding_box)
109    root.attrs["sample"] = sample
110    root.attrs["resolution_nm"] = list(info["resolution_nm"])
111    root.attrs["labels_are_exhaustive"] = False
112
113    _make_array("raw", raw, shuffle="shuffle")
114    _make_array("labels", labels, shuffle="bitshuffle")
115
116    return zarr_path

Stream a subvolume of the NeuromastConnectomics dataset and cache it as a zarr v3 store.

Arguments:
  • path: Filepath to a folder where the cached zarr store will be saved.
  • bounding_box: The region to fetch as (z_min, z_max, y_min, y_max, x_min, x_max) in voxel coordinates, at the dataset's native resolution (see SAMPLES).
  • sample: Which neuromast to use. See SAMPLES for the available keys.
  • download: Whether to stream and cache the data if it is not present.
Returns:

The filepath to the cached zarr store.

def get_neuromast_connectomics_paths( path: Union[os.PathLike, str], bounding_boxes: List[Tuple[int, int, int, int, int, int]], sample: Literal['wt1', 'wt2', 'wt6', 'wt7', 'wt8', 't1', 't3', 't4', 'n1', 'n2'] = 'wt1', download: bool = False) -> List[str]:
119def get_neuromast_connectomics_paths(
120    path: Union[os.PathLike, str],
121    bounding_boxes: List[Tuple[int, int, int, int, int, int]],
122    sample: Literal["wt1", "wt2", "wt6", "wt7", "wt8", "t1", "t3", "t4", "n1", "n2"] = "wt1",
123    download: bool = False,
124) -> List[str]:
125    """Get paths to cached NeuromastConnectomics zarr stores.
126
127    Args:
128        path: Filepath to a folder where the cached zarr stores will be saved.
129        bounding_boxes: List of regions to fetch, each as (z_min, z_max, y_min, y_max, x_min, x_max).
130        sample: Which neuromast to use. See `SAMPLES` for the available keys.
131        download: Whether to stream and cache the data if it is not present.
132
133    Returns:
134        List of filepaths to the cached zarr stores.
135    """
136    return [get_neuromast_connectomics_data(path, bbox, sample, download) for bbox in bounding_boxes]

Get paths to cached NeuromastConnectomics zarr stores.

Arguments:
  • path: Filepath to a folder where the cached zarr stores will be saved.
  • bounding_boxes: List of regions to fetch, each as (z_min, z_max, y_min, y_max, x_min, x_max).
  • sample: Which neuromast to use. See SAMPLES for the available keys.
  • download: Whether to stream and cache the data if it is not present.
Returns:

List of filepaths to the cached zarr stores.

def get_neuromast_connectomics_dataset( path: Union[os.PathLike, str], patch_shape: Tuple[int, int, int], bounding_boxes: List[Tuple[int, int, int, int, int, int]], sample: Literal['wt1', 'wt2', 'wt6', 'wt7', 'wt8', 't1', 't3', 't4', 'n1', 'n2'] = 'wt1', download: bool = False, **kwargs) -> torch.utils.data.dataset.Dataset:
139def get_neuromast_connectomics_dataset(
140    path: Union[os.PathLike, str],
141    patch_shape: Tuple[int, int, int],
142    bounding_boxes: List[Tuple[int, int, int, int, int, int]],
143    sample: Literal["wt1", "wt2", "wt6", "wt7", "wt8", "t1", "t3", "t4", "n1", "n2"] = "wt1",
144    download: bool = False,
145    **kwargs,
146) -> Dataset:
147    """Get the NeuromastConnectomics dataset for cell instance segmentation.
148
149    Args:
150        path: Filepath to a folder where the cached zarr stores will be saved.
151        patch_shape: The patch shape (z, y, x) to use for training.
152        bounding_boxes: List of subvolumes to use, each as (z_min, z_max, y_min, y_max, x_min, x_max).
153        sample: Which neuromast to use. See `SAMPLES` for the available keys.
154        download: Whether to stream and cache data if not already present.
155        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`.
156
157    Returns:
158        The segmentation dataset.
159    """
160    assert len(patch_shape) == 3
161
162    paths = get_neuromast_connectomics_paths(path, bounding_boxes, sample, download)
163    kwargs = util.update_kwargs(kwargs, "is_seg_dataset", True)
164
165    return torch_em.default_segmentation_dataset(
166        raw_paths=paths,
167        raw_key="raw",
168        label_paths=paths,
169        label_key="labels",
170        patch_shape=patch_shape,
171        **kwargs,
172    )

Get the NeuromastConnectomics dataset for cell instance segmentation.

Arguments:
  • path: Filepath to a folder where the cached zarr stores will be saved.
  • patch_shape: The patch shape (z, y, x) to use for training.
  • bounding_boxes: List of subvolumes to use, each as (z_min, z_max, y_min, y_max, x_min, x_max).
  • sample: Which neuromast to use. See SAMPLES for the available keys.
  • download: Whether to stream and cache data if not already present.
  • kwargs: Additional keyword arguments for torch_em.default_segmentation_dataset.
Returns:

The segmentation dataset.

def get_neuromast_connectomics_loader( path: Union[os.PathLike, str], patch_shape: Tuple[int, int, int], batch_size: int, bounding_boxes: List[Tuple[int, int, int, int, int, int]], sample: Literal['wt1', 'wt2', 'wt6', 'wt7', 'wt8', 't1', 't3', 't4', 'n1', 'n2'] = 'wt1', download: bool = False, **kwargs) -> torch.utils.data.dataloader.DataLoader:
175def get_neuromast_connectomics_loader(
176    path: Union[os.PathLike, str],
177    patch_shape: Tuple[int, int, int],
178    batch_size: int,
179    bounding_boxes: List[Tuple[int, int, int, int, int, int]],
180    sample: Literal["wt1", "wt2", "wt6", "wt7", "wt8", "t1", "t3", "t4", "n1", "n2"] = "wt1",
181    download: bool = False,
182    **kwargs,
183) -> DataLoader:
184    """Get the DataLoader for cell instance segmentation in the NeuromastConnectomics dataset.
185
186    Args:
187        path: Filepath to a folder where the cached zarr stores will be saved.
188        patch_shape: The patch shape (z, y, x) to use for training.
189        batch_size: The batch size for training.
190        bounding_boxes: List of subvolumes to use, each as (z_min, z_max, y_min, y_max, x_min, x_max).
191        sample: Which neuromast to use. See `SAMPLES` for the available keys.
192        download: Whether to stream and cache data if not already present.
193        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the
194            PyTorch DataLoader.
195
196    Returns:
197        The DataLoader.
198    """
199    ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs)
200    dataset = get_neuromast_connectomics_dataset(path, patch_shape, bounding_boxes, sample, download, **ds_kwargs)
201    return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)

Get the DataLoader for cell instance segmentation in the NeuromastConnectomics dataset.

Arguments:
  • path: Filepath to a folder where the cached zarr stores will be saved.
  • patch_shape: The patch shape (z, y, x) to use for training.
  • batch_size: The batch size for training.
  • bounding_boxes: List of subvolumes to use, each as (z_min, z_max, y_min, y_max, x_min, x_max).
  • sample: Which neuromast to use. See SAMPLES for the available keys.
  • download: Whether to stream and cache data if not already present.
  • kwargs: Additional keyword arguments for torch_em.default_segmentation_dataset or for the PyTorch DataLoader.
Returns:

The DataLoader.