torch_em.data.datasets.electron_microscopy.fluoem

The FluoEM dataset contains an electron microscopy volume of mouse cortex with proofread axon instance segmentation, generated to validate a fluorescence-guided axon labeling method (FluoEM) for long-range connectomics.

Segmentation is dense but not uniformly proofread throughout the volume: some regions still show block-stitching seams from the automated reconstruction, visible as brief instance ID mismatches at internal block boundaries.

The data is available via the Helmstaedter Lab's public WEBKNOSSOS instance (organization MPI_Brain_Research). No license is specified by the data provider. The dataset was published in https://doi.org/10.7554/eLife.38976. Please cite this publication if you use the dataset in your research.

  1"""The FluoEM dataset contains an electron microscopy volume of mouse cortex with proofread
  2axon instance segmentation, generated to validate a fluorescence-guided axon labeling method
  3(FluoEM) for long-range connectomics.
  4
  5Segmentation is dense but not uniformly proofread throughout the volume: some regions still show
  6block-stitching seams from the automated reconstruction, visible as brief instance ID mismatches
  7at internal block boundaries.
  8
  9The data is available via the Helmstaedter Lab's public WEBKNOSSOS instance (organization
 10MPI_Brain_Research). No license is specified by the data provider.
 11The dataset was published in https://doi.org/10.7554/eLife.38976.
 12Please cite this publication if you use the dataset in your research.
 13"""
 14
 15import os
 16from typing import Tuple, List, Union
 17
 18import numpy as np
 19
 20from torch.utils.data import DataLoader, Dataset
 21
 22import torch_em
 23
 24from .. import util
 25
 26
 27DATASTORE_URL = "https://demo.wk1.connectomics.hpccloud.mpg.de/data/zarr"
 28DATASET_ID = "5abb89f348d7a73cea448019"
 29SHAPE_ZYX = (2767, 26242, 29000)
 30RESOLUTION_NM = (30, 11.24, 11.24)
 31
 32
 33def _bbox_to_str(bounding_box):
 34    import hashlib
 35    return hashlib.md5("_".join(str(v) for v in bounding_box).encode()).hexdigest()[:12]
 36
 37
 38def _open_remote_array(layer):
 39    import fsspec
 40    import zarr
 41    return zarr.open(fsspec.get_mapper(f"{DATASTORE_URL}/{DATASET_ID}/{layer}/1-1-1"), mode="r", zarr_format=2)
 42
 43
 44def get_fluoem_data(
 45    path: Union[os.PathLike, str],
 46    bounding_box: Tuple[int, int, int, int, int, int],
 47    download: bool = False,
 48) -> str:
 49    """Stream a subvolume of the FluoEM dataset and cache it as a zarr v3 store.
 50
 51    Args:
 52        path: Filepath to a folder where the cached zarr store will be saved.
 53        bounding_box: The region to fetch as (z_min, z_max, y_min, y_max, x_min, x_max) in voxel
 54            coordinates, at the dataset's native resolution.
 55        download: Whether to stream and cache the data if it is not present.
 56
 57    Returns:
 58        The filepath to the cached zarr store.
 59    """
 60    import zarr
 61    from zarr.codecs import BloscCodec
 62
 63    os.makedirs(str(path), exist_ok=True)
 64    zarr_path = os.path.join(str(path), f"{_bbox_to_str(bounding_box)}.zarr")
 65
 66    root = zarr.open_group(zarr_path, mode="a")
 67    if "raw" in root and "labels" in root:
 68        return zarr_path
 69
 70    if not download:
 71        raise RuntimeError(f"No cached data found at '{zarr_path}'. Set download=True to stream it.")
 72
 73    z_min, z_max, y_min, y_max, x_min, x_max = bounding_box
 74    shape_z, shape_y, shape_x = SHAPE_ZYX
 75    assert z_max <= shape_z and y_max <= shape_y and x_max <= shape_x, \
 76        f"Bounding box exceeds the FluoEM volume {SHAPE_ZYX}"
 77
 78    color = _open_remote_array("color")
 79    seg = _open_remote_array("segmentation")
 80
 81    # the remote arrays are indexed as (channel, x, y, z); torch_em volumes use (z, y, x).
 82    raw = np.transpose(color[0, x_min:x_max, y_min:y_max, z_min:z_max], (2, 1, 0))
 83    labels = np.transpose(seg[0, x_min:x_max, y_min:y_max, z_min:z_max], (2, 1, 0))
 84
 85    def _make_array(name, data, shuffle):
 86        array = root.create_array(
 87            name, shape=data.shape, chunks=(32, 512, 512), dtype=data.dtype,
 88            compressors=BloscCodec(cname="zstd", clevel=6, shuffle=shuffle),
 89        )
 90        array[:] = data
 91
 92    root.attrs["bounding_box"] = list(bounding_box)
 93    root.attrs["resolution_nm"] = list(RESOLUTION_NM)
 94    root.attrs["labels_are_exhaustive"] = False
 95
 96    _make_array("raw", raw, shuffle="shuffle")
 97    _make_array("labels", labels, shuffle="bitshuffle")
 98
 99    return zarr_path
100
101
102def get_fluoem_paths(
103    path: Union[os.PathLike, str],
104    bounding_boxes: List[Tuple[int, int, int, int, int, int]],
105    download: bool = False,
106) -> List[str]:
107    """Get paths to cached FluoEM zarr stores.
108
109    Args:
110        path: Filepath to a folder where the cached zarr stores will be saved.
111        bounding_boxes: List of regions to fetch, each as (z_min, z_max, y_min, y_max, x_min, x_max).
112        download: Whether to stream and cache the data if it is not present.
113
114    Returns:
115        List of filepaths to the cached zarr stores.
116    """
117    return [get_fluoem_data(path, bbox, download) for bbox in bounding_boxes]
118
119
120def get_fluoem_dataset(
121    path: Union[os.PathLike, str],
122    patch_shape: Tuple[int, int, int],
123    bounding_boxes: List[Tuple[int, int, int, int, int, int]],
124    download: bool = False,
125    **kwargs,
126) -> Dataset:
127    """Get the FluoEM dataset for axon instance segmentation.
128
129    Args:
130        path: Filepath to a folder where the cached zarr stores will be saved.
131        patch_shape: The patch shape (z, y, x) to use for training.
132        bounding_boxes: List of subvolumes to use, each as (z_min, z_max, y_min, y_max, x_min, x_max).
133        download: Whether to stream and cache data if not already present.
134        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`.
135
136    Returns:
137        The segmentation dataset.
138    """
139    assert len(patch_shape) == 3
140
141    paths = get_fluoem_paths(path, bounding_boxes, download)
142    kwargs = util.update_kwargs(kwargs, "is_seg_dataset", True)
143
144    return torch_em.default_segmentation_dataset(
145        raw_paths=paths,
146        raw_key="raw",
147        label_paths=paths,
148        label_key="labels",
149        patch_shape=patch_shape,
150        **kwargs,
151    )
152
153
154def get_fluoem_loader(
155    path: Union[os.PathLike, str],
156    patch_shape: Tuple[int, int, int],
157    batch_size: int,
158    bounding_boxes: List[Tuple[int, int, int, int, int, int]],
159    download: bool = False,
160    **kwargs,
161) -> DataLoader:
162    """Get the DataLoader for axon instance segmentation in the FluoEM dataset.
163
164    Args:
165        path: Filepath to a folder where the cached zarr stores will be saved.
166        patch_shape: The patch shape (z, y, x) to use for training.
167        batch_size: The batch size for training.
168        bounding_boxes: List of subvolumes to use, each as (z_min, z_max, y_min, y_max, x_min, x_max).
169        download: Whether to stream and cache data if not already present.
170        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the
171            PyTorch DataLoader.
172
173    Returns:
174        The DataLoader.
175    """
176    ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs)
177    dataset = get_fluoem_dataset(path, patch_shape, bounding_boxes, download, **ds_kwargs)
178    return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)
DATASTORE_URL = 'https://demo.wk1.connectomics.hpccloud.mpg.de/data/zarr'
DATASET_ID = '5abb89f348d7a73cea448019'
SHAPE_ZYX = (2767, 26242, 29000)
RESOLUTION_NM = (30, 11.24, 11.24)
def get_fluoem_data( path: Union[os.PathLike, str], bounding_box: Tuple[int, int, int, int, int, int], download: bool = False) -> str:
 45def get_fluoem_data(
 46    path: Union[os.PathLike, str],
 47    bounding_box: Tuple[int, int, int, int, int, int],
 48    download: bool = False,
 49) -> str:
 50    """Stream a subvolume of the FluoEM dataset and cache it as a zarr v3 store.
 51
 52    Args:
 53        path: Filepath to a folder where the cached zarr store will be saved.
 54        bounding_box: The region to fetch as (z_min, z_max, y_min, y_max, x_min, x_max) in voxel
 55            coordinates, at the dataset's native resolution.
 56        download: Whether to stream and cache the data if it is not present.
 57
 58    Returns:
 59        The filepath to the cached zarr store.
 60    """
 61    import zarr
 62    from zarr.codecs import BloscCodec
 63
 64    os.makedirs(str(path), exist_ok=True)
 65    zarr_path = os.path.join(str(path), f"{_bbox_to_str(bounding_box)}.zarr")
 66
 67    root = zarr.open_group(zarr_path, mode="a")
 68    if "raw" in root and "labels" in root:
 69        return zarr_path
 70
 71    if not download:
 72        raise RuntimeError(f"No cached data found at '{zarr_path}'. Set download=True to stream it.")
 73
 74    z_min, z_max, y_min, y_max, x_min, x_max = bounding_box
 75    shape_z, shape_y, shape_x = SHAPE_ZYX
 76    assert z_max <= shape_z and y_max <= shape_y and x_max <= shape_x, \
 77        f"Bounding box exceeds the FluoEM volume {SHAPE_ZYX}"
 78
 79    color = _open_remote_array("color")
 80    seg = _open_remote_array("segmentation")
 81
 82    # the remote arrays are indexed as (channel, x, y, z); torch_em volumes use (z, y, x).
 83    raw = np.transpose(color[0, x_min:x_max, y_min:y_max, z_min:z_max], (2, 1, 0))
 84    labels = np.transpose(seg[0, x_min:x_max, y_min:y_max, z_min:z_max], (2, 1, 0))
 85
 86    def _make_array(name, data, shuffle):
 87        array = root.create_array(
 88            name, shape=data.shape, chunks=(32, 512, 512), dtype=data.dtype,
 89            compressors=BloscCodec(cname="zstd", clevel=6, shuffle=shuffle),
 90        )
 91        array[:] = data
 92
 93    root.attrs["bounding_box"] = list(bounding_box)
 94    root.attrs["resolution_nm"] = list(RESOLUTION_NM)
 95    root.attrs["labels_are_exhaustive"] = False
 96
 97    _make_array("raw", raw, shuffle="shuffle")
 98    _make_array("labels", labels, shuffle="bitshuffle")
 99
100    return zarr_path

Stream a subvolume of the FluoEM dataset and cache it as a zarr v3 store.

Arguments:
  • path: Filepath to a folder where the cached zarr store will be saved.
  • bounding_box: The region to fetch as (z_min, z_max, y_min, y_max, x_min, x_max) in voxel coordinates, at the dataset's native resolution.
  • download: Whether to stream and cache the data if it is not present.
Returns:

The filepath to the cached zarr store.

def get_fluoem_paths( path: Union[os.PathLike, str], bounding_boxes: List[Tuple[int, int, int, int, int, int]], download: bool = False) -> List[str]:
103def get_fluoem_paths(
104    path: Union[os.PathLike, str],
105    bounding_boxes: List[Tuple[int, int, int, int, int, int]],
106    download: bool = False,
107) -> List[str]:
108    """Get paths to cached FluoEM zarr stores.
109
110    Args:
111        path: Filepath to a folder where the cached zarr stores will be saved.
112        bounding_boxes: List of regions to fetch, each as (z_min, z_max, y_min, y_max, x_min, x_max).
113        download: Whether to stream and cache the data if it is not present.
114
115    Returns:
116        List of filepaths to the cached zarr stores.
117    """
118    return [get_fluoem_data(path, bbox, download) for bbox in bounding_boxes]

Get paths to cached FluoEM zarr stores.

Arguments:
  • path: Filepath to a folder where the cached zarr stores will be saved.
  • bounding_boxes: List of regions to fetch, each as (z_min, z_max, y_min, y_max, x_min, x_max).
  • download: Whether to stream and cache the data if it is not present.
Returns:

List of filepaths to the cached zarr stores.

def get_fluoem_dataset( path: Union[os.PathLike, str], patch_shape: Tuple[int, int, int], bounding_boxes: List[Tuple[int, int, int, int, int, int]], download: bool = False, **kwargs) -> torch.utils.data.dataset.Dataset:
121def get_fluoem_dataset(
122    path: Union[os.PathLike, str],
123    patch_shape: Tuple[int, int, int],
124    bounding_boxes: List[Tuple[int, int, int, int, int, int]],
125    download: bool = False,
126    **kwargs,
127) -> Dataset:
128    """Get the FluoEM dataset for axon instance segmentation.
129
130    Args:
131        path: Filepath to a folder where the cached zarr stores will be saved.
132        patch_shape: The patch shape (z, y, x) to use for training.
133        bounding_boxes: List of subvolumes to use, each as (z_min, z_max, y_min, y_max, x_min, x_max).
134        download: Whether to stream and cache data if not already present.
135        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`.
136
137    Returns:
138        The segmentation dataset.
139    """
140    assert len(patch_shape) == 3
141
142    paths = get_fluoem_paths(path, bounding_boxes, download)
143    kwargs = util.update_kwargs(kwargs, "is_seg_dataset", True)
144
145    return torch_em.default_segmentation_dataset(
146        raw_paths=paths,
147        raw_key="raw",
148        label_paths=paths,
149        label_key="labels",
150        patch_shape=patch_shape,
151        **kwargs,
152    )

Get the FluoEM dataset for axon instance segmentation.

Arguments:
  • path: Filepath to a folder where the cached zarr stores will be saved.
  • patch_shape: The patch shape (z, y, x) to use for training.
  • bounding_boxes: List of subvolumes to use, each as (z_min, z_max, y_min, y_max, x_min, x_max).
  • download: Whether to stream and cache data if not already present.
  • kwargs: Additional keyword arguments for torch_em.default_segmentation_dataset.
Returns:

The segmentation dataset.

def get_fluoem_loader( path: Union[os.PathLike, str], patch_shape: Tuple[int, int, int], batch_size: int, bounding_boxes: List[Tuple[int, int, int, int, int, int]], download: bool = False, **kwargs) -> torch.utils.data.dataloader.DataLoader:
155def get_fluoem_loader(
156    path: Union[os.PathLike, str],
157    patch_shape: Tuple[int, int, int],
158    batch_size: int,
159    bounding_boxes: List[Tuple[int, int, int, int, int, int]],
160    download: bool = False,
161    **kwargs,
162) -> DataLoader:
163    """Get the DataLoader for axon instance segmentation in the FluoEM dataset.
164
165    Args:
166        path: Filepath to a folder where the cached zarr stores will be saved.
167        patch_shape: The patch shape (z, y, x) to use for training.
168        batch_size: The batch size for training.
169        bounding_boxes: List of subvolumes to use, each as (z_min, z_max, y_min, y_max, x_min, x_max).
170        download: Whether to stream and cache data if not already present.
171        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the
172            PyTorch DataLoader.
173
174    Returns:
175        The DataLoader.
176    """
177    ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs)
178    dataset = get_fluoem_dataset(path, patch_shape, bounding_boxes, download, **ds_kwargs)
179    return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)

Get the DataLoader for axon instance segmentation in the FluoEM dataset.

Arguments:
  • path: Filepath to a folder where the cached zarr stores will be saved.
  • patch_shape: The patch shape (z, y, x) to use for training.
  • batch_size: The batch size for training.
  • bounding_boxes: List of subvolumes to use, each as (z_min, z_max, y_min, y_max, x_min, x_max).
  • download: Whether to stream and cache data if not already present.
  • kwargs: Additional keyword arguments for torch_em.default_segmentation_dataset or for the PyTorch DataLoader.
Returns:

The DataLoader.