torch_em.data.datasets.electron_microscopy.mito_anf

The Mito-ANFs dataset contains mitochondria instance segmentation for two mouse cochlear serial block-face EM (SBEM) volumes of auditory nerve fibers in the inner spiral bundle region, at 12x12x50 nm (XYZ) resolution.

Labels come from a 3D U-Net prediction followed by manual proofreading. Annotation coverage is sparse, not exhaustive: only mitochondria belonging to traced auditory nerve fibers are labeled.

The data is available at https://webknossos.org/datasets/b2275d664e4c2a96 (organization 'Yunfeng Hua Lab'). No license is specified by the data provider. The dataset was published in https://doi.org/10.1007/s10162-024-00957-y. Please cite this publication if you use the dataset in your research.

  1"""The Mito-ANFs dataset contains mitochondria instance segmentation for two mouse cochlear
  2serial block-face EM (SBEM) volumes of auditory nerve fibers in the inner spiral bundle region,
  3at 12x12x50 nm (XYZ) resolution.
  4
  5Labels come from a 3D U-Net prediction followed by manual proofreading. Annotation coverage is
  6sparse, not exhaustive: only mitochondria belonging to traced auditory nerve fibers are labeled.
  7
  8The data is available at https://webknossos.org/datasets/b2275d664e4c2a96 (organization
  9'Yunfeng Hua Lab'). No license is specified by the data provider.
 10The dataset was published in https://doi.org/10.1007/s10162-024-00957-y.
 11Please cite this publication if you use the dataset in your research.
 12"""
 13
 14import os
 15from typing import Dict, List, Literal, Tuple, Union
 16
 17import numpy as np
 18
 19from torch.utils.data import DataLoader, Dataset
 20
 21import torch_em
 22
 23from .. import util
 24
 25
 26BASE_URL = "https://data-humerus.webknossos.org/data/zarr"
 27
 28SAMPLES: Dict[str, dict] = {
 29    "M1": {
 30        "dataset_id": "652d442501000053049c0270",
 31        "seg_layer": "mito_0_890_proof",
 32        "shape_zyx": (2494, 23573, 11333),
 33        "resolution_nm": (50, 12, 12),
 34    },
 35    "M2": {
 36        "dataset_id": "652d563301000068049c066e",
 37        "seg_layer": "mito_650_1849_proof",
 38        "shape_zyx": (2558, 25183, 10329),
 39        "resolution_nm": (50, 12, 12),
 40    },
 41}
 42
 43
 44def _bbox_to_str(bounding_box):
 45    import hashlib
 46    return hashlib.md5("_".join(str(v) for v in bounding_box).encode()).hexdigest()[:12]
 47
 48
 49def _open_remote_array(dataset_id, layer):
 50    import fsspec
 51    import zarr
 52    return zarr.open(fsspec.get_mapper(f"{BASE_URL}/{dataset_id}/{layer}/1-1-1"), mode="r", zarr_format=2)
 53
 54
 55def get_mito_anf_data(
 56    path: Union[os.PathLike, str],
 57    bounding_box: Tuple[int, int, int, int, int, int],
 58    sample: Literal["M1", "M2"] = "M1",
 59    download: bool = False,
 60) -> str:
 61    """Stream a subvolume of the Mito-ANFs dataset and cache it as a zarr v3 store.
 62
 63    Args:
 64        path: Filepath to a folder where the cached zarr store will be saved.
 65        bounding_box: The region to fetch as (z_min, z_max, y_min, y_max, x_min, x_max) in voxel
 66            coordinates, at the dataset's native resolution (see `SAMPLES`).
 67        sample: Which cochlear sample to use. One of 'M1', 'M2'.
 68        download: Whether to stream and cache the data if it is not present.
 69
 70    Returns:
 71        The filepath to the cached zarr store.
 72    """
 73    import zarr
 74    from zarr.codecs import BloscCodec
 75
 76    if sample not in SAMPLES:
 77        raise ValueError(f"sample must be one of {list(SAMPLES)}, got {sample!r}")
 78
 79    os.makedirs(str(path), exist_ok=True)
 80    zarr_path = os.path.join(str(path), f"{sample}_{_bbox_to_str(bounding_box)}.zarr")
 81
 82    root = zarr.open_group(zarr_path, mode="a")
 83    if "raw" in root and "labels" in root:
 84        return zarr_path
 85
 86    if not download:
 87        raise RuntimeError(f"No cached data found at '{zarr_path}'. Set download=True to stream it.")
 88
 89    z_min, z_max, y_min, y_max, x_min, x_max = bounding_box
 90    info = SAMPLES[sample]
 91    shape_z, shape_y, shape_x = info["shape_zyx"]
 92    assert z_max <= shape_z and y_max <= shape_y and x_max <= shape_x, \
 93        f"Bounding box exceeds the {sample} volume {info['shape_zyx']}"
 94
 95    color = _open_remote_array(info["dataset_id"], "color")
 96    seg = _open_remote_array(info["dataset_id"], info["seg_layer"])
 97
 98    # the remote arrays are indexed as (channel, x, y, z); torch_em volumes use (z, y, x).
 99    raw = np.transpose(color[0, x_min:x_max, y_min:y_max, z_min:z_max], (2, 1, 0))
100    labels = np.transpose(seg[0, x_min:x_max, y_min:y_max, z_min:z_max], (2, 1, 0))
101
102    def _make_array(name, data, shuffle):
103        array = root.create_array(
104            name, shape=data.shape, chunks=(32, 512, 512), dtype=data.dtype,
105            compressors=BloscCodec(cname="zstd", clevel=6, shuffle=shuffle),
106        )
107        array[:] = data
108
109    root.attrs["bounding_box"] = list(bounding_box)
110    root.attrs["sample"] = sample
111    root.attrs["resolution_nm"] = list(info["resolution_nm"])
112    root.attrs["labels_are_exhaustive"] = False
113
114    _make_array("raw", raw, shuffle="shuffle")
115    _make_array("labels", labels, shuffle="bitshuffle")
116
117    return zarr_path
118
119
120def get_mito_anf_paths(
121    path: Union[os.PathLike, str],
122    bounding_boxes: List[Tuple[int, int, int, int, int, int]],
123    sample: Literal["M1", "M2"] = "M1",
124    download: bool = False,
125) -> List[str]:
126    """Get paths to cached Mito-ANFs zarr stores.
127
128    Args:
129        path: Filepath to a folder where the cached zarr stores will be saved.
130        bounding_boxes: List of regions to fetch, each as (z_min, z_max, y_min, y_max, x_min, x_max).
131        sample: Which cochlear sample to use. One of 'M1', 'M2'.
132        download: Whether to stream and cache the data if it is not present.
133
134    Returns:
135        List of filepaths to the cached zarr stores.
136    """
137    return [get_mito_anf_data(path, bbox, sample, download) for bbox in bounding_boxes]
138
139
140def get_mito_anf_dataset(
141    path: Union[os.PathLike, str],
142    patch_shape: Tuple[int, int, int],
143    bounding_boxes: List[Tuple[int, int, int, int, int, int]],
144    sample: Literal["M1", "M2"] = "M1",
145    download: bool = False,
146    **kwargs,
147) -> Dataset:
148    """Get the Mito-ANFs dataset for mitochondria instance segmentation.
149
150    Labels are not exhaustive manual ground truth: only mitochondria belonging to traced
151    auditory nerve fibers are annotated.
152
153    Args:
154        path: Filepath to a folder where the cached zarr stores will be saved.
155        patch_shape: The patch shape (z, y, x) to use for training.
156        bounding_boxes: List of subvolumes to use, each as (z_min, z_max, y_min, y_max, x_min, x_max).
157        sample: Which cochlear sample to use. One of 'M1', 'M2'.
158        download: Whether to stream and cache data if not already present.
159        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`.
160
161    Returns:
162        The segmentation dataset.
163    """
164    assert len(patch_shape) == 3
165
166    paths = get_mito_anf_paths(path, bounding_boxes, sample, download)
167    kwargs = util.update_kwargs(kwargs, "is_seg_dataset", True)
168
169    return torch_em.default_segmentation_dataset(
170        raw_paths=paths,
171        raw_key="raw",
172        label_paths=paths,
173        label_key="labels",
174        patch_shape=patch_shape,
175        **kwargs,
176    )
177
178
179def get_mito_anf_loader(
180    path: Union[os.PathLike, str],
181    patch_shape: Tuple[int, int, int],
182    batch_size: int,
183    bounding_boxes: List[Tuple[int, int, int, int, int, int]],
184    sample: Literal["M1", "M2"] = "M1",
185    download: bool = False,
186    **kwargs,
187) -> DataLoader:
188    """Get the DataLoader for mitochondria instance segmentation in the Mito-ANFs dataset.
189
190    Labels are not exhaustive manual ground truth: only mitochondria belonging to traced
191    auditory nerve fibers are annotated.
192
193    Args:
194        path: Filepath to a folder where the cached zarr stores will be saved.
195        patch_shape: The patch shape (z, y, x) to use for training.
196        batch_size: The batch size for training.
197        bounding_boxes: List of subvolumes to use, each as (z_min, z_max, y_min, y_max, x_min, x_max).
198        sample: Which cochlear sample to use. One of 'M1', 'M2'.
199        download: Whether to stream and cache data if not already present.
200        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the
201            PyTorch DataLoader.
202
203    Returns:
204        The DataLoader.
205    """
206    ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs)
207    dataset = get_mito_anf_dataset(path, patch_shape, bounding_boxes, sample, download, **ds_kwargs)
208    return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)
BASE_URL = 'https://data-humerus.webknossos.org/data/zarr'
SAMPLES: Dict[str, dict] = {'M1': {'dataset_id': '652d442501000053049c0270', 'seg_layer': 'mito_0_890_proof', 'shape_zyx': (2494, 23573, 11333), 'resolution_nm': (50, 12, 12)}, 'M2': {'dataset_id': '652d563301000068049c066e', 'seg_layer': 'mito_650_1849_proof', 'shape_zyx': (2558, 25183, 10329), 'resolution_nm': (50, 12, 12)}}
def get_mito_anf_data( path: Union[os.PathLike, str], bounding_box: Tuple[int, int, int, int, int, int], sample: Literal['M1', 'M2'] = 'M1', download: bool = False) -> str:
 56def get_mito_anf_data(
 57    path: Union[os.PathLike, str],
 58    bounding_box: Tuple[int, int, int, int, int, int],
 59    sample: Literal["M1", "M2"] = "M1",
 60    download: bool = False,
 61) -> str:
 62    """Stream a subvolume of the Mito-ANFs dataset and cache it as a zarr v3 store.
 63
 64    Args:
 65        path: Filepath to a folder where the cached zarr store will be saved.
 66        bounding_box: The region to fetch as (z_min, z_max, y_min, y_max, x_min, x_max) in voxel
 67            coordinates, at the dataset's native resolution (see `SAMPLES`).
 68        sample: Which cochlear sample to use. One of 'M1', 'M2'.
 69        download: Whether to stream and cache the data if it is not present.
 70
 71    Returns:
 72        The filepath to the cached zarr store.
 73    """
 74    import zarr
 75    from zarr.codecs import BloscCodec
 76
 77    if sample not in SAMPLES:
 78        raise ValueError(f"sample must be one of {list(SAMPLES)}, got {sample!r}")
 79
 80    os.makedirs(str(path), exist_ok=True)
 81    zarr_path = os.path.join(str(path), f"{sample}_{_bbox_to_str(bounding_box)}.zarr")
 82
 83    root = zarr.open_group(zarr_path, mode="a")
 84    if "raw" in root and "labels" in root:
 85        return zarr_path
 86
 87    if not download:
 88        raise RuntimeError(f"No cached data found at '{zarr_path}'. Set download=True to stream it.")
 89
 90    z_min, z_max, y_min, y_max, x_min, x_max = bounding_box
 91    info = SAMPLES[sample]
 92    shape_z, shape_y, shape_x = info["shape_zyx"]
 93    assert z_max <= shape_z and y_max <= shape_y and x_max <= shape_x, \
 94        f"Bounding box exceeds the {sample} volume {info['shape_zyx']}"
 95
 96    color = _open_remote_array(info["dataset_id"], "color")
 97    seg = _open_remote_array(info["dataset_id"], info["seg_layer"])
 98
 99    # the remote arrays are indexed as (channel, x, y, z); torch_em volumes use (z, y, x).
100    raw = np.transpose(color[0, x_min:x_max, y_min:y_max, z_min:z_max], (2, 1, 0))
101    labels = np.transpose(seg[0, x_min:x_max, y_min:y_max, z_min:z_max], (2, 1, 0))
102
103    def _make_array(name, data, shuffle):
104        array = root.create_array(
105            name, shape=data.shape, chunks=(32, 512, 512), dtype=data.dtype,
106            compressors=BloscCodec(cname="zstd", clevel=6, shuffle=shuffle),
107        )
108        array[:] = data
109
110    root.attrs["bounding_box"] = list(bounding_box)
111    root.attrs["sample"] = sample
112    root.attrs["resolution_nm"] = list(info["resolution_nm"])
113    root.attrs["labels_are_exhaustive"] = False
114
115    _make_array("raw", raw, shuffle="shuffle")
116    _make_array("labels", labels, shuffle="bitshuffle")
117
118    return zarr_path

Stream a subvolume of the Mito-ANFs dataset and cache it as a zarr v3 store.

Arguments:
  • path: Filepath to a folder where the cached zarr store will be saved.
  • bounding_box: The region to fetch as (z_min, z_max, y_min, y_max, x_min, x_max) in voxel coordinates, at the dataset's native resolution (see SAMPLES).
  • sample: Which cochlear sample to use. One of 'M1', 'M2'.
  • download: Whether to stream and cache the data if it is not present.
Returns:

The filepath to the cached zarr store.

def get_mito_anf_paths( path: Union[os.PathLike, str], bounding_boxes: List[Tuple[int, int, int, int, int, int]], sample: Literal['M1', 'M2'] = 'M1', download: bool = False) -> List[str]:
121def get_mito_anf_paths(
122    path: Union[os.PathLike, str],
123    bounding_boxes: List[Tuple[int, int, int, int, int, int]],
124    sample: Literal["M1", "M2"] = "M1",
125    download: bool = False,
126) -> List[str]:
127    """Get paths to cached Mito-ANFs zarr stores.
128
129    Args:
130        path: Filepath to a folder where the cached zarr stores will be saved.
131        bounding_boxes: List of regions to fetch, each as (z_min, z_max, y_min, y_max, x_min, x_max).
132        sample: Which cochlear sample to use. One of 'M1', 'M2'.
133        download: Whether to stream and cache the data if it is not present.
134
135    Returns:
136        List of filepaths to the cached zarr stores.
137    """
138    return [get_mito_anf_data(path, bbox, sample, download) for bbox in bounding_boxes]

Get paths to cached Mito-ANFs zarr stores.

Arguments:
  • path: Filepath to a folder where the cached zarr stores will be saved.
  • bounding_boxes: List of regions to fetch, each as (z_min, z_max, y_min, y_max, x_min, x_max).
  • sample: Which cochlear sample to use. One of 'M1', 'M2'.
  • download: Whether to stream and cache the data if it is not present.
Returns:

List of filepaths to the cached zarr stores.

def get_mito_anf_dataset( path: Union[os.PathLike, str], patch_shape: Tuple[int, int, int], bounding_boxes: List[Tuple[int, int, int, int, int, int]], sample: Literal['M1', 'M2'] = 'M1', download: bool = False, **kwargs) -> torch.utils.data.dataset.Dataset:
141def get_mito_anf_dataset(
142    path: Union[os.PathLike, str],
143    patch_shape: Tuple[int, int, int],
144    bounding_boxes: List[Tuple[int, int, int, int, int, int]],
145    sample: Literal["M1", "M2"] = "M1",
146    download: bool = False,
147    **kwargs,
148) -> Dataset:
149    """Get the Mito-ANFs dataset for mitochondria instance segmentation.
150
151    Labels are not exhaustive manual ground truth: only mitochondria belonging to traced
152    auditory nerve fibers are annotated.
153
154    Args:
155        path: Filepath to a folder where the cached zarr stores will be saved.
156        patch_shape: The patch shape (z, y, x) to use for training.
157        bounding_boxes: List of subvolumes to use, each as (z_min, z_max, y_min, y_max, x_min, x_max).
158        sample: Which cochlear sample to use. One of 'M1', 'M2'.
159        download: Whether to stream and cache data if not already present.
160        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`.
161
162    Returns:
163        The segmentation dataset.
164    """
165    assert len(patch_shape) == 3
166
167    paths = get_mito_anf_paths(path, bounding_boxes, sample, download)
168    kwargs = util.update_kwargs(kwargs, "is_seg_dataset", True)
169
170    return torch_em.default_segmentation_dataset(
171        raw_paths=paths,
172        raw_key="raw",
173        label_paths=paths,
174        label_key="labels",
175        patch_shape=patch_shape,
176        **kwargs,
177    )

Get the Mito-ANFs dataset for mitochondria instance segmentation.

Labels are not exhaustive manual ground truth: only mitochondria belonging to traced auditory nerve fibers are annotated.

Arguments:
  • path: Filepath to a folder where the cached zarr stores will be saved.
  • patch_shape: The patch shape (z, y, x) to use for training.
  • bounding_boxes: List of subvolumes to use, each as (z_min, z_max, y_min, y_max, x_min, x_max).
  • sample: Which cochlear sample to use. One of 'M1', 'M2'.
  • download: Whether to stream and cache data if not already present.
  • kwargs: Additional keyword arguments for torch_em.default_segmentation_dataset.
Returns:

The segmentation dataset.

def get_mito_anf_loader( path: Union[os.PathLike, str], patch_shape: Tuple[int, int, int], batch_size: int, bounding_boxes: List[Tuple[int, int, int, int, int, int]], sample: Literal['M1', 'M2'] = 'M1', download: bool = False, **kwargs) -> torch.utils.data.dataloader.DataLoader:
180def get_mito_anf_loader(
181    path: Union[os.PathLike, str],
182    patch_shape: Tuple[int, int, int],
183    batch_size: int,
184    bounding_boxes: List[Tuple[int, int, int, int, int, int]],
185    sample: Literal["M1", "M2"] = "M1",
186    download: bool = False,
187    **kwargs,
188) -> DataLoader:
189    """Get the DataLoader for mitochondria instance segmentation in the Mito-ANFs dataset.
190
191    Labels are not exhaustive manual ground truth: only mitochondria belonging to traced
192    auditory nerve fibers are annotated.
193
194    Args:
195        path: Filepath to a folder where the cached zarr stores will be saved.
196        patch_shape: The patch shape (z, y, x) to use for training.
197        batch_size: The batch size for training.
198        bounding_boxes: List of subvolumes to use, each as (z_min, z_max, y_min, y_max, x_min, x_max).
199        sample: Which cochlear sample to use. One of 'M1', 'M2'.
200        download: Whether to stream and cache data if not already present.
201        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the
202            PyTorch DataLoader.
203
204    Returns:
205        The DataLoader.
206    """
207    ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs)
208    dataset = get_mito_anf_dataset(path, patch_shape, bounding_boxes, sample, download, **ds_kwargs)
209    return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)

Get the DataLoader for mitochondria instance segmentation in the Mito-ANFs dataset.

Labels are not exhaustive manual ground truth: only mitochondria belonging to traced auditory nerve fibers are annotated.

Arguments:
  • path: Filepath to a folder where the cached zarr stores will be saved.
  • patch_shape: The patch shape (z, y, x) to use for training.
  • batch_size: The batch size for training.
  • bounding_boxes: List of subvolumes to use, each as (z_min, z_max, y_min, y_max, x_min, x_max).
  • sample: Which cochlear sample to use. One of 'M1', 'M2'.
  • download: Whether to stream and cache data if not already present.
  • kwargs: Additional keyword arguments for torch_em.default_segmentation_dataset or for the PyTorch DataLoader.
Returns:

The DataLoader.