torch_em.data.datasets.electron_microscopy.mito_segem

The MitoSegEM dataset contains mitochondria instance segmentation for three large mouse SBF-SEM/ serial-section EM volumes: intestine, testis, and dorsal cochlear nucleus (brainstem).

Labels come from mito-segEM: a pretrained network prediction followed by webKnossos path-based instance assignment. Annotation coverage is sparse, not exhaustive: some visible mitochondria in the raw data are left unannotated.

The data is available at https://www.ebi.ac.uk/empiar/EMPIAR-12535/ under the CC0 license. The dataset was published in https://doi.org/10.1016/j.crmeth.2025.100989. Please cite this publication if you use the dataset in your research.

  1"""The MitoSegEM dataset contains mitochondria instance segmentation for three large mouse SBF-SEM/
  2serial-section EM volumes: intestine, testis, and dorsal cochlear nucleus (brainstem).
  3
  4Labels come from mito-segEM: a pretrained network prediction followed by webKnossos path-based
  5instance assignment. Annotation coverage is sparse, not exhaustive: some visible mitochondria in the
  6raw data are left unannotated.
  7
  8The data is available at https://www.ebi.ac.uk/empiar/EMPIAR-12535/ under the CC0 license.
  9The dataset was published in https://doi.org/10.1016/j.crmeth.2025.100989.
 10Please cite this publication if you use the dataset in your research.
 11"""
 12
 13import os
 14from io import BytesIO
 15from typing import Dict, List, Literal, Tuple, Union
 16
 17import numpy as np
 18
 19from torch.utils.data import DataLoader, Dataset
 20
 21import torch_em
 22
 23from .. import util
 24
 25
 26BASE_URL = "https://ftp.ebi.ac.uk/empiar/world_availability/12535/data"
 27
 28TISSUES: Dict[str, dict] = {
 29    "intestine": {
 30        "raw_dir": "HuaLab-mouse-intestinal-tissue/raw",
 31        "seg_dir": "HuaLab-mouse-intestinal-tissue/intestinal_mito_segmentation_h5",
 32        "raw_pattern": "HuaLab-mouse-intestinal-tissue-raw-_{z:06d}.tiff",
 33        "first_slice_index": 1,
 34        "num_slices": 1000,
 35        "shape_yx": (5024, 6808),
 36        "resolution_nm": (50, 8, 8),
 37    },
 38    "testis": {
 39        "raw_dir": "HuaLab_mouse_testis_tissue/raw",
 40        "seg_dir": "HuaLab_mouse_testis_tissue/testis_mito_segmentation_h5",
 41        "raw_pattern": "HuaLab_mouse_testis_tissue-raw-_{z:06d}.tiff",
 42        "first_slice_index": 1,
 43        "num_slices": 2605,
 44        "shape_yx": (4884, 10419),
 45        "resolution_nm": (50, 15, 15),
 46    },
 47    "brainstem": {
 48        "raw_dir": "Hualab-DCN-A23-CBA-2M/raw",
 49        "seg_dir": "Hualab-DCN-A23-CBA-2M/brainstem_mito_segmentation_h5",
 50        "raw_pattern": "DUP_aligned_Prefix_ONPOINT_slice_{z:04d}.tif",
 51        "first_slice_index": 0,
 52        "num_slices": 3893,
 53        "shape_yx": (16449, 15303),
 54        "resolution_nm": (50, 15, 15),
 55    },
 56}
 57
 58
 59def _bbox_to_str(bounding_box):
 60    import hashlib
 61    return hashlib.md5("_".join(str(v) for v in bounding_box).encode()).hexdigest()[:12]
 62
 63
 64def _read_raw_slice(url, y_min, y_max, x_min, x_max):
 65    """Download one full raw slice and crop it to the requested region."""
 66    import time
 67    import requests
 68    import tifffile
 69
 70    for attempt in range(5):
 71        try:
 72            response = requests.get(url, timeout=180)
 73            response.raise_for_status()
 74            image = tifffile.imread(BytesIO(response.content))
 75            return image[y_min:y_max, x_min:x_max]
 76        except Exception:
 77            if attempt == 4:
 78                raise
 79            time.sleep(2 ** attempt)
 80
 81
 82def _read_label_slice(url, y_min, y_max, x_min, x_max):
 83    """Download one full label slice (h5) and crop it to the requested region."""
 84    import time
 85    import h5py
 86    import requests
 87
 88    for attempt in range(5):
 89        try:
 90            response = requests.get(url, timeout=180)
 91            response.raise_for_status()
 92            with h5py.File(BytesIO(response.content), "r") as f:
 93                return f["data"][y_min:y_max, x_min:x_max]
 94        except Exception:
 95            if attempt == 4:
 96                raise
 97            time.sleep(2 ** attempt)
 98
 99
100def get_mito_segem_data(
101    path: Union[os.PathLike, str],
102    bounding_box: Tuple[int, int, int, int, int, int],
103    tissue: Literal["intestine", "testis", "brainstem"] = "intestine",
104    download: bool = False,
105) -> str:
106    """Stream a subvolume of the MitoSegEM dataset and cache it as a zarr v3 store.
107
108    Args:
109        path: Filepath to a folder where the cached zarr store will be saved.
110        bounding_box: The region to fetch as (z_min, z_max, y_min, y_max, x_min, x_max) in voxel
111            coordinates, at each tissue's native resolution (see `TISSUES`).
112        tissue: Which tissue volume to use. One of 'intestine', 'testis', 'brainstem'.
113        download: Whether to stream and cache the data if it is not present.
114
115    Returns:
116        The filepath to the cached zarr store.
117    """
118    import zarr
119    from zarr.codecs import BloscCodec
120
121    if tissue not in TISSUES:
122        raise ValueError(f"tissue must be one of {list(TISSUES)}, got {tissue!r}")
123
124    os.makedirs(str(path), exist_ok=True)
125    zarr_path = os.path.join(str(path), f"{tissue}_{_bbox_to_str(bounding_box)}.zarr")
126
127    root = zarr.open_group(zarr_path, mode="a")
128    if "raw" in root and "labels" in root:
129        return zarr_path
130
131    if not download:
132        raise RuntimeError(f"No cached data found at '{zarr_path}'. Set download=True to stream it.")
133
134    z_min, z_max, y_min, y_max, x_min, x_max = bounding_box
135    info = TISSUES[tissue]
136    assert z_max <= info["num_slices"], f"z_max exceeds the {tissue} volume ({info['num_slices']} slices)"
137    max_y, max_x = info["shape_yx"]
138    assert y_max <= max_y and x_max <= max_x, f"Bounding box exceeds the {tissue} slice shape {info['shape_yx']}"
139
140    shape = (z_max - z_min, y_max - y_min, x_max - x_min)
141    raw = np.zeros(shape, dtype=np.uint8)
142    labels = np.zeros(shape, dtype=np.uint32)
143
144    for i, z in enumerate(range(z_min, z_max)):
145        raw_url = f"{BASE_URL}/{info['raw_dir']}/{info['raw_pattern'].format(z=z + info['first_slice_index'])}"
146        label_url = f"{BASE_URL}/{info['seg_dir']}/{z:04d}.h5"
147        raw[i] = _read_raw_slice(raw_url, y_min, y_max, x_min, x_max)
148        labels[i] = _read_label_slice(label_url, y_min, y_max, x_min, x_max)
149
150    def _make_array(name, data, shuffle):
151        array = root.create_array(
152            name, shape=data.shape, chunks=(32, 512, 512), dtype=data.dtype,
153            compressors=BloscCodec(cname="zstd", clevel=6, shuffle=shuffle),
154        )
155        array[:] = data
156
157    root.attrs["bounding_box"] = list(bounding_box)
158    root.attrs["tissue"] = tissue
159    root.attrs["resolution_nm"] = list(info["resolution_nm"])
160    root.attrs["labels_are_exhaustive"] = False
161
162    _make_array("raw", raw, shuffle="shuffle")
163    _make_array("labels", labels, shuffle="bitshuffle")
164
165    return zarr_path
166
167
168def get_mito_segem_paths(
169    path: Union[os.PathLike, str],
170    bounding_boxes: List[Tuple[int, int, int, int, int, int]],
171    tissue: Literal["intestine", "testis", "brainstem"] = "intestine",
172    download: bool = False,
173) -> List[str]:
174    """Get paths to cached MitoSegEM zarr stores.
175
176    Args:
177        path: Filepath to a folder where the cached zarr stores will be saved.
178        bounding_boxes: List of regions to fetch, each as (z_min, z_max, y_min, y_max, x_min, x_max).
179        tissue: Which tissue volume to use. One of 'intestine', 'testis', 'brainstem'.
180        download: Whether to stream and cache the data if it is not present.
181
182    Returns:
183        List of filepaths to the cached zarr stores.
184    """
185    return [get_mito_segem_data(path, bbox, tissue, download) for bbox in bounding_boxes]
186
187
188def get_mito_segem_dataset(
189    path: Union[os.PathLike, str],
190    patch_shape: Tuple[int, int, int],
191    bounding_boxes: List[Tuple[int, int, int, int, int, int]],
192    tissue: Literal["intestine", "testis", "brainstem"] = "intestine",
193    download: bool = False,
194    **kwargs,
195) -> Dataset:
196    """Get the MitoSegEM dataset for mitochondria instance segmentation.
197
198    Labels are not exhaustive manual ground truth: some mitochondria may be unannotated.
199
200    Args:
201        path: Filepath to a folder where the cached zarr stores will be saved.
202        patch_shape: The patch shape (z, y, x) to use for training.
203        bounding_boxes: List of subvolumes to use, each as (z_min, z_max, y_min, y_max, x_min, x_max).
204        tissue: Which tissue volume to use. One of 'intestine', 'testis', 'brainstem'.
205        download: Whether to stream and cache data if not already present.
206        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`.
207
208    Returns:
209        The segmentation dataset.
210    """
211    assert len(patch_shape) == 3
212
213    paths = get_mito_segem_paths(path, bounding_boxes, tissue, download)
214    kwargs = util.update_kwargs(kwargs, "is_seg_dataset", True)
215
216    return torch_em.default_segmentation_dataset(
217        raw_paths=paths,
218        raw_key="raw",
219        label_paths=paths,
220        label_key="labels",
221        patch_shape=patch_shape,
222        **kwargs,
223    )
224
225
226def get_mito_segem_loader(
227    path: Union[os.PathLike, str],
228    patch_shape: Tuple[int, int, int],
229    batch_size: int,
230    bounding_boxes: List[Tuple[int, int, int, int, int, int]],
231    tissue: Literal["intestine", "testis", "brainstem"] = "intestine",
232    download: bool = False,
233    **kwargs,
234) -> DataLoader:
235    """Get the DataLoader for mitochondria instance segmentation in the MitoSegEM dataset.
236
237    Labels are not exhaustive manual ground truth: some mitochondria may be unannotated.
238
239    Args:
240        path: Filepath to a folder where the cached zarr stores will be saved.
241        patch_shape: The patch shape (z, y, x) to use for training.
242        batch_size: The batch size for training.
243        bounding_boxes: List of subvolumes to use, each as (z_min, z_max, y_min, y_max, x_min, x_max).
244        tissue: Which tissue volume to use. One of 'intestine', 'testis', 'brainstem'.
245        download: Whether to stream and cache data if not already present.
246        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the
247            PyTorch DataLoader.
248
249    Returns:
250        The DataLoader.
251    """
252    ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs)
253    dataset = get_mito_segem_dataset(path, patch_shape, bounding_boxes, tissue, download, **ds_kwargs)
254    return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)
BASE_URL = 'https://ftp.ebi.ac.uk/empiar/world_availability/12535/data'
TISSUES: Dict[str, dict] = {'intestine': {'raw_dir': 'HuaLab-mouse-intestinal-tissue/raw', 'seg_dir': 'HuaLab-mouse-intestinal-tissue/intestinal_mito_segmentation_h5', 'raw_pattern': 'HuaLab-mouse-intestinal-tissue-raw-_{z:06d}.tiff', 'first_slice_index': 1, 'num_slices': 1000, 'shape_yx': (5024, 6808), 'resolution_nm': (50, 8, 8)}, 'testis': {'raw_dir': 'HuaLab_mouse_testis_tissue/raw', 'seg_dir': 'HuaLab_mouse_testis_tissue/testis_mito_segmentation_h5', 'raw_pattern': 'HuaLab_mouse_testis_tissue-raw-_{z:06d}.tiff', 'first_slice_index': 1, 'num_slices': 2605, 'shape_yx': (4884, 10419), 'resolution_nm': (50, 15, 15)}, 'brainstem': {'raw_dir': 'Hualab-DCN-A23-CBA-2M/raw', 'seg_dir': 'Hualab-DCN-A23-CBA-2M/brainstem_mito_segmentation_h5', 'raw_pattern': 'DUP_aligned_Prefix_ONPOINT_slice_{z:04d}.tif', 'first_slice_index': 0, 'num_slices': 3893, 'shape_yx': (16449, 15303), 'resolution_nm': (50, 15, 15)}}
def get_mito_segem_data( path: Union[os.PathLike, str], bounding_box: Tuple[int, int, int, int, int, int], tissue: Literal['intestine', 'testis', 'brainstem'] = 'intestine', download: bool = False) -> str:
101def get_mito_segem_data(
102    path: Union[os.PathLike, str],
103    bounding_box: Tuple[int, int, int, int, int, int],
104    tissue: Literal["intestine", "testis", "brainstem"] = "intestine",
105    download: bool = False,
106) -> str:
107    """Stream a subvolume of the MitoSegEM dataset and cache it as a zarr v3 store.
108
109    Args:
110        path: Filepath to a folder where the cached zarr store will be saved.
111        bounding_box: The region to fetch as (z_min, z_max, y_min, y_max, x_min, x_max) in voxel
112            coordinates, at each tissue's native resolution (see `TISSUES`).
113        tissue: Which tissue volume to use. One of 'intestine', 'testis', 'brainstem'.
114        download: Whether to stream and cache the data if it is not present.
115
116    Returns:
117        The filepath to the cached zarr store.
118    """
119    import zarr
120    from zarr.codecs import BloscCodec
121
122    if tissue not in TISSUES:
123        raise ValueError(f"tissue must be one of {list(TISSUES)}, got {tissue!r}")
124
125    os.makedirs(str(path), exist_ok=True)
126    zarr_path = os.path.join(str(path), f"{tissue}_{_bbox_to_str(bounding_box)}.zarr")
127
128    root = zarr.open_group(zarr_path, mode="a")
129    if "raw" in root and "labels" in root:
130        return zarr_path
131
132    if not download:
133        raise RuntimeError(f"No cached data found at '{zarr_path}'. Set download=True to stream it.")
134
135    z_min, z_max, y_min, y_max, x_min, x_max = bounding_box
136    info = TISSUES[tissue]
137    assert z_max <= info["num_slices"], f"z_max exceeds the {tissue} volume ({info['num_slices']} slices)"
138    max_y, max_x = info["shape_yx"]
139    assert y_max <= max_y and x_max <= max_x, f"Bounding box exceeds the {tissue} slice shape {info['shape_yx']}"
140
141    shape = (z_max - z_min, y_max - y_min, x_max - x_min)
142    raw = np.zeros(shape, dtype=np.uint8)
143    labels = np.zeros(shape, dtype=np.uint32)
144
145    for i, z in enumerate(range(z_min, z_max)):
146        raw_url = f"{BASE_URL}/{info['raw_dir']}/{info['raw_pattern'].format(z=z + info['first_slice_index'])}"
147        label_url = f"{BASE_URL}/{info['seg_dir']}/{z:04d}.h5"
148        raw[i] = _read_raw_slice(raw_url, y_min, y_max, x_min, x_max)
149        labels[i] = _read_label_slice(label_url, y_min, y_max, x_min, x_max)
150
151    def _make_array(name, data, shuffle):
152        array = root.create_array(
153            name, shape=data.shape, chunks=(32, 512, 512), dtype=data.dtype,
154            compressors=BloscCodec(cname="zstd", clevel=6, shuffle=shuffle),
155        )
156        array[:] = data
157
158    root.attrs["bounding_box"] = list(bounding_box)
159    root.attrs["tissue"] = tissue
160    root.attrs["resolution_nm"] = list(info["resolution_nm"])
161    root.attrs["labels_are_exhaustive"] = False
162
163    _make_array("raw", raw, shuffle="shuffle")
164    _make_array("labels", labels, shuffle="bitshuffle")
165
166    return zarr_path

Stream a subvolume of the MitoSegEM dataset and cache it as a zarr v3 store.

Arguments:
  • path: Filepath to a folder where the cached zarr store will be saved.
  • bounding_box: The region to fetch as (z_min, z_max, y_min, y_max, x_min, x_max) in voxel coordinates, at each tissue's native resolution (see TISSUES).
  • tissue: Which tissue volume to use. One of 'intestine', 'testis', 'brainstem'.
  • download: Whether to stream and cache the data if it is not present.
Returns:

The filepath to the cached zarr store.

def get_mito_segem_paths( path: Union[os.PathLike, str], bounding_boxes: List[Tuple[int, int, int, int, int, int]], tissue: Literal['intestine', 'testis', 'brainstem'] = 'intestine', download: bool = False) -> List[str]:
169def get_mito_segem_paths(
170    path: Union[os.PathLike, str],
171    bounding_boxes: List[Tuple[int, int, int, int, int, int]],
172    tissue: Literal["intestine", "testis", "brainstem"] = "intestine",
173    download: bool = False,
174) -> List[str]:
175    """Get paths to cached MitoSegEM zarr stores.
176
177    Args:
178        path: Filepath to a folder where the cached zarr stores will be saved.
179        bounding_boxes: List of regions to fetch, each as (z_min, z_max, y_min, y_max, x_min, x_max).
180        tissue: Which tissue volume to use. One of 'intestine', 'testis', 'brainstem'.
181        download: Whether to stream and cache the data if it is not present.
182
183    Returns:
184        List of filepaths to the cached zarr stores.
185    """
186    return [get_mito_segem_data(path, bbox, tissue, download) for bbox in bounding_boxes]

Get paths to cached MitoSegEM zarr stores.

Arguments:
  • path: Filepath to a folder where the cached zarr stores will be saved.
  • bounding_boxes: List of regions to fetch, each as (z_min, z_max, y_min, y_max, x_min, x_max).
  • tissue: Which tissue volume to use. One of 'intestine', 'testis', 'brainstem'.
  • download: Whether to stream and cache the data if it is not present.
Returns:

List of filepaths to the cached zarr stores.

def get_mito_segem_dataset( path: Union[os.PathLike, str], patch_shape: Tuple[int, int, int], bounding_boxes: List[Tuple[int, int, int, int, int, int]], tissue: Literal['intestine', 'testis', 'brainstem'] = 'intestine', download: bool = False, **kwargs) -> torch.utils.data.dataset.Dataset:
189def get_mito_segem_dataset(
190    path: Union[os.PathLike, str],
191    patch_shape: Tuple[int, int, int],
192    bounding_boxes: List[Tuple[int, int, int, int, int, int]],
193    tissue: Literal["intestine", "testis", "brainstem"] = "intestine",
194    download: bool = False,
195    **kwargs,
196) -> Dataset:
197    """Get the MitoSegEM dataset for mitochondria instance segmentation.
198
199    Labels are not exhaustive manual ground truth: some mitochondria may be unannotated.
200
201    Args:
202        path: Filepath to a folder where the cached zarr stores will be saved.
203        patch_shape: The patch shape (z, y, x) to use for training.
204        bounding_boxes: List of subvolumes to use, each as (z_min, z_max, y_min, y_max, x_min, x_max).
205        tissue: Which tissue volume to use. One of 'intestine', 'testis', 'brainstem'.
206        download: Whether to stream and cache data if not already present.
207        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`.
208
209    Returns:
210        The segmentation dataset.
211    """
212    assert len(patch_shape) == 3
213
214    paths = get_mito_segem_paths(path, bounding_boxes, tissue, download)
215    kwargs = util.update_kwargs(kwargs, "is_seg_dataset", True)
216
217    return torch_em.default_segmentation_dataset(
218        raw_paths=paths,
219        raw_key="raw",
220        label_paths=paths,
221        label_key="labels",
222        patch_shape=patch_shape,
223        **kwargs,
224    )

Get the MitoSegEM dataset for mitochondria instance segmentation.

Labels are not exhaustive manual ground truth: some mitochondria may be unannotated.

Arguments:
  • path: Filepath to a folder where the cached zarr stores will be saved.
  • patch_shape: The patch shape (z, y, x) to use for training.
  • bounding_boxes: List of subvolumes to use, each as (z_min, z_max, y_min, y_max, x_min, x_max).
  • tissue: Which tissue volume to use. One of 'intestine', 'testis', 'brainstem'.
  • download: Whether to stream and cache data if not already present.
  • kwargs: Additional keyword arguments for torch_em.default_segmentation_dataset.
Returns:

The segmentation dataset.

def get_mito_segem_loader( path: Union[os.PathLike, str], patch_shape: Tuple[int, int, int], batch_size: int, bounding_boxes: List[Tuple[int, int, int, int, int, int]], tissue: Literal['intestine', 'testis', 'brainstem'] = 'intestine', download: bool = False, **kwargs) -> torch.utils.data.dataloader.DataLoader:
227def get_mito_segem_loader(
228    path: Union[os.PathLike, str],
229    patch_shape: Tuple[int, int, int],
230    batch_size: int,
231    bounding_boxes: List[Tuple[int, int, int, int, int, int]],
232    tissue: Literal["intestine", "testis", "brainstem"] = "intestine",
233    download: bool = False,
234    **kwargs,
235) -> DataLoader:
236    """Get the DataLoader for mitochondria instance segmentation in the MitoSegEM dataset.
237
238    Labels are not exhaustive manual ground truth: some mitochondria may be unannotated.
239
240    Args:
241        path: Filepath to a folder where the cached zarr stores will be saved.
242        patch_shape: The patch shape (z, y, x) to use for training.
243        batch_size: The batch size for training.
244        bounding_boxes: List of subvolumes to use, each as (z_min, z_max, y_min, y_max, x_min, x_max).
245        tissue: Which tissue volume to use. One of 'intestine', 'testis', 'brainstem'.
246        download: Whether to stream and cache data if not already present.
247        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the
248            PyTorch DataLoader.
249
250    Returns:
251        The DataLoader.
252    """
253    ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs)
254    dataset = get_mito_segem_dataset(path, patch_shape, bounding_boxes, tissue, download, **ds_kwargs)
255    return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)

Get the DataLoader for mitochondria instance segmentation in the MitoSegEM dataset.

Labels are not exhaustive manual ground truth: some mitochondria may be unannotated.

Arguments:
  • path: Filepath to a folder where the cached zarr stores will be saved.
  • patch_shape: The patch shape (z, y, x) to use for training.
  • batch_size: The batch size for training.
  • bounding_boxes: List of subvolumes to use, each as (z_min, z_max, y_min, y_max, x_min, x_max).
  • tissue: Which tissue volume to use. One of 'intestine', 'testis', 'brainstem'.
  • download: Whether to stream and cache data if not already present.
  • kwargs: Additional keyword arguments for torch_em.default_segmentation_dataset or for the PyTorch DataLoader.
Returns:

The DataLoader.