torch_em.data.datasets.electron_microscopy.hela_mito

The HeLa-Mito dataset contains mitochondria instance segmentation for 2D electron microscopy slices of HeLa cells.

The raw slices are from the same acquisition (EMPIAR-10094) already used by the cefa_hela module, but that module only provides nuclear-envelope annotations; this module adds a separate, complementary layer of mitochondria instance masks on 9 of the 24 raw slices released alongside it.

Annotation coverage is sparse, not exhaustive: only 9 of the 24 available slices carry labels, and some mitochondria visible in the raw data may be left unannotated even on labeled slices.

The dataset is available at https://github.com/reyesaldasoro/MitoEM. The dataset was published in https://doi.org/10.1101/2023.11.14.567016. Please cite this publication if you use the dataset in your research.

  1"""The HeLa-Mito dataset contains mitochondria instance segmentation for 2D electron
  2microscopy slices of HeLa cells.
  3
  4The raw slices are from the same acquisition (EMPIAR-10094) already used by the
  5`cefa_hela` module, but that module only provides nuclear-envelope annotations; this
  6module adds a separate, complementary layer of mitochondria instance masks on 9 of the
  724 raw slices released alongside it.
  8
  9Annotation coverage is sparse, not exhaustive: only 9 of the 24 available slices carry
 10labels, and some mitochondria visible in the raw data may be left unannotated even on
 11labeled slices.
 12
 13The dataset is available at https://github.com/reyesaldasoro/MitoEM.
 14The dataset was published in https://doi.org/10.1101/2023.11.14.567016.
 15Please cite this publication if you use the dataset in your research.
 16"""
 17
 18import os
 19from typing import List, Tuple, Union
 20
 21from torch.utils.data import DataLoader, Dataset
 22
 23import torch_em
 24
 25from .. import util
 26
 27
 28BASE_URL = "https://raw.githubusercontent.com/reyesaldasoro/MitoEM/main/CODE"
 29RAW_URL = BASE_URL + "/OriginalImages/ROI_6005_4739_81_z{slice_id}.tif"
 30LABEL_URL = BASE_URL + "/{prefix}_ROI_6005_4739_81_z{slice_id}.mat"
 31
 32# The annotated slice ids and the ground-truth file prefix to use for each. GT and GT2
 33# both annotate z0001 (0.94 IoU between them); GT is used there and GT2 only for the
 34# 4 additional slices it uniquely covers.
 35SLICES = {
 36    "0001": "GT",
 37    "0002": "GT2",
 38    "0004": "GT2",
 39    "0026": "GT2",
 40    "0028": "GT2",
 41    "0030": "GT",
 42    "0060": "GT",
 43    "0116": "GT",
 44    "0150": "GT",
 45}
 46
 47
 48def _convert_slice(raw_path, mat_path, out_path):
 49    import h5py
 50    import tifffile
 51    import scipy.io as sio
 52
 53    raw = tifffile.imread(raw_path)
 54    labels = sio.loadmat(mat_path)["groundTruthM"].max(axis=2)
 55    assert raw.shape == labels.shape, f"{raw.shape} != {labels.shape}"
 56
 57    tmp_path = out_path + ".incomplete"
 58    with h5py.File(tmp_path, "w") as f:
 59        f.create_dataset("raw", data=raw, compression="gzip")
 60        f.create_dataset("labels", data=labels, compression="gzip")
 61    os.rename(tmp_path, out_path)
 62
 63
 64def get_hela_mito_data(path: Union[os.PathLike, str], download: bool = False) -> str:
 65    """Download the HeLa-Mito dataset.
 66
 67    Args:
 68        path: Filepath to a folder where the data will be downloaded.
 69        download: Whether to download the data if it is not present.
 70
 71    Returns:
 72        Filepath to the folder with the per-slice HDF5 files.
 73    """
 74    os.makedirs(path, exist_ok=True)
 75
 76    for slice_id, prefix in SLICES.items():
 77        out_path = os.path.join(path, f"z{slice_id}.h5")
 78        if os.path.exists(out_path):
 79            continue
 80        if not download:
 81            raise RuntimeError(f"Cannot find the data at {out_path}, but download was set to False.")
 82
 83        raw_path = os.path.join(path, f"raw_z{slice_id}.tif")
 84        mat_path = os.path.join(path, f"{prefix}_z{slice_id}.mat")
 85        util.download_source(raw_path, RAW_URL.format(slice_id=slice_id), download)
 86        util.download_source(mat_path, LABEL_URL.format(prefix=prefix, slice_id=slice_id), download)
 87
 88        _convert_slice(raw_path, mat_path, out_path)
 89        os.remove(raw_path)
 90        os.remove(mat_path)
 91
 92    return path
 93
 94
 95def get_hela_mito_paths(path: Union[os.PathLike, str], download: bool = False) -> List[str]:
 96    """Get paths to the HeLa-Mito data.
 97
 98    Args:
 99        path: Filepath to a folder where the data will be downloaded.
100        download: Whether to download the data if it is not present.
101
102    Returns:
103        List of filepaths to the per-slice HDF5 files.
104    """
105    data_dir = get_hela_mito_data(path, download)
106    return [os.path.join(data_dir, f"z{slice_id}.h5") for slice_id in SLICES]
107
108
109def get_hela_mito_dataset(
110    path: Union[os.PathLike, str],
111    patch_shape: Tuple[int, int],
112    download: bool = False,
113    **kwargs
114) -> Dataset:
115    """Get the dataset for mitochondria instance segmentation in HeLa cell EM slices.
116
117    Args:
118        path: Filepath to a folder where the data will be downloaded.
119        patch_shape: The patch shape to use for training.
120        download: Whether to download the data if it is not present.
121        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`.
122
123    Returns:
124        The segmentation dataset.
125    """
126    assert len(patch_shape) == 2
127
128    paths = get_hela_mito_paths(path, download)
129
130    kwargs = util.update_kwargs(kwargs, "is_seg_dataset", True)
131    kwargs, _ = util.add_instance_label_transform(kwargs, add_binary_target=True)
132
133    return torch_em.default_segmentation_dataset(
134        raw_paths=paths,
135        raw_key="raw",
136        label_paths=paths,
137        label_key="labels",
138        patch_shape=patch_shape,
139        **kwargs
140    )
141
142
143def get_hela_mito_loader(
144    path: Union[os.PathLike, str],
145    patch_shape: Tuple[int, int],
146    batch_size: int,
147    download: bool = False,
148    **kwargs
149) -> DataLoader:
150    """Get the DataLoader for mitochondria instance segmentation in HeLa cell EM slices.
151
152    Args:
153        path: Filepath to a folder where the data will be downloaded.
154        patch_shape: The patch shape to use for training.
155        batch_size: The batch size for training.
156        download: Whether to download the data if it is not present.
157        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader.
158
159    Returns:
160        The DataLoader.
161    """
162    ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs)
163    dataset = get_hela_mito_dataset(path, patch_shape, download=download, **ds_kwargs)
164    return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)
BASE_URL = 'https://raw.githubusercontent.com/reyesaldasoro/MitoEM/main/CODE'
RAW_URL = 'https://raw.githubusercontent.com/reyesaldasoro/MitoEM/main/CODE/OriginalImages/ROI_6005_4739_81_z{slice_id}.tif'
LABEL_URL = 'https://raw.githubusercontent.com/reyesaldasoro/MitoEM/main/CODE/{prefix}_ROI_6005_4739_81_z{slice_id}.mat'
SLICES = {'0001': 'GT', '0002': 'GT2', '0004': 'GT2', '0026': 'GT2', '0028': 'GT2', '0030': 'GT', '0060': 'GT', '0116': 'GT', '0150': 'GT'}
def get_hela_mito_data(path: Union[os.PathLike, str], download: bool = False) -> str:
65def get_hela_mito_data(path: Union[os.PathLike, str], download: bool = False) -> str:
66    """Download the HeLa-Mito dataset.
67
68    Args:
69        path: Filepath to a folder where the data will be downloaded.
70        download: Whether to download the data if it is not present.
71
72    Returns:
73        Filepath to the folder with the per-slice HDF5 files.
74    """
75    os.makedirs(path, exist_ok=True)
76
77    for slice_id, prefix in SLICES.items():
78        out_path = os.path.join(path, f"z{slice_id}.h5")
79        if os.path.exists(out_path):
80            continue
81        if not download:
82            raise RuntimeError(f"Cannot find the data at {out_path}, but download was set to False.")
83
84        raw_path = os.path.join(path, f"raw_z{slice_id}.tif")
85        mat_path = os.path.join(path, f"{prefix}_z{slice_id}.mat")
86        util.download_source(raw_path, RAW_URL.format(slice_id=slice_id), download)
87        util.download_source(mat_path, LABEL_URL.format(prefix=prefix, slice_id=slice_id), download)
88
89        _convert_slice(raw_path, mat_path, out_path)
90        os.remove(raw_path)
91        os.remove(mat_path)
92
93    return path

Download the HeLa-Mito dataset.

Arguments:
  • path: Filepath to a folder where the data will be downloaded.
  • download: Whether to download the data if it is not present.
Returns:

Filepath to the folder with the per-slice HDF5 files.

def get_hela_mito_paths(path: Union[os.PathLike, str], download: bool = False) -> List[str]:
 96def get_hela_mito_paths(path: Union[os.PathLike, str], download: bool = False) -> List[str]:
 97    """Get paths to the HeLa-Mito data.
 98
 99    Args:
100        path: Filepath to a folder where the data will be downloaded.
101        download: Whether to download the data if it is not present.
102
103    Returns:
104        List of filepaths to the per-slice HDF5 files.
105    """
106    data_dir = get_hela_mito_data(path, download)
107    return [os.path.join(data_dir, f"z{slice_id}.h5") for slice_id in SLICES]

Get paths to the HeLa-Mito data.

Arguments:
  • path: Filepath to a folder where the data will be downloaded.
  • download: Whether to download the data if it is not present.
Returns:

List of filepaths to the per-slice HDF5 files.

def get_hela_mito_dataset( path: Union[os.PathLike, str], patch_shape: Tuple[int, int], download: bool = False, **kwargs) -> torch.utils.data.dataset.Dataset:
110def get_hela_mito_dataset(
111    path: Union[os.PathLike, str],
112    patch_shape: Tuple[int, int],
113    download: bool = False,
114    **kwargs
115) -> Dataset:
116    """Get the dataset for mitochondria instance segmentation in HeLa cell EM slices.
117
118    Args:
119        path: Filepath to a folder where the data will be downloaded.
120        patch_shape: The patch shape to use for training.
121        download: Whether to download the data if it is not present.
122        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`.
123
124    Returns:
125        The segmentation dataset.
126    """
127    assert len(patch_shape) == 2
128
129    paths = get_hela_mito_paths(path, download)
130
131    kwargs = util.update_kwargs(kwargs, "is_seg_dataset", True)
132    kwargs, _ = util.add_instance_label_transform(kwargs, add_binary_target=True)
133
134    return torch_em.default_segmentation_dataset(
135        raw_paths=paths,
136        raw_key="raw",
137        label_paths=paths,
138        label_key="labels",
139        patch_shape=patch_shape,
140        **kwargs
141    )

Get the dataset for mitochondria instance segmentation in HeLa cell EM slices.

Arguments:
  • path: Filepath to a folder where the data will be downloaded.
  • patch_shape: The patch shape to use for training.
  • download: Whether to download the data if it is not present.
  • kwargs: Additional keyword arguments for torch_em.default_segmentation_dataset.
Returns:

The segmentation dataset.

def get_hela_mito_loader( path: Union[os.PathLike, str], patch_shape: Tuple[int, int], batch_size: int, download: bool = False, **kwargs) -> torch.utils.data.dataloader.DataLoader:
144def get_hela_mito_loader(
145    path: Union[os.PathLike, str],
146    patch_shape: Tuple[int, int],
147    batch_size: int,
148    download: bool = False,
149    **kwargs
150) -> DataLoader:
151    """Get the DataLoader for mitochondria instance segmentation in HeLa cell EM slices.
152
153    Args:
154        path: Filepath to a folder where the data will be downloaded.
155        patch_shape: The patch shape to use for training.
156        batch_size: The batch size for training.
157        download: Whether to download the data if it is not present.
158        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader.
159
160    Returns:
161        The DataLoader.
162    """
163    ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs)
164    dataset = get_hela_mito_dataset(path, patch_shape, download=download, **ds_kwargs)
165    return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)

Get the DataLoader for mitochondria instance segmentation in HeLa cell EM slices.

Arguments:
  • path: Filepath to a folder where the data will be downloaded.
  • patch_shape: The patch shape to use for training.
  • batch_size: The batch size for training.
  • download: Whether to download the data if it is not present.
  • kwargs: Additional keyword arguments for torch_em.default_segmentation_dataset or for the PyTorch DataLoader.
Returns:

The DataLoader.