torch_em.data.datasets.electron_microscopy.surface_morphometrics

The SurfaceMorphometrics dataset contains voxel segmentations of mitochondria and endoplasmic reticulum membranes in cryo-electron tomograms of mouse embryonic fibroblasts, treated with vehicle or thapsigargin.

Each tomogram has a multi-class semantic label volume (0=background, 1-3=membrane classes). The source deposit does not document a fixed name for each label value; inspect a representative volume to determine which value corresponds to which membrane in your use case.

The data is available at https://www.ebi.ac.uk/empiar/EMPIAR-11370/. The dataset was published in https://doi.org/10.1101/2022.01.23.477440. Please cite this publication if you use the dataset in your research.

  1"""The SurfaceMorphometrics dataset contains voxel segmentations of mitochondria and endoplasmic
  2reticulum membranes in cryo-electron tomograms of mouse embryonic fibroblasts, treated with vehicle
  3or thapsigargin.
  4
  5Each tomogram has a multi-class semantic label volume (0=background, 1-3=membrane classes). The
  6source deposit does not document a fixed name for each label value; inspect a representative volume
  7to determine which value corresponds to which membrane in your use case.
  8
  9The data is available at https://www.ebi.ac.uk/empiar/EMPIAR-11370/.
 10The dataset was published in https://doi.org/10.1101/2022.01.23.477440.
 11Please cite this publication if you use the dataset in your research.
 12"""
 13
 14import os
 15from typing import List, Tuple, Union
 16
 17import numpy as np
 18
 19from torch.utils.data import DataLoader, Dataset
 20
 21import torch_em
 22
 23from .. import util
 24
 25
 26BASE_URL = "https://ftp.ebi.ac.uk/empiar/world_availability/11370/data"
 27
 28TOMOGRAM_IDS = (
 29    "TE2", "TE3", "TE4", "TE5", "TE6", "TE7", "TE8", "TE9", "TE10", "TE11", "TE12", "TE13", "TE14",
 30    "TF1", "TF2", "TF3", "TF5", "TF6",
 31    "UE1", "UE2", "UE3", "UE4", "UE5", "UE6", "UE7", "UE8", "UE9", "UE10",
 32    "UF1", "UF2", "UF3", "UF4", "UF5", "UF6",
 33)
 34
 35# UE9 and UF3 were deposited without the usual '.mrc' suffix on their label file.
 36_LABEL_SUFFIX_EXCEPTIONS = {"UE9": "_labels.rec", "UF3": "_labels.rec"}
 37
 38
 39def _label_url(tomogram_id):
 40    suffix = _LABEL_SUFFIX_EXCEPTIONS.get(tomogram_id, "_labels.rec.mrc")
 41    return f"{BASE_URL}/Voxel_Segmentations/{tomogram_id}{suffix}"
 42
 43
 44def _read_raw_mrc(path):
 45    """Read an MRC volume, tolerating the non-standard headers in this deposit."""
 46    import mrcfile
 47    with mrcfile.open(path, permissive=True) as mrc:
 48        return np.asarray(mrc.data)
 49
 50
 51def _read_label_mrc(path, shape):
 52    """Read a label MRC volume as raw uint8 bytes after the standard 1024-byte header.
 53
 54    `mrcfile` rejects these files outright (non-standard machine stamp, missing map ID), so the
 55    label data is read directly instead.
 56    """
 57    with open(path, "rb") as f:
 58        content = f.read()
 59    voxel_count = int(np.prod(shape))
 60    return np.frombuffer(content[1024:1024 + voxel_count], dtype=np.uint8).reshape(shape)
 61
 62
 63def get_surface_morphometrics_data(
 64    path: Union[os.PathLike, str], tomogram_id: str, download: bool = False,
 65) -> Tuple[str, str]:
 66    """Download one tomogram and its label volume from the SurfaceMorphometrics dataset.
 67
 68    Args:
 69        path: Filepath to a folder where the downloaded data will be saved.
 70        tomogram_id: Which tomogram to download. One of `TOMOGRAM_IDS`.
 71        download: Whether to download the data if it is not present.
 72
 73    Returns:
 74        The filepath to the downloaded tomogram.
 75        The filepath to the downloaded label volume.
 76    """
 77    if tomogram_id not in TOMOGRAM_IDS:
 78        raise ValueError(f"tomogram_id must be one of {TOMOGRAM_IDS}, got {tomogram_id!r}")
 79
 80    os.makedirs(path, exist_ok=True)
 81    raw_path = os.path.join(path, f"{tomogram_id}_tomo.rec.mrc")
 82    label_path = os.path.join(path, f"{tomogram_id}_labels.rec")
 83
 84    util.download_source(raw_path, f"{BASE_URL}/Tomograms/{tomogram_id}_tomo.rec.mrc", download, checksum=None)
 85    util.download_source(label_path, _label_url(tomogram_id), download, checksum=None)
 86
 87    return raw_path, label_path
 88
 89
 90def get_surface_morphometrics_paths(
 91    path: Union[os.PathLike, str],
 92    tomogram_ids: List[str],
 93    cache_path: Union[os.PathLike, str, None] = None,
 94    download: bool = False,
 95) -> List[str]:
 96    """Get paths to cached SurfaceMorphometrics zarr stores, one per tomogram.
 97
 98    Args:
 99        path: Filepath to a folder where the downloaded raw MRC files will be saved.
100        tomogram_ids: Which tomograms to use. See `TOMOGRAM_IDS`.
101        cache_path: Filepath to a folder where the converted zarr stores will be saved. Defaults to
102            `path` if not given.
103        download: Whether to download the data if it is not present.
104
105    Returns:
106        List of filepaths to the cached zarr stores.
107    """
108    import zarr
109    from zarr.codecs import BloscCodec
110
111    cache_path = cache_path or path
112    os.makedirs(cache_path, exist_ok=True)
113
114    zarr_paths = []
115    for tomogram_id in tomogram_ids:
116        zarr_path = os.path.join(str(cache_path), f"{tomogram_id}.zarr")
117        root = zarr.open_group(zarr_path, mode="a")
118        if "raw" not in root or "labels" not in root:
119            raw_path, label_path = get_surface_morphometrics_data(path, tomogram_id, download)
120            raw = _read_raw_mrc(raw_path)
121            labels = _read_label_mrc(label_path, raw.shape)
122
123            def _make_array(name, data, shuffle):
124                array = root.create_array(
125                    name, shape=data.shape, chunks=(32, 256, 256), dtype=data.dtype,
126                    compressors=BloscCodec(cname="zstd", clevel=6, shuffle=shuffle),
127                )
128                array[:] = data
129
130            _make_array("raw", raw, shuffle="shuffle")
131            _make_array("labels", labels, shuffle="bitshuffle")
132        zarr_paths.append(zarr_path)
133
134    return zarr_paths
135
136
137def get_surface_morphometrics_dataset(
138    path: Union[os.PathLike, str],
139    patch_shape: Tuple[int, int, int],
140    tomogram_ids: List[str] = TOMOGRAM_IDS,
141    cache_path: Union[os.PathLike, str, None] = None,
142    download: bool = False,
143    **kwargs,
144) -> Dataset:
145    """Get the SurfaceMorphometrics dataset for mitochondria and ER membrane segmentation.
146
147    Args:
148        path: Filepath to a folder where the downloaded raw MRC files will be saved.
149        patch_shape: The patch shape (z, y, x) to use for training.
150        tomogram_ids: Which tomograms to use. See `TOMOGRAM_IDS`. Defaults to all of them.
151        cache_path: Filepath to a folder where the converted zarr stores will be saved. Defaults to
152            `path` if not given.
153        download: Whether to download the data if it is not present.
154        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`.
155
156    Returns:
157        The segmentation dataset.
158    """
159    assert len(patch_shape) == 3
160
161    paths = get_surface_morphometrics_paths(path, tomogram_ids, cache_path, download)
162    kwargs = util.update_kwargs(kwargs, "is_seg_dataset", True)
163
164    return torch_em.default_segmentation_dataset(
165        raw_paths=paths,
166        raw_key="raw",
167        label_paths=paths,
168        label_key="labels",
169        patch_shape=patch_shape,
170        **kwargs,
171    )
172
173
174def get_surface_morphometrics_loader(
175    path: Union[os.PathLike, str],
176    patch_shape: Tuple[int, int, int],
177    batch_size: int,
178    tomogram_ids: List[str] = TOMOGRAM_IDS,
179    cache_path: Union[os.PathLike, str, None] = None,
180    download: bool = False,
181    **kwargs,
182) -> DataLoader:
183    """Get the DataLoader for mitochondria and ER membrane segmentation in the SurfaceMorphometrics
184    dataset.
185
186    Args:
187        path: Filepath to a folder where the downloaded raw MRC files will be saved.
188        patch_shape: The patch shape (z, y, x) to use for training.
189        batch_size: The batch size for training.
190        tomogram_ids: Which tomograms to use. See `TOMOGRAM_IDS`. Defaults to all of them.
191        cache_path: Filepath to a folder where the converted zarr stores will be saved. Defaults to
192            `path` if not given.
193        download: Whether to download the data if it is not present.
194        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the
195            PyTorch DataLoader.
196
197    Returns:
198        The DataLoader.
199    """
200    ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs)
201    dataset = get_surface_morphometrics_dataset(
202        path, patch_shape, tomogram_ids, cache_path, download, **ds_kwargs
203    )
204    return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)
BASE_URL = 'https://ftp.ebi.ac.uk/empiar/world_availability/11370/data'
TOMOGRAM_IDS = ('TE2', 'TE3', 'TE4', 'TE5', 'TE6', 'TE7', 'TE8', 'TE9', 'TE10', 'TE11', 'TE12', 'TE13', 'TE14', 'TF1', 'TF2', 'TF3', 'TF5', 'TF6', 'UE1', 'UE2', 'UE3', 'UE4', 'UE5', 'UE6', 'UE7', 'UE8', 'UE9', 'UE10', 'UF1', 'UF2', 'UF3', 'UF4', 'UF5', 'UF6')
def get_surface_morphometrics_data( path: Union[os.PathLike, str], tomogram_id: str, download: bool = False) -> Tuple[str, str]:
64def get_surface_morphometrics_data(
65    path: Union[os.PathLike, str], tomogram_id: str, download: bool = False,
66) -> Tuple[str, str]:
67    """Download one tomogram and its label volume from the SurfaceMorphometrics dataset.
68
69    Args:
70        path: Filepath to a folder where the downloaded data will be saved.
71        tomogram_id: Which tomogram to download. One of `TOMOGRAM_IDS`.
72        download: Whether to download the data if it is not present.
73
74    Returns:
75        The filepath to the downloaded tomogram.
76        The filepath to the downloaded label volume.
77    """
78    if tomogram_id not in TOMOGRAM_IDS:
79        raise ValueError(f"tomogram_id must be one of {TOMOGRAM_IDS}, got {tomogram_id!r}")
80
81    os.makedirs(path, exist_ok=True)
82    raw_path = os.path.join(path, f"{tomogram_id}_tomo.rec.mrc")
83    label_path = os.path.join(path, f"{tomogram_id}_labels.rec")
84
85    util.download_source(raw_path, f"{BASE_URL}/Tomograms/{tomogram_id}_tomo.rec.mrc", download, checksum=None)
86    util.download_source(label_path, _label_url(tomogram_id), download, checksum=None)
87
88    return raw_path, label_path

Download one tomogram and its label volume from the SurfaceMorphometrics dataset.

Arguments:
  • path: Filepath to a folder where the downloaded data will be saved.
  • tomogram_id: Which tomogram to download. One of TOMOGRAM_IDS.
  • download: Whether to download the data if it is not present.
Returns:

The filepath to the downloaded tomogram. The filepath to the downloaded label volume.

def get_surface_morphometrics_paths( path: Union[os.PathLike, str], tomogram_ids: List[str], cache_path: Union[os.PathLike, str, NoneType] = None, download: bool = False) -> List[str]:
 91def get_surface_morphometrics_paths(
 92    path: Union[os.PathLike, str],
 93    tomogram_ids: List[str],
 94    cache_path: Union[os.PathLike, str, None] = None,
 95    download: bool = False,
 96) -> List[str]:
 97    """Get paths to cached SurfaceMorphometrics zarr stores, one per tomogram.
 98
 99    Args:
100        path: Filepath to a folder where the downloaded raw MRC files will be saved.
101        tomogram_ids: Which tomograms to use. See `TOMOGRAM_IDS`.
102        cache_path: Filepath to a folder where the converted zarr stores will be saved. Defaults to
103            `path` if not given.
104        download: Whether to download the data if it is not present.
105
106    Returns:
107        List of filepaths to the cached zarr stores.
108    """
109    import zarr
110    from zarr.codecs import BloscCodec
111
112    cache_path = cache_path or path
113    os.makedirs(cache_path, exist_ok=True)
114
115    zarr_paths = []
116    for tomogram_id in tomogram_ids:
117        zarr_path = os.path.join(str(cache_path), f"{tomogram_id}.zarr")
118        root = zarr.open_group(zarr_path, mode="a")
119        if "raw" not in root or "labels" not in root:
120            raw_path, label_path = get_surface_morphometrics_data(path, tomogram_id, download)
121            raw = _read_raw_mrc(raw_path)
122            labels = _read_label_mrc(label_path, raw.shape)
123
124            def _make_array(name, data, shuffle):
125                array = root.create_array(
126                    name, shape=data.shape, chunks=(32, 256, 256), dtype=data.dtype,
127                    compressors=BloscCodec(cname="zstd", clevel=6, shuffle=shuffle),
128                )
129                array[:] = data
130
131            _make_array("raw", raw, shuffle="shuffle")
132            _make_array("labels", labels, shuffle="bitshuffle")
133        zarr_paths.append(zarr_path)
134
135    return zarr_paths

Get paths to cached SurfaceMorphometrics zarr stores, one per tomogram.

Arguments:
  • path: Filepath to a folder where the downloaded raw MRC files will be saved.
  • tomogram_ids: Which tomograms to use. See TOMOGRAM_IDS.
  • cache_path: Filepath to a folder where the converted zarr stores will be saved. Defaults to path if not given.
  • download: Whether to download the data if it is not present.
Returns:

List of filepaths to the cached zarr stores.

def get_surface_morphometrics_dataset( path: Union[os.PathLike, str], patch_shape: Tuple[int, int, int], tomogram_ids: List[str] = ('TE2', 'TE3', 'TE4', 'TE5', 'TE6', 'TE7', 'TE8', 'TE9', 'TE10', 'TE11', 'TE12', 'TE13', 'TE14', 'TF1', 'TF2', 'TF3', 'TF5', 'TF6', 'UE1', 'UE2', 'UE3', 'UE4', 'UE5', 'UE6', 'UE7', 'UE8', 'UE9', 'UE10', 'UF1', 'UF2', 'UF3', 'UF4', 'UF5', 'UF6'), cache_path: Union[os.PathLike, str, NoneType] = None, download: bool = False, **kwargs) -> torch.utils.data.dataset.Dataset:
138def get_surface_morphometrics_dataset(
139    path: Union[os.PathLike, str],
140    patch_shape: Tuple[int, int, int],
141    tomogram_ids: List[str] = TOMOGRAM_IDS,
142    cache_path: Union[os.PathLike, str, None] = None,
143    download: bool = False,
144    **kwargs,
145) -> Dataset:
146    """Get the SurfaceMorphometrics dataset for mitochondria and ER membrane segmentation.
147
148    Args:
149        path: Filepath to a folder where the downloaded raw MRC files will be saved.
150        patch_shape: The patch shape (z, y, x) to use for training.
151        tomogram_ids: Which tomograms to use. See `TOMOGRAM_IDS`. Defaults to all of them.
152        cache_path: Filepath to a folder where the converted zarr stores will be saved. Defaults to
153            `path` if not given.
154        download: Whether to download the data if it is not present.
155        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`.
156
157    Returns:
158        The segmentation dataset.
159    """
160    assert len(patch_shape) == 3
161
162    paths = get_surface_morphometrics_paths(path, tomogram_ids, cache_path, download)
163    kwargs = util.update_kwargs(kwargs, "is_seg_dataset", True)
164
165    return torch_em.default_segmentation_dataset(
166        raw_paths=paths,
167        raw_key="raw",
168        label_paths=paths,
169        label_key="labels",
170        patch_shape=patch_shape,
171        **kwargs,
172    )

Get the SurfaceMorphometrics dataset for mitochondria and ER membrane segmentation.

Arguments:
  • path: Filepath to a folder where the downloaded raw MRC files will be saved.
  • patch_shape: The patch shape (z, y, x) to use for training.
  • tomogram_ids: Which tomograms to use. See TOMOGRAM_IDS. Defaults to all of them.
  • cache_path: Filepath to a folder where the converted zarr stores will be saved. Defaults to path if not given.
  • download: Whether to download the data if it is not present.
  • kwargs: Additional keyword arguments for torch_em.default_segmentation_dataset.
Returns:

The segmentation dataset.

def get_surface_morphometrics_loader( path: Union[os.PathLike, str], patch_shape: Tuple[int, int, int], batch_size: int, tomogram_ids: List[str] = ('TE2', 'TE3', 'TE4', 'TE5', 'TE6', 'TE7', 'TE8', 'TE9', 'TE10', 'TE11', 'TE12', 'TE13', 'TE14', 'TF1', 'TF2', 'TF3', 'TF5', 'TF6', 'UE1', 'UE2', 'UE3', 'UE4', 'UE5', 'UE6', 'UE7', 'UE8', 'UE9', 'UE10', 'UF1', 'UF2', 'UF3', 'UF4', 'UF5', 'UF6'), cache_path: Union[os.PathLike, str, NoneType] = None, download: bool = False, **kwargs) -> torch.utils.data.dataloader.DataLoader:
175def get_surface_morphometrics_loader(
176    path: Union[os.PathLike, str],
177    patch_shape: Tuple[int, int, int],
178    batch_size: int,
179    tomogram_ids: List[str] = TOMOGRAM_IDS,
180    cache_path: Union[os.PathLike, str, None] = None,
181    download: bool = False,
182    **kwargs,
183) -> DataLoader:
184    """Get the DataLoader for mitochondria and ER membrane segmentation in the SurfaceMorphometrics
185    dataset.
186
187    Args:
188        path: Filepath to a folder where the downloaded raw MRC files will be saved.
189        patch_shape: The patch shape (z, y, x) to use for training.
190        batch_size: The batch size for training.
191        tomogram_ids: Which tomograms to use. See `TOMOGRAM_IDS`. Defaults to all of them.
192        cache_path: Filepath to a folder where the converted zarr stores will be saved. Defaults to
193            `path` if not given.
194        download: Whether to download the data if it is not present.
195        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the
196            PyTorch DataLoader.
197
198    Returns:
199        The DataLoader.
200    """
201    ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs)
202    dataset = get_surface_morphometrics_dataset(
203        path, patch_shape, tomogram_ids, cache_path, download, **ds_kwargs
204    )
205    return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)

Get the DataLoader for mitochondria and ER membrane segmentation in the SurfaceMorphometrics dataset.

Arguments:
  • path: Filepath to a folder where the downloaded raw MRC files will be saved.
  • patch_shape: The patch shape (z, y, x) to use for training.
  • batch_size: The batch size for training.
  • tomogram_ids: Which tomograms to use. See TOMOGRAM_IDS. Defaults to all of them.
  • cache_path: Filepath to a folder where the converted zarr stores will be saved. Defaults to path if not given.
  • download: Whether to download the data if it is not present.
  • kwargs: Additional keyword arguments for torch_em.default_segmentation_dataset or for the PyTorch DataLoader.
Returns:

The DataLoader.