torch_em.data.datasets.electron_microscopy.surface_morphometrics
The SurfaceMorphometrics dataset contains voxel segmentations of mitochondria and endoplasmic reticulum membranes in cryo-electron tomograms of mouse embryonic fibroblasts, treated with vehicle or thapsigargin.
Each tomogram has a multi-class semantic label volume (0=background, 1-3=membrane classes). The source deposit does not document a fixed name for each label value; inspect a representative volume to determine which value corresponds to which membrane in your use case.
The data is available at https://www.ebi.ac.uk/empiar/EMPIAR-11370/. The dataset was published in https://doi.org/10.1101/2022.01.23.477440. Please cite this publication if you use the dataset in your research.
1"""The SurfaceMorphometrics dataset contains voxel segmentations of mitochondria and endoplasmic 2reticulum membranes in cryo-electron tomograms of mouse embryonic fibroblasts, treated with vehicle 3or thapsigargin. 4 5Each tomogram has a multi-class semantic label volume (0=background, 1-3=membrane classes). The 6source deposit does not document a fixed name for each label value; inspect a representative volume 7to determine which value corresponds to which membrane in your use case. 8 9The data is available at https://www.ebi.ac.uk/empiar/EMPIAR-11370/. 10The dataset was published in https://doi.org/10.1101/2022.01.23.477440. 11Please cite this publication if you use the dataset in your research. 12""" 13 14import os 15from typing import List, Tuple, Union 16 17import numpy as np 18 19from torch.utils.data import DataLoader, Dataset 20 21import torch_em 22 23from .. import util 24 25 26BASE_URL = "https://ftp.ebi.ac.uk/empiar/world_availability/11370/data" 27 28TOMOGRAM_IDS = ( 29 "TE2", "TE3", "TE4", "TE5", "TE6", "TE7", "TE8", "TE9", "TE10", "TE11", "TE12", "TE13", "TE14", 30 "TF1", "TF2", "TF3", "TF5", "TF6", 31 "UE1", "UE2", "UE3", "UE4", "UE5", "UE6", "UE7", "UE8", "UE9", "UE10", 32 "UF1", "UF2", "UF3", "UF4", "UF5", "UF6", 33) 34 35# UE9 and UF3 were deposited without the usual '.mrc' suffix on their label file. 36_LABEL_SUFFIX_EXCEPTIONS = {"UE9": "_labels.rec", "UF3": "_labels.rec"} 37 38 39def _label_url(tomogram_id): 40 suffix = _LABEL_SUFFIX_EXCEPTIONS.get(tomogram_id, "_labels.rec.mrc") 41 return f"{BASE_URL}/Voxel_Segmentations/{tomogram_id}{suffix}" 42 43 44def _read_raw_mrc(path): 45 """Read an MRC volume, tolerating the non-standard headers in this deposit.""" 46 import mrcfile 47 with mrcfile.open(path, permissive=True) as mrc: 48 return np.asarray(mrc.data) 49 50 51def _read_label_mrc(path, shape): 52 """Read a label MRC volume as raw uint8 bytes after the standard 1024-byte header. 53 54 `mrcfile` rejects these files outright (non-standard machine stamp, missing map ID), so the 55 label data is read directly instead. 56 """ 57 with open(path, "rb") as f: 58 content = f.read() 59 voxel_count = int(np.prod(shape)) 60 return np.frombuffer(content[1024:1024 + voxel_count], dtype=np.uint8).reshape(shape) 61 62 63def get_surface_morphometrics_data( 64 path: Union[os.PathLike, str], tomogram_id: str, download: bool = False, 65) -> Tuple[str, str]: 66 """Download one tomogram and its label volume from the SurfaceMorphometrics dataset. 67 68 Args: 69 path: Filepath to a folder where the downloaded data will be saved. 70 tomogram_id: Which tomogram to download. One of `TOMOGRAM_IDS`. 71 download: Whether to download the data if it is not present. 72 73 Returns: 74 The filepath to the downloaded tomogram. 75 The filepath to the downloaded label volume. 76 """ 77 if tomogram_id not in TOMOGRAM_IDS: 78 raise ValueError(f"tomogram_id must be one of {TOMOGRAM_IDS}, got {tomogram_id!r}") 79 80 os.makedirs(path, exist_ok=True) 81 raw_path = os.path.join(path, f"{tomogram_id}_tomo.rec.mrc") 82 label_path = os.path.join(path, f"{tomogram_id}_labels.rec") 83 84 util.download_source(raw_path, f"{BASE_URL}/Tomograms/{tomogram_id}_tomo.rec.mrc", download, checksum=None) 85 util.download_source(label_path, _label_url(tomogram_id), download, checksum=None) 86 87 return raw_path, label_path 88 89 90def get_surface_morphometrics_paths( 91 path: Union[os.PathLike, str], 92 tomogram_ids: List[str], 93 cache_path: Union[os.PathLike, str, None] = None, 94 download: bool = False, 95) -> List[str]: 96 """Get paths to cached SurfaceMorphometrics zarr stores, one per tomogram. 97 98 Args: 99 path: Filepath to a folder where the downloaded raw MRC files will be saved. 100 tomogram_ids: Which tomograms to use. See `TOMOGRAM_IDS`. 101 cache_path: Filepath to a folder where the converted zarr stores will be saved. Defaults to 102 `path` if not given. 103 download: Whether to download the data if it is not present. 104 105 Returns: 106 List of filepaths to the cached zarr stores. 107 """ 108 import zarr 109 from zarr.codecs import BloscCodec 110 111 cache_path = cache_path or path 112 os.makedirs(cache_path, exist_ok=True) 113 114 zarr_paths = [] 115 for tomogram_id in tomogram_ids: 116 zarr_path = os.path.join(str(cache_path), f"{tomogram_id}.zarr") 117 root = zarr.open_group(zarr_path, mode="a") 118 if "raw" not in root or "labels" not in root: 119 raw_path, label_path = get_surface_morphometrics_data(path, tomogram_id, download) 120 raw = _read_raw_mrc(raw_path) 121 labels = _read_label_mrc(label_path, raw.shape) 122 123 def _make_array(name, data, shuffle): 124 array = root.create_array( 125 name, shape=data.shape, chunks=(32, 256, 256), dtype=data.dtype, 126 compressors=BloscCodec(cname="zstd", clevel=6, shuffle=shuffle), 127 ) 128 array[:] = data 129 130 _make_array("raw", raw, shuffle="shuffle") 131 _make_array("labels", labels, shuffle="bitshuffle") 132 zarr_paths.append(zarr_path) 133 134 return zarr_paths 135 136 137def get_surface_morphometrics_dataset( 138 path: Union[os.PathLike, str], 139 patch_shape: Tuple[int, int, int], 140 tomogram_ids: List[str] = TOMOGRAM_IDS, 141 cache_path: Union[os.PathLike, str, None] = None, 142 download: bool = False, 143 **kwargs, 144) -> Dataset: 145 """Get the SurfaceMorphometrics dataset for mitochondria and ER membrane segmentation. 146 147 Args: 148 path: Filepath to a folder where the downloaded raw MRC files will be saved. 149 patch_shape: The patch shape (z, y, x) to use for training. 150 tomogram_ids: Which tomograms to use. See `TOMOGRAM_IDS`. Defaults to all of them. 151 cache_path: Filepath to a folder where the converted zarr stores will be saved. Defaults to 152 `path` if not given. 153 download: Whether to download the data if it is not present. 154 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`. 155 156 Returns: 157 The segmentation dataset. 158 """ 159 assert len(patch_shape) == 3 160 161 paths = get_surface_morphometrics_paths(path, tomogram_ids, cache_path, download) 162 kwargs = util.update_kwargs(kwargs, "is_seg_dataset", True) 163 164 return torch_em.default_segmentation_dataset( 165 raw_paths=paths, 166 raw_key="raw", 167 label_paths=paths, 168 label_key="labels", 169 patch_shape=patch_shape, 170 **kwargs, 171 ) 172 173 174def get_surface_morphometrics_loader( 175 path: Union[os.PathLike, str], 176 patch_shape: Tuple[int, int, int], 177 batch_size: int, 178 tomogram_ids: List[str] = TOMOGRAM_IDS, 179 cache_path: Union[os.PathLike, str, None] = None, 180 download: bool = False, 181 **kwargs, 182) -> DataLoader: 183 """Get the DataLoader for mitochondria and ER membrane segmentation in the SurfaceMorphometrics 184 dataset. 185 186 Args: 187 path: Filepath to a folder where the downloaded raw MRC files will be saved. 188 patch_shape: The patch shape (z, y, x) to use for training. 189 batch_size: The batch size for training. 190 tomogram_ids: Which tomograms to use. See `TOMOGRAM_IDS`. Defaults to all of them. 191 cache_path: Filepath to a folder where the converted zarr stores will be saved. Defaults to 192 `path` if not given. 193 download: Whether to download the data if it is not present. 194 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the 195 PyTorch DataLoader. 196 197 Returns: 198 The DataLoader. 199 """ 200 ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs) 201 dataset = get_surface_morphometrics_dataset( 202 path, patch_shape, tomogram_ids, cache_path, download, **ds_kwargs 203 ) 204 return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)
64def get_surface_morphometrics_data( 65 path: Union[os.PathLike, str], tomogram_id: str, download: bool = False, 66) -> Tuple[str, str]: 67 """Download one tomogram and its label volume from the SurfaceMorphometrics dataset. 68 69 Args: 70 path: Filepath to a folder where the downloaded data will be saved. 71 tomogram_id: Which tomogram to download. One of `TOMOGRAM_IDS`. 72 download: Whether to download the data if it is not present. 73 74 Returns: 75 The filepath to the downloaded tomogram. 76 The filepath to the downloaded label volume. 77 """ 78 if tomogram_id not in TOMOGRAM_IDS: 79 raise ValueError(f"tomogram_id must be one of {TOMOGRAM_IDS}, got {tomogram_id!r}") 80 81 os.makedirs(path, exist_ok=True) 82 raw_path = os.path.join(path, f"{tomogram_id}_tomo.rec.mrc") 83 label_path = os.path.join(path, f"{tomogram_id}_labels.rec") 84 85 util.download_source(raw_path, f"{BASE_URL}/Tomograms/{tomogram_id}_tomo.rec.mrc", download, checksum=None) 86 util.download_source(label_path, _label_url(tomogram_id), download, checksum=None) 87 88 return raw_path, label_path
Download one tomogram and its label volume from the SurfaceMorphometrics dataset.
Arguments:
- path: Filepath to a folder where the downloaded data will be saved.
- tomogram_id: Which tomogram to download. One of
TOMOGRAM_IDS. - download: Whether to download the data if it is not present.
Returns:
The filepath to the downloaded tomogram. The filepath to the downloaded label volume.
91def get_surface_morphometrics_paths( 92 path: Union[os.PathLike, str], 93 tomogram_ids: List[str], 94 cache_path: Union[os.PathLike, str, None] = None, 95 download: bool = False, 96) -> List[str]: 97 """Get paths to cached SurfaceMorphometrics zarr stores, one per tomogram. 98 99 Args: 100 path: Filepath to a folder where the downloaded raw MRC files will be saved. 101 tomogram_ids: Which tomograms to use. See `TOMOGRAM_IDS`. 102 cache_path: Filepath to a folder where the converted zarr stores will be saved. Defaults to 103 `path` if not given. 104 download: Whether to download the data if it is not present. 105 106 Returns: 107 List of filepaths to the cached zarr stores. 108 """ 109 import zarr 110 from zarr.codecs import BloscCodec 111 112 cache_path = cache_path or path 113 os.makedirs(cache_path, exist_ok=True) 114 115 zarr_paths = [] 116 for tomogram_id in tomogram_ids: 117 zarr_path = os.path.join(str(cache_path), f"{tomogram_id}.zarr") 118 root = zarr.open_group(zarr_path, mode="a") 119 if "raw" not in root or "labels" not in root: 120 raw_path, label_path = get_surface_morphometrics_data(path, tomogram_id, download) 121 raw = _read_raw_mrc(raw_path) 122 labels = _read_label_mrc(label_path, raw.shape) 123 124 def _make_array(name, data, shuffle): 125 array = root.create_array( 126 name, shape=data.shape, chunks=(32, 256, 256), dtype=data.dtype, 127 compressors=BloscCodec(cname="zstd", clevel=6, shuffle=shuffle), 128 ) 129 array[:] = data 130 131 _make_array("raw", raw, shuffle="shuffle") 132 _make_array("labels", labels, shuffle="bitshuffle") 133 zarr_paths.append(zarr_path) 134 135 return zarr_paths
Get paths to cached SurfaceMorphometrics zarr stores, one per tomogram.
Arguments:
- path: Filepath to a folder where the downloaded raw MRC files will be saved.
- tomogram_ids: Which tomograms to use. See
TOMOGRAM_IDS. - cache_path: Filepath to a folder where the converted zarr stores will be saved. Defaults to
pathif not given. - download: Whether to download the data if it is not present.
Returns:
List of filepaths to the cached zarr stores.
138def get_surface_morphometrics_dataset( 139 path: Union[os.PathLike, str], 140 patch_shape: Tuple[int, int, int], 141 tomogram_ids: List[str] = TOMOGRAM_IDS, 142 cache_path: Union[os.PathLike, str, None] = None, 143 download: bool = False, 144 **kwargs, 145) -> Dataset: 146 """Get the SurfaceMorphometrics dataset for mitochondria and ER membrane segmentation. 147 148 Args: 149 path: Filepath to a folder where the downloaded raw MRC files will be saved. 150 patch_shape: The patch shape (z, y, x) to use for training. 151 tomogram_ids: Which tomograms to use. See `TOMOGRAM_IDS`. Defaults to all of them. 152 cache_path: Filepath to a folder where the converted zarr stores will be saved. Defaults to 153 `path` if not given. 154 download: Whether to download the data if it is not present. 155 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`. 156 157 Returns: 158 The segmentation dataset. 159 """ 160 assert len(patch_shape) == 3 161 162 paths = get_surface_morphometrics_paths(path, tomogram_ids, cache_path, download) 163 kwargs = util.update_kwargs(kwargs, "is_seg_dataset", True) 164 165 return torch_em.default_segmentation_dataset( 166 raw_paths=paths, 167 raw_key="raw", 168 label_paths=paths, 169 label_key="labels", 170 patch_shape=patch_shape, 171 **kwargs, 172 )
Get the SurfaceMorphometrics dataset for mitochondria and ER membrane segmentation.
Arguments:
- path: Filepath to a folder where the downloaded raw MRC files will be saved.
- patch_shape: The patch shape (z, y, x) to use for training.
- tomogram_ids: Which tomograms to use. See
TOMOGRAM_IDS. Defaults to all of them. - cache_path: Filepath to a folder where the converted zarr stores will be saved. Defaults to
pathif not given. - download: Whether to download the data if it is not present.
- kwargs: Additional keyword arguments for
torch_em.default_segmentation_dataset.
Returns:
The segmentation dataset.
175def get_surface_morphometrics_loader( 176 path: Union[os.PathLike, str], 177 patch_shape: Tuple[int, int, int], 178 batch_size: int, 179 tomogram_ids: List[str] = TOMOGRAM_IDS, 180 cache_path: Union[os.PathLike, str, None] = None, 181 download: bool = False, 182 **kwargs, 183) -> DataLoader: 184 """Get the DataLoader for mitochondria and ER membrane segmentation in the SurfaceMorphometrics 185 dataset. 186 187 Args: 188 path: Filepath to a folder where the downloaded raw MRC files will be saved. 189 patch_shape: The patch shape (z, y, x) to use for training. 190 batch_size: The batch size for training. 191 tomogram_ids: Which tomograms to use. See `TOMOGRAM_IDS`. Defaults to all of them. 192 cache_path: Filepath to a folder where the converted zarr stores will be saved. Defaults to 193 `path` if not given. 194 download: Whether to download the data if it is not present. 195 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the 196 PyTorch DataLoader. 197 198 Returns: 199 The DataLoader. 200 """ 201 ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs) 202 dataset = get_surface_morphometrics_dataset( 203 path, patch_shape, tomogram_ids, cache_path, download, **ds_kwargs 204 ) 205 return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)
Get the DataLoader for mitochondria and ER membrane segmentation in the SurfaceMorphometrics dataset.
Arguments:
- path: Filepath to a folder where the downloaded raw MRC files will be saved.
- patch_shape: The patch shape (z, y, x) to use for training.
- batch_size: The batch size for training.
- tomogram_ids: Which tomograms to use. See
TOMOGRAM_IDS. Defaults to all of them. - cache_path: Filepath to a folder where the converted zarr stores will be saved. Defaults to
pathif not given. - download: Whether to download the data if it is not present.
- kwargs: Additional keyword arguments for
torch_em.default_segmentation_datasetor for the PyTorch DataLoader.
Returns:
The DataLoader.