torch_em.data.datasets.electron_microscopy.npc1_mito

The NPC1 mitochondria dataset provides a mitochondrion segmentation mask for one cryo-ET tomogram of an NPC1-deficient HEK293T cell.

The data is hosted on the CryoET Data Portal at https://cryoetdataportal.czscience.com/datasets/10456, a standalone Chan Zuckerberg Initiative-funded data release (grant CZII-2023-327779) with no linked publication. The dataset covers 4 runs (8 tomograms); only this one run carries any annotation.

The mitochondrion mask (as well as the lysosome and membrane masks also present on this run, not provided here) is a fully automated prediction (nnInteractive followed by mcm-cryoET smoothing), not expert-verified ground truth. Treat it the same way as the MitoNet auto-labels in mitonet_predicted_kidney.py: a useful pseudo-label, not verified ground truth.

The data is released under CC0-1.0, per the CryoET Data Portal's portal-wide terms of use. No public corresponding-author email could be found for this dataset (the corresponding authors, Daniel Serwas and Utz Heinrich Ermel, have no email listed on the portal or their ORCID records).

  1"""The NPC1 mitochondria dataset provides a mitochondrion segmentation mask for one cryo-ET
  2tomogram of an NPC1-deficient HEK293T cell.
  3
  4The data is hosted on the CryoET Data Portal at https://cryoetdataportal.czscience.com/datasets/10456,
  5a standalone Chan Zuckerberg Initiative-funded data release (grant CZII-2023-327779) with no linked
  6publication. The dataset covers 4 runs (8 tomograms); only this one run carries any annotation.
  7
  8The mitochondrion mask (as well as the lysosome and membrane masks also present on this run, not
  9provided here) is a fully automated prediction (nnInteractive followed by mcm-cryoET smoothing),
 10not expert-verified ground truth. Treat it the same way as the MitoNet auto-labels in
 11`mitonet_predicted_kidney.py`: a useful pseudo-label, not verified ground truth.
 12
 13The data is released under CC0-1.0, per the CryoET Data Portal's portal-wide terms of use. No
 14public corresponding-author email could be found for this dataset (the corresponding authors,
 15Daniel Serwas and Utz Heinrich Ermel, have no email listed on the portal or their ORCID records).
 16"""
 17
 18import os
 19import json
 20from typing import Union, Tuple
 21
 22import requests
 23
 24from torch.utils.data import Dataset, DataLoader
 25
 26import torch_em
 27
 28from .. import util
 29
 30
 31DATASET_ID = 10456
 32RUN = "25jul29a_Position_3"
 33VOXEL_SPACING = "VoxelSpacing14.985"
 34ANNOTATION_FOLDER = "102"
 35
 36BASE_URL = f"https://files.cryoetdataportal.cziscience.com/{DATASET_ID}/{RUN}/Reconstructions/{VOXEL_SPACING}/"
 37RAW_URL = BASE_URL + f"Tomograms/100/{RUN}.zarr"
 38LABEL_URL = BASE_URL + f"Annotations/{ANNOTATION_FOLDER}/mitochondrion-1.0_segmentationmask.zarr"
 39
 40
 41def _fetch(url, path, optional=False):
 42    if os.path.exists(path):
 43        return True
 44
 45    with requests.get(url, stream=True, timeout=(20, 300)) as response:
 46        # A chunk that holds only the fill value is not written by the portal.
 47        if optional and response.status_code == 404:
 48            return False
 49        response.raise_for_status()
 50        # The chunk is renamed only once it is complete, so an interrupted download is not reused.
 51        tmp_path = path + ".partial"
 52        with open(tmp_path, "wb") as f:
 53            for block in response.iter_content(8 * 1024 ** 2):
 54                f.write(block)
 55
 56    os.rename(tmp_path, path)
 57    return True
 58
 59
 60def _download_ome_zarr(url, out_path, download):
 61    array_path = os.path.join(out_path, "0")
 62    if os.path.exists(array_path):
 63        return array_path
 64
 65    if not download:
 66        raise RuntimeError(f"Cannot find the data at {out_path}, but download was set to False.")
 67
 68    os.makedirs(out_path, exist_ok=True)
 69    for name in (".zattrs", ".zgroup"):
 70        if not os.path.exists(os.path.join(out_path, name)):
 71            _fetch(f"{url}/{name}", os.path.join(out_path, name))
 72
 73    tmp_path = os.path.join(out_path, "0.partial")
 74    os.makedirs(tmp_path, exist_ok=True)
 75    _fetch(f"{url}/0/.zarray", os.path.join(tmp_path, ".zarray"))
 76    with open(os.path.join(tmp_path, ".zarray")) as f:
 77        meta = json.load(f)
 78
 79    grid = [-(-size // chunk) for size, chunk in zip(meta["shape"], meta["chunks"])]
 80    for z in range(grid[0]):
 81        for y in range(grid[1]):
 82            for x in range(grid[2]):
 83                chunk_dir = os.path.join(tmp_path, str(z), str(y))
 84                os.makedirs(chunk_dir, exist_ok=True)
 85                _fetch(f"{url}/0/{z}/{y}/{x}", os.path.join(chunk_dir, str(x)), optional=True)
 86
 87    os.rename(tmp_path, array_path)
 88    return array_path
 89
 90
 91def get_npc1_mito_data(path: Union[os.PathLike, str], download: bool = False) -> Tuple[str, str]:
 92    """Download the NPC1 mitochondria cryo-ET tomogram and its automated mitochondrion mask.
 93
 94    Args:
 95        path: Filepath to a folder where the data will be downloaded.
 96        download: Whether to download the data if it is not present.
 97
 98    Returns:
 99        Filepath to the tomogram.
100        Filepath to the mitochondrion mask.
101    """
102    run_dir = os.path.join(path, RUN)
103    os.makedirs(run_dir, exist_ok=True)
104
105    raw_path = _download_ome_zarr(RAW_URL, os.path.join(run_dir, "raw.zarr"), download)
106    label_path = _download_ome_zarr(LABEL_URL, os.path.join(run_dir, "labels.zarr"), download)
107    return os.path.dirname(raw_path), os.path.dirname(label_path)
108
109
110def get_npc1_mito_paths(path: Union[os.PathLike, str], download: bool = False) -> Tuple[str, str]:
111    """Get paths to the NPC1 mitochondria data.
112
113    Args:
114        path: Filepath to a folder where the data will be downloaded.
115        download: Whether to download the data if it is not present.
116
117    Returns:
118        Filepath to the tomogram.
119        Filepath to the mitochondrion mask.
120    """
121    return get_npc1_mito_data(path, download)
122
123
124def get_npc1_mito_dataset(
125    path: Union[os.PathLike, str], patch_shape: Tuple[int, int, int], download: bool = False, **kwargs
126) -> Dataset:
127    """Get the dataset for mitochondrion segmentation in the NPC1 cryo-ET tomogram.
128
129    Args:
130        path: Filepath to a folder where the data will be downloaded.
131        patch_shape: The patch shape to use for training.
132        download: Whether to download the data if it is not present.
133        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`.
134
135    Returns:
136        The segmentation dataset.
137    """
138    assert len(patch_shape) == 3
139
140    raw_path, label_path = get_npc1_mito_paths(path, download)
141
142    return torch_em.default_segmentation_dataset(
143        raw_paths=raw_path,
144        raw_key="0",
145        label_paths=label_path,
146        label_key="0",
147        patch_shape=patch_shape,
148        is_seg_dataset=True,
149        **kwargs
150    )
151
152
153def get_npc1_mito_loader(
154    path: Union[os.PathLike, str],
155    patch_shape: Tuple[int, int, int],
156    batch_size: int,
157    download: bool = False,
158    **kwargs
159) -> DataLoader:
160    """Get the DataLoader for mitochondrion segmentation in the NPC1 cryo-ET tomogram.
161
162    Args:
163        path: Filepath to a folder where the data will be downloaded.
164        patch_shape: The patch shape to use for training.
165        batch_size: The batch size for training.
166        download: Whether to download the data if it is not present.
167        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`
168            or for the PyTorch DataLoader.
169
170    Returns:
171        The DataLoader.
172    """
173    ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs)
174    dataset = get_npc1_mito_dataset(path, patch_shape, download=download, **ds_kwargs)
175    return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)
DATASET_ID = 10456
RUN = '25jul29a_Position_3'
VOXEL_SPACING = 'VoxelSpacing14.985'
ANNOTATION_FOLDER = '102'
BASE_URL = 'https://files.cryoetdataportal.cziscience.com/10456/25jul29a_Position_3/Reconstructions/VoxelSpacing14.985/'
RAW_URL = 'https://files.cryoetdataportal.cziscience.com/10456/25jul29a_Position_3/Reconstructions/VoxelSpacing14.985/Tomograms/100/25jul29a_Position_3.zarr'
LABEL_URL = 'https://files.cryoetdataportal.cziscience.com/10456/25jul29a_Position_3/Reconstructions/VoxelSpacing14.985/Annotations/102/mitochondrion-1.0_segmentationmask.zarr'
def get_npc1_mito_data(path: Union[os.PathLike, str], download: bool = False) -> Tuple[str, str]:
 92def get_npc1_mito_data(path: Union[os.PathLike, str], download: bool = False) -> Tuple[str, str]:
 93    """Download the NPC1 mitochondria cryo-ET tomogram and its automated mitochondrion mask.
 94
 95    Args:
 96        path: Filepath to a folder where the data will be downloaded.
 97        download: Whether to download the data if it is not present.
 98
 99    Returns:
100        Filepath to the tomogram.
101        Filepath to the mitochondrion mask.
102    """
103    run_dir = os.path.join(path, RUN)
104    os.makedirs(run_dir, exist_ok=True)
105
106    raw_path = _download_ome_zarr(RAW_URL, os.path.join(run_dir, "raw.zarr"), download)
107    label_path = _download_ome_zarr(LABEL_URL, os.path.join(run_dir, "labels.zarr"), download)
108    return os.path.dirname(raw_path), os.path.dirname(label_path)

Download the NPC1 mitochondria cryo-ET tomogram and its automated mitochondrion mask.

Arguments:
  • path: Filepath to a folder where the data will be downloaded.
  • download: Whether to download the data if it is not present.
Returns:

Filepath to the tomogram. Filepath to the mitochondrion mask.

def get_npc1_mito_paths(path: Union[os.PathLike, str], download: bool = False) -> Tuple[str, str]:
111def get_npc1_mito_paths(path: Union[os.PathLike, str], download: bool = False) -> Tuple[str, str]:
112    """Get paths to the NPC1 mitochondria data.
113
114    Args:
115        path: Filepath to a folder where the data will be downloaded.
116        download: Whether to download the data if it is not present.
117
118    Returns:
119        Filepath to the tomogram.
120        Filepath to the mitochondrion mask.
121    """
122    return get_npc1_mito_data(path, download)

Get paths to the NPC1 mitochondria data.

Arguments:
  • path: Filepath to a folder where the data will be downloaded.
  • download: Whether to download the data if it is not present.
Returns:

Filepath to the tomogram. Filepath to the mitochondrion mask.

def get_npc1_mito_dataset( path: Union[os.PathLike, str], patch_shape: Tuple[int, int, int], download: bool = False, **kwargs) -> torch.utils.data.dataset.Dataset:
125def get_npc1_mito_dataset(
126    path: Union[os.PathLike, str], patch_shape: Tuple[int, int, int], download: bool = False, **kwargs
127) -> Dataset:
128    """Get the dataset for mitochondrion segmentation in the NPC1 cryo-ET tomogram.
129
130    Args:
131        path: Filepath to a folder where the data will be downloaded.
132        patch_shape: The patch shape to use for training.
133        download: Whether to download the data if it is not present.
134        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`.
135
136    Returns:
137        The segmentation dataset.
138    """
139    assert len(patch_shape) == 3
140
141    raw_path, label_path = get_npc1_mito_paths(path, download)
142
143    return torch_em.default_segmentation_dataset(
144        raw_paths=raw_path,
145        raw_key="0",
146        label_paths=label_path,
147        label_key="0",
148        patch_shape=patch_shape,
149        is_seg_dataset=True,
150        **kwargs
151    )

Get the dataset for mitochondrion segmentation in the NPC1 cryo-ET tomogram.

Arguments:
  • path: Filepath to a folder where the data will be downloaded.
  • patch_shape: The patch shape to use for training.
  • download: Whether to download the data if it is not present.
  • kwargs: Additional keyword arguments for torch_em.default_segmentation_dataset.
Returns:

The segmentation dataset.

def get_npc1_mito_loader( path: Union[os.PathLike, str], patch_shape: Tuple[int, int, int], batch_size: int, download: bool = False, **kwargs) -> torch.utils.data.dataloader.DataLoader:
154def get_npc1_mito_loader(
155    path: Union[os.PathLike, str],
156    patch_shape: Tuple[int, int, int],
157    batch_size: int,
158    download: bool = False,
159    **kwargs
160) -> DataLoader:
161    """Get the DataLoader for mitochondrion segmentation in the NPC1 cryo-ET tomogram.
162
163    Args:
164        path: Filepath to a folder where the data will be downloaded.
165        patch_shape: The patch shape to use for training.
166        batch_size: The batch size for training.
167        download: Whether to download the data if it is not present.
168        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`
169            or for the PyTorch DataLoader.
170
171    Returns:
172        The DataLoader.
173    """
174    ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs)
175    dataset = get_npc1_mito_dataset(path, patch_shape, download=download, **ds_kwargs)
176    return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)

Get the DataLoader for mitochondrion segmentation in the NPC1 cryo-ET tomogram.

Arguments:
  • path: Filepath to a folder where the data will be downloaded.
  • patch_shape: The patch shape to use for training.
  • batch_size: The batch size for training.
  • download: Whether to download the data if it is not present.
  • kwargs: Additional keyword arguments for torch_em.default_segmentation_dataset or for the PyTorch DataLoader.
Returns:

The DataLoader.