torch_em.data.datasets.electron_microscopy.sxt_ins1e_mito

The SXT-INS1E-Mito dataset contains organelle segmentation for whole INS-1E pancreatic beta cells imaged by cryo-hydrated soft X-ray tomography (SXT), across unstimulated and glucose/Exendin-4 stimulated conditions.

IMPORTANT: The imaging modality is soft X-ray tomography, not electron microscopy. It is filed here because it shares the same volumetric organelle-segmentation role as the other datasets in this module, not because it is EM.

Each of the 55 cells has a semantic label volume with four classes: 0: exterior, 1: cell, 2: nucleus, 5: mitochondria

The data is available at https://doi.org/10.5281/zenodo.20513085 under the CC-BY-4.0 license. The dataset was published in https://doi.org/10.64898/2026.03.19.712811. Please cite this publication if you use the dataset in your research.

  1"""The SXT-INS1E-Mito dataset contains organelle segmentation for whole INS-1E pancreatic beta cells
  2imaged by cryo-hydrated soft X-ray tomography (SXT), across unstimulated and glucose/Exendin-4
  3stimulated conditions.
  4
  5IMPORTANT: The imaging modality is soft X-ray tomography, not electron microscopy. It is filed here
  6because it shares the same volumetric organelle-segmentation role as the other datasets in this module,
  7not because it is EM.
  8
  9Each of the 55 cells has a semantic label volume with four classes:
 10    0: exterior, 1: cell, 2: nucleus, 5: mitochondria
 11
 12The data is available at https://doi.org/10.5281/zenodo.20513085 under the CC-BY-4.0 license.
 13The dataset was published in https://doi.org/10.64898/2026.03.19.712811.
 14Please cite this publication if you use the dataset in your research.
 15"""
 16
 17import os
 18import zipfile
 19import tempfile
 20from typing import List, Tuple, Union
 21
 22import numpy as np
 23
 24from torch.utils.data import DataLoader, Dataset
 25
 26import torch_em
 27
 28from .. import util
 29
 30
 31BASE_URL = "https://zenodo.org/api/records/20513085/files"
 32RAW_ZIP_URL = f"{BASE_URL}/Raw_tomograms.zip/content"
 33LABEL_ZIP_URL = f"{BASE_URL}/Labels.zip/content"
 34METADATA_URL = f"{BASE_URL}/Cell_Metadata.csv/content"
 35
 36LABEL_NAMES = {0: "exterior", 1: "cell", 2: "nucleus", 5: "mitochondria"}
 37
 38
 39def get_sxt_ins1e_mito_cell_names(path: Union[os.PathLike, str], download: bool = False) -> List[str]:
 40    """Get the list of cell names in the SXT-INS1E-Mito dataset.
 41
 42    Args:
 43        path: Filepath to a folder where the downloaded metadata will be saved.
 44        download: Whether to download the metadata if it is not present.
 45
 46    Returns:
 47        The list of cell names.
 48    """
 49    os.makedirs(path, exist_ok=True)
 50    metadata_path = os.path.join(path, "Cell_Metadata.csv")
 51    util.download_source(metadata_path, METADATA_URL, download, checksum=None)
 52
 53    with open(metadata_path) as f:
 54        next(f)  # Skip the header line.
 55        return [line.split(",")[0] for line in f if line.strip()]
 56
 57
 58def _read_mrc_member(zip_url, member_name):
 59    """Read one MRC member of a remote ZIP archive via HTTP range requests, without downloading the rest.
 60
 61    `mrcfile` only accepts a filesystem path, so the member is written to a temporary file first.
 62    """
 63    import fsspec
 64    import mrcfile
 65
 66    fs = fsspec.filesystem("http")
 67    with fs.open(zip_url, "rb") as f, zipfile.ZipFile(f) as zf:
 68        content = zf.read(member_name)
 69
 70    with tempfile.NamedTemporaryFile(suffix=".mrc") as tmp:
 71        tmp.write(content)
 72        tmp.flush()
 73        with mrcfile.open(tmp.name, permissive=True) as mrc:
 74            return np.asarray(mrc.data)
 75
 76
 77def get_sxt_ins1e_mito_data(path: Union[os.PathLike, str], cell_name: str, download: bool = False) -> str:
 78    """Stream one cell's tomogram and label volume and cache it as a zarr v3 store.
 79
 80    Args:
 81        path: Filepath to a folder where the cached zarr store will be saved.
 82        cell_name: The cell to fetch. See `get_sxt_ins1e_mito_cell_names`.
 83        download: Whether to stream and cache the data if it is not present.
 84
 85    Returns:
 86        The filepath to the cached zarr store.
 87    """
 88    import zarr
 89    from zarr.codecs import BloscCodec
 90
 91    os.makedirs(path, exist_ok=True)
 92    zarr_path = os.path.join(path, f"{cell_name}.zarr")
 93
 94    root = zarr.open_group(zarr_path, mode="a")
 95    if "raw" in root and "labels" in root:
 96        return zarr_path
 97
 98    if not download:
 99        raise RuntimeError(f"No cached data found at '{zarr_path}'. Set download=True to stream it.")
100
101    raw = _read_mrc_member(RAW_ZIP_URL, f"Raw_tomograms/{cell_name}_scaled.mrc")
102    labels = _read_mrc_member(LABEL_ZIP_URL, f"Labels/{cell_name}_labels.mrc")
103
104    assert raw.shape == labels.shape, f"Shape mismatch for '{cell_name}': {raw.shape} vs {labels.shape}"
105
106    def _make_array(name, data, shuffle):
107        array = root.create_array(
108            name, shape=data.shape, chunks=(32, 256, 256), dtype=data.dtype,
109            compressors=BloscCodec(cname="zstd", clevel=6, shuffle=shuffle),
110        )
111        array[:] = data
112
113    root.attrs["cell_name"] = cell_name
114    root.attrs["label_names"] = LABEL_NAMES
115
116    _make_array("raw", raw, shuffle="shuffle")
117    _make_array("labels", labels, shuffle="bitshuffle")
118
119    return zarr_path
120
121
122def get_sxt_ins1e_mito_paths(
123    path: Union[os.PathLike, str], cell_names: List[str], download: bool = False,
124) -> List[str]:
125    """Get paths to cached SXT-INS1E-Mito zarr stores, one per cell.
126
127    Args:
128        path: Filepath to a folder where the cached zarr stores will be saved.
129        cell_names: Which cells to use. See `get_sxt_ins1e_mito_cell_names`.
130        download: Whether to stream and cache the data if it is not present.
131
132    Returns:
133        List of filepaths to the cached zarr stores.
134    """
135    return [get_sxt_ins1e_mito_data(path, cell_name, download) for cell_name in cell_names]
136
137
138def get_sxt_ins1e_mito_dataset(
139    path: Union[os.PathLike, str],
140    patch_shape: Tuple[int, int, int],
141    cell_names: List[str],
142    download: bool = False,
143    **kwargs,
144) -> Dataset:
145    """Get the SXT-INS1E-Mito dataset for mitochondria and organelle segmentation.
146
147    Args:
148        path: Filepath to a folder where the cached zarr stores will be saved.
149        patch_shape: The patch shape (z, y, x) to use for training.
150        cell_names: Which cells to use. See `get_sxt_ins1e_mito_cell_names`.
151        download: Whether to stream and cache data if not already present.
152        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`.
153
154    Returns:
155        The segmentation dataset.
156    """
157    assert len(patch_shape) == 3
158
159    paths = get_sxt_ins1e_mito_paths(path, cell_names, download)
160    kwargs = util.update_kwargs(kwargs, "is_seg_dataset", True)
161
162    return torch_em.default_segmentation_dataset(
163        raw_paths=paths,
164        raw_key="raw",
165        label_paths=paths,
166        label_key="labels",
167        patch_shape=patch_shape,
168        **kwargs,
169    )
170
171
172def get_sxt_ins1e_mito_loader(
173    path: Union[os.PathLike, str],
174    patch_shape: Tuple[int, int, int],
175    batch_size: int,
176    cell_names: List[str],
177    download: bool = False,
178    **kwargs,
179) -> DataLoader:
180    """Get the DataLoader for mitochondria and organelle segmentation in the SXT-INS1E-Mito dataset.
181
182    Args:
183        path: Filepath to a folder where the cached zarr stores will be saved.
184        patch_shape: The patch shape (z, y, x) to use for training.
185        batch_size: The batch size for training.
186        cell_names: Which cells to use. See `get_sxt_ins1e_mito_cell_names`.
187        download: Whether to stream and cache data if not already present.
188        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the
189            PyTorch DataLoader.
190
191    Returns:
192        The DataLoader.
193    """
194    ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs)
195    dataset = get_sxt_ins1e_mito_dataset(path, patch_shape, cell_names, download, **ds_kwargs)
196    return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)
BASE_URL = 'https://zenodo.org/api/records/20513085/files'
RAW_ZIP_URL = 'https://zenodo.org/api/records/20513085/files/Raw_tomograms.zip/content'
LABEL_ZIP_URL = 'https://zenodo.org/api/records/20513085/files/Labels.zip/content'
METADATA_URL = 'https://zenodo.org/api/records/20513085/files/Cell_Metadata.csv/content'
LABEL_NAMES = {0: 'exterior', 1: 'cell', 2: 'nucleus', 5: 'mitochondria'}
def get_sxt_ins1e_mito_cell_names(path: Union[os.PathLike, str], download: bool = False) -> List[str]:
40def get_sxt_ins1e_mito_cell_names(path: Union[os.PathLike, str], download: bool = False) -> List[str]:
41    """Get the list of cell names in the SXT-INS1E-Mito dataset.
42
43    Args:
44        path: Filepath to a folder where the downloaded metadata will be saved.
45        download: Whether to download the metadata if it is not present.
46
47    Returns:
48        The list of cell names.
49    """
50    os.makedirs(path, exist_ok=True)
51    metadata_path = os.path.join(path, "Cell_Metadata.csv")
52    util.download_source(metadata_path, METADATA_URL, download, checksum=None)
53
54    with open(metadata_path) as f:
55        next(f)  # Skip the header line.
56        return [line.split(",")[0] for line in f if line.strip()]

Get the list of cell names in the SXT-INS1E-Mito dataset.

Arguments:
  • path: Filepath to a folder where the downloaded metadata will be saved.
  • download: Whether to download the metadata if it is not present.
Returns:

The list of cell names.

def get_sxt_ins1e_mito_data( path: Union[os.PathLike, str], cell_name: str, download: bool = False) -> str:
 78def get_sxt_ins1e_mito_data(path: Union[os.PathLike, str], cell_name: str, download: bool = False) -> str:
 79    """Stream one cell's tomogram and label volume and cache it as a zarr v3 store.
 80
 81    Args:
 82        path: Filepath to a folder where the cached zarr store will be saved.
 83        cell_name: The cell to fetch. See `get_sxt_ins1e_mito_cell_names`.
 84        download: Whether to stream and cache the data if it is not present.
 85
 86    Returns:
 87        The filepath to the cached zarr store.
 88    """
 89    import zarr
 90    from zarr.codecs import BloscCodec
 91
 92    os.makedirs(path, exist_ok=True)
 93    zarr_path = os.path.join(path, f"{cell_name}.zarr")
 94
 95    root = zarr.open_group(zarr_path, mode="a")
 96    if "raw" in root and "labels" in root:
 97        return zarr_path
 98
 99    if not download:
100        raise RuntimeError(f"No cached data found at '{zarr_path}'. Set download=True to stream it.")
101
102    raw = _read_mrc_member(RAW_ZIP_URL, f"Raw_tomograms/{cell_name}_scaled.mrc")
103    labels = _read_mrc_member(LABEL_ZIP_URL, f"Labels/{cell_name}_labels.mrc")
104
105    assert raw.shape == labels.shape, f"Shape mismatch for '{cell_name}': {raw.shape} vs {labels.shape}"
106
107    def _make_array(name, data, shuffle):
108        array = root.create_array(
109            name, shape=data.shape, chunks=(32, 256, 256), dtype=data.dtype,
110            compressors=BloscCodec(cname="zstd", clevel=6, shuffle=shuffle),
111        )
112        array[:] = data
113
114    root.attrs["cell_name"] = cell_name
115    root.attrs["label_names"] = LABEL_NAMES
116
117    _make_array("raw", raw, shuffle="shuffle")
118    _make_array("labels", labels, shuffle="bitshuffle")
119
120    return zarr_path

Stream one cell's tomogram and label volume and cache it as a zarr v3 store.

Arguments:
  • path: Filepath to a folder where the cached zarr store will be saved.
  • cell_name: The cell to fetch. See get_sxt_ins1e_mito_cell_names.
  • download: Whether to stream and cache the data if it is not present.
Returns:

The filepath to the cached zarr store.

def get_sxt_ins1e_mito_paths( path: Union[os.PathLike, str], cell_names: List[str], download: bool = False) -> List[str]:
123def get_sxt_ins1e_mito_paths(
124    path: Union[os.PathLike, str], cell_names: List[str], download: bool = False,
125) -> List[str]:
126    """Get paths to cached SXT-INS1E-Mito zarr stores, one per cell.
127
128    Args:
129        path: Filepath to a folder where the cached zarr stores will be saved.
130        cell_names: Which cells to use. See `get_sxt_ins1e_mito_cell_names`.
131        download: Whether to stream and cache the data if it is not present.
132
133    Returns:
134        List of filepaths to the cached zarr stores.
135    """
136    return [get_sxt_ins1e_mito_data(path, cell_name, download) for cell_name in cell_names]

Get paths to cached SXT-INS1E-Mito zarr stores, one per cell.

Arguments:
  • path: Filepath to a folder where the cached zarr stores will be saved.
  • cell_names: Which cells to use. See get_sxt_ins1e_mito_cell_names.
  • download: Whether to stream and cache the data if it is not present.
Returns:

List of filepaths to the cached zarr stores.

def get_sxt_ins1e_mito_dataset( path: Union[os.PathLike, str], patch_shape: Tuple[int, int, int], cell_names: List[str], download: bool = False, **kwargs) -> torch.utils.data.dataset.Dataset:
139def get_sxt_ins1e_mito_dataset(
140    path: Union[os.PathLike, str],
141    patch_shape: Tuple[int, int, int],
142    cell_names: List[str],
143    download: bool = False,
144    **kwargs,
145) -> Dataset:
146    """Get the SXT-INS1E-Mito dataset for mitochondria and organelle segmentation.
147
148    Args:
149        path: Filepath to a folder where the cached zarr stores will be saved.
150        patch_shape: The patch shape (z, y, x) to use for training.
151        cell_names: Which cells to use. See `get_sxt_ins1e_mito_cell_names`.
152        download: Whether to stream and cache data if not already present.
153        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`.
154
155    Returns:
156        The segmentation dataset.
157    """
158    assert len(patch_shape) == 3
159
160    paths = get_sxt_ins1e_mito_paths(path, cell_names, download)
161    kwargs = util.update_kwargs(kwargs, "is_seg_dataset", True)
162
163    return torch_em.default_segmentation_dataset(
164        raw_paths=paths,
165        raw_key="raw",
166        label_paths=paths,
167        label_key="labels",
168        patch_shape=patch_shape,
169        **kwargs,
170    )

Get the SXT-INS1E-Mito dataset for mitochondria and organelle segmentation.

Arguments:
  • path: Filepath to a folder where the cached zarr stores will be saved.
  • patch_shape: The patch shape (z, y, x) to use for training.
  • cell_names: Which cells to use. See get_sxt_ins1e_mito_cell_names.
  • download: Whether to stream and cache data if not already present.
  • kwargs: Additional keyword arguments for torch_em.default_segmentation_dataset.
Returns:

The segmentation dataset.

def get_sxt_ins1e_mito_loader( path: Union[os.PathLike, str], patch_shape: Tuple[int, int, int], batch_size: int, cell_names: List[str], download: bool = False, **kwargs) -> torch.utils.data.dataloader.DataLoader:
173def get_sxt_ins1e_mito_loader(
174    path: Union[os.PathLike, str],
175    patch_shape: Tuple[int, int, int],
176    batch_size: int,
177    cell_names: List[str],
178    download: bool = False,
179    **kwargs,
180) -> DataLoader:
181    """Get the DataLoader for mitochondria and organelle segmentation in the SXT-INS1E-Mito dataset.
182
183    Args:
184        path: Filepath to a folder where the cached zarr stores will be saved.
185        patch_shape: The patch shape (z, y, x) to use for training.
186        batch_size: The batch size for training.
187        cell_names: Which cells to use. See `get_sxt_ins1e_mito_cell_names`.
188        download: Whether to stream and cache data if not already present.
189        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the
190            PyTorch DataLoader.
191
192    Returns:
193        The DataLoader.
194    """
195    ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs)
196    dataset = get_sxt_ins1e_mito_dataset(path, patch_shape, cell_names, download, **ds_kwargs)
197    return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)

Get the DataLoader for mitochondria and organelle segmentation in the SXT-INS1E-Mito dataset.

Arguments:
  • path: Filepath to a folder where the cached zarr stores will be saved.
  • patch_shape: The patch shape (z, y, x) to use for training.
  • batch_size: The batch size for training.
  • cell_names: Which cells to use. See get_sxt_ins1e_mito_cell_names.
  • download: Whether to stream and cache data if not already present.
  • kwargs: Additional keyword arguments for torch_em.default_segmentation_dataset or for the PyTorch DataLoader.
Returns:

The DataLoader.