torch_em.data.datasets.light_microscopy.neurons_3d_multispecies

This dataset contains 3D label-free microscopy volumes of neuronal cell bodies (somata) from human, mouse and rat brain tissue, imaged with oblique illumination or Dodt gradient contrast (DGC) microscopy, with manually annotated instance segmentation masks and an official train / val / test split.

The dataset is located at https://zenodo.org/records/20797635. Please cite it if you use this dataset in your research.

  1"""This dataset contains 3D label-free microscopy volumes of neuronal cell
  2bodies (somata) from human, mouse and rat brain tissue, imaged with oblique
  3illumination or Dodt gradient contrast (DGC) microscopy, with manually
  4annotated instance segmentation masks and an official train / val / test split.
  5
  6The dataset is located at https://zenodo.org/records/20797635.
  7Please cite it if you use this dataset in your research.
  8"""
  9
 10import os
 11import json
 12from glob import glob
 13from natsort import natsorted
 14from typing import List, Literal, Optional, Tuple, Union
 15
 16from torch.utils.data import Dataset, DataLoader
 17
 18import torch_em
 19
 20from .. import util
 21
 22
 23URLS = {
 24    "human_oblique": "https://zenodo.org/records/20797635/files/human_neurons_oblique.zip",
 25    "human_dodt": "https://zenodo.org/records/20797635/files/human_neurons_dodt.zip",
 26    "mouse_oblique": "https://zenodo.org/records/20797635/files/mice_neurons_oblique.zip",
 27    "mouse_dodt": "https://zenodo.org/records/20797635/files/mice_neurons_dodt.zip",
 28    "rat_oblique": "https://zenodo.org/records/20797635/files/rat_neurons_oblique.zip",
 29    "rat_dodt": "https://zenodo.org/records/20797635/files/rat_neurons_dodt.zip",
 30}
 31
 32CHECKSUMS = {
 33    "human_oblique": "8b2c37b74fe2b890a3d941097d619651d56fa457a37c010c1c31d24604acdebc",
 34    "human_dodt": "1b4a72302e0e85dfb3169fd753dcfca6c4d200cea135ce7b2d8401af009d3621",
 35    "mouse_oblique": "31dd601cf114359a66e39266b688e51448f70bf6a516474b8905f3071ce372fa",
 36    "mouse_dodt": "3debfdd5534509b8161bbf2f952cba1bd7629aa22c78d5f32bf585d10edb3b1c",
 37    "rat_oblique": "0b7be7b0a9436d74ed1f2f2676256c9ad0072996f89fd95ce2b01a4a2cc64f86",
 38    "rat_dodt": "431269c513bb558f7434018fc087c22e31c1e7cbdccc4da2d399d0bf3b85715a",
 39}
 40
 41# Zenodo archive names differ from the species / modality keys used here.
 42ARCHIVE_NAMES = {
 43    "human_oblique": "human_neurons_oblique",
 44    "human_dodt": "human_neurons_dodt",
 45    "mouse_oblique": "mice_neurons_oblique",
 46    "mouse_dodt": "mice_neurons_dodt",
 47    "rat_oblique": "rat_neurons_oblique",
 48    "rat_dodt": "rat_neurons_dodt",
 49}
 50
 51SPECIES = ["human", "mouse", "rat"]
 52MODALITIES = ["oblique", "dodt"]
 53
 54
 55def get_neurons_3d_multispecies_data(
 56    path: Union[os.PathLike, str],
 57    species: Optional[List[Literal["human", "mouse", "rat"]]] = None,
 58    modality: Optional[List[Literal["oblique", "dodt"]]] = None,
 59    download: bool = False,
 60) -> List[str]:
 61    """Download the multi-species 3D neuron segmentation dataset.
 62
 63    Args:
 64        path: Filepath to a folder where the downloaded data will be saved.
 65        species: The species subset(s) to download. Defaults to all species.
 66        modality: The imaging modality subset(s) to download. Defaults to all modalities.
 67        download: Whether to download the data if it is not present.
 68
 69    Returns:
 70        List of filepaths to the extracted data directories.
 71    """
 72    species = SPECIES if species is None else species
 73    modality = MODALITIES if modality is None else modality
 74
 75    os.makedirs(path, exist_ok=True)
 76
 77    data_dirs = []
 78    for sp in species:
 79        for mod in modality:
 80            key = f"{sp}_{mod}"
 81            data_dir = os.path.join(path, ARCHIVE_NAMES[key])
 82            if not os.path.exists(data_dir):
 83                zip_path = os.path.join(path, f"{ARCHIVE_NAMES[key]}.zip")
 84                util.download_source(zip_path, URLS[key], download, checksum=CHECKSUMS[key])
 85                util.unzip(zip_path, path)
 86            data_dirs.append(data_dir)
 87
 88    return data_dirs
 89
 90
 91def get_neurons_3d_multispecies_paths(
 92    path: Union[os.PathLike, str],
 93    species: Optional[List[Literal["human", "mouse", "rat"]]] = None,
 94    modality: Optional[List[Literal["oblique", "dodt"]]] = None,
 95    split: Optional[Literal["train", "val", "test"]] = None,
 96    download: bool = False,
 97) -> Tuple[List[str], List[str]]:
 98    """Get paths to the multi-species 3D neuron segmentation data.
 99
100    Args:
101        path: Filepath to a folder where the downloaded data will be saved.
102        species: The species subset(s) to use. Defaults to all species.
103        modality: The imaging modality subset(s) to use. Defaults to all modalities.
104        split: The data split to use. Either 'train', 'val' or 'test'. Defaults to using all splits.
105        download: Whether to download the data if it is not present.
106
107    Returns:
108        List of filepaths for the image data.
109        List of filepaths for the label data.
110    """
111    data_dirs = get_neurons_3d_multispecies_data(path, species, modality, download)
112
113    raw_paths, label_paths = [], []
114    for data_dir in data_dirs:
115        if split is None:
116            fnames = natsorted(os.path.basename(p) for p in glob(os.path.join(data_dir, "images", "*.tif")))
117        else:
118            split_file = os.path.join(data_dir, "split.json")
119            with open(split_file) as f:
120                fnames = natsorted(json.load(f)[split])
121
122        for fname in fnames:
123            raw_path = os.path.join(data_dir, "images", fname)
124            label_path = os.path.join(data_dir, "masks", fname)
125            if os.path.exists(raw_path) and os.path.exists(label_path):
126                raw_paths.append(raw_path)
127                label_paths.append(label_path)
128
129    if len(raw_paths) == 0:
130        raise RuntimeError(f"No image files found under {path}. Please check the dataset structure.")
131
132    return raw_paths, label_paths
133
134
135def get_neurons_3d_multispecies_dataset(
136    path: Union[os.PathLike, str],
137    patch_shape: Tuple[int, int, int],
138    species: Optional[List[Literal["human", "mouse", "rat"]]] = None,
139    modality: Optional[List[Literal["oblique", "dodt"]]] = None,
140    split: Optional[Literal["train", "val", "test"]] = None,
141    download: bool = False,
142    **kwargs,
143) -> Dataset:
144    """Get the multi-species 3D neuron segmentation dataset.
145
146    Args:
147        path: Filepath to a folder where the downloaded data will be saved.
148        patch_shape: The patch shape to use for training.
149        species: The species subset(s) to use. Defaults to all species.
150        modality: The imaging modality subset(s) to use. Defaults to all modalities.
151        split: The data split to use. Either 'train', 'val' or 'test'. Defaults to using all splits.
152        download: Whether to download the data if it is not present.
153        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`.
154
155    Returns:
156        The segmentation dataset.
157    """
158    raw_paths, label_paths = get_neurons_3d_multispecies_paths(path, species, modality, split, download)
159
160    return torch_em.default_segmentation_dataset(
161        raw_paths=raw_paths,
162        raw_key=None,
163        label_paths=label_paths,
164        label_key=None,
165        patch_shape=patch_shape,
166        **kwargs,
167    )
168
169
170def get_neurons_3d_multispecies_loader(
171    path: Union[os.PathLike, str],
172    batch_size: int,
173    patch_shape: Tuple[int, int, int],
174    species: Optional[List[Literal["human", "mouse", "rat"]]] = None,
175    modality: Optional[List[Literal["oblique", "dodt"]]] = None,
176    split: Optional[Literal["train", "val", "test"]] = None,
177    download: bool = False,
178    **kwargs,
179) -> DataLoader:
180    """Get the multi-species 3D neuron segmentation dataloader.
181
182    Args:
183        path: Filepath to a folder where the downloaded data will be saved.
184        batch_size: The batch size for training.
185        patch_shape: The patch shape to use for training.
186        species: The species subset(s) to use. Defaults to all species.
187        modality: The imaging modality subset(s) to use. Defaults to all modalities.
188        split: The data split to use. Either 'train', 'val' or 'test'. Defaults to using all splits.
189        download: Whether to download the data if it is not present.
190        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader.
191
192    Returns:
193        The DataLoader.
194    """
195    ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs)
196    dataset = get_neurons_3d_multispecies_dataset(path, patch_shape, species, modality, split, download, **ds_kwargs)
197    return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)
URLS = {'human_oblique': 'https://zenodo.org/records/20797635/files/human_neurons_oblique.zip', 'human_dodt': 'https://zenodo.org/records/20797635/files/human_neurons_dodt.zip', 'mouse_oblique': 'https://zenodo.org/records/20797635/files/mice_neurons_oblique.zip', 'mouse_dodt': 'https://zenodo.org/records/20797635/files/mice_neurons_dodt.zip', 'rat_oblique': 'https://zenodo.org/records/20797635/files/rat_neurons_oblique.zip', 'rat_dodt': 'https://zenodo.org/records/20797635/files/rat_neurons_dodt.zip'}
CHECKSUMS = {'human_oblique': '8b2c37b74fe2b890a3d941097d619651d56fa457a37c010c1c31d24604acdebc', 'human_dodt': '1b4a72302e0e85dfb3169fd753dcfca6c4d200cea135ce7b2d8401af009d3621', 'mouse_oblique': '31dd601cf114359a66e39266b688e51448f70bf6a516474b8905f3071ce372fa', 'mouse_dodt': '3debfdd5534509b8161bbf2f952cba1bd7629aa22c78d5f32bf585d10edb3b1c', 'rat_oblique': '0b7be7b0a9436d74ed1f2f2676256c9ad0072996f89fd95ce2b01a4a2cc64f86', 'rat_dodt': '431269c513bb558f7434018fc087c22e31c1e7cbdccc4da2d399d0bf3b85715a'}
ARCHIVE_NAMES = {'human_oblique': 'human_neurons_oblique', 'human_dodt': 'human_neurons_dodt', 'mouse_oblique': 'mice_neurons_oblique', 'mouse_dodt': 'mice_neurons_dodt', 'rat_oblique': 'rat_neurons_oblique', 'rat_dodt': 'rat_neurons_dodt'}
SPECIES = ['human', 'mouse', 'rat']
MODALITIES = ['oblique', 'dodt']
def get_neurons_3d_multispecies_data( path: Union[os.PathLike, str], species: Optional[List[Literal['human', 'mouse', 'rat']]] = None, modality: Optional[List[Literal['oblique', 'dodt']]] = None, download: bool = False) -> List[str]:
56def get_neurons_3d_multispecies_data(
57    path: Union[os.PathLike, str],
58    species: Optional[List[Literal["human", "mouse", "rat"]]] = None,
59    modality: Optional[List[Literal["oblique", "dodt"]]] = None,
60    download: bool = False,
61) -> List[str]:
62    """Download the multi-species 3D neuron segmentation dataset.
63
64    Args:
65        path: Filepath to a folder where the downloaded data will be saved.
66        species: The species subset(s) to download. Defaults to all species.
67        modality: The imaging modality subset(s) to download. Defaults to all modalities.
68        download: Whether to download the data if it is not present.
69
70    Returns:
71        List of filepaths to the extracted data directories.
72    """
73    species = SPECIES if species is None else species
74    modality = MODALITIES if modality is None else modality
75
76    os.makedirs(path, exist_ok=True)
77
78    data_dirs = []
79    for sp in species:
80        for mod in modality:
81            key = f"{sp}_{mod}"
82            data_dir = os.path.join(path, ARCHIVE_NAMES[key])
83            if not os.path.exists(data_dir):
84                zip_path = os.path.join(path, f"{ARCHIVE_NAMES[key]}.zip")
85                util.download_source(zip_path, URLS[key], download, checksum=CHECKSUMS[key])
86                util.unzip(zip_path, path)
87            data_dirs.append(data_dir)
88
89    return data_dirs

Download the multi-species 3D neuron segmentation dataset.

Arguments:
  • path: Filepath to a folder where the downloaded data will be saved.
  • species: The species subset(s) to download. Defaults to all species.
  • modality: The imaging modality subset(s) to download. Defaults to all modalities.
  • download: Whether to download the data if it is not present.
Returns:

List of filepaths to the extracted data directories.

def get_neurons_3d_multispecies_paths( path: Union[os.PathLike, str], species: Optional[List[Literal['human', 'mouse', 'rat']]] = None, modality: Optional[List[Literal['oblique', 'dodt']]] = None, split: Optional[Literal['train', 'val', 'test']] = None, download: bool = False) -> Tuple[List[str], List[str]]:
 92def get_neurons_3d_multispecies_paths(
 93    path: Union[os.PathLike, str],
 94    species: Optional[List[Literal["human", "mouse", "rat"]]] = None,
 95    modality: Optional[List[Literal["oblique", "dodt"]]] = None,
 96    split: Optional[Literal["train", "val", "test"]] = None,
 97    download: bool = False,
 98) -> Tuple[List[str], List[str]]:
 99    """Get paths to the multi-species 3D neuron segmentation data.
100
101    Args:
102        path: Filepath to a folder where the downloaded data will be saved.
103        species: The species subset(s) to use. Defaults to all species.
104        modality: The imaging modality subset(s) to use. Defaults to all modalities.
105        split: The data split to use. Either 'train', 'val' or 'test'. Defaults to using all splits.
106        download: Whether to download the data if it is not present.
107
108    Returns:
109        List of filepaths for the image data.
110        List of filepaths for the label data.
111    """
112    data_dirs = get_neurons_3d_multispecies_data(path, species, modality, download)
113
114    raw_paths, label_paths = [], []
115    for data_dir in data_dirs:
116        if split is None:
117            fnames = natsorted(os.path.basename(p) for p in glob(os.path.join(data_dir, "images", "*.tif")))
118        else:
119            split_file = os.path.join(data_dir, "split.json")
120            with open(split_file) as f:
121                fnames = natsorted(json.load(f)[split])
122
123        for fname in fnames:
124            raw_path = os.path.join(data_dir, "images", fname)
125            label_path = os.path.join(data_dir, "masks", fname)
126            if os.path.exists(raw_path) and os.path.exists(label_path):
127                raw_paths.append(raw_path)
128                label_paths.append(label_path)
129
130    if len(raw_paths) == 0:
131        raise RuntimeError(f"No image files found under {path}. Please check the dataset structure.")
132
133    return raw_paths, label_paths

Get paths to the multi-species 3D neuron segmentation data.

Arguments:
  • path: Filepath to a folder where the downloaded data will be saved.
  • species: The species subset(s) to use. Defaults to all species.
  • modality: The imaging modality subset(s) to use. Defaults to all modalities.
  • split: The data split to use. Either 'train', 'val' or 'test'. Defaults to using all splits.
  • download: Whether to download the data if it is not present.
Returns:

List of filepaths for the image data. List of filepaths for the label data.

def get_neurons_3d_multispecies_dataset( path: Union[os.PathLike, str], patch_shape: Tuple[int, int, int], species: Optional[List[Literal['human', 'mouse', 'rat']]] = None, modality: Optional[List[Literal['oblique', 'dodt']]] = None, split: Optional[Literal['train', 'val', 'test']] = None, download: bool = False, **kwargs) -> torch.utils.data.dataset.Dataset:
136def get_neurons_3d_multispecies_dataset(
137    path: Union[os.PathLike, str],
138    patch_shape: Tuple[int, int, int],
139    species: Optional[List[Literal["human", "mouse", "rat"]]] = None,
140    modality: Optional[List[Literal["oblique", "dodt"]]] = None,
141    split: Optional[Literal["train", "val", "test"]] = None,
142    download: bool = False,
143    **kwargs,
144) -> Dataset:
145    """Get the multi-species 3D neuron segmentation dataset.
146
147    Args:
148        path: Filepath to a folder where the downloaded data will be saved.
149        patch_shape: The patch shape to use for training.
150        species: The species subset(s) to use. Defaults to all species.
151        modality: The imaging modality subset(s) to use. Defaults to all modalities.
152        split: The data split to use. Either 'train', 'val' or 'test'. Defaults to using all splits.
153        download: Whether to download the data if it is not present.
154        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`.
155
156    Returns:
157        The segmentation dataset.
158    """
159    raw_paths, label_paths = get_neurons_3d_multispecies_paths(path, species, modality, split, download)
160
161    return torch_em.default_segmentation_dataset(
162        raw_paths=raw_paths,
163        raw_key=None,
164        label_paths=label_paths,
165        label_key=None,
166        patch_shape=patch_shape,
167        **kwargs,
168    )

Get the multi-species 3D neuron segmentation dataset.

Arguments:
  • path: Filepath to a folder where the downloaded data will be saved.
  • patch_shape: The patch shape to use for training.
  • species: The species subset(s) to use. Defaults to all species.
  • modality: The imaging modality subset(s) to use. Defaults to all modalities.
  • split: The data split to use. Either 'train', 'val' or 'test'. Defaults to using all splits.
  • download: Whether to download the data if it is not present.
  • kwargs: Additional keyword arguments for torch_em.default_segmentation_dataset.
Returns:

The segmentation dataset.

def get_neurons_3d_multispecies_loader( path: Union[os.PathLike, str], batch_size: int, patch_shape: Tuple[int, int, int], species: Optional[List[Literal['human', 'mouse', 'rat']]] = None, modality: Optional[List[Literal['oblique', 'dodt']]] = None, split: Optional[Literal['train', 'val', 'test']] = None, download: bool = False, **kwargs) -> torch.utils.data.dataloader.DataLoader:
171def get_neurons_3d_multispecies_loader(
172    path: Union[os.PathLike, str],
173    batch_size: int,
174    patch_shape: Tuple[int, int, int],
175    species: Optional[List[Literal["human", "mouse", "rat"]]] = None,
176    modality: Optional[List[Literal["oblique", "dodt"]]] = None,
177    split: Optional[Literal["train", "val", "test"]] = None,
178    download: bool = False,
179    **kwargs,
180) -> DataLoader:
181    """Get the multi-species 3D neuron segmentation dataloader.
182
183    Args:
184        path: Filepath to a folder where the downloaded data will be saved.
185        batch_size: The batch size for training.
186        patch_shape: The patch shape to use for training.
187        species: The species subset(s) to use. Defaults to all species.
188        modality: The imaging modality subset(s) to use. Defaults to all modalities.
189        split: The data split to use. Either 'train', 'val' or 'test'. Defaults to using all splits.
190        download: Whether to download the data if it is not present.
191        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader.
192
193    Returns:
194        The DataLoader.
195    """
196    ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs)
197    dataset = get_neurons_3d_multispecies_dataset(path, patch_shape, species, modality, split, download, **ds_kwargs)
198    return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)

Get the multi-species 3D neuron segmentation dataloader.

Arguments:
  • path: Filepath to a folder where the downloaded data will be saved.
  • batch_size: The batch size for training.
  • patch_shape: The patch shape to use for training.
  • species: The species subset(s) to use. Defaults to all species.
  • modality: The imaging modality subset(s) to use. Defaults to all modalities.
  • split: The data split to use. Either 'train', 'val' or 'test'. Defaults to using all splits.
  • download: Whether to download the data if it is not present.
  • kwargs: Additional keyword arguments for torch_em.default_segmentation_dataset or for the PyTorch DataLoader.
Returns:

The DataLoader.