torch_em.data.datasets.medical.bobs

The BOBs (Baby Open Brains) dataset contains manually curated and expert-reviewed brain segmentations for infant MRI.

The dataset consists of 71 longitudinal imaging visits from 51 children aged 1-9 months, part of the UNC/UMN Baby Connectome Project. Each visit has paired T1-weighted and T2-weighted MRI scans, together with a semantic segmentation of the cerebral tissues and 23 subcortical structures (FreeSurfer-style label ids). The segmentations were initially generated by the BIBSNet deep learning pipeline and subsequently manually corrected by trained specialists.

NOTE: This module uses the T2-weighted scans by default (modality="T2w"), the primary modality for the infant-abcd-bids-pipeline / BIBSNet segmentation protocol. The complete label lookup table (id, RGB color, name) is shipped with the data in 'dseg.tsv', which is downloaded next to the volumes.

The dataset is located at https://bobsrepository.readthedocs.io/ (hosted as a public AWS S3 bucket, and archived on OSF at https://doi.org/10.17605/OSF.IO/WDR78) and is distributed under the CC BY 4.0 license.

This dataset is from the publication https://doi.org/10.1038/s41597-025-05404-y. Please cite it if you use this dataset in your research.

  1"""The BOBs (Baby Open Brains) dataset contains manually curated and expert-reviewed brain
  2segmentations for infant MRI.
  3
  4The dataset consists of 71 longitudinal imaging visits from 51 children aged 1-9 months, part
  5of the UNC/UMN Baby Connectome Project. Each visit has paired T1-weighted and T2-weighted MRI
  6scans, together with a semantic segmentation of the cerebral tissues and 23 subcortical
  7structures (FreeSurfer-style label ids). The segmentations were initially generated by the
  8BIBSNet deep learning pipeline and subsequently manually corrected by trained specialists.
  9
 10NOTE: This module uses the T2-weighted scans by default (`modality="T2w"`), the primary
 11modality for the infant-abcd-bids-pipeline / BIBSNet segmentation protocol. The complete label
 12lookup table (id, RGB color, name) is shipped with the data in 'dseg.tsv', which is downloaded
 13next to the volumes.
 14
 15The dataset is located at https://bobsrepository.readthedocs.io/ (hosted as a public AWS S3
 16bucket, and archived on OSF at https://doi.org/10.17605/OSF.IO/WDR78) and is distributed under
 17the CC BY 4.0 license.
 18
 19This dataset is from the publication https://doi.org/10.1038/s41597-025-05404-y.
 20Please cite it if you use this dataset in your research.
 21"""
 22
 23import os
 24from glob import glob
 25from natsort import natsorted
 26from typing import Union, Tuple, Literal, List
 27
 28from torch.utils.data import Dataset, DataLoader
 29
 30import torch_em
 31
 32from .. import util
 33
 34
 35BASE_URL = "https://bobsrepository.s3.us-east-2.amazonaws.com"
 36ARCHIVE_URL = f"{BASE_URL}/V1.0.zip"
 37DSEG_URL = f"{BASE_URL}/dseg.tsv"
 38
 39
 40def get_bobs_data(path: Union[os.PathLike, str], download: bool = False) -> str:
 41    """Download the BOBs dataset.
 42
 43    Args:
 44        path: Filepath to a folder where the data is downloaded for further processing.
 45        download: Whether to download the data if it is not present.
 46
 47    Returns:
 48        Filepath where the data is downloaded.
 49    """
 50    data_dir = os.path.join(path, "data")
 51    if os.path.exists(data_dir):
 52        return data_dir
 53
 54    os.makedirs(path, exist_ok=True)
 55
 56    zip_path = os.path.join(path, "V1.0.zip")
 57    util.download_source(path=zip_path, url=ARCHIVE_URL, download=download, checksum=None)
 58    util.unzip(zip_path=zip_path, dst=data_dir)
 59
 60    dseg_path = os.path.join(path, "dseg.tsv")
 61    util.download_source(path=dseg_path, url=DSEG_URL, download=download, checksum=None)
 62
 63    return data_dir
 64
 65
 66def get_bobs_paths(
 67    path: Union[os.PathLike, str], modality: Literal["T1w", "T2w"] = "T2w", download: bool = False
 68) -> Tuple[List[str], List[str]]:
 69    """Get paths to the BOBs data.
 70
 71    Args:
 72        path: Filepath to a folder where the data is downloaded for further processing.
 73        modality: The MRI modality. Either 'T1w' or 'T2w'.
 74        download: Whether to download the data if it is not present.
 75
 76    Returns:
 77        List of filepaths for the image data.
 78        List of filepaths for the label data.
 79    """
 80    if modality not in ("T1w", "T2w"):
 81        raise ValueError(f"'{modality}' is not a valid modality. Choose either 'T1w' or 'T2w'.")
 82
 83    data_dir = get_bobs_data(path, download)
 84
 85    label_paths = natsorted(glob(os.path.join(data_dir, "sub-*", "ses-*", "anat", "*_desc-aseg_dseg.nii.gz")))
 86    raw_paths = [p.replace("_desc-aseg_dseg.nii.gz", f"_{modality}.nii.gz") for p in label_paths]
 87
 88    keep = [i for i, p in enumerate(raw_paths) if os.path.exists(p)]
 89    raw_paths = [raw_paths[i] for i in keep]
 90    label_paths = [label_paths[i] for i in keep]
 91
 92    assert len(raw_paths) == len(label_paths) and len(raw_paths) > 0
 93
 94    return raw_paths, label_paths
 95
 96
 97def get_bobs_dataset(
 98    path: Union[os.PathLike, str],
 99    patch_shape: Tuple[int, ...],
100    modality: Literal["T1w", "T2w"] = "T2w",
101    resize_inputs: bool = False,
102    download: bool = False,
103    **kwargs
104) -> Dataset:
105    """Get the BOBs dataset for infant brain tissue and subcortical structure segmentation.
106
107    Args:
108        path: Filepath to a folder where the data is downloaded for further processing.
109        patch_shape: The patch shape to use for training.
110        modality: The MRI modality. Either 'T1w' or 'T2w'.
111        resize_inputs: Whether to resize inputs to the desired patch shape.
112        download: Whether to download the data if it is not present.
113        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`.
114
115    Returns:
116        The segmentation dataset.
117    """
118    raw_paths, label_paths = get_bobs_paths(path, modality, download)
119
120    if resize_inputs:
121        resize_kwargs = {"patch_shape": patch_shape, "is_rgb": False}
122        kwargs, patch_shape = util.update_kwargs_for_resize_trafo(
123            kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs
124        )
125
126    return torch_em.default_segmentation_dataset(
127        raw_paths=raw_paths,
128        raw_key="data",
129        label_paths=label_paths,
130        label_key="data",
131        patch_shape=patch_shape,
132        is_seg_dataset=True,
133        **kwargs
134    )
135
136
137def get_bobs_loader(
138    path: Union[os.PathLike, str],
139    batch_size: int,
140    patch_shape: Tuple[int, ...],
141    modality: Literal["T1w", "T2w"] = "T2w",
142    resize_inputs: bool = False,
143    download: bool = False,
144    **kwargs
145) -> DataLoader:
146    """Get the BOBs dataloader for infant brain tissue and subcortical structure segmentation.
147
148    Args:
149        path: Filepath to a folder where the data is downloaded for further processing.
150        batch_size: The batch size for training.
151        patch_shape: The patch shape to use for training.
152        modality: The MRI modality. Either 'T1w' or 'T2w'.
153        resize_inputs: Whether to resize inputs to the desired patch shape.
154        download: Whether to download the data if it is not present.
155        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader.
156
157    Returns:
158        The DataLoader.
159    """
160    ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs)
161    dataset = get_bobs_dataset(path, patch_shape, modality, resize_inputs, download, **ds_kwargs)
162    return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)
BASE_URL = 'https://bobsrepository.s3.us-east-2.amazonaws.com'
ARCHIVE_URL = 'https://bobsrepository.s3.us-east-2.amazonaws.com/V1.0.zip'
DSEG_URL = 'https://bobsrepository.s3.us-east-2.amazonaws.com/dseg.tsv'
def get_bobs_data(path: Union[os.PathLike, str], download: bool = False) -> str:
41def get_bobs_data(path: Union[os.PathLike, str], download: bool = False) -> str:
42    """Download the BOBs dataset.
43
44    Args:
45        path: Filepath to a folder where the data is downloaded for further processing.
46        download: Whether to download the data if it is not present.
47
48    Returns:
49        Filepath where the data is downloaded.
50    """
51    data_dir = os.path.join(path, "data")
52    if os.path.exists(data_dir):
53        return data_dir
54
55    os.makedirs(path, exist_ok=True)
56
57    zip_path = os.path.join(path, "V1.0.zip")
58    util.download_source(path=zip_path, url=ARCHIVE_URL, download=download, checksum=None)
59    util.unzip(zip_path=zip_path, dst=data_dir)
60
61    dseg_path = os.path.join(path, "dseg.tsv")
62    util.download_source(path=dseg_path, url=DSEG_URL, download=download, checksum=None)
63
64    return data_dir

Download the BOBs dataset.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • download: Whether to download the data if it is not present.
Returns:

Filepath where the data is downloaded.

def get_bobs_paths( path: Union[os.PathLike, str], modality: Literal['T1w', 'T2w'] = 'T2w', download: bool = False) -> Tuple[List[str], List[str]]:
67def get_bobs_paths(
68    path: Union[os.PathLike, str], modality: Literal["T1w", "T2w"] = "T2w", download: bool = False
69) -> Tuple[List[str], List[str]]:
70    """Get paths to the BOBs data.
71
72    Args:
73        path: Filepath to a folder where the data is downloaded for further processing.
74        modality: The MRI modality. Either 'T1w' or 'T2w'.
75        download: Whether to download the data if it is not present.
76
77    Returns:
78        List of filepaths for the image data.
79        List of filepaths for the label data.
80    """
81    if modality not in ("T1w", "T2w"):
82        raise ValueError(f"'{modality}' is not a valid modality. Choose either 'T1w' or 'T2w'.")
83
84    data_dir = get_bobs_data(path, download)
85
86    label_paths = natsorted(glob(os.path.join(data_dir, "sub-*", "ses-*", "anat", "*_desc-aseg_dseg.nii.gz")))
87    raw_paths = [p.replace("_desc-aseg_dseg.nii.gz", f"_{modality}.nii.gz") for p in label_paths]
88
89    keep = [i for i, p in enumerate(raw_paths) if os.path.exists(p)]
90    raw_paths = [raw_paths[i] for i in keep]
91    label_paths = [label_paths[i] for i in keep]
92
93    assert len(raw_paths) == len(label_paths) and len(raw_paths) > 0
94
95    return raw_paths, label_paths

Get paths to the BOBs data.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • modality: The MRI modality. Either 'T1w' or 'T2w'.
  • download: Whether to download the data if it is not present.
Returns:

List of filepaths for the image data. List of filepaths for the label data.

def get_bobs_dataset( path: Union[os.PathLike, str], patch_shape: Tuple[int, ...], modality: Literal['T1w', 'T2w'] = 'T2w', resize_inputs: bool = False, download: bool = False, **kwargs) -> torch.utils.data.dataset.Dataset:
 98def get_bobs_dataset(
 99    path: Union[os.PathLike, str],
100    patch_shape: Tuple[int, ...],
101    modality: Literal["T1w", "T2w"] = "T2w",
102    resize_inputs: bool = False,
103    download: bool = False,
104    **kwargs
105) -> Dataset:
106    """Get the BOBs dataset for infant brain tissue and subcortical structure segmentation.
107
108    Args:
109        path: Filepath to a folder where the data is downloaded for further processing.
110        patch_shape: The patch shape to use for training.
111        modality: The MRI modality. Either 'T1w' or 'T2w'.
112        resize_inputs: Whether to resize inputs to the desired patch shape.
113        download: Whether to download the data if it is not present.
114        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`.
115
116    Returns:
117        The segmentation dataset.
118    """
119    raw_paths, label_paths = get_bobs_paths(path, modality, download)
120
121    if resize_inputs:
122        resize_kwargs = {"patch_shape": patch_shape, "is_rgb": False}
123        kwargs, patch_shape = util.update_kwargs_for_resize_trafo(
124            kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs
125        )
126
127    return torch_em.default_segmentation_dataset(
128        raw_paths=raw_paths,
129        raw_key="data",
130        label_paths=label_paths,
131        label_key="data",
132        patch_shape=patch_shape,
133        is_seg_dataset=True,
134        **kwargs
135    )

Get the BOBs dataset for infant brain tissue and subcortical structure segmentation.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • patch_shape: The patch shape to use for training.
  • modality: The MRI modality. Either 'T1w' or 'T2w'.
  • resize_inputs: Whether to resize inputs to the desired patch shape.
  • download: Whether to download the data if it is not present.
  • kwargs: Additional keyword arguments for torch_em.default_segmentation_dataset.
Returns:

The segmentation dataset.

def get_bobs_loader( path: Union[os.PathLike, str], batch_size: int, patch_shape: Tuple[int, ...], modality: Literal['T1w', 'T2w'] = 'T2w', resize_inputs: bool = False, download: bool = False, **kwargs) -> torch.utils.data.dataloader.DataLoader:
138def get_bobs_loader(
139    path: Union[os.PathLike, str],
140    batch_size: int,
141    patch_shape: Tuple[int, ...],
142    modality: Literal["T1w", "T2w"] = "T2w",
143    resize_inputs: bool = False,
144    download: bool = False,
145    **kwargs
146) -> DataLoader:
147    """Get the BOBs dataloader for infant brain tissue and subcortical structure segmentation.
148
149    Args:
150        path: Filepath to a folder where the data is downloaded for further processing.
151        batch_size: The batch size for training.
152        patch_shape: The patch shape to use for training.
153        modality: The MRI modality. Either 'T1w' or 'T2w'.
154        resize_inputs: Whether to resize inputs to the desired patch shape.
155        download: Whether to download the data if it is not present.
156        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader.
157
158    Returns:
159        The DataLoader.
160    """
161    ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs)
162    dataset = get_bobs_dataset(path, patch_shape, modality, resize_inputs, download, **ds_kwargs)
163    return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)

Get the BOBs dataloader for infant brain tissue and subcortical structure segmentation.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • batch_size: The batch size for training.
  • patch_shape: The patch shape to use for training.
  • modality: The MRI modality. Either 'T1w' or 'T2w'.
  • resize_inputs: Whether to resize inputs to the desired patch shape.
  • download: Whether to download the data if it is not present.
  • kwargs: Additional keyword arguments for torch_em.default_segmentation_dataset or for the PyTorch DataLoader.
Returns:

The DataLoader.