torch_em.data.datasets.medical.bobs
The BOBs (Baby Open Brains) dataset contains manually curated and expert-reviewed brain segmentations for infant MRI.
The dataset consists of 71 longitudinal imaging visits from 51 children aged 1-9 months, part of the UNC/UMN Baby Connectome Project. Each visit has paired T1-weighted and T2-weighted MRI scans, together with a semantic segmentation of the cerebral tissues and 23 subcortical structures (FreeSurfer-style label ids). The segmentations were initially generated by the BIBSNet deep learning pipeline and subsequently manually corrected by trained specialists.
NOTE: This module uses the T2-weighted scans by default (modality="T2w"), the primary
modality for the infant-abcd-bids-pipeline / BIBSNet segmentation protocol. The complete label
lookup table (id, RGB color, name) is shipped with the data in 'dseg.tsv', which is downloaded
next to the volumes.
The dataset is located at https://bobsrepository.readthedocs.io/ (hosted as a public AWS S3 bucket, and archived on OSF at https://doi.org/10.17605/OSF.IO/WDR78) and is distributed under the CC BY 4.0 license.
This dataset is from the publication https://doi.org/10.1038/s41597-025-05404-y. Please cite it if you use this dataset in your research.
1"""The BOBs (Baby Open Brains) dataset contains manually curated and expert-reviewed brain 2segmentations for infant MRI. 3 4The dataset consists of 71 longitudinal imaging visits from 51 children aged 1-9 months, part 5of the UNC/UMN Baby Connectome Project. Each visit has paired T1-weighted and T2-weighted MRI 6scans, together with a semantic segmentation of the cerebral tissues and 23 subcortical 7structures (FreeSurfer-style label ids). The segmentations were initially generated by the 8BIBSNet deep learning pipeline and subsequently manually corrected by trained specialists. 9 10NOTE: This module uses the T2-weighted scans by default (`modality="T2w"`), the primary 11modality for the infant-abcd-bids-pipeline / BIBSNet segmentation protocol. The complete label 12lookup table (id, RGB color, name) is shipped with the data in 'dseg.tsv', which is downloaded 13next to the volumes. 14 15The dataset is located at https://bobsrepository.readthedocs.io/ (hosted as a public AWS S3 16bucket, and archived on OSF at https://doi.org/10.17605/OSF.IO/WDR78) and is distributed under 17the CC BY 4.0 license. 18 19This dataset is from the publication https://doi.org/10.1038/s41597-025-05404-y. 20Please cite it if you use this dataset in your research. 21""" 22 23import os 24from glob import glob 25from natsort import natsorted 26from typing import Union, Tuple, Literal, List 27 28from torch.utils.data import Dataset, DataLoader 29 30import torch_em 31 32from .. import util 33 34 35BASE_URL = "https://bobsrepository.s3.us-east-2.amazonaws.com" 36ARCHIVE_URL = f"{BASE_URL}/V1.0.zip" 37DSEG_URL = f"{BASE_URL}/dseg.tsv" 38 39 40def get_bobs_data(path: Union[os.PathLike, str], download: bool = False) -> str: 41 """Download the BOBs dataset. 42 43 Args: 44 path: Filepath to a folder where the data is downloaded for further processing. 45 download: Whether to download the data if it is not present. 46 47 Returns: 48 Filepath where the data is downloaded. 49 """ 50 data_dir = os.path.join(path, "data") 51 if os.path.exists(data_dir): 52 return data_dir 53 54 os.makedirs(path, exist_ok=True) 55 56 zip_path = os.path.join(path, "V1.0.zip") 57 util.download_source(path=zip_path, url=ARCHIVE_URL, download=download, checksum=None) 58 util.unzip(zip_path=zip_path, dst=data_dir) 59 60 dseg_path = os.path.join(path, "dseg.tsv") 61 util.download_source(path=dseg_path, url=DSEG_URL, download=download, checksum=None) 62 63 return data_dir 64 65 66def get_bobs_paths( 67 path: Union[os.PathLike, str], modality: Literal["T1w", "T2w"] = "T2w", download: bool = False 68) -> Tuple[List[str], List[str]]: 69 """Get paths to the BOBs data. 70 71 Args: 72 path: Filepath to a folder where the data is downloaded for further processing. 73 modality: The MRI modality. Either 'T1w' or 'T2w'. 74 download: Whether to download the data if it is not present. 75 76 Returns: 77 List of filepaths for the image data. 78 List of filepaths for the label data. 79 """ 80 if modality not in ("T1w", "T2w"): 81 raise ValueError(f"'{modality}' is not a valid modality. Choose either 'T1w' or 'T2w'.") 82 83 data_dir = get_bobs_data(path, download) 84 85 label_paths = natsorted(glob(os.path.join(data_dir, "sub-*", "ses-*", "anat", "*_desc-aseg_dseg.nii.gz"))) 86 raw_paths = [p.replace("_desc-aseg_dseg.nii.gz", f"_{modality}.nii.gz") for p in label_paths] 87 88 keep = [i for i, p in enumerate(raw_paths) if os.path.exists(p)] 89 raw_paths = [raw_paths[i] for i in keep] 90 label_paths = [label_paths[i] for i in keep] 91 92 assert len(raw_paths) == len(label_paths) and len(raw_paths) > 0 93 94 return raw_paths, label_paths 95 96 97def get_bobs_dataset( 98 path: Union[os.PathLike, str], 99 patch_shape: Tuple[int, ...], 100 modality: Literal["T1w", "T2w"] = "T2w", 101 resize_inputs: bool = False, 102 download: bool = False, 103 **kwargs 104) -> Dataset: 105 """Get the BOBs dataset for infant brain tissue and subcortical structure segmentation. 106 107 Args: 108 path: Filepath to a folder where the data is downloaded for further processing. 109 patch_shape: The patch shape to use for training. 110 modality: The MRI modality. Either 'T1w' or 'T2w'. 111 resize_inputs: Whether to resize inputs to the desired patch shape. 112 download: Whether to download the data if it is not present. 113 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`. 114 115 Returns: 116 The segmentation dataset. 117 """ 118 raw_paths, label_paths = get_bobs_paths(path, modality, download) 119 120 if resize_inputs: 121 resize_kwargs = {"patch_shape": patch_shape, "is_rgb": False} 122 kwargs, patch_shape = util.update_kwargs_for_resize_trafo( 123 kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs 124 ) 125 126 return torch_em.default_segmentation_dataset( 127 raw_paths=raw_paths, 128 raw_key="data", 129 label_paths=label_paths, 130 label_key="data", 131 patch_shape=patch_shape, 132 is_seg_dataset=True, 133 **kwargs 134 ) 135 136 137def get_bobs_loader( 138 path: Union[os.PathLike, str], 139 batch_size: int, 140 patch_shape: Tuple[int, ...], 141 modality: Literal["T1w", "T2w"] = "T2w", 142 resize_inputs: bool = False, 143 download: bool = False, 144 **kwargs 145) -> DataLoader: 146 """Get the BOBs dataloader for infant brain tissue and subcortical structure segmentation. 147 148 Args: 149 path: Filepath to a folder where the data is downloaded for further processing. 150 batch_size: The batch size for training. 151 patch_shape: The patch shape to use for training. 152 modality: The MRI modality. Either 'T1w' or 'T2w'. 153 resize_inputs: Whether to resize inputs to the desired patch shape. 154 download: Whether to download the data if it is not present. 155 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader. 156 157 Returns: 158 The DataLoader. 159 """ 160 ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs) 161 dataset = get_bobs_dataset(path, patch_shape, modality, resize_inputs, download, **ds_kwargs) 162 return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)
41def get_bobs_data(path: Union[os.PathLike, str], download: bool = False) -> str: 42 """Download the BOBs dataset. 43 44 Args: 45 path: Filepath to a folder where the data is downloaded for further processing. 46 download: Whether to download the data if it is not present. 47 48 Returns: 49 Filepath where the data is downloaded. 50 """ 51 data_dir = os.path.join(path, "data") 52 if os.path.exists(data_dir): 53 return data_dir 54 55 os.makedirs(path, exist_ok=True) 56 57 zip_path = os.path.join(path, "V1.0.zip") 58 util.download_source(path=zip_path, url=ARCHIVE_URL, download=download, checksum=None) 59 util.unzip(zip_path=zip_path, dst=data_dir) 60 61 dseg_path = os.path.join(path, "dseg.tsv") 62 util.download_source(path=dseg_path, url=DSEG_URL, download=download, checksum=None) 63 64 return data_dir
Download the BOBs dataset.
Arguments:
- path: Filepath to a folder where the data is downloaded for further processing.
- download: Whether to download the data if it is not present.
Returns:
Filepath where the data is downloaded.
67def get_bobs_paths( 68 path: Union[os.PathLike, str], modality: Literal["T1w", "T2w"] = "T2w", download: bool = False 69) -> Tuple[List[str], List[str]]: 70 """Get paths to the BOBs data. 71 72 Args: 73 path: Filepath to a folder where the data is downloaded for further processing. 74 modality: The MRI modality. Either 'T1w' or 'T2w'. 75 download: Whether to download the data if it is not present. 76 77 Returns: 78 List of filepaths for the image data. 79 List of filepaths for the label data. 80 """ 81 if modality not in ("T1w", "T2w"): 82 raise ValueError(f"'{modality}' is not a valid modality. Choose either 'T1w' or 'T2w'.") 83 84 data_dir = get_bobs_data(path, download) 85 86 label_paths = natsorted(glob(os.path.join(data_dir, "sub-*", "ses-*", "anat", "*_desc-aseg_dseg.nii.gz"))) 87 raw_paths = [p.replace("_desc-aseg_dseg.nii.gz", f"_{modality}.nii.gz") for p in label_paths] 88 89 keep = [i for i, p in enumerate(raw_paths) if os.path.exists(p)] 90 raw_paths = [raw_paths[i] for i in keep] 91 label_paths = [label_paths[i] for i in keep] 92 93 assert len(raw_paths) == len(label_paths) and len(raw_paths) > 0 94 95 return raw_paths, label_paths
Get paths to the BOBs data.
Arguments:
- path: Filepath to a folder where the data is downloaded for further processing.
- modality: The MRI modality. Either 'T1w' or 'T2w'.
- download: Whether to download the data if it is not present.
Returns:
List of filepaths for the image data. List of filepaths for the label data.
98def get_bobs_dataset( 99 path: Union[os.PathLike, str], 100 patch_shape: Tuple[int, ...], 101 modality: Literal["T1w", "T2w"] = "T2w", 102 resize_inputs: bool = False, 103 download: bool = False, 104 **kwargs 105) -> Dataset: 106 """Get the BOBs dataset for infant brain tissue and subcortical structure segmentation. 107 108 Args: 109 path: Filepath to a folder where the data is downloaded for further processing. 110 patch_shape: The patch shape to use for training. 111 modality: The MRI modality. Either 'T1w' or 'T2w'. 112 resize_inputs: Whether to resize inputs to the desired patch shape. 113 download: Whether to download the data if it is not present. 114 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`. 115 116 Returns: 117 The segmentation dataset. 118 """ 119 raw_paths, label_paths = get_bobs_paths(path, modality, download) 120 121 if resize_inputs: 122 resize_kwargs = {"patch_shape": patch_shape, "is_rgb": False} 123 kwargs, patch_shape = util.update_kwargs_for_resize_trafo( 124 kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs 125 ) 126 127 return torch_em.default_segmentation_dataset( 128 raw_paths=raw_paths, 129 raw_key="data", 130 label_paths=label_paths, 131 label_key="data", 132 patch_shape=patch_shape, 133 is_seg_dataset=True, 134 **kwargs 135 )
Get the BOBs dataset for infant brain tissue and subcortical structure segmentation.
Arguments:
- path: Filepath to a folder where the data is downloaded for further processing.
- patch_shape: The patch shape to use for training.
- modality: The MRI modality. Either 'T1w' or 'T2w'.
- resize_inputs: Whether to resize inputs to the desired patch shape.
- download: Whether to download the data if it is not present.
- kwargs: Additional keyword arguments for
torch_em.default_segmentation_dataset.
Returns:
The segmentation dataset.
138def get_bobs_loader( 139 path: Union[os.PathLike, str], 140 batch_size: int, 141 patch_shape: Tuple[int, ...], 142 modality: Literal["T1w", "T2w"] = "T2w", 143 resize_inputs: bool = False, 144 download: bool = False, 145 **kwargs 146) -> DataLoader: 147 """Get the BOBs dataloader for infant brain tissue and subcortical structure segmentation. 148 149 Args: 150 path: Filepath to a folder where the data is downloaded for further processing. 151 batch_size: The batch size for training. 152 patch_shape: The patch shape to use for training. 153 modality: The MRI modality. Either 'T1w' or 'T2w'. 154 resize_inputs: Whether to resize inputs to the desired patch shape. 155 download: Whether to download the data if it is not present. 156 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader. 157 158 Returns: 159 The DataLoader. 160 """ 161 ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs) 162 dataset = get_bobs_dataset(path, patch_shape, modality, resize_inputs, download, **ds_kwargs) 163 return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)
Get the BOBs dataloader for infant brain tissue and subcortical structure segmentation.
Arguments:
- path: Filepath to a folder where the data is downloaded for further processing.
- batch_size: The batch size for training.
- patch_shape: The patch shape to use for training.
- modality: The MRI modality. Either 'T1w' or 'T2w'.
- resize_inputs: Whether to resize inputs to the desired patch shape.
- download: Whether to download the data if it is not present.
- kwargs: Additional keyword arguments for
torch_em.default_segmentation_datasetor for the PyTorch DataLoader.
Returns:
The DataLoader.