torch_em.data.datasets.medical.bonbid_hie

The BONBID-HIE dataset contains annotations for lesion segmentation in neonatal brain diffusion MRI of patients with hypoxic-ischemic encephalopathy (HIE).

The dataset is released as part of the BONBID-HIE 2023 MICCAI challenge (https://bonbid-hie2023.grand-challenge.org) and is hosted at https://doi.org/10.5281/zenodo.10602767. It provides 85 training, 4 validation and 44 test cases (see SPLITS). Each case has a skull-stripped apparent diffusion coefficient (ADC) map and the derived Z-ADC map (the ADC map normalized against a healthy reference population, smoothed and clipped to the range [-6, 10]), both registered to a binary lesion segmentation mask.

NOTE: The scans are distributed as MetaImage (.mha) files. This module converts them to compressed nifti volumes on first use, which requires the SimpleITK python package.

NOTE: The Zenodo record's structured license field lists CC-BY-NC-ND 2.5, even though the record's description text states that the data is released under the CC BY 4.0 license. Please check the record for the authoritative license terms before further use.

This dataset is from the publication https://doi.org/10.1038/s41597-024-03986-7. Please cite it if you use this dataset in your research.

  1"""The BONBID-HIE dataset contains annotations for lesion segmentation in neonatal brain
  2diffusion MRI of patients with hypoxic-ischemic encephalopathy (HIE).
  3
  4The dataset is released as part of the BONBID-HIE 2023 MICCAI challenge
  5(https://bonbid-hie2023.grand-challenge.org) and is hosted at
  6https://doi.org/10.5281/zenodo.10602767. It provides 85 training, 4 validation and 44
  7test cases (see `SPLITS`). Each case has a skull-stripped apparent diffusion coefficient
  8(ADC) map and the derived Z-ADC map (the ADC map normalized against a healthy reference
  9population, smoothed and clipped to the range [-6, 10]), both registered to a binary
 10lesion segmentation mask.
 11
 12NOTE: The scans are distributed as MetaImage (.mha) files. This module converts them to
 13compressed nifti volumes on first use, which requires the SimpleITK python package.
 14
 15NOTE: The Zenodo record's structured license field lists CC-BY-NC-ND 2.5, even though the
 16record's description text states that the data is released under the CC BY 4.0 license.
 17Please check the record for the authoritative license terms before further use.
 18
 19This dataset is from the publication https://doi.org/10.1038/s41597-024-03986-7.
 20Please cite it if you use this dataset in your research.
 21"""
 22
 23import os
 24from glob import glob
 25from tqdm import tqdm
 26from natsort import natsorted
 27from typing import Union, Tuple, Optional, Literal, List
 28
 29from torch.utils.data import Dataset, DataLoader
 30
 31import torch_em
 32
 33from .. import util
 34
 35
 36URLS = {
 37    "train": "https://zenodo.org/records/10602767/files/BONBID2023_Train.zip",
 38    "val": "https://zenodo.org/records/10602767/files/BONBID2023_Val.zip",
 39    "test": "https://zenodo.org/records/10602767/files/BONBID2023_Test.zip",
 40}
 41
 42CHECKSUMS = {
 43    "train": "f1058093887daea1ebc07368b0642d3c40a3f70da5e2e225558ec32de3fa345b",
 44    "val": "8eaaa05cad00d5d5112583ca01be9ec685e15ca87b946b19a4010b3da35f4786",
 45    "test": "ac3988c57f035ee74a20dd70194e6fd51fe9aecc3a68e093f487620b24fcbfa3",
 46}
 47
 48SPLITS = ("train", "val", "test")
 49
 50
 51def _convert_mha_to_nifti(mha_path, nifti_path):
 52    if os.path.exists(nifti_path):
 53        return
 54
 55    import SimpleITK as sitk
 56
 57    volume = sitk.ReadImage(mha_path)
 58    sitk.WriteImage(volume, nifti_path, useCompression=True)
 59
 60
 61def _preprocess_bonbid_hie(raw_dir, preprocessed_dir, split):
 62    adc_paths = natsorted(glob(os.path.join(raw_dir, "**", "1ADC_ss", "*-ADC_ss.mha"), recursive=True))
 63    if len(adc_paths) == 0:
 64        raise RuntimeError(f"Did not find any '{split}' scans at '{raw_dir}'.")
 65
 66    for folder in ["adc", "zadc", "labels"]:
 67        os.makedirs(os.path.join(preprocessed_dir, folder), exist_ok=True)
 68
 69    for adc_path in tqdm(adc_paths, desc=f"Preprocess BONBID-HIE '{split}' scans"):
 70        case_dir = os.path.dirname(os.path.dirname(adc_path))
 71        case_id = os.path.basename(adc_path)[:-len("-ADC_ss.mha")]
 72
 73        zadc_path = os.path.join(case_dir, "2Z_ADC", f"Zmap_{case_id}-ADC_smooth2mm_clipped10.mha")
 74        label_path = os.path.join(case_dir, "3LABEL", f"{case_id}_lesion.mha")
 75        if not os.path.exists(zadc_path) or not os.path.exists(label_path):
 76            raise RuntimeError(f"Could not find the Z-ADC map or the label for the case '{case_id}'.")
 77
 78        _convert_mha_to_nifti(adc_path, os.path.join(preprocessed_dir, "adc", f"{case_id}.nii.gz"))
 79        _convert_mha_to_nifti(zadc_path, os.path.join(preprocessed_dir, "zadc", f"{case_id}.nii.gz"))
 80        _convert_mha_to_nifti(label_path, os.path.join(preprocessed_dir, "labels", f"{case_id}.nii.gz"))
 81
 82
 83def get_bonbid_hie_data(path: Union[os.PathLike, str], split: Literal["train", "val", "test"], download: bool = False) -> str:  # noqa
 84    """Download the BONBID-HIE dataset and preprocess it to nifti volumes.
 85
 86    Args:
 87        path: Filepath to a folder where the data is downloaded for further processing.
 88        split: The choice of data split. Either 'train', 'val' or 'test'.
 89        download: Whether to download the data if it is not present.
 90
 91    Returns:
 92        Filepath to the folder where the preprocessed data is stored.
 93    """
 94    if split not in SPLITS:
 95        raise ValueError(f"'{split}' is not a valid split. Choose from {SPLITS}.")
 96
 97    preprocessed_dir = os.path.join(path, "preprocessed", split)
 98    if os.path.exists(preprocessed_dir):
 99        return preprocessed_dir
100
101    os.makedirs(path, exist_ok=True)
102
103    zip_path = os.path.join(path, f"BONBID2023_{split.capitalize()}.zip")
104    util.download_source(path=zip_path, url=URLS[split], download=download, checksum=CHECKSUMS[split])
105
106    raw_dir = os.path.join(path, "raw", split)
107    util.unzip(zip_path=zip_path, dst=raw_dir)
108
109    _preprocess_bonbid_hie(raw_dir, preprocessed_dir, split)
110
111    return preprocessed_dir
112
113
114def get_bonbid_hie_paths(
115    path: Union[os.PathLike, str],
116    split: Literal["train", "val", "test"],
117    modality: Optional[Literal["adc", "zadc"]] = None,
118    download: bool = False,
119) -> Tuple[List, List[str]]:
120    """Get paths to the BONBID-HIE data.
121
122    Args:
123        path: Filepath to a folder where the data is downloaded for further processing.
124        split: The choice of data split. Either 'train', 'val' or 'test'.
125        modality: The choice of modality for MRIs. Either 'adc' or 'zadc'.
126        download: Whether to download the data if it is not present.
127
128    Returns:
129        List of filepaths for the image data.
130        List of filepaths for the label data.
131    """
132    preprocessed_dir = get_bonbid_hie_data(path, split, download)
133
134    label_paths = natsorted(glob(os.path.join(preprocessed_dir, "labels", "*.nii.gz")))
135    adc_paths = natsorted(glob(os.path.join(preprocessed_dir, "adc", "*.nii.gz")))
136    zadc_paths = natsorted(glob(os.path.join(preprocessed_dir, "zadc", "*.nii.gz")))
137
138    if modality is None:
139        image_paths = [(adc_path, zadc_path) for adc_path, zadc_path in zip(adc_paths, zadc_paths)]
140    elif modality == "adc":
141        image_paths = adc_paths
142    elif modality == "zadc":
143        image_paths = zadc_paths
144    else:
145        raise ValueError(f"'{modality}' is not a valid modality.")
146
147    return image_paths, label_paths
148
149
150def get_bonbid_hie_dataset(
151    path: Union[os.PathLike, str],
152    patch_shape: Tuple[int, ...],
153    split: Literal["train", "val", "test"] = "train",
154    modality: Optional[Literal["adc", "zadc"]] = None,
155    download: bool = False,
156    **kwargs
157) -> Dataset:
158    """Get the BONBID-HIE dataset for segmentation of HIE-related brain lesions.
159
160    Args:
161        path: Filepath to a folder where the data is downloaded for further processing.
162        patch_shape: The patch shape to use for training.
163        split: The choice of data split. Either 'train', 'val' or 'test'.
164        modality: The choice of modality for MRIs. Either 'adc' or 'zadc'.
165        download: Whether to download the data if it is not present.
166        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`.
167
168    Returns:
169        The segmentation dataset.
170    """
171    image_paths, label_paths = get_bonbid_hie_paths(path, split, modality, download)
172
173    return torch_em.default_segmentation_dataset(
174        raw_paths=image_paths,
175        raw_key="data",
176        label_paths=label_paths,
177        label_key="data",
178        patch_shape=patch_shape,
179        with_channels=modality is None,
180        is_seg_dataset=True,
181        **kwargs
182    )
183
184
185def get_bonbid_hie_loader(
186    path: Union[os.PathLike, str],
187    batch_size: int,
188    patch_shape: Tuple[int, ...],
189    split: Literal["train", "val", "test"] = "train",
190    modality: Optional[Literal["adc", "zadc"]] = None,
191    download: bool = False,
192    **kwargs
193) -> DataLoader:
194    """Get the BONBID-HIE dataloader for segmentation of HIE-related brain lesions.
195
196    Args:
197        path: Filepath to a folder where the data is downloaded for further processing.
198        batch_size: The batch size for training.
199        patch_shape: The patch shape to use for training.
200        split: The choice of data split. Either 'train', 'val' or 'test'.
201        modality: The choice of modality for MRIs. Either 'adc' or 'zadc'.
202        download: Whether to download the data if it is not present.
203        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader.
204
205    Returns:
206        The DataLoader.
207    """
208    ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs)
209    dataset = get_bonbid_hie_dataset(path, patch_shape, split, modality, download, **ds_kwargs)
210    return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)
URLS = {'train': 'https://zenodo.org/records/10602767/files/BONBID2023_Train.zip', 'val': 'https://zenodo.org/records/10602767/files/BONBID2023_Val.zip', 'test': 'https://zenodo.org/records/10602767/files/BONBID2023_Test.zip'}
CHECKSUMS = {'train': 'f1058093887daea1ebc07368b0642d3c40a3f70da5e2e225558ec32de3fa345b', 'val': '8eaaa05cad00d5d5112583ca01be9ec685e15ca87b946b19a4010b3da35f4786', 'test': 'ac3988c57f035ee74a20dd70194e6fd51fe9aecc3a68e093f487620b24fcbfa3'}
SPLITS = ('train', 'val', 'test')
def get_bonbid_hie_data( path: Union[os.PathLike, str], split: Literal['train', 'val', 'test'], download: bool = False) -> str:
 84def get_bonbid_hie_data(path: Union[os.PathLike, str], split: Literal["train", "val", "test"], download: bool = False) -> str:  # noqa
 85    """Download the BONBID-HIE dataset and preprocess it to nifti volumes.
 86
 87    Args:
 88        path: Filepath to a folder where the data is downloaded for further processing.
 89        split: The choice of data split. Either 'train', 'val' or 'test'.
 90        download: Whether to download the data if it is not present.
 91
 92    Returns:
 93        Filepath to the folder where the preprocessed data is stored.
 94    """
 95    if split not in SPLITS:
 96        raise ValueError(f"'{split}' is not a valid split. Choose from {SPLITS}.")
 97
 98    preprocessed_dir = os.path.join(path, "preprocessed", split)
 99    if os.path.exists(preprocessed_dir):
100        return preprocessed_dir
101
102    os.makedirs(path, exist_ok=True)
103
104    zip_path = os.path.join(path, f"BONBID2023_{split.capitalize()}.zip")
105    util.download_source(path=zip_path, url=URLS[split], download=download, checksum=CHECKSUMS[split])
106
107    raw_dir = os.path.join(path, "raw", split)
108    util.unzip(zip_path=zip_path, dst=raw_dir)
109
110    _preprocess_bonbid_hie(raw_dir, preprocessed_dir, split)
111
112    return preprocessed_dir

Download the BONBID-HIE dataset and preprocess it to nifti volumes.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • split: The choice of data split. Either 'train', 'val' or 'test'.
  • download: Whether to download the data if it is not present.
Returns:

Filepath to the folder where the preprocessed data is stored.

def get_bonbid_hie_paths( path: Union[os.PathLike, str], split: Literal['train', 'val', 'test'], modality: Optional[Literal['adc', 'zadc']] = None, download: bool = False) -> Tuple[List, List[str]]:
115def get_bonbid_hie_paths(
116    path: Union[os.PathLike, str],
117    split: Literal["train", "val", "test"],
118    modality: Optional[Literal["adc", "zadc"]] = None,
119    download: bool = False,
120) -> Tuple[List, List[str]]:
121    """Get paths to the BONBID-HIE data.
122
123    Args:
124        path: Filepath to a folder where the data is downloaded for further processing.
125        split: The choice of data split. Either 'train', 'val' or 'test'.
126        modality: The choice of modality for MRIs. Either 'adc' or 'zadc'.
127        download: Whether to download the data if it is not present.
128
129    Returns:
130        List of filepaths for the image data.
131        List of filepaths for the label data.
132    """
133    preprocessed_dir = get_bonbid_hie_data(path, split, download)
134
135    label_paths = natsorted(glob(os.path.join(preprocessed_dir, "labels", "*.nii.gz")))
136    adc_paths = natsorted(glob(os.path.join(preprocessed_dir, "adc", "*.nii.gz")))
137    zadc_paths = natsorted(glob(os.path.join(preprocessed_dir, "zadc", "*.nii.gz")))
138
139    if modality is None:
140        image_paths = [(adc_path, zadc_path) for adc_path, zadc_path in zip(adc_paths, zadc_paths)]
141    elif modality == "adc":
142        image_paths = adc_paths
143    elif modality == "zadc":
144        image_paths = zadc_paths
145    else:
146        raise ValueError(f"'{modality}' is not a valid modality.")
147
148    return image_paths, label_paths

Get paths to the BONBID-HIE data.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • split: The choice of data split. Either 'train', 'val' or 'test'.
  • modality: The choice of modality for MRIs. Either 'adc' or 'zadc'.
  • download: Whether to download the data if it is not present.
Returns:

List of filepaths for the image data. List of filepaths for the label data.

def get_bonbid_hie_dataset( path: Union[os.PathLike, str], patch_shape: Tuple[int, ...], split: Literal['train', 'val', 'test'] = 'train', modality: Optional[Literal['adc', 'zadc']] = None, download: bool = False, **kwargs) -> torch.utils.data.dataset.Dataset:
151def get_bonbid_hie_dataset(
152    path: Union[os.PathLike, str],
153    patch_shape: Tuple[int, ...],
154    split: Literal["train", "val", "test"] = "train",
155    modality: Optional[Literal["adc", "zadc"]] = None,
156    download: bool = False,
157    **kwargs
158) -> Dataset:
159    """Get the BONBID-HIE dataset for segmentation of HIE-related brain lesions.
160
161    Args:
162        path: Filepath to a folder where the data is downloaded for further processing.
163        patch_shape: The patch shape to use for training.
164        split: The choice of data split. Either 'train', 'val' or 'test'.
165        modality: The choice of modality for MRIs. Either 'adc' or 'zadc'.
166        download: Whether to download the data if it is not present.
167        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`.
168
169    Returns:
170        The segmentation dataset.
171    """
172    image_paths, label_paths = get_bonbid_hie_paths(path, split, modality, download)
173
174    return torch_em.default_segmentation_dataset(
175        raw_paths=image_paths,
176        raw_key="data",
177        label_paths=label_paths,
178        label_key="data",
179        patch_shape=patch_shape,
180        with_channels=modality is None,
181        is_seg_dataset=True,
182        **kwargs
183    )

Get the BONBID-HIE dataset for segmentation of HIE-related brain lesions.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • patch_shape: The patch shape to use for training.
  • split: The choice of data split. Either 'train', 'val' or 'test'.
  • modality: The choice of modality for MRIs. Either 'adc' or 'zadc'.
  • download: Whether to download the data if it is not present.
  • kwargs: Additional keyword arguments for torch_em.default_segmentation_dataset.
Returns:

The segmentation dataset.

def get_bonbid_hie_loader( path: Union[os.PathLike, str], batch_size: int, patch_shape: Tuple[int, ...], split: Literal['train', 'val', 'test'] = 'train', modality: Optional[Literal['adc', 'zadc']] = None, download: bool = False, **kwargs) -> torch.utils.data.dataloader.DataLoader:
186def get_bonbid_hie_loader(
187    path: Union[os.PathLike, str],
188    batch_size: int,
189    patch_shape: Tuple[int, ...],
190    split: Literal["train", "val", "test"] = "train",
191    modality: Optional[Literal["adc", "zadc"]] = None,
192    download: bool = False,
193    **kwargs
194) -> DataLoader:
195    """Get the BONBID-HIE dataloader for segmentation of HIE-related brain lesions.
196
197    Args:
198        path: Filepath to a folder where the data is downloaded for further processing.
199        batch_size: The batch size for training.
200        patch_shape: The patch shape to use for training.
201        split: The choice of data split. Either 'train', 'val' or 'test'.
202        modality: The choice of modality for MRIs. Either 'adc' or 'zadc'.
203        download: Whether to download the data if it is not present.
204        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader.
205
206    Returns:
207        The DataLoader.
208    """
209    ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs)
210    dataset = get_bonbid_hie_dataset(path, patch_shape, split, modality, download, **ds_kwargs)
211    return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)

Get the BONBID-HIE dataloader for segmentation of HIE-related brain lesions.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • batch_size: The batch size for training.
  • patch_shape: The patch shape to use for training.
  • split: The choice of data split. Either 'train', 'val' or 'test'.
  • modality: The choice of modality for MRIs. Either 'adc' or 'zadc'.
  • download: Whether to download the data if it is not present.
  • kwargs: Additional keyword arguments for torch_em.default_segmentation_dataset or for the PyTorch DataLoader.
Returns:

The DataLoader.