torch_em.data.datasets.medical.bonbid_hie
The BONBID-HIE dataset contains annotations for lesion segmentation in neonatal brain diffusion MRI of patients with hypoxic-ischemic encephalopathy (HIE).
The dataset is released as part of the BONBID-HIE 2023 MICCAI challenge
(https://bonbid-hie2023.grand-challenge.org) and is hosted at
https://doi.org/10.5281/zenodo.10602767. It provides 85 training, 4 validation and 44
test cases (see SPLITS). Each case has a skull-stripped apparent diffusion coefficient
(ADC) map and the derived Z-ADC map (the ADC map normalized against a healthy reference
population, smoothed and clipped to the range [-6, 10]), both registered to a binary
lesion segmentation mask.
NOTE: The scans are distributed as MetaImage (.mha) files. This module converts them to compressed nifti volumes on first use, which requires the SimpleITK python package.
NOTE: The Zenodo record's structured license field lists CC-BY-NC-ND 2.5, even though the record's description text states that the data is released under the CC BY 4.0 license. Please check the record for the authoritative license terms before further use.
This dataset is from the publication https://doi.org/10.1038/s41597-024-03986-7. Please cite it if you use this dataset in your research.
1"""The BONBID-HIE dataset contains annotations for lesion segmentation in neonatal brain 2diffusion MRI of patients with hypoxic-ischemic encephalopathy (HIE). 3 4The dataset is released as part of the BONBID-HIE 2023 MICCAI challenge 5(https://bonbid-hie2023.grand-challenge.org) and is hosted at 6https://doi.org/10.5281/zenodo.10602767. It provides 85 training, 4 validation and 44 7test cases (see `SPLITS`). Each case has a skull-stripped apparent diffusion coefficient 8(ADC) map and the derived Z-ADC map (the ADC map normalized against a healthy reference 9population, smoothed and clipped to the range [-6, 10]), both registered to a binary 10lesion segmentation mask. 11 12NOTE: The scans are distributed as MetaImage (.mha) files. This module converts them to 13compressed nifti volumes on first use, which requires the SimpleITK python package. 14 15NOTE: The Zenodo record's structured license field lists CC-BY-NC-ND 2.5, even though the 16record's description text states that the data is released under the CC BY 4.0 license. 17Please check the record for the authoritative license terms before further use. 18 19This dataset is from the publication https://doi.org/10.1038/s41597-024-03986-7. 20Please cite it if you use this dataset in your research. 21""" 22 23import os 24from glob import glob 25from tqdm import tqdm 26from natsort import natsorted 27from typing import Union, Tuple, Optional, Literal, List 28 29from torch.utils.data import Dataset, DataLoader 30 31import torch_em 32 33from .. import util 34 35 36URLS = { 37 "train": "https://zenodo.org/records/10602767/files/BONBID2023_Train.zip", 38 "val": "https://zenodo.org/records/10602767/files/BONBID2023_Val.zip", 39 "test": "https://zenodo.org/records/10602767/files/BONBID2023_Test.zip", 40} 41 42CHECKSUMS = { 43 "train": "f1058093887daea1ebc07368b0642d3c40a3f70da5e2e225558ec32de3fa345b", 44 "val": "8eaaa05cad00d5d5112583ca01be9ec685e15ca87b946b19a4010b3da35f4786", 45 "test": "ac3988c57f035ee74a20dd70194e6fd51fe9aecc3a68e093f487620b24fcbfa3", 46} 47 48SPLITS = ("train", "val", "test") 49 50 51def _convert_mha_to_nifti(mha_path, nifti_path): 52 if os.path.exists(nifti_path): 53 return 54 55 import SimpleITK as sitk 56 57 volume = sitk.ReadImage(mha_path) 58 sitk.WriteImage(volume, nifti_path, useCompression=True) 59 60 61def _preprocess_bonbid_hie(raw_dir, preprocessed_dir, split): 62 adc_paths = natsorted(glob(os.path.join(raw_dir, "**", "1ADC_ss", "*-ADC_ss.mha"), recursive=True)) 63 if len(adc_paths) == 0: 64 raise RuntimeError(f"Did not find any '{split}' scans at '{raw_dir}'.") 65 66 for folder in ["adc", "zadc", "labels"]: 67 os.makedirs(os.path.join(preprocessed_dir, folder), exist_ok=True) 68 69 for adc_path in tqdm(adc_paths, desc=f"Preprocess BONBID-HIE '{split}' scans"): 70 case_dir = os.path.dirname(os.path.dirname(adc_path)) 71 case_id = os.path.basename(adc_path)[:-len("-ADC_ss.mha")] 72 73 zadc_path = os.path.join(case_dir, "2Z_ADC", f"Zmap_{case_id}-ADC_smooth2mm_clipped10.mha") 74 label_path = os.path.join(case_dir, "3LABEL", f"{case_id}_lesion.mha") 75 if not os.path.exists(zadc_path) or not os.path.exists(label_path): 76 raise RuntimeError(f"Could not find the Z-ADC map or the label for the case '{case_id}'.") 77 78 _convert_mha_to_nifti(adc_path, os.path.join(preprocessed_dir, "adc", f"{case_id}.nii.gz")) 79 _convert_mha_to_nifti(zadc_path, os.path.join(preprocessed_dir, "zadc", f"{case_id}.nii.gz")) 80 _convert_mha_to_nifti(label_path, os.path.join(preprocessed_dir, "labels", f"{case_id}.nii.gz")) 81 82 83def get_bonbid_hie_data(path: Union[os.PathLike, str], split: Literal["train", "val", "test"], download: bool = False) -> str: # noqa 84 """Download the BONBID-HIE dataset and preprocess it to nifti volumes. 85 86 Args: 87 path: Filepath to a folder where the data is downloaded for further processing. 88 split: The choice of data split. Either 'train', 'val' or 'test'. 89 download: Whether to download the data if it is not present. 90 91 Returns: 92 Filepath to the folder where the preprocessed data is stored. 93 """ 94 if split not in SPLITS: 95 raise ValueError(f"'{split}' is not a valid split. Choose from {SPLITS}.") 96 97 preprocessed_dir = os.path.join(path, "preprocessed", split) 98 if os.path.exists(preprocessed_dir): 99 return preprocessed_dir 100 101 os.makedirs(path, exist_ok=True) 102 103 zip_path = os.path.join(path, f"BONBID2023_{split.capitalize()}.zip") 104 util.download_source(path=zip_path, url=URLS[split], download=download, checksum=CHECKSUMS[split]) 105 106 raw_dir = os.path.join(path, "raw", split) 107 util.unzip(zip_path=zip_path, dst=raw_dir) 108 109 _preprocess_bonbid_hie(raw_dir, preprocessed_dir, split) 110 111 return preprocessed_dir 112 113 114def get_bonbid_hie_paths( 115 path: Union[os.PathLike, str], 116 split: Literal["train", "val", "test"], 117 modality: Optional[Literal["adc", "zadc"]] = None, 118 download: bool = False, 119) -> Tuple[List, List[str]]: 120 """Get paths to the BONBID-HIE data. 121 122 Args: 123 path: Filepath to a folder where the data is downloaded for further processing. 124 split: The choice of data split. Either 'train', 'val' or 'test'. 125 modality: The choice of modality for MRIs. Either 'adc' or 'zadc'. 126 download: Whether to download the data if it is not present. 127 128 Returns: 129 List of filepaths for the image data. 130 List of filepaths for the label data. 131 """ 132 preprocessed_dir = get_bonbid_hie_data(path, split, download) 133 134 label_paths = natsorted(glob(os.path.join(preprocessed_dir, "labels", "*.nii.gz"))) 135 adc_paths = natsorted(glob(os.path.join(preprocessed_dir, "adc", "*.nii.gz"))) 136 zadc_paths = natsorted(glob(os.path.join(preprocessed_dir, "zadc", "*.nii.gz"))) 137 138 if modality is None: 139 image_paths = [(adc_path, zadc_path) for adc_path, zadc_path in zip(adc_paths, zadc_paths)] 140 elif modality == "adc": 141 image_paths = adc_paths 142 elif modality == "zadc": 143 image_paths = zadc_paths 144 else: 145 raise ValueError(f"'{modality}' is not a valid modality.") 146 147 return image_paths, label_paths 148 149 150def get_bonbid_hie_dataset( 151 path: Union[os.PathLike, str], 152 patch_shape: Tuple[int, ...], 153 split: Literal["train", "val", "test"] = "train", 154 modality: Optional[Literal["adc", "zadc"]] = None, 155 download: bool = False, 156 **kwargs 157) -> Dataset: 158 """Get the BONBID-HIE dataset for segmentation of HIE-related brain lesions. 159 160 Args: 161 path: Filepath to a folder where the data is downloaded for further processing. 162 patch_shape: The patch shape to use for training. 163 split: The choice of data split. Either 'train', 'val' or 'test'. 164 modality: The choice of modality for MRIs. Either 'adc' or 'zadc'. 165 download: Whether to download the data if it is not present. 166 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`. 167 168 Returns: 169 The segmentation dataset. 170 """ 171 image_paths, label_paths = get_bonbid_hie_paths(path, split, modality, download) 172 173 return torch_em.default_segmentation_dataset( 174 raw_paths=image_paths, 175 raw_key="data", 176 label_paths=label_paths, 177 label_key="data", 178 patch_shape=patch_shape, 179 with_channels=modality is None, 180 is_seg_dataset=True, 181 **kwargs 182 ) 183 184 185def get_bonbid_hie_loader( 186 path: Union[os.PathLike, str], 187 batch_size: int, 188 patch_shape: Tuple[int, ...], 189 split: Literal["train", "val", "test"] = "train", 190 modality: Optional[Literal["adc", "zadc"]] = None, 191 download: bool = False, 192 **kwargs 193) -> DataLoader: 194 """Get the BONBID-HIE dataloader for segmentation of HIE-related brain lesions. 195 196 Args: 197 path: Filepath to a folder where the data is downloaded for further processing. 198 batch_size: The batch size for training. 199 patch_shape: The patch shape to use for training. 200 split: The choice of data split. Either 'train', 'val' or 'test'. 201 modality: The choice of modality for MRIs. Either 'adc' or 'zadc'. 202 download: Whether to download the data if it is not present. 203 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader. 204 205 Returns: 206 The DataLoader. 207 """ 208 ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs) 209 dataset = get_bonbid_hie_dataset(path, patch_shape, split, modality, download, **ds_kwargs) 210 return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)
84def get_bonbid_hie_data(path: Union[os.PathLike, str], split: Literal["train", "val", "test"], download: bool = False) -> str: # noqa 85 """Download the BONBID-HIE dataset and preprocess it to nifti volumes. 86 87 Args: 88 path: Filepath to a folder where the data is downloaded for further processing. 89 split: The choice of data split. Either 'train', 'val' or 'test'. 90 download: Whether to download the data if it is not present. 91 92 Returns: 93 Filepath to the folder where the preprocessed data is stored. 94 """ 95 if split not in SPLITS: 96 raise ValueError(f"'{split}' is not a valid split. Choose from {SPLITS}.") 97 98 preprocessed_dir = os.path.join(path, "preprocessed", split) 99 if os.path.exists(preprocessed_dir): 100 return preprocessed_dir 101 102 os.makedirs(path, exist_ok=True) 103 104 zip_path = os.path.join(path, f"BONBID2023_{split.capitalize()}.zip") 105 util.download_source(path=zip_path, url=URLS[split], download=download, checksum=CHECKSUMS[split]) 106 107 raw_dir = os.path.join(path, "raw", split) 108 util.unzip(zip_path=zip_path, dst=raw_dir) 109 110 _preprocess_bonbid_hie(raw_dir, preprocessed_dir, split) 111 112 return preprocessed_dir
Download the BONBID-HIE dataset and preprocess it to nifti volumes.
Arguments:
- path: Filepath to a folder where the data is downloaded for further processing.
- split: The choice of data split. Either 'train', 'val' or 'test'.
- download: Whether to download the data if it is not present.
Returns:
Filepath to the folder where the preprocessed data is stored.
115def get_bonbid_hie_paths( 116 path: Union[os.PathLike, str], 117 split: Literal["train", "val", "test"], 118 modality: Optional[Literal["adc", "zadc"]] = None, 119 download: bool = False, 120) -> Tuple[List, List[str]]: 121 """Get paths to the BONBID-HIE data. 122 123 Args: 124 path: Filepath to a folder where the data is downloaded for further processing. 125 split: The choice of data split. Either 'train', 'val' or 'test'. 126 modality: The choice of modality for MRIs. Either 'adc' or 'zadc'. 127 download: Whether to download the data if it is not present. 128 129 Returns: 130 List of filepaths for the image data. 131 List of filepaths for the label data. 132 """ 133 preprocessed_dir = get_bonbid_hie_data(path, split, download) 134 135 label_paths = natsorted(glob(os.path.join(preprocessed_dir, "labels", "*.nii.gz"))) 136 adc_paths = natsorted(glob(os.path.join(preprocessed_dir, "adc", "*.nii.gz"))) 137 zadc_paths = natsorted(glob(os.path.join(preprocessed_dir, "zadc", "*.nii.gz"))) 138 139 if modality is None: 140 image_paths = [(adc_path, zadc_path) for adc_path, zadc_path in zip(adc_paths, zadc_paths)] 141 elif modality == "adc": 142 image_paths = adc_paths 143 elif modality == "zadc": 144 image_paths = zadc_paths 145 else: 146 raise ValueError(f"'{modality}' is not a valid modality.") 147 148 return image_paths, label_paths
Get paths to the BONBID-HIE data.
Arguments:
- path: Filepath to a folder where the data is downloaded for further processing.
- split: The choice of data split. Either 'train', 'val' or 'test'.
- modality: The choice of modality for MRIs. Either 'adc' or 'zadc'.
- download: Whether to download the data if it is not present.
Returns:
List of filepaths for the image data. List of filepaths for the label data.
151def get_bonbid_hie_dataset( 152 path: Union[os.PathLike, str], 153 patch_shape: Tuple[int, ...], 154 split: Literal["train", "val", "test"] = "train", 155 modality: Optional[Literal["adc", "zadc"]] = None, 156 download: bool = False, 157 **kwargs 158) -> Dataset: 159 """Get the BONBID-HIE dataset for segmentation of HIE-related brain lesions. 160 161 Args: 162 path: Filepath to a folder where the data is downloaded for further processing. 163 patch_shape: The patch shape to use for training. 164 split: The choice of data split. Either 'train', 'val' or 'test'. 165 modality: The choice of modality for MRIs. Either 'adc' or 'zadc'. 166 download: Whether to download the data if it is not present. 167 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`. 168 169 Returns: 170 The segmentation dataset. 171 """ 172 image_paths, label_paths = get_bonbid_hie_paths(path, split, modality, download) 173 174 return torch_em.default_segmentation_dataset( 175 raw_paths=image_paths, 176 raw_key="data", 177 label_paths=label_paths, 178 label_key="data", 179 patch_shape=patch_shape, 180 with_channels=modality is None, 181 is_seg_dataset=True, 182 **kwargs 183 )
Get the BONBID-HIE dataset for segmentation of HIE-related brain lesions.
Arguments:
- path: Filepath to a folder where the data is downloaded for further processing.
- patch_shape: The patch shape to use for training.
- split: The choice of data split. Either 'train', 'val' or 'test'.
- modality: The choice of modality for MRIs. Either 'adc' or 'zadc'.
- download: Whether to download the data if it is not present.
- kwargs: Additional keyword arguments for
torch_em.default_segmentation_dataset.
Returns:
The segmentation dataset.
186def get_bonbid_hie_loader( 187 path: Union[os.PathLike, str], 188 batch_size: int, 189 patch_shape: Tuple[int, ...], 190 split: Literal["train", "val", "test"] = "train", 191 modality: Optional[Literal["adc", "zadc"]] = None, 192 download: bool = False, 193 **kwargs 194) -> DataLoader: 195 """Get the BONBID-HIE dataloader for segmentation of HIE-related brain lesions. 196 197 Args: 198 path: Filepath to a folder where the data is downloaded for further processing. 199 batch_size: The batch size for training. 200 patch_shape: The patch shape to use for training. 201 split: The choice of data split. Either 'train', 'val' or 'test'. 202 modality: The choice of modality for MRIs. Either 'adc' or 'zadc'. 203 download: Whether to download the data if it is not present. 204 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader. 205 206 Returns: 207 The DataLoader. 208 """ 209 ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs) 210 dataset = get_bonbid_hie_dataset(path, patch_shape, split, modality, download, **ds_kwargs) 211 return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)
Get the BONBID-HIE dataloader for segmentation of HIE-related brain lesions.
Arguments:
- path: Filepath to a folder where the data is downloaded for further processing.
- batch_size: The batch size for training.
- patch_shape: The patch shape to use for training.
- split: The choice of data split. Either 'train', 'val' or 'test'.
- modality: The choice of modality for MRIs. Either 'adc' or 'zadc'.
- download: Whether to download the data if it is not present.
- kwargs: Additional keyword arguments for
torch_em.default_segmentation_datasetor for the PyTorch DataLoader.
Returns:
The DataLoader.