torch_em.data.datasets.medical.bratious
The BraTioUS dataset contains 1,669 B-mode 2D intraoperative brain-tumor ultrasound images from 142 glioma patients, collected across 6 hospitals in 5 countries. Every image has a corresponding binary tumor segmentation mask: the masks started as nnU-Net pseudo-labels and were then manually reviewed and corrected by neurosurgeons.
The dataset is located at https://doi.org/10.5281/zenodo.16887362, released under a CC-BY-4.0 license. This loader uses the latest Zenodo version (record 18130394), which includes an additional pass of label review and correction over the initial release.
This dataset is used in the publications https://doi.org/10.3390/cancers17020280 and https://doi.org/10.3390/cancers17020315. Please cite them if you use this dataset for your research.
NOTE: A handful of label volumes (4 out of 1,669) ship with an extra trailing singleton axis compared to their matching image. This loader squeezes them in-place on first use.
1"""The BraTioUS dataset contains 1,669 B-mode 2D intraoperative brain-tumor ultrasound images from 2142 glioma patients, collected across 6 hospitals in 5 countries. Every image has a corresponding 3binary tumor segmentation mask: the masks started as nnU-Net pseudo-labels and were then manually 4reviewed and corrected by neurosurgeons. 5 6The dataset is located at https://doi.org/10.5281/zenodo.16887362, released under a CC-BY-4.0 license. 7This loader uses the latest Zenodo version (record 18130394), which includes an additional pass of 8label review and correction over the initial release. 9 10This dataset is used in the publications https://doi.org/10.3390/cancers17020280 and 11https://doi.org/10.3390/cancers17020315. Please cite them if you use this dataset for your research. 12 13NOTE: A handful of label volumes (4 out of 1,669) ship with an extra trailing singleton axis 14compared to their matching image. This loader squeezes them in-place on first use. 15""" 16 17import os 18from glob import glob 19from natsort import natsorted 20from typing import Union, Tuple, List 21 22from torch.utils.data import Dataset, DataLoader 23 24import torch_em 25 26from .. import util 27 28 29URLS = { 30 "images": "https://zenodo.org/records/18130394/files/ioUS-BraTioUS-dataset.zip", 31 "labels": "https://zenodo.org/records/18130394/files/tumor-segmentation-BraTioUS-dataset.zip", 32} 33 34CHECKSUMS = { 35 "images": "a26e4cda539a9f8520d7f5032a4a81a72f0978a22e80c47bec62fc8463d93016", 36 "labels": "1448b9f28ab28a8a9db8ceb3a9235be54cb6dbc26d48c592b384c42d25ee8d6d", 37} 38 39 40def get_bratious_data(path: Union[os.PathLike, str], download: bool = False) -> Tuple[str, str]: 41 """Download the BraTioUS dataset. 42 43 Args: 44 path: Filepath to a folder where the data is downloaded for further processing. 45 download: Whether to download the data if it is not present. 46 47 Returns: 48 Filepath to the folder with the image data. 49 Filepath to the folder with the label data. 50 """ 51 image_dir = os.path.join(path, "images", "imagesBraTioUS-public-dataset") 52 label_dir = os.path.join(path, "labels", "labelsBraTioUS-public-dataset") 53 if os.path.exists(image_dir) and os.path.exists(label_dir): 54 _squeeze_odd_label_shapes(label_dir) 55 return image_dir, label_dir 56 57 os.makedirs(path, exist_ok=True) 58 59 image_zip = os.path.join(path, "images.zip") 60 util.download_source(path=image_zip, url=URLS["images"], download=download, checksum=CHECKSUMS["images"]) 61 util.unzip(zip_path=image_zip, dst=os.path.join(path, "images")) 62 63 label_zip = os.path.join(path, "labels.zip") 64 util.download_source(path=label_zip, url=URLS["labels"], download=download, checksum=CHECKSUMS["labels"]) 65 util.unzip(zip_path=label_zip, dst=os.path.join(path, "labels")) 66 67 assert os.path.exists(image_dir) and os.path.exists(label_dir), \ 68 f"The extraction of the BraTioUS archives did not create the expected folders in '{path}'." 69 70 _squeeze_odd_label_shapes(label_dir) 71 72 return image_dir, label_dir 73 74 75def _squeeze_odd_label_shapes(label_dir): 76 """A handful of label volumes ship with an extra trailing singleton axis, e.g. (800, 600, 1) 77 instead of (800, 600). Squeeze them in-place so that they match their raw image's shape.""" 78 import nibabel as nib 79 80 for label_path in glob(os.path.join(label_dir, "*.nii.gz")): 81 image = nib.load(label_path) 82 if len(image.shape) == 2: 83 continue 84 squeezed = image.get_fdata().squeeze() 85 nib.save(nib.Nifti1Image(squeezed, image.affine, image.header), label_path) 86 87 88def get_bratious_paths(path: Union[os.PathLike, str], download: bool = False) -> Tuple[List[str], List[str]]: 89 """Get paths to the BraTioUS data. 90 91 Args: 92 path: Filepath to a folder where the data is downloaded for further processing. 93 download: Whether to download the data if it is not present. 94 95 Returns: 96 List of filepaths for the image data. 97 List of filepaths for the label data. 98 """ 99 image_dir, label_dir = get_bratious_data(path, download) 100 101 label_paths = natsorted(glob(os.path.join(label_dir, "*.nii.gz"))) 102 raw_paths = [os.path.join(image_dir, os.path.basename(p)) for p in label_paths] 103 104 assert len(raw_paths) == len(label_paths) and len(raw_paths) > 0 105 assert all(os.path.exists(p) for p in raw_paths) 106 107 return raw_paths, label_paths 108 109 110def get_bratious_dataset( 111 path: Union[os.PathLike, str], 112 patch_shape: Tuple[int, int], 113 resize_inputs: bool = False, 114 download: bool = False, 115 **kwargs 116) -> Dataset: 117 """Get the BraTioUS dataset for tumor segmentation in intraoperative brain ultrasound. 118 119 Args: 120 path: Filepath to a folder where the data is downloaded for further processing. 121 patch_shape: The patch shape to use for training. 122 resize_inputs: Whether to resize the inputs to the patch shape. 123 download: Whether to download the data if it is not present. 124 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`. 125 126 Returns: 127 The segmentation dataset. 128 """ 129 raw_paths, label_paths = get_bratious_paths(path, download) 130 131 if resize_inputs: 132 resize_kwargs = {"patch_shape": patch_shape, "is_rgb": False} 133 kwargs, patch_shape = util.update_kwargs_for_resize_trafo( 134 kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs 135 ) 136 137 return torch_em.default_segmentation_dataset( 138 raw_paths=raw_paths, 139 raw_key="data", 140 label_paths=label_paths, 141 label_key="data", 142 is_seg_dataset=True, 143 patch_shape=patch_shape, 144 ndim=2, 145 **kwargs 146 ) 147 148 149def get_bratious_loader( 150 path: Union[os.PathLike, str], 151 batch_size: int, 152 patch_shape: Tuple[int, int], 153 resize_inputs: bool = False, 154 download: bool = False, 155 **kwargs 156) -> DataLoader: 157 """Get the BraTioUS dataloader for tumor segmentation in intraoperative brain ultrasound. 158 159 Args: 160 path: Filepath to a folder where the data is downloaded for further processing. 161 batch_size: The batch size for training. 162 patch_shape: The patch shape to use for training. 163 resize_inputs: Whether to resize the inputs to the patch shape. 164 download: Whether to download the data if it is not present. 165 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader. 166 167 Returns: 168 The DataLoader. 169 """ 170 ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs) 171 dataset = get_bratious_dataset(path, patch_shape, resize_inputs, download, **ds_kwargs) 172 return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)
41def get_bratious_data(path: Union[os.PathLike, str], download: bool = False) -> Tuple[str, str]: 42 """Download the BraTioUS dataset. 43 44 Args: 45 path: Filepath to a folder where the data is downloaded for further processing. 46 download: Whether to download the data if it is not present. 47 48 Returns: 49 Filepath to the folder with the image data. 50 Filepath to the folder with the label data. 51 """ 52 image_dir = os.path.join(path, "images", "imagesBraTioUS-public-dataset") 53 label_dir = os.path.join(path, "labels", "labelsBraTioUS-public-dataset") 54 if os.path.exists(image_dir) and os.path.exists(label_dir): 55 _squeeze_odd_label_shapes(label_dir) 56 return image_dir, label_dir 57 58 os.makedirs(path, exist_ok=True) 59 60 image_zip = os.path.join(path, "images.zip") 61 util.download_source(path=image_zip, url=URLS["images"], download=download, checksum=CHECKSUMS["images"]) 62 util.unzip(zip_path=image_zip, dst=os.path.join(path, "images")) 63 64 label_zip = os.path.join(path, "labels.zip") 65 util.download_source(path=label_zip, url=URLS["labels"], download=download, checksum=CHECKSUMS["labels"]) 66 util.unzip(zip_path=label_zip, dst=os.path.join(path, "labels")) 67 68 assert os.path.exists(image_dir) and os.path.exists(label_dir), \ 69 f"The extraction of the BraTioUS archives did not create the expected folders in '{path}'." 70 71 _squeeze_odd_label_shapes(label_dir) 72 73 return image_dir, label_dir
Download the BraTioUS dataset.
Arguments:
- path: Filepath to a folder where the data is downloaded for further processing.
- download: Whether to download the data if it is not present.
Returns:
Filepath to the folder with the image data. Filepath to the folder with the label data.
89def get_bratious_paths(path: Union[os.PathLike, str], download: bool = False) -> Tuple[List[str], List[str]]: 90 """Get paths to the BraTioUS data. 91 92 Args: 93 path: Filepath to a folder where the data is downloaded for further processing. 94 download: Whether to download the data if it is not present. 95 96 Returns: 97 List of filepaths for the image data. 98 List of filepaths for the label data. 99 """ 100 image_dir, label_dir = get_bratious_data(path, download) 101 102 label_paths = natsorted(glob(os.path.join(label_dir, "*.nii.gz"))) 103 raw_paths = [os.path.join(image_dir, os.path.basename(p)) for p in label_paths] 104 105 assert len(raw_paths) == len(label_paths) and len(raw_paths) > 0 106 assert all(os.path.exists(p) for p in raw_paths) 107 108 return raw_paths, label_paths
Get paths to the BraTioUS data.
Arguments:
- path: Filepath to a folder where the data is downloaded for further processing.
- download: Whether to download the data if it is not present.
Returns:
List of filepaths for the image data. List of filepaths for the label data.
111def get_bratious_dataset( 112 path: Union[os.PathLike, str], 113 patch_shape: Tuple[int, int], 114 resize_inputs: bool = False, 115 download: bool = False, 116 **kwargs 117) -> Dataset: 118 """Get the BraTioUS dataset for tumor segmentation in intraoperative brain ultrasound. 119 120 Args: 121 path: Filepath to a folder where the data is downloaded for further processing. 122 patch_shape: The patch shape to use for training. 123 resize_inputs: Whether to resize the inputs to the patch shape. 124 download: Whether to download the data if it is not present. 125 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`. 126 127 Returns: 128 The segmentation dataset. 129 """ 130 raw_paths, label_paths = get_bratious_paths(path, download) 131 132 if resize_inputs: 133 resize_kwargs = {"patch_shape": patch_shape, "is_rgb": False} 134 kwargs, patch_shape = util.update_kwargs_for_resize_trafo( 135 kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs 136 ) 137 138 return torch_em.default_segmentation_dataset( 139 raw_paths=raw_paths, 140 raw_key="data", 141 label_paths=label_paths, 142 label_key="data", 143 is_seg_dataset=True, 144 patch_shape=patch_shape, 145 ndim=2, 146 **kwargs 147 )
Get the BraTioUS dataset for tumor segmentation in intraoperative brain ultrasound.
Arguments:
- path: Filepath to a folder where the data is downloaded for further processing.
- patch_shape: The patch shape to use for training.
- resize_inputs: Whether to resize the inputs to the patch shape.
- download: Whether to download the data if it is not present.
- kwargs: Additional keyword arguments for
torch_em.default_segmentation_dataset.
Returns:
The segmentation dataset.
150def get_bratious_loader( 151 path: Union[os.PathLike, str], 152 batch_size: int, 153 patch_shape: Tuple[int, int], 154 resize_inputs: bool = False, 155 download: bool = False, 156 **kwargs 157) -> DataLoader: 158 """Get the BraTioUS dataloader for tumor segmentation in intraoperative brain ultrasound. 159 160 Args: 161 path: Filepath to a folder where the data is downloaded for further processing. 162 batch_size: The batch size for training. 163 patch_shape: The patch shape to use for training. 164 resize_inputs: Whether to resize the inputs to the patch shape. 165 download: Whether to download the data if it is not present. 166 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader. 167 168 Returns: 169 The DataLoader. 170 """ 171 ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs) 172 dataset = get_bratious_dataset(path, patch_shape, resize_inputs, download, **ds_kwargs) 173 return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)
Get the BraTioUS dataloader for tumor segmentation in intraoperative brain ultrasound.
Arguments:
- path: Filepath to a folder where the data is downloaded for further processing.
- batch_size: The batch size for training.
- patch_shape: The patch shape to use for training.
- resize_inputs: Whether to resize the inputs to the patch shape.
- download: Whether to download the data if it is not present.
- kwargs: Additional keyword arguments for
torch_em.default_segmentation_datasetor for the PyTorch DataLoader.
Returns:
The DataLoader.