torch_em.data.datasets.medical.bratious

The BraTioUS dataset contains 1,669 B-mode 2D intraoperative brain-tumor ultrasound images from 142 glioma patients, collected across 6 hospitals in 5 countries. Every image has a corresponding binary tumor segmentation mask: the masks started as nnU-Net pseudo-labels and were then manually reviewed and corrected by neurosurgeons.

The dataset is located at https://doi.org/10.5281/zenodo.16887362, released under a CC-BY-4.0 license. This loader uses the latest Zenodo version (record 18130394), which includes an additional pass of label review and correction over the initial release.

This dataset is used in the publications https://doi.org/10.3390/cancers17020280 and https://doi.org/10.3390/cancers17020315. Please cite them if you use this dataset for your research.

NOTE: A handful of label volumes (4 out of 1,669) ship with an extra trailing singleton axis compared to their matching image. This loader squeezes them in-place on first use.

  1"""The BraTioUS dataset contains 1,669 B-mode 2D intraoperative brain-tumor ultrasound images from
  2142 glioma patients, collected across 6 hospitals in 5 countries. Every image has a corresponding
  3binary tumor segmentation mask: the masks started as nnU-Net pseudo-labels and were then manually
  4reviewed and corrected by neurosurgeons.
  5
  6The dataset is located at https://doi.org/10.5281/zenodo.16887362, released under a CC-BY-4.0 license.
  7This loader uses the latest Zenodo version (record 18130394), which includes an additional pass of
  8label review and correction over the initial release.
  9
 10This dataset is used in the publications https://doi.org/10.3390/cancers17020280 and
 11https://doi.org/10.3390/cancers17020315. Please cite them if you use this dataset for your research.
 12
 13NOTE: A handful of label volumes (4 out of 1,669) ship with an extra trailing singleton axis
 14compared to their matching image. This loader squeezes them in-place on first use.
 15"""
 16
 17import os
 18from glob import glob
 19from natsort import natsorted
 20from typing import Union, Tuple, List
 21
 22from torch.utils.data import Dataset, DataLoader
 23
 24import torch_em
 25
 26from .. import util
 27
 28
 29URLS = {
 30    "images": "https://zenodo.org/records/18130394/files/ioUS-BraTioUS-dataset.zip",
 31    "labels": "https://zenodo.org/records/18130394/files/tumor-segmentation-BraTioUS-dataset.zip",
 32}
 33
 34CHECKSUMS = {
 35    "images": "a26e4cda539a9f8520d7f5032a4a81a72f0978a22e80c47bec62fc8463d93016",
 36    "labels": "1448b9f28ab28a8a9db8ceb3a9235be54cb6dbc26d48c592b384c42d25ee8d6d",
 37}
 38
 39
 40def get_bratious_data(path: Union[os.PathLike, str], download: bool = False) -> Tuple[str, str]:
 41    """Download the BraTioUS dataset.
 42
 43    Args:
 44        path: Filepath to a folder where the data is downloaded for further processing.
 45        download: Whether to download the data if it is not present.
 46
 47    Returns:
 48        Filepath to the folder with the image data.
 49        Filepath to the folder with the label data.
 50    """
 51    image_dir = os.path.join(path, "images", "imagesBraTioUS-public-dataset")
 52    label_dir = os.path.join(path, "labels", "labelsBraTioUS-public-dataset")
 53    if os.path.exists(image_dir) and os.path.exists(label_dir):
 54        _squeeze_odd_label_shapes(label_dir)
 55        return image_dir, label_dir
 56
 57    os.makedirs(path, exist_ok=True)
 58
 59    image_zip = os.path.join(path, "images.zip")
 60    util.download_source(path=image_zip, url=URLS["images"], download=download, checksum=CHECKSUMS["images"])
 61    util.unzip(zip_path=image_zip, dst=os.path.join(path, "images"))
 62
 63    label_zip = os.path.join(path, "labels.zip")
 64    util.download_source(path=label_zip, url=URLS["labels"], download=download, checksum=CHECKSUMS["labels"])
 65    util.unzip(zip_path=label_zip, dst=os.path.join(path, "labels"))
 66
 67    assert os.path.exists(image_dir) and os.path.exists(label_dir), \
 68        f"The extraction of the BraTioUS archives did not create the expected folders in '{path}'."
 69
 70    _squeeze_odd_label_shapes(label_dir)
 71
 72    return image_dir, label_dir
 73
 74
 75def _squeeze_odd_label_shapes(label_dir):
 76    """A handful of label volumes ship with an extra trailing singleton axis, e.g. (800, 600, 1)
 77    instead of (800, 600). Squeeze them in-place so that they match their raw image's shape."""
 78    import nibabel as nib
 79
 80    for label_path in glob(os.path.join(label_dir, "*.nii.gz")):
 81        image = nib.load(label_path)
 82        if len(image.shape) == 2:
 83            continue
 84        squeezed = image.get_fdata().squeeze()
 85        nib.save(nib.Nifti1Image(squeezed, image.affine, image.header), label_path)
 86
 87
 88def get_bratious_paths(path: Union[os.PathLike, str], download: bool = False) -> Tuple[List[str], List[str]]:
 89    """Get paths to the BraTioUS data.
 90
 91    Args:
 92        path: Filepath to a folder where the data is downloaded for further processing.
 93        download: Whether to download the data if it is not present.
 94
 95    Returns:
 96        List of filepaths for the image data.
 97        List of filepaths for the label data.
 98    """
 99    image_dir, label_dir = get_bratious_data(path, download)
100
101    label_paths = natsorted(glob(os.path.join(label_dir, "*.nii.gz")))
102    raw_paths = [os.path.join(image_dir, os.path.basename(p)) for p in label_paths]
103
104    assert len(raw_paths) == len(label_paths) and len(raw_paths) > 0
105    assert all(os.path.exists(p) for p in raw_paths)
106
107    return raw_paths, label_paths
108
109
110def get_bratious_dataset(
111    path: Union[os.PathLike, str],
112    patch_shape: Tuple[int, int],
113    resize_inputs: bool = False,
114    download: bool = False,
115    **kwargs
116) -> Dataset:
117    """Get the BraTioUS dataset for tumor segmentation in intraoperative brain ultrasound.
118
119    Args:
120        path: Filepath to a folder where the data is downloaded for further processing.
121        patch_shape: The patch shape to use for training.
122        resize_inputs: Whether to resize the inputs to the patch shape.
123        download: Whether to download the data if it is not present.
124        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`.
125
126    Returns:
127        The segmentation dataset.
128    """
129    raw_paths, label_paths = get_bratious_paths(path, download)
130
131    if resize_inputs:
132        resize_kwargs = {"patch_shape": patch_shape, "is_rgb": False}
133        kwargs, patch_shape = util.update_kwargs_for_resize_trafo(
134            kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs
135        )
136
137    return torch_em.default_segmentation_dataset(
138        raw_paths=raw_paths,
139        raw_key="data",
140        label_paths=label_paths,
141        label_key="data",
142        is_seg_dataset=True,
143        patch_shape=patch_shape,
144        ndim=2,
145        **kwargs
146    )
147
148
149def get_bratious_loader(
150    path: Union[os.PathLike, str],
151    batch_size: int,
152    patch_shape: Tuple[int, int],
153    resize_inputs: bool = False,
154    download: bool = False,
155    **kwargs
156) -> DataLoader:
157    """Get the BraTioUS dataloader for tumor segmentation in intraoperative brain ultrasound.
158
159    Args:
160        path: Filepath to a folder where the data is downloaded for further processing.
161        batch_size: The batch size for training.
162        patch_shape: The patch shape to use for training.
163        resize_inputs: Whether to resize the inputs to the patch shape.
164        download: Whether to download the data if it is not present.
165        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader.
166
167    Returns:
168        The DataLoader.
169    """
170    ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs)
171    dataset = get_bratious_dataset(path, patch_shape, resize_inputs, download, **ds_kwargs)
172    return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)
URLS = {'images': 'https://zenodo.org/records/18130394/files/ioUS-BraTioUS-dataset.zip', 'labels': 'https://zenodo.org/records/18130394/files/tumor-segmentation-BraTioUS-dataset.zip'}
CHECKSUMS = {'images': 'a26e4cda539a9f8520d7f5032a4a81a72f0978a22e80c47bec62fc8463d93016', 'labels': '1448b9f28ab28a8a9db8ceb3a9235be54cb6dbc26d48c592b384c42d25ee8d6d'}
def get_bratious_data(path: Union[os.PathLike, str], download: bool = False) -> Tuple[str, str]:
41def get_bratious_data(path: Union[os.PathLike, str], download: bool = False) -> Tuple[str, str]:
42    """Download the BraTioUS dataset.
43
44    Args:
45        path: Filepath to a folder where the data is downloaded for further processing.
46        download: Whether to download the data if it is not present.
47
48    Returns:
49        Filepath to the folder with the image data.
50        Filepath to the folder with the label data.
51    """
52    image_dir = os.path.join(path, "images", "imagesBraTioUS-public-dataset")
53    label_dir = os.path.join(path, "labels", "labelsBraTioUS-public-dataset")
54    if os.path.exists(image_dir) and os.path.exists(label_dir):
55        _squeeze_odd_label_shapes(label_dir)
56        return image_dir, label_dir
57
58    os.makedirs(path, exist_ok=True)
59
60    image_zip = os.path.join(path, "images.zip")
61    util.download_source(path=image_zip, url=URLS["images"], download=download, checksum=CHECKSUMS["images"])
62    util.unzip(zip_path=image_zip, dst=os.path.join(path, "images"))
63
64    label_zip = os.path.join(path, "labels.zip")
65    util.download_source(path=label_zip, url=URLS["labels"], download=download, checksum=CHECKSUMS["labels"])
66    util.unzip(zip_path=label_zip, dst=os.path.join(path, "labels"))
67
68    assert os.path.exists(image_dir) and os.path.exists(label_dir), \
69        f"The extraction of the BraTioUS archives did not create the expected folders in '{path}'."
70
71    _squeeze_odd_label_shapes(label_dir)
72
73    return image_dir, label_dir

Download the BraTioUS dataset.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • download: Whether to download the data if it is not present.
Returns:

Filepath to the folder with the image data. Filepath to the folder with the label data.

def get_bratious_paths( path: Union[os.PathLike, str], download: bool = False) -> Tuple[List[str], List[str]]:
 89def get_bratious_paths(path: Union[os.PathLike, str], download: bool = False) -> Tuple[List[str], List[str]]:
 90    """Get paths to the BraTioUS data.
 91
 92    Args:
 93        path: Filepath to a folder where the data is downloaded for further processing.
 94        download: Whether to download the data if it is not present.
 95
 96    Returns:
 97        List of filepaths for the image data.
 98        List of filepaths for the label data.
 99    """
100    image_dir, label_dir = get_bratious_data(path, download)
101
102    label_paths = natsorted(glob(os.path.join(label_dir, "*.nii.gz")))
103    raw_paths = [os.path.join(image_dir, os.path.basename(p)) for p in label_paths]
104
105    assert len(raw_paths) == len(label_paths) and len(raw_paths) > 0
106    assert all(os.path.exists(p) for p in raw_paths)
107
108    return raw_paths, label_paths

Get paths to the BraTioUS data.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • download: Whether to download the data if it is not present.
Returns:

List of filepaths for the image data. List of filepaths for the label data.

def get_bratious_dataset( path: Union[os.PathLike, str], patch_shape: Tuple[int, int], resize_inputs: bool = False, download: bool = False, **kwargs) -> torch.utils.data.dataset.Dataset:
111def get_bratious_dataset(
112    path: Union[os.PathLike, str],
113    patch_shape: Tuple[int, int],
114    resize_inputs: bool = False,
115    download: bool = False,
116    **kwargs
117) -> Dataset:
118    """Get the BraTioUS dataset for tumor segmentation in intraoperative brain ultrasound.
119
120    Args:
121        path: Filepath to a folder where the data is downloaded for further processing.
122        patch_shape: The patch shape to use for training.
123        resize_inputs: Whether to resize the inputs to the patch shape.
124        download: Whether to download the data if it is not present.
125        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`.
126
127    Returns:
128        The segmentation dataset.
129    """
130    raw_paths, label_paths = get_bratious_paths(path, download)
131
132    if resize_inputs:
133        resize_kwargs = {"patch_shape": patch_shape, "is_rgb": False}
134        kwargs, patch_shape = util.update_kwargs_for_resize_trafo(
135            kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs
136        )
137
138    return torch_em.default_segmentation_dataset(
139        raw_paths=raw_paths,
140        raw_key="data",
141        label_paths=label_paths,
142        label_key="data",
143        is_seg_dataset=True,
144        patch_shape=patch_shape,
145        ndim=2,
146        **kwargs
147    )

Get the BraTioUS dataset for tumor segmentation in intraoperative brain ultrasound.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • patch_shape: The patch shape to use for training.
  • resize_inputs: Whether to resize the inputs to the patch shape.
  • download: Whether to download the data if it is not present.
  • kwargs: Additional keyword arguments for torch_em.default_segmentation_dataset.
Returns:

The segmentation dataset.

def get_bratious_loader( path: Union[os.PathLike, str], batch_size: int, patch_shape: Tuple[int, int], resize_inputs: bool = False, download: bool = False, **kwargs) -> torch.utils.data.dataloader.DataLoader:
150def get_bratious_loader(
151    path: Union[os.PathLike, str],
152    batch_size: int,
153    patch_shape: Tuple[int, int],
154    resize_inputs: bool = False,
155    download: bool = False,
156    **kwargs
157) -> DataLoader:
158    """Get the BraTioUS dataloader for tumor segmentation in intraoperative brain ultrasound.
159
160    Args:
161        path: Filepath to a folder where the data is downloaded for further processing.
162        batch_size: The batch size for training.
163        patch_shape: The patch shape to use for training.
164        resize_inputs: Whether to resize the inputs to the patch shape.
165        download: Whether to download the data if it is not present.
166        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader.
167
168    Returns:
169        The DataLoader.
170    """
171    ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs)
172    dataset = get_bratious_dataset(path, patch_shape, resize_inputs, download, **ds_kwargs)
173    return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)

Get the BraTioUS dataloader for tumor segmentation in intraoperative brain ultrasound.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • batch_size: The batch size for training.
  • patch_shape: The patch shape to use for training.
  • resize_inputs: Whether to resize the inputs to the patch shape.
  • download: Whether to download the data if it is not present.
  • kwargs: Additional keyword arguments for torch_em.default_segmentation_dataset or for the PyTorch DataLoader.
Returns:

The DataLoader.