torch_em.data.datasets.medical.brisc

BRISC (BRain tumor Image Segmentation and Classification) is a curated, expert-annotated dataset of contrast-enhanced T1-weighted brain MRI slices for brain tumor segmentation and classification.

The dataset consists of 6,000 slices (5,000 train / 1,000 test) collated from existing public MRI collections, spanning axial, coronal and sagittal planes. It provides physician-reviewed pixel-wise segmentation masks for three tumor types (glioma, meningioma, pituitary tumor), in addition to image-level labels for a 'no tumor' class (which has no corresponding segmentation mask). While the raw images originate from other public collections, the segmentation masks are a new, original annotation contribution.

The dataset is located at https://doi.org/10.6084/m9.figshare.30533120 and is distributed under the CC BY 4.0 license.

This dataset is from the publication https://doi.org/10.1038/s41597-026-06753-y. Please cite it if you use this dataset in your research.

  1"""BRISC (BRain tumor Image Segmentation and Classification) is a curated, expert-annotated dataset
  2of contrast-enhanced T1-weighted brain MRI slices for brain tumor segmentation and classification.
  3
  4The dataset consists of 6,000 slices (5,000 train / 1,000 test) collated from existing public MRI
  5collections, spanning axial, coronal and sagittal planes. It provides physician-reviewed pixel-wise
  6segmentation masks for three tumor types (glioma, meningioma, pituitary tumor), in addition to
  7image-level labels for a 'no tumor' class (which has no corresponding segmentation mask). While the
  8raw images originate from other public collections, the segmentation masks are a new, original
  9annotation contribution.
 10
 11The dataset is located at https://doi.org/10.6084/m9.figshare.30533120 and is distributed under the
 12CC BY 4.0 license.
 13
 14This dataset is from the publication https://doi.org/10.1038/s41597-026-06753-y.
 15Please cite it if you use this dataset in your research.
 16"""
 17
 18import os
 19from glob import glob
 20from natsort import natsorted
 21from typing import Union, Tuple, Optional, Literal, List
 22
 23from torch.utils.data import Dataset, DataLoader
 24
 25import torch_em
 26
 27from .. import util
 28
 29
 30URL = "https://ndownloader.figshare.com/files/59298329"
 31CHECKSUM = "3167e9bdfcb3b2a2502091f41c9bd3f1e2eeb64ec3ad9849680fa760e2e7633d"
 32
 33TUMOR_TYPES = {"glioma": "gl", "meningioma": "me", "pituitary": "pi"}
 34
 35
 36def get_brisc_data(path: Union[os.PathLike, str], download: bool = False) -> str:
 37    """Download the BRISC dataset.
 38
 39    Args:
 40        path: Filepath to a folder where the data is downloaded for further processing.
 41        download: Whether to download the data if it is not present.
 42
 43    Returns:
 44        Filepath where the data is downloaded.
 45    """
 46    data_dir = os.path.join(path, "brisc2025")
 47    if os.path.exists(data_dir):
 48        return data_dir
 49
 50    os.makedirs(path, exist_ok=True)
 51
 52    zip_path = os.path.join(path, "brisc2025.zip")
 53    util.download_source(path=zip_path, url=URL, download=download, checksum=CHECKSUM)
 54    util.unzip(zip_path=zip_path, dst=path)
 55
 56    return data_dir
 57
 58
 59def get_brisc_paths(
 60    path: Union[os.PathLike, str],
 61    split: Literal["train", "test"] = "train",
 62    tumor_type: Optional[Literal["glioma", "meningioma", "pituitary"]] = None,
 63    download: bool = False,
 64) -> Tuple[List[str], List[str]]:
 65    """Get paths to the BRISC data.
 66
 67    Args:
 68        path: Filepath to a folder where the data is downloaded for further processing.
 69        split: The choice of data split. Either 'train' or 'test'.
 70        tumor_type: The choice of tumor type. One of 'glioma', 'meningioma', 'pituitary'. By default, all are used.
 71        download: Whether to download the data if it is not present.
 72
 73    Returns:
 74        List of filepaths for the image data.
 75        List of filepaths for the label data.
 76    """
 77    data_dir = get_brisc_data(path=path, download=download)
 78
 79    if split not in ["train", "test"]:
 80        raise ValueError(f"'{split}' is not a valid split choice.")
 81
 82    image_dir = os.path.join(data_dir, "segmentation_task", split, "images")
 83    label_dir = os.path.join(data_dir, "segmentation_task", split, "masks")
 84
 85    if tumor_type is None:
 86        pattern = "*.jpg"
 87    elif tumor_type in TUMOR_TYPES:
 88        pattern = f"*_{TUMOR_TYPES[tumor_type]}_*.jpg"
 89    else:
 90        raise ValueError(f"'{tumor_type}' is not a valid tumor type. Choose from {list(TUMOR_TYPES.keys())}.")
 91
 92    image_paths = natsorted(glob(os.path.join(image_dir, pattern)))
 93    label_paths = natsorted(
 94        os.path.join(label_dir, os.path.splitext(os.path.basename(p))[0] + ".png") for p in image_paths
 95    )
 96    assert len(image_paths) > 0 and len(image_paths) == len(label_paths)
 97
 98    return image_paths, label_paths
 99
100
101def get_brisc_dataset(
102    path: Union[os.PathLike, str],
103    patch_shape: Tuple[int, int],
104    split: Literal["train", "test"] = "train",
105    tumor_type: Optional[Literal["glioma", "meningioma", "pituitary"]] = None,
106    resize_inputs: bool = False,
107    download: bool = False,
108    **kwargs
109) -> Dataset:
110    """Get the BRISC dataset for brain tumor segmentation.
111
112    Args:
113        path: Filepath to a folder where the data is downloaded for further processing.
114        patch_shape: The patch shape to use for training.
115        split: The choice of data split. Either 'train' or 'test'.
116        tumor_type: The choice of tumor type. One of 'glioma', 'meningioma', 'pituitary'. By default, all are used.
117        resize_inputs: Whether to resize the inputs.
118        download: Whether to download the data if it is not present.
119        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`.
120
121    Returns:
122        The segmentation dataset.
123    """
124    image_paths, label_paths = get_brisc_paths(path, split, tumor_type, download)
125
126    if resize_inputs:
127        resize_kwargs = {"patch_shape": patch_shape, "is_rgb": True}
128        kwargs, patch_shape = util.update_kwargs_for_resize_trafo(
129            kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs
130        )
131
132    return torch_em.default_segmentation_dataset(
133        raw_paths=image_paths,
134        raw_key=None,
135        label_paths=label_paths,
136        label_key=None,
137        patch_shape=patch_shape,
138        is_seg_dataset=False,
139        with_channels=True,
140        **kwargs
141    )
142
143
144def get_brisc_loader(
145    path: Union[os.PathLike, str],
146    batch_size: int,
147    patch_shape: Tuple[int, int],
148    split: Literal["train", "test"] = "train",
149    tumor_type: Optional[Literal["glioma", "meningioma", "pituitary"]] = None,
150    resize_inputs: bool = False,
151    download: bool = False,
152    **kwargs
153) -> DataLoader:
154    """Get the BRISC dataloader for brain tumor segmentation.
155
156    Args:
157        path: Filepath to a folder where the data is downloaded for further processing.
158        batch_size: The batch size for training.
159        patch_shape: The patch shape to use for training.
160        split: The choice of data split. Either 'train' or 'test'.
161        tumor_type: The choice of tumor type. One of 'glioma', 'meningioma', 'pituitary'. By default, all are used.
162        resize_inputs: Whether to resize the inputs.
163        download: Whether to download the data if it is not present.
164        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader.
165
166    Returns:
167        The DataLoader.
168    """
169    ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs)
170    dataset = get_brisc_dataset(path, patch_shape, split, tumor_type, resize_inputs, download, **ds_kwargs)
171    return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)
URL = 'https://ndownloader.figshare.com/files/59298329'
CHECKSUM = '3167e9bdfcb3b2a2502091f41c9bd3f1e2eeb64ec3ad9849680fa760e2e7633d'
TUMOR_TYPES = {'glioma': 'gl', 'meningioma': 'me', 'pituitary': 'pi'}
def get_brisc_data(path: Union[os.PathLike, str], download: bool = False) -> str:
37def get_brisc_data(path: Union[os.PathLike, str], download: bool = False) -> str:
38    """Download the BRISC dataset.
39
40    Args:
41        path: Filepath to a folder where the data is downloaded for further processing.
42        download: Whether to download the data if it is not present.
43
44    Returns:
45        Filepath where the data is downloaded.
46    """
47    data_dir = os.path.join(path, "brisc2025")
48    if os.path.exists(data_dir):
49        return data_dir
50
51    os.makedirs(path, exist_ok=True)
52
53    zip_path = os.path.join(path, "brisc2025.zip")
54    util.download_source(path=zip_path, url=URL, download=download, checksum=CHECKSUM)
55    util.unzip(zip_path=zip_path, dst=path)
56
57    return data_dir

Download the BRISC dataset.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • download: Whether to download the data if it is not present.
Returns:

Filepath where the data is downloaded.

def get_brisc_paths( path: Union[os.PathLike, str], split: Literal['train', 'test'] = 'train', tumor_type: Optional[Literal['glioma', 'meningioma', 'pituitary']] = None, download: bool = False) -> Tuple[List[str], List[str]]:
60def get_brisc_paths(
61    path: Union[os.PathLike, str],
62    split: Literal["train", "test"] = "train",
63    tumor_type: Optional[Literal["glioma", "meningioma", "pituitary"]] = None,
64    download: bool = False,
65) -> Tuple[List[str], List[str]]:
66    """Get paths to the BRISC data.
67
68    Args:
69        path: Filepath to a folder where the data is downloaded for further processing.
70        split: The choice of data split. Either 'train' or 'test'.
71        tumor_type: The choice of tumor type. One of 'glioma', 'meningioma', 'pituitary'. By default, all are used.
72        download: Whether to download the data if it is not present.
73
74    Returns:
75        List of filepaths for the image data.
76        List of filepaths for the label data.
77    """
78    data_dir = get_brisc_data(path=path, download=download)
79
80    if split not in ["train", "test"]:
81        raise ValueError(f"'{split}' is not a valid split choice.")
82
83    image_dir = os.path.join(data_dir, "segmentation_task", split, "images")
84    label_dir = os.path.join(data_dir, "segmentation_task", split, "masks")
85
86    if tumor_type is None:
87        pattern = "*.jpg"
88    elif tumor_type in TUMOR_TYPES:
89        pattern = f"*_{TUMOR_TYPES[tumor_type]}_*.jpg"
90    else:
91        raise ValueError(f"'{tumor_type}' is not a valid tumor type. Choose from {list(TUMOR_TYPES.keys())}.")
92
93    image_paths = natsorted(glob(os.path.join(image_dir, pattern)))
94    label_paths = natsorted(
95        os.path.join(label_dir, os.path.splitext(os.path.basename(p))[0] + ".png") for p in image_paths
96    )
97    assert len(image_paths) > 0 and len(image_paths) == len(label_paths)
98
99    return image_paths, label_paths

Get paths to the BRISC data.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • split: The choice of data split. Either 'train' or 'test'.
  • tumor_type: The choice of tumor type. One of 'glioma', 'meningioma', 'pituitary'. By default, all are used.
  • download: Whether to download the data if it is not present.
Returns:

List of filepaths for the image data. List of filepaths for the label data.

def get_brisc_dataset( path: Union[os.PathLike, str], patch_shape: Tuple[int, int], split: Literal['train', 'test'] = 'train', tumor_type: Optional[Literal['glioma', 'meningioma', 'pituitary']] = None, resize_inputs: bool = False, download: bool = False, **kwargs) -> torch.utils.data.dataset.Dataset:
102def get_brisc_dataset(
103    path: Union[os.PathLike, str],
104    patch_shape: Tuple[int, int],
105    split: Literal["train", "test"] = "train",
106    tumor_type: Optional[Literal["glioma", "meningioma", "pituitary"]] = None,
107    resize_inputs: bool = False,
108    download: bool = False,
109    **kwargs
110) -> Dataset:
111    """Get the BRISC dataset for brain tumor segmentation.
112
113    Args:
114        path: Filepath to a folder where the data is downloaded for further processing.
115        patch_shape: The patch shape to use for training.
116        split: The choice of data split. Either 'train' or 'test'.
117        tumor_type: The choice of tumor type. One of 'glioma', 'meningioma', 'pituitary'. By default, all are used.
118        resize_inputs: Whether to resize the inputs.
119        download: Whether to download the data if it is not present.
120        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`.
121
122    Returns:
123        The segmentation dataset.
124    """
125    image_paths, label_paths = get_brisc_paths(path, split, tumor_type, download)
126
127    if resize_inputs:
128        resize_kwargs = {"patch_shape": patch_shape, "is_rgb": True}
129        kwargs, patch_shape = util.update_kwargs_for_resize_trafo(
130            kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs
131        )
132
133    return torch_em.default_segmentation_dataset(
134        raw_paths=image_paths,
135        raw_key=None,
136        label_paths=label_paths,
137        label_key=None,
138        patch_shape=patch_shape,
139        is_seg_dataset=False,
140        with_channels=True,
141        **kwargs
142    )

Get the BRISC dataset for brain tumor segmentation.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • patch_shape: The patch shape to use for training.
  • split: The choice of data split. Either 'train' or 'test'.
  • tumor_type: The choice of tumor type. One of 'glioma', 'meningioma', 'pituitary'. By default, all are used.
  • resize_inputs: Whether to resize the inputs.
  • download: Whether to download the data if it is not present.
  • kwargs: Additional keyword arguments for torch_em.default_segmentation_dataset.
Returns:

The segmentation dataset.

def get_brisc_loader( path: Union[os.PathLike, str], batch_size: int, patch_shape: Tuple[int, int], split: Literal['train', 'test'] = 'train', tumor_type: Optional[Literal['glioma', 'meningioma', 'pituitary']] = None, resize_inputs: bool = False, download: bool = False, **kwargs) -> torch.utils.data.dataloader.DataLoader:
145def get_brisc_loader(
146    path: Union[os.PathLike, str],
147    batch_size: int,
148    patch_shape: Tuple[int, int],
149    split: Literal["train", "test"] = "train",
150    tumor_type: Optional[Literal["glioma", "meningioma", "pituitary"]] = None,
151    resize_inputs: bool = False,
152    download: bool = False,
153    **kwargs
154) -> DataLoader:
155    """Get the BRISC dataloader for brain tumor segmentation.
156
157    Args:
158        path: Filepath to a folder where the data is downloaded for further processing.
159        batch_size: The batch size for training.
160        patch_shape: The patch shape to use for training.
161        split: The choice of data split. Either 'train' or 'test'.
162        tumor_type: The choice of tumor type. One of 'glioma', 'meningioma', 'pituitary'. By default, all are used.
163        resize_inputs: Whether to resize the inputs.
164        download: Whether to download the data if it is not present.
165        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader.
166
167    Returns:
168        The DataLoader.
169    """
170    ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs)
171    dataset = get_brisc_dataset(path, patch_shape, split, tumor_type, resize_inputs, download, **ds_kwargs)
172    return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)

Get the BRISC dataloader for brain tumor segmentation.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • batch_size: The batch size for training.
  • patch_shape: The patch shape to use for training.
  • split: The choice of data split. Either 'train' or 'test'.
  • tumor_type: The choice of tumor type. One of 'glioma', 'meningioma', 'pituitary'. By default, all are used.
  • resize_inputs: Whether to resize the inputs.
  • download: Whether to download the data if it is not present.
  • kwargs: Additional keyword arguments for torch_em.default_segmentation_dataset or for the PyTorch DataLoader.
Returns:

The DataLoader.