torch_em.data.datasets.medical.brisc
BRISC (BRain tumor Image Segmentation and Classification) is a curated, expert-annotated dataset of contrast-enhanced T1-weighted brain MRI slices for brain tumor segmentation and classification.
The dataset consists of 6,000 slices (5,000 train / 1,000 test) collated from existing public MRI collections, spanning axial, coronal and sagittal planes. It provides physician-reviewed pixel-wise segmentation masks for three tumor types (glioma, meningioma, pituitary tumor), in addition to image-level labels for a 'no tumor' class (which has no corresponding segmentation mask). While the raw images originate from other public collections, the segmentation masks are a new, original annotation contribution.
The dataset is located at https://doi.org/10.6084/m9.figshare.30533120 and is distributed under the CC BY 4.0 license.
This dataset is from the publication https://doi.org/10.1038/s41597-026-06753-y. Please cite it if you use this dataset in your research.
1"""BRISC (BRain tumor Image Segmentation and Classification) is a curated, expert-annotated dataset 2of contrast-enhanced T1-weighted brain MRI slices for brain tumor segmentation and classification. 3 4The dataset consists of 6,000 slices (5,000 train / 1,000 test) collated from existing public MRI 5collections, spanning axial, coronal and sagittal planes. It provides physician-reviewed pixel-wise 6segmentation masks for three tumor types (glioma, meningioma, pituitary tumor), in addition to 7image-level labels for a 'no tumor' class (which has no corresponding segmentation mask). While the 8raw images originate from other public collections, the segmentation masks are a new, original 9annotation contribution. 10 11The dataset is located at https://doi.org/10.6084/m9.figshare.30533120 and is distributed under the 12CC BY 4.0 license. 13 14This dataset is from the publication https://doi.org/10.1038/s41597-026-06753-y. 15Please cite it if you use this dataset in your research. 16""" 17 18import os 19from glob import glob 20from natsort import natsorted 21from typing import Union, Tuple, Optional, Literal, List 22 23from torch.utils.data import Dataset, DataLoader 24 25import torch_em 26 27from .. import util 28 29 30URL = "https://ndownloader.figshare.com/files/59298329" 31CHECKSUM = "3167e9bdfcb3b2a2502091f41c9bd3f1e2eeb64ec3ad9849680fa760e2e7633d" 32 33TUMOR_TYPES = {"glioma": "gl", "meningioma": "me", "pituitary": "pi"} 34 35 36def get_brisc_data(path: Union[os.PathLike, str], download: bool = False) -> str: 37 """Download the BRISC dataset. 38 39 Args: 40 path: Filepath to a folder where the data is downloaded for further processing. 41 download: Whether to download the data if it is not present. 42 43 Returns: 44 Filepath where the data is downloaded. 45 """ 46 data_dir = os.path.join(path, "brisc2025") 47 if os.path.exists(data_dir): 48 return data_dir 49 50 os.makedirs(path, exist_ok=True) 51 52 zip_path = os.path.join(path, "brisc2025.zip") 53 util.download_source(path=zip_path, url=URL, download=download, checksum=CHECKSUM) 54 util.unzip(zip_path=zip_path, dst=path) 55 56 return data_dir 57 58 59def get_brisc_paths( 60 path: Union[os.PathLike, str], 61 split: Literal["train", "test"] = "train", 62 tumor_type: Optional[Literal["glioma", "meningioma", "pituitary"]] = None, 63 download: bool = False, 64) -> Tuple[List[str], List[str]]: 65 """Get paths to the BRISC data. 66 67 Args: 68 path: Filepath to a folder where the data is downloaded for further processing. 69 split: The choice of data split. Either 'train' or 'test'. 70 tumor_type: The choice of tumor type. One of 'glioma', 'meningioma', 'pituitary'. By default, all are used. 71 download: Whether to download the data if it is not present. 72 73 Returns: 74 List of filepaths for the image data. 75 List of filepaths for the label data. 76 """ 77 data_dir = get_brisc_data(path=path, download=download) 78 79 if split not in ["train", "test"]: 80 raise ValueError(f"'{split}' is not a valid split choice.") 81 82 image_dir = os.path.join(data_dir, "segmentation_task", split, "images") 83 label_dir = os.path.join(data_dir, "segmentation_task", split, "masks") 84 85 if tumor_type is None: 86 pattern = "*.jpg" 87 elif tumor_type in TUMOR_TYPES: 88 pattern = f"*_{TUMOR_TYPES[tumor_type]}_*.jpg" 89 else: 90 raise ValueError(f"'{tumor_type}' is not a valid tumor type. Choose from {list(TUMOR_TYPES.keys())}.") 91 92 image_paths = natsorted(glob(os.path.join(image_dir, pattern))) 93 label_paths = natsorted( 94 os.path.join(label_dir, os.path.splitext(os.path.basename(p))[0] + ".png") for p in image_paths 95 ) 96 assert len(image_paths) > 0 and len(image_paths) == len(label_paths) 97 98 return image_paths, label_paths 99 100 101def get_brisc_dataset( 102 path: Union[os.PathLike, str], 103 patch_shape: Tuple[int, int], 104 split: Literal["train", "test"] = "train", 105 tumor_type: Optional[Literal["glioma", "meningioma", "pituitary"]] = None, 106 resize_inputs: bool = False, 107 download: bool = False, 108 **kwargs 109) -> Dataset: 110 """Get the BRISC dataset for brain tumor segmentation. 111 112 Args: 113 path: Filepath to a folder where the data is downloaded for further processing. 114 patch_shape: The patch shape to use for training. 115 split: The choice of data split. Either 'train' or 'test'. 116 tumor_type: The choice of tumor type. One of 'glioma', 'meningioma', 'pituitary'. By default, all are used. 117 resize_inputs: Whether to resize the inputs. 118 download: Whether to download the data if it is not present. 119 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`. 120 121 Returns: 122 The segmentation dataset. 123 """ 124 image_paths, label_paths = get_brisc_paths(path, split, tumor_type, download) 125 126 if resize_inputs: 127 resize_kwargs = {"patch_shape": patch_shape, "is_rgb": True} 128 kwargs, patch_shape = util.update_kwargs_for_resize_trafo( 129 kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs 130 ) 131 132 return torch_em.default_segmentation_dataset( 133 raw_paths=image_paths, 134 raw_key=None, 135 label_paths=label_paths, 136 label_key=None, 137 patch_shape=patch_shape, 138 is_seg_dataset=False, 139 with_channels=True, 140 **kwargs 141 ) 142 143 144def get_brisc_loader( 145 path: Union[os.PathLike, str], 146 batch_size: int, 147 patch_shape: Tuple[int, int], 148 split: Literal["train", "test"] = "train", 149 tumor_type: Optional[Literal["glioma", "meningioma", "pituitary"]] = None, 150 resize_inputs: bool = False, 151 download: bool = False, 152 **kwargs 153) -> DataLoader: 154 """Get the BRISC dataloader for brain tumor segmentation. 155 156 Args: 157 path: Filepath to a folder where the data is downloaded for further processing. 158 batch_size: The batch size for training. 159 patch_shape: The patch shape to use for training. 160 split: The choice of data split. Either 'train' or 'test'. 161 tumor_type: The choice of tumor type. One of 'glioma', 'meningioma', 'pituitary'. By default, all are used. 162 resize_inputs: Whether to resize the inputs. 163 download: Whether to download the data if it is not present. 164 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader. 165 166 Returns: 167 The DataLoader. 168 """ 169 ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs) 170 dataset = get_brisc_dataset(path, patch_shape, split, tumor_type, resize_inputs, download, **ds_kwargs) 171 return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)
37def get_brisc_data(path: Union[os.PathLike, str], download: bool = False) -> str: 38 """Download the BRISC dataset. 39 40 Args: 41 path: Filepath to a folder where the data is downloaded for further processing. 42 download: Whether to download the data if it is not present. 43 44 Returns: 45 Filepath where the data is downloaded. 46 """ 47 data_dir = os.path.join(path, "brisc2025") 48 if os.path.exists(data_dir): 49 return data_dir 50 51 os.makedirs(path, exist_ok=True) 52 53 zip_path = os.path.join(path, "brisc2025.zip") 54 util.download_source(path=zip_path, url=URL, download=download, checksum=CHECKSUM) 55 util.unzip(zip_path=zip_path, dst=path) 56 57 return data_dir
Download the BRISC dataset.
Arguments:
- path: Filepath to a folder where the data is downloaded for further processing.
- download: Whether to download the data if it is not present.
Returns:
Filepath where the data is downloaded.
60def get_brisc_paths( 61 path: Union[os.PathLike, str], 62 split: Literal["train", "test"] = "train", 63 tumor_type: Optional[Literal["glioma", "meningioma", "pituitary"]] = None, 64 download: bool = False, 65) -> Tuple[List[str], List[str]]: 66 """Get paths to the BRISC data. 67 68 Args: 69 path: Filepath to a folder where the data is downloaded for further processing. 70 split: The choice of data split. Either 'train' or 'test'. 71 tumor_type: The choice of tumor type. One of 'glioma', 'meningioma', 'pituitary'. By default, all are used. 72 download: Whether to download the data if it is not present. 73 74 Returns: 75 List of filepaths for the image data. 76 List of filepaths for the label data. 77 """ 78 data_dir = get_brisc_data(path=path, download=download) 79 80 if split not in ["train", "test"]: 81 raise ValueError(f"'{split}' is not a valid split choice.") 82 83 image_dir = os.path.join(data_dir, "segmentation_task", split, "images") 84 label_dir = os.path.join(data_dir, "segmentation_task", split, "masks") 85 86 if tumor_type is None: 87 pattern = "*.jpg" 88 elif tumor_type in TUMOR_TYPES: 89 pattern = f"*_{TUMOR_TYPES[tumor_type]}_*.jpg" 90 else: 91 raise ValueError(f"'{tumor_type}' is not a valid tumor type. Choose from {list(TUMOR_TYPES.keys())}.") 92 93 image_paths = natsorted(glob(os.path.join(image_dir, pattern))) 94 label_paths = natsorted( 95 os.path.join(label_dir, os.path.splitext(os.path.basename(p))[0] + ".png") for p in image_paths 96 ) 97 assert len(image_paths) > 0 and len(image_paths) == len(label_paths) 98 99 return image_paths, label_paths
Get paths to the BRISC data.
Arguments:
- path: Filepath to a folder where the data is downloaded for further processing.
- split: The choice of data split. Either 'train' or 'test'.
- tumor_type: The choice of tumor type. One of 'glioma', 'meningioma', 'pituitary'. By default, all are used.
- download: Whether to download the data if it is not present.
Returns:
List of filepaths for the image data. List of filepaths for the label data.
102def get_brisc_dataset( 103 path: Union[os.PathLike, str], 104 patch_shape: Tuple[int, int], 105 split: Literal["train", "test"] = "train", 106 tumor_type: Optional[Literal["glioma", "meningioma", "pituitary"]] = None, 107 resize_inputs: bool = False, 108 download: bool = False, 109 **kwargs 110) -> Dataset: 111 """Get the BRISC dataset for brain tumor segmentation. 112 113 Args: 114 path: Filepath to a folder where the data is downloaded for further processing. 115 patch_shape: The patch shape to use for training. 116 split: The choice of data split. Either 'train' or 'test'. 117 tumor_type: The choice of tumor type. One of 'glioma', 'meningioma', 'pituitary'. By default, all are used. 118 resize_inputs: Whether to resize the inputs. 119 download: Whether to download the data if it is not present. 120 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`. 121 122 Returns: 123 The segmentation dataset. 124 """ 125 image_paths, label_paths = get_brisc_paths(path, split, tumor_type, download) 126 127 if resize_inputs: 128 resize_kwargs = {"patch_shape": patch_shape, "is_rgb": True} 129 kwargs, patch_shape = util.update_kwargs_for_resize_trafo( 130 kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs 131 ) 132 133 return torch_em.default_segmentation_dataset( 134 raw_paths=image_paths, 135 raw_key=None, 136 label_paths=label_paths, 137 label_key=None, 138 patch_shape=patch_shape, 139 is_seg_dataset=False, 140 with_channels=True, 141 **kwargs 142 )
Get the BRISC dataset for brain tumor segmentation.
Arguments:
- path: Filepath to a folder where the data is downloaded for further processing.
- patch_shape: The patch shape to use for training.
- split: The choice of data split. Either 'train' or 'test'.
- tumor_type: The choice of tumor type. One of 'glioma', 'meningioma', 'pituitary'. By default, all are used.
- resize_inputs: Whether to resize the inputs.
- download: Whether to download the data if it is not present.
- kwargs: Additional keyword arguments for
torch_em.default_segmentation_dataset.
Returns:
The segmentation dataset.
145def get_brisc_loader( 146 path: Union[os.PathLike, str], 147 batch_size: int, 148 patch_shape: Tuple[int, int], 149 split: Literal["train", "test"] = "train", 150 tumor_type: Optional[Literal["glioma", "meningioma", "pituitary"]] = None, 151 resize_inputs: bool = False, 152 download: bool = False, 153 **kwargs 154) -> DataLoader: 155 """Get the BRISC dataloader for brain tumor segmentation. 156 157 Args: 158 path: Filepath to a folder where the data is downloaded for further processing. 159 batch_size: The batch size for training. 160 patch_shape: The patch shape to use for training. 161 split: The choice of data split. Either 'train' or 'test'. 162 tumor_type: The choice of tumor type. One of 'glioma', 'meningioma', 'pituitary'. By default, all are used. 163 resize_inputs: Whether to resize the inputs. 164 download: Whether to download the data if it is not present. 165 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader. 166 167 Returns: 168 The DataLoader. 169 """ 170 ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs) 171 dataset = get_brisc_dataset(path, patch_shape, split, tumor_type, resize_inputs, download, **ds_kwargs) 172 return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)
Get the BRISC dataloader for brain tumor segmentation.
Arguments:
- path: Filepath to a folder where the data is downloaded for further processing.
- batch_size: The batch size for training.
- patch_shape: The patch shape to use for training.
- split: The choice of data split. Either 'train' or 'test'.
- tumor_type: The choice of tumor type. One of 'glioma', 'meningioma', 'pituitary'. By default, all are used.
- resize_inputs: Whether to resize the inputs.
- download: Whether to download the data if it is not present.
- kwargs: Additional keyword arguments for
torch_em.default_segmentation_datasetor for the PyTorch DataLoader.
Returns:
The DataLoader.